|
@@ -0,0 +1,260 @@
|
|
|
|
|
+#include "common/mime_words.hpp"
|
|
|
|
|
+
|
|
|
|
|
+#include "common/string_utils.hpp"
|
|
|
|
|
+
|
|
|
|
|
+#include <algorithm>
|
|
|
|
|
+#include <cctype>
|
|
|
|
|
+#include <sstream>
|
|
|
|
|
+
|
|
|
|
|
+namespace smartbotic::common {
|
|
|
|
|
+
|
|
|
|
|
+namespace {
|
|
|
|
|
+
|
|
|
|
|
+int hexValue(char c) {
|
|
|
|
|
+ if (c >= '0' && c <= '9') return c - '0';
|
|
|
|
|
+ if (c >= 'a' && c <= 'f') return c - 'a' + 10;
|
|
|
|
|
+ if (c >= 'A' && c <= 'F') return c - 'A' + 10;
|
|
|
|
|
+ return -1;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string upper(std::string text) {
|
|
|
|
|
+ std::transform(text.begin(), text.end(), text.begin(),
|
|
|
|
|
+ [](unsigned char c) { return std::toupper(c); });
|
|
|
|
|
+ return text;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// ISO-8859-1 and -2 are still what a good deal of European mail says it is.
|
|
|
|
|
+// Only 8859-1 maps trivially - each byte is the code point - so that is the one
|
|
|
|
|
+// converted; anything else is left as it arrived rather than guessed at.
|
|
|
|
|
+std::string latin1ToUtf8(const std::string& input) {
|
|
|
|
|
+ std::string out;
|
|
|
|
|
+ out.reserve(input.size() * 2);
|
|
|
|
|
+ for (unsigned char c : input) {
|
|
|
|
|
+ if (c < 0x80) {
|
|
|
|
|
+ out.push_back(static_cast<char>(c));
|
|
|
|
|
+ } else {
|
|
|
|
|
+ out.push_back(static_cast<char>(0xC0 | (c >> 6)));
|
|
|
|
|
+ out.push_back(static_cast<char>(0x80 | (c & 0x3F)));
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ return out;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string decodeQ(const std::string& payload) {
|
|
|
|
|
+ std::string out;
|
|
|
|
|
+ out.reserve(payload.size());
|
|
|
|
|
+ for (size_t i = 0; i < payload.size(); ++i) {
|
|
|
|
|
+ const char c = payload[i];
|
|
|
|
|
+ if (c == '_') {
|
|
|
|
|
+ // In an encoded word an underscore is a space, not an underscore.
|
|
|
|
|
+ out.push_back(' ');
|
|
|
|
|
+ } else if (c == '=' && i + 2 < payload.size()) {
|
|
|
|
|
+ const int hi = hexValue(payload[i + 1]);
|
|
|
|
|
+ const int lo = hexValue(payload[i + 2]);
|
|
|
|
|
+ if (hi >= 0 && lo >= 0) {
|
|
|
|
|
+ out.push_back(static_cast<char>((hi << 4) | lo));
|
|
|
|
|
+ i += 2;
|
|
|
|
|
+ } else {
|
|
|
|
|
+ out.push_back(c);
|
|
|
|
|
+ }
|
|
|
|
|
+ } else {
|
|
|
|
|
+ out.push_back(c);
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ return out;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string toUtf8(const std::string& charset, const std::string& bytes) {
|
|
|
|
|
+ const std::string cs = upper(charset);
|
|
|
|
|
+ if (cs == "UTF-8" || cs == "UTF8" || cs == "US-ASCII" || cs == "ASCII") {
|
|
|
|
|
+ return bytes;
|
|
|
|
|
+ }
|
|
|
|
|
+ if (cs == "ISO-8859-1" || cs == "LATIN1" || cs == "ISO8859-1" || cs == "WINDOWS-1252") {
|
|
|
|
|
+ return latin1ToUtf8(bytes);
|
|
|
|
|
+ }
|
|
|
|
|
+ // An unknown charset is passed through. It may render oddly; it will not
|
|
|
|
|
+ // disappear, and the caller can still see what arrived.
|
|
|
|
|
+ return bytes;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+} // namespace
|
|
|
|
|
+
|
|
|
|
|
+bool MimeWords::isAscii(const std::string& value) {
|
|
|
|
|
+ return std::all_of(value.begin(), value.end(),
|
|
|
|
|
+ [](unsigned char c) { return c < 0x80; });
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string MimeWords::decode(const std::string& value) {
|
|
|
|
|
+ if (value.find("=?") == std::string::npos) {
|
|
|
|
|
+ return value;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ std::string out;
|
|
|
|
|
+ out.reserve(value.size());
|
|
|
|
|
+
|
|
|
|
|
+ size_t pos = 0;
|
|
|
|
|
+ bool previous_was_word = false;
|
|
|
|
|
+ while (pos < value.size()) {
|
|
|
|
|
+ const size_t start = value.find("=?", pos);
|
|
|
|
|
+ if (start == std::string::npos) {
|
|
|
|
|
+ out.append(value, pos, std::string::npos);
|
|
|
|
|
+ break;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Whitespace between two encoded words is separator, not content, and
|
|
|
|
|
+ // is dropped - that is how a value too long for one word is rejoined.
|
|
|
|
|
+ std::string between = value.substr(pos, start - pos);
|
|
|
|
|
+ const bool only_space = !between.empty() &&
|
|
|
|
|
+ std::all_of(between.begin(), between.end(),
|
|
|
|
|
+ [](unsigned char c) { return std::isspace(c); });
|
|
|
|
|
+ if (!(previous_was_word && only_space)) {
|
|
|
|
|
+ out += between;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ const size_t charset_end = value.find('?', start + 2);
|
|
|
|
|
+ if (charset_end == std::string::npos) { out += value.substr(start); break; }
|
|
|
|
|
+ const size_t encoding_end = value.find('?', charset_end + 1);
|
|
|
|
|
+ if (encoding_end == std::string::npos) { out += value.substr(start); break; }
|
|
|
|
|
+ const size_t word_end = value.find("?=", encoding_end + 1);
|
|
|
|
|
+ if (word_end == std::string::npos) { out += value.substr(start); break; }
|
|
|
|
|
+
|
|
|
|
|
+ const std::string charset = value.substr(start + 2, charset_end - start - 2);
|
|
|
|
|
+ const std::string encoding = value.substr(charset_end + 1, encoding_end - charset_end - 1);
|
|
|
|
|
+ const std::string payload = value.substr(encoding_end + 1, word_end - encoding_end - 1);
|
|
|
|
|
+
|
|
|
|
|
+ std::string bytes;
|
|
|
|
|
+ if (encoding == "B" || encoding == "b") {
|
|
|
|
|
+ bytes = StringUtils::base64Decode(payload);
|
|
|
|
|
+ } else if (encoding == "Q" || encoding == "q") {
|
|
|
|
|
+ bytes = decodeQ(payload);
|
|
|
|
|
+ } else {
|
|
|
|
|
+ // Not an encoding we know - keep the word as written.
|
|
|
|
|
+ out += value.substr(start, word_end + 2 - start);
|
|
|
|
|
+ pos = word_end + 2;
|
|
|
|
|
+ previous_was_word = false;
|
|
|
|
|
+ continue;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ out += toUtf8(charset, bytes);
|
|
|
|
|
+ pos = word_end + 2;
|
|
|
|
|
+ previous_was_word = true;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ return out;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string MimeWords::encode(const std::string& value) {
|
|
|
|
|
+ if (isAscii(value)) {
|
|
|
|
|
+ return value;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Base64 rather than Q: the values that need encoding here are mostly
|
|
|
|
|
+ // non-ASCII throughout, where Q would escape nearly every character.
|
|
|
|
|
+ //
|
|
|
|
|
+ // A word may be 75 characters including the `=?UTF-8?B?` and `?=` around
|
|
|
|
|
+ // it, so the payload is capped at 45 bytes - a multiple of 3, so no word
|
|
|
|
|
+ // ends mid-base64-group and each decodes on its own. Splitting is done on
|
|
|
|
|
+ // UTF-8 character boundaries so a multi-byte character is never cut in two.
|
|
|
|
|
+ constexpr size_t kMaxBytes = 45;
|
|
|
|
|
+ std::string out;
|
|
|
|
|
+ size_t pos = 0;
|
|
|
|
|
+ while (pos < value.size()) {
|
|
|
|
|
+ size_t take = std::min(kMaxBytes, value.size() - pos);
|
|
|
|
|
+ // Do not split inside a UTF-8 sequence: back off while the next byte is
|
|
|
|
|
+ // a continuation byte.
|
|
|
|
|
+ while (take > 0 && pos + take < value.size() &&
|
|
|
|
|
+ (static_cast<unsigned char>(value[pos + take]) & 0xC0) == 0x80) {
|
|
|
|
|
+ --take;
|
|
|
|
|
+ }
|
|
|
|
|
+ if (take == 0) take = std::min(kMaxBytes, value.size() - pos);
|
|
|
|
|
+
|
|
|
|
|
+ if (!out.empty()) out += "\r\n "; // fold, as a long header must
|
|
|
|
|
+ out += "=?UTF-8?B?" + StringUtils::base64Encode(value.substr(pos, take)) + "?=";
|
|
|
|
|
+ pos += take;
|
|
|
|
|
+ }
|
|
|
|
|
+ return out;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string MimeWords::encodeAddress(const std::string& address) {
|
|
|
|
|
+ // `Display Name <local@domain>`. Only the display name may be touched; the
|
|
|
|
|
+ // address inside the angle brackets must go out exactly as it came.
|
|
|
|
|
+ const size_t open = address.rfind('<');
|
|
|
|
|
+ const size_t close = address.rfind('>');
|
|
|
|
|
+ if (open == std::string::npos || close == std::string::npos || close < open) {
|
|
|
|
|
+ // A bare address, or something we cannot split with confidence.
|
|
|
|
|
+ // Encoding the whole of it would corrupt the address itself, so it is
|
|
|
|
|
+ // left alone even when it is not ASCII.
|
|
|
|
|
+ return address;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ std::string name = address.substr(0, open);
|
|
|
|
|
+ const std::string addr = address.substr(open, close - open + 1);
|
|
|
|
|
+
|
|
|
|
|
+ while (!name.empty() && std::isspace(static_cast<unsigned char>(name.back()))) {
|
|
|
|
|
+ name.pop_back();
|
|
|
|
|
+ }
|
|
|
|
|
+ // Often already quoted; the quotes must not end up inside the encoded word
|
|
|
|
|
+ // or doubled.
|
|
|
|
|
+ if (name.size() >= 2 && name.front() == '"' && name.back() == '"') {
|
|
|
|
|
+ name = name.substr(1, name.size() - 2);
|
|
|
|
|
+ }
|
|
|
|
|
+ if (name.empty()) {
|
|
|
|
|
+ return addr;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ if (!isAscii(name)) {
|
|
|
|
|
+ // An encoded word is its own token and must NOT be wrapped in quotes -
|
|
|
|
|
+ // a quoted encoded word is treated as literal text by a conforming
|
|
|
|
|
+ // reader, which is how a name arrives looking like =?UTF-8?B?...?=.
|
|
|
|
|
+ return encode(name) + " " + addr;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // ASCII, but a display name carrying any of the specials RFC 5322 reserves
|
|
|
|
|
+ // has to be quoted or it breaks the address list it sits in - a comma in
|
|
|
|
|
+ // "Doe, John" otherwise reads as the end of one address and the start of
|
|
|
|
|
+ // another.
|
|
|
|
|
+ const std::string specials = "()<>[]:;@\\,.\"";
|
|
|
|
|
+ if (name.find_first_of(specials) != std::string::npos) {
|
|
|
|
|
+ std::string escaped;
|
|
|
|
|
+ escaped.reserve(name.size() + 2);
|
|
|
|
|
+ for (char c : name) {
|
|
|
|
|
+ if (c == '"' || c == '\\') escaped.push_back('\\');
|
|
|
|
|
+ escaped.push_back(c);
|
|
|
|
|
+ }
|
|
|
|
|
+ return "\"" + escaped + "\" " + addr;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ return name + " " + addr;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string MimeWords::encodeParameter(const std::string& name, const std::string& value) {
|
|
|
|
|
+ if (isAscii(value)) {
|
|
|
|
|
+ return name + "=\"" + value + "\"";
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // RFC 2231. An ASCII-only fallback is emitted alongside for readers that do
|
|
|
|
|
+ // not implement it; a reader that does prefers the starred form.
|
|
|
|
|
+ std::string ascii;
|
|
|
|
|
+ ascii.reserve(value.size());
|
|
|
|
|
+ for (unsigned char c : value) {
|
|
|
|
|
+ ascii.push_back(c < 0x80 ? static_cast<char>(c) : '_');
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ std::ostringstream percent;
|
|
|
|
|
+ percent << std::hex;
|
|
|
|
|
+ for (unsigned char c : value) {
|
|
|
|
|
+ const bool safe = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') ||
|
|
|
|
|
+ (c >= '0' && c <= '9') || c == '.' || c == '-' || c == '_';
|
|
|
|
|
+ if (safe) {
|
|
|
|
|
+ percent << static_cast<char>(c);
|
|
|
|
|
+ } else {
|
|
|
|
|
+ percent << '%' << (c < 16 ? "0" : "")
|
|
|
|
|
+ << upper(std::string(1, "0123456789abcdef"[c >> 4]))
|
|
|
|
|
+ << upper(std::string(1, "0123456789abcdef"[c & 0x0F]));
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ return name + "=\"" + ascii + "\"; " + name + "*=UTF-8''" + percent.str();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+} // namespace smartbotic::common
|