diff --git a/src/core/config/Categories.json b/src/core/config/Categories.json index d3e7648a..d2a7567b 100644 --- a/src/core/config/Categories.json +++ b/src/core/config/Categories.json @@ -83,6 +83,7 @@ "Rison Decode", "To Modhex", "From Modhex", + "MIME Encoding", "MIME Decoding" ] }, diff --git a/src/core/operations/MIMEEncoding.mjs b/src/core/operations/MIMEEncoding.mjs new file mode 100644 index 00000000..84f7d38a --- /dev/null +++ b/src/core/operations/MIMEEncoding.mjs @@ -0,0 +1,149 @@ +/** + * @author skyswordw + * @copyright Crown Copyright 2026 + * @license Apache-2.0 + */ + +import Operation from "../Operation.mjs"; +import OperationError from "../errors/OperationError.mjs"; +import {toBase64} from "../lib/Base64.mjs"; +import cptable from "codepage"; + +const CHARSET_CODEPAGES = { + "UTF-8": 65001, + "US-ASCII": 20127, + "ISO-8859-1": 28591, + "ISO-8859-2": 28592, + "ISO-8859-3": 28593, + "ISO-8859-4": 28594, + "ISO-8859-5": 28595, + "ISO-8859-6": 28596, + "ISO-8859-7": 28597, + "ISO-8859-8": 28598, + "ISO-8859-9": 28599, + "ISO-8859-10": 28600, + "ISO-8859-11": 28601, + "ISO-8859-13": 28603, + "ISO-8859-14": 28604, + "ISO-8859-15": 28605, + "ISO-8859-16": 28606, + }, + TRANSFER_ENCODINGS = ["Base64", "Q-encoding"], + MAX_ENCODED_WORD_LENGTH = 75; + +/** + * MIME Encoding operation + */ +class MIMEEncoding extends Operation { + + /** + * MIMEEncoding constructor + */ + constructor() { + super(); + + this.name = "MIME Encoding"; + this.module = "Default"; + this.description = "Encodes text as MIME encoded-words for non-ASCII email header values."; + this.infoURL = "https://tools.ietf.org/html/rfc2047"; + this.inputType = "string"; + this.outputType = "string"; + this.args = [ + { + name: "Charset", + type: "option", + value: Object.keys(CHARSET_CODEPAGES) + }, + { + name: "Transfer encoding", + type: "option", + value: TRANSFER_ENCODINGS + } + ]; + } + + /** + * @param {string} input + * @param {Object[]} args + * @returns {string} + */ + run(input, args) { + const [charset, encoding] = args, + codepage = CHARSET_CODEPAGES[charset]; + + if (!codepage) throw new OperationError("Invalid charset"); + if (!TRANSFER_ENCODINGS.includes(encoding)) throw new OperationError("Invalid transfer encoding"); + if (!input.length) return ""; + + return input.split(/\r\n|\n|\r/).map(line => { + return this.encodeLine(line, charset, codepage, encoding); + }).join("\r\n"); + } + + /** + * @param {string} input + * @param {string} charset + * @param {number} codepage + * @param {string} encoding + * @returns {string} + */ + encodeLine(input, charset, codepage, encoding) { + const encodedWords = []; + let chunk = ""; + + for (const char of input) { + const next = chunk + char; + if (chunk && this.encodedWord(charset, codepage, encoding, next).length > MAX_ENCODED_WORD_LENGTH) { + encodedWords.push(this.encodedWord(charset, codepage, encoding, chunk)); + chunk = char; + } else { + chunk = next; + } + } + + if (chunk) encodedWords.push(this.encodedWord(charset, codepage, encoding, chunk)); + return encodedWords.join("\r\n "); + } + + /** + * @param {string} charset + * @param {number} codepage + * @param {string} encoding + * @param {string} input + * @returns {string} + */ + encodedWord(charset, codepage, encoding, input) { + const encodedText = encoding === "Base64" ? + toBase64(cptable.utils.encode(codepage, input)) : + this.qEncode(cptable.utils.encode(codepage, input)); + + return `=?${charset}?${encoding === "Base64" ? "B" : "Q"}?${encodedText}?=`; + } + + /** + * @param {Uint8Array|byteArray} input + * @returns {string} + */ + qEncode(input) { + let output = ""; + for (const byte of input) { + if (byte === 0x20) { + output += "_"; + } else if ( + byte >= 0x21 && + byte <= 0x7e && + byte !== 0x3d && + byte !== 0x3f && + byte !== 0x5f + ) { + output += String.fromCharCode(byte); + } else { + output += `=${byte.toString(16).toUpperCase().padStart(2, "0")}`; + } + } + return output; + } + +} + +export default MIMEEncoding; diff --git a/tests/node/tests/operations.mjs b/tests/node/tests/operations.mjs index 6cf85718..b4496bd9 100644 --- a/tests/node/tests/operations.mjs +++ b/tests/node/tests/operations.mjs @@ -681,6 +681,13 @@ WWFkYSBZYWRh\r assert.strictEqual(chef.LZNT1Decompress("\x1a\xb0\x00compress\x00edtestda\x04ta\x07\x88alot").toString(), "compressedtestdatacompressedalot"); }), + it("MIME Encoding", () => { + assert.strictEqual(chef.MIMEEncoding("Keld Jørn Simonsen", { + charset: "ISO-8859-1", + transferEncoding: "Q-encoding", + }).toString(), "=?ISO-8859-1?Q?Keld_J=F8rn_Simonsen?="); + }), + it("MD6", () => { assert.strictEqual(chef.MD6("Head Over Heels", {key: "arty"}).toString(), "d8f7fe4931fbaa37316f76283d5f615f50ddd54afdc794b61da522556aee99ad"); }), @@ -1178,4 +1185,3 @@ ExifImageHeight: 57`); ]); - diff --git a/tests/operations/tests/MIMEDecoding.mjs b/tests/operations/tests/MIMEDecoding.mjs index b99fc489..bfa6516c 100644 --- a/tests/operations/tests/MIMEDecoding.mjs +++ b/tests/operations/tests/MIMEDecoding.mjs @@ -1,5 +1,5 @@ /** - * MIME Header Decoding tests + * MIME Header Encoding and Decoding tests * * @author mshwed [m@ttshwed.com] * @copyright Crown Copyright 2019 @@ -9,6 +9,65 @@ import TestRegister from "../../lib/TestRegister.mjs"; TestRegister.addTests([ + { + name: "MIME Encoding: UTF-8 Base64", + input: "Éric ", + expectedOutput: "=?UTF-8?B?w4lyaWMgPGVyaWNAZXhhbXBsZS5vcmc+?=", + recipeConfig: [ + { + "op": "MIME Encoding", + "args": ["UTF-8", "Base64"] + } + ] + }, + { + name: "MIME Encoding: UTF-8 Q-encoding", + input: "Éric ", + expectedOutput: "=?UTF-8?Q?=C3=89ric_?=", + recipeConfig: [ + { + "op": "MIME Encoding", + "args": ["UTF-8", "Q-encoding"] + } + ] + }, + { + name: "MIME Encoding: ISO-8859-1 Q-encoding", + input: "Keld Jørn Simonsen", + expectedOutput: "=?ISO-8859-1?Q?Keld_J=F8rn_Simonsen?=", + recipeConfig: [ + { + "op": "MIME Encoding", + "args": ["ISO-8859-1", "Q-encoding"] + } + ] + }, + { + name: "MIME Encoding: long Q-encoded header round trip", + input: "This is a deliberately long MIME header value with Éric and Anaïs in it.", + expectedOutput: "This is a deliberately long MIME header value with Éric and Anaïs in it.", + recipeConfig: [ + { + "op": "MIME Encoding", + "args": ["UTF-8", "Q-encoding"] + }, + { + "op": "MIME Decoding", + "args": [] + } + ] + }, + { + name: "MIME Encoding: long Base64 header is folded", + input: "This is a deliberately long MIME header value with Eric and Anais in it.", + expectedOutput: "=?US-ASCII?B?VGhpcyBpcyBhIGRlbGliZXJhdGVseSBsb25nIE1JTUUgaGVhZGVyIHZhbHVl?=\r\n =?US-ASCII?B?IHdpdGggRXJpYyBhbmQgQW5haXMgaW4gaXQu?=", + recipeConfig: [ + { + "op": "MIME Encoding", + "args": ["US-ASCII", "Base64"] + } + ] + }, { name: "Encoded comments", input: "(=?ISO-8859-1?Q?a?=)",