From 402ac6e884b5e3783217c75cc190a8b7d698a7cf Mon Sep 17 00:00:00 2001 From: Anubhav Harsh Sinha Date: Fri, 2 Oct 2026 00:01:52 +0530 Subject: [PATCH] fix(v8): accept UTF-8 labels and decode UTF-16LE --- .../src/host/v8/builtins/text_encoding.js | 85 +++++++++- .../host/v8/builtins/text_encoding.test.js | 151 ++++++++++++++++++ crates/core/src/host/v8/mod.rs | 12 ++ 3 files changed, 244 insertions(+), 4 deletions(-) create mode 100644 crates/core/src/host/v8/builtins/text_encoding.test.js diff --git a/crates/core/src/host/v8/builtins/text_encoding.js b/crates/core/src/host/v8/builtins/text_encoding.js index 1a343e04d34..f1990ffda57 100644 --- a/crates/core/src/host/v8/builtins/text_encoding.js +++ b/crates/core/src/host/v8/builtins/text_encoding.js @@ -28,10 +28,31 @@ globalThis.TextDecoder = class TextDecoder { * @argument {any} options */ constructor(label = 'utf-8', options = {}) { - if (label !== 'utf-8') { - throw new RangeError('The encoding label provided is invalid'); + // Encoding labels are ASCII case-insensitive, with ASCII whitespace trimmed. + // See https://encoding.spec.whatwg.org/#names-and-labels. + switch ( + `${label}`.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, '').toLowerCase() + ) { + case 'unicode-1-1-utf-8': + case 'unicode11utf8': + case 'unicode20utf8': + case 'utf-8': + case 'utf8': + case 'x-unicode20utf8': + this.#encoding = 'utf-8'; + break; + case 'csunicode': + case 'iso-10646-ucs-2': + case 'ucs-2': + case 'unicode': + case 'unicodefeff': + case 'utf-16': + case 'utf-16le': + this.#encoding = 'utf-16le'; + break; + default: + throw new RangeError('The encoding label provided is invalid'); } - this.#encoding = label; this.#fatal = !!options.fatal; if (options.ignoreBOM) { throw new TypeError("Option 'ignoreBOM' not supported"); @@ -52,13 +73,69 @@ globalThis.TextDecoder = class TextDecoder { * @argument {any} input * @argument {any} options */ - decode(input, options = {}) { + decode(input = new Uint8Array(), options = {}) { if (options.stream) { throw new TypeError("Option 'stream' not supported"); } if (input instanceof ArrayBuffer || input instanceof SharedArrayBuffer) { input = new Uint8Array(input); } + if (this.#encoding === 'utf-16le') { + return utf16le_decode(input, this.#fatal); + } return utf8_decode(input, this.#fatal); } }; + +/** + * Non-streaming UTF-16LE decoding. Do not reinterpret the input as a Uint16Array: + * views can start at odd offsets and the host's byte order need not be LE. + * @argument {ArrayBufferView} input + * @argument {boolean} fatal + */ +function utf16le_decode(input, fatal) { + if (!ArrayBuffer.isView(input)) { + throw new TypeError('argument is not an `ArrayBuffer` or a view on one'); + } + const bytes = new DataView(input.buffer, input.byteOffset, input.byteLength); + let output = ''; + const error = () => { + if (fatal) { + throw new TypeError('The encoded data is not valid UTF-16LE'); + } + output += '\uFFFD'; + }; + let offset = 0; + while (offset + 1 < bytes.byteLength) { + const unit = bytes.getUint16(offset, true); + offset += 2; + // Strip only a leading BOM, on each independent decode call. + if (offset === 2 && unit === 0xfeff) { + continue; + } + if (unit >= 0xd800 && unit <= 0xdbff) { + if (offset + 1 >= bytes.byteLength) { + // A pending surrogate and an odd trailing byte are one EOF error. + error(); + offset = bytes.byteLength; + break; + } + const next = bytes.getUint16(offset, true); + if (next >= 0xdc00 && next <= 0xdfff) { + output += String.fromCharCode(unit, next); + offset += 2; + } else { + // Leave the next code unit to be processed again after the error. + error(); + } + } else if (unit >= 0xdc00 && unit <= 0xdfff) { + error(); + } else { + output += String.fromCharCode(unit); + } + } + if (offset < bytes.byteLength) { + error(); + } + return output; +} diff --git a/crates/core/src/host/v8/builtins/text_encoding.test.js b/crates/core/src/host/v8/builtins/text_encoding.test.js new file mode 100644 index 00000000000..7adff0d8681 --- /dev/null +++ b/crates/core/src/host/v8/builtins/text_encoding.test.js @@ -0,0 +1,151 @@ +// Evaluated as a user module by the V8 host test, after installing the builtins. +function equal(actual, expected) { + if (actual !== expected) { + throw new Error( + `Expected ${JSON.stringify(expected)}, got ${JSON.stringify(actual)}` + ); + } +} + +function throws(action, type) { + try { + action(); + } catch (error) { + if (error instanceof type) { + return; + } + throw error; + } + throw new Error(`Expected ${type.name}`); +} + +// h3-js's generated browser bundle constructs both of these eagerly during +// module evaluation, including the unused UTF16Decoder (issue #5994). +const UTF8Decoder = new TextDecoder('utf8'); +const UTF16Decoder = new TextDecoder('utf-16le'); +equal(UTF8Decoder.encoding, 'utf-8'); +equal(UTF16Decoder.encoding, 'utf-16le'); +equal(UTF8Decoder.decode(new Uint8Array([0x48, 0xc3, 0xa9])), 'Hé'); +equal(UTF16Decoder.decode(new Uint8Array([0x48, 0, 0xe9, 0])), 'Hé'); + +for (const [encoding, labels] of [ + [ + 'utf-8', + [ + 'unicode-1-1-utf-8', + 'unicode11utf8', + 'unicode20utf8', + 'utf-8', + 'utf8', + 'x-unicode20utf8', + ], + ], + [ + 'utf-16le', + [ + 'csunicode', + 'iso-10646-ucs-2', + 'ucs-2', + 'unicode', + 'unicodefeff', + 'utf-16', + 'utf-16le', + ], + ], +]) { + for (const label of labels) { + equal(new TextDecoder(label).encoding, encoding); + equal( + new TextDecoder(` \t\n\f\r${label.toUpperCase()}\r\f\n\t `).encoding, + encoding + ); + } +} +equal(new TextDecoder().encoding, 'utf-8'); +equal(new TextDecoder({ toString: () => 'UTF8' }).encoding, 'utf-8'); +throws(() => new TextDecoder(Symbol()), TypeError); +for (const label of [ + '', + 'utf-16be', + 'latin1', + 'replacement', + 'utf_8', + '\u00a0utf8', + 'utf8\u000b', + 'utf8\uFEFF', +]) { + throws(() => new TextDecoder(label), RangeError); +} + +for (const decoder of [UTF8Decoder, UTF16Decoder]) { + equal(decoder.fatal, false); + equal(decoder.ignoreBOM, false); + equal(decoder.decode(), ''); + equal(decoder.decode(new Uint8Array()), ''); + throws(() => decoder.decode(null), TypeError); + throws(() => decoder.decode([0x41, 0]), TypeError); + throws(() => decoder.decode(new Uint8Array(), { stream: true }), TypeError); + throws( + () => new TextDecoder(decoder.encoding, { ignoreBOM: true }), + TypeError + ); +} + +// Buffer sources must decode the bytes in the view, including odd offsets. +const buffer = new Uint8Array([0xff, 0x41, 0, 0x3d, 0xd8, 0, 0xde, 0xff]); +equal(UTF16Decoder.decode(buffer.subarray(1, 7)), 'A😀'); +equal(UTF16Decoder.decode(new DataView(buffer.buffer, 1, 6)), 'A😀'); +equal(UTF16Decoder.decode(buffer.slice(1, 7).buffer), 'A😀'); +const words = new Uint16Array(3); +new Uint8Array(words.buffer).set([0x41, 0, 0x3d, 0xd8, 0, 0xde]); +equal(UTF16Decoder.decode(words), 'A😀'); +const shared = new SharedArrayBuffer(6); +new Uint8Array(shared).set([0x41, 0, 0x3d, 0xd8, 0, 0xde]); +equal(UTF16Decoder.decode(shared), 'A😀'); +equal(UTF16Decoder.decode(new DataView(shared, 2, 4)), '😀'); + +equal( + UTF16Decoder.decode(new Uint8Array([0xff, 0xfe, 0x41, 0, 0xff, 0xfe])), + 'A\uFEFF' +); +equal(UTF16Decoder.decode(new Uint8Array([0xff, 0xfe])), ''); +equal(UTF16Decoder.decode(new Uint8Array([0xff, 0xfe, 0x42, 0])), 'B'); +// A big-endian BOM does not change the selected encoding. +equal( + UTF16Decoder.decode(new Uint8Array([0xfe, 0xff, 0, 0x41])), + '\uFFFE\u4100' +); +equal(UTF16Decoder.decode(new Uint8Array([0, 0, 0xff, 0xff])), '\0\uFFFF'); + +const fatal = new TextDecoder('utf-16le', { fatal: true }); +equal(fatal.fatal, true); +equal(fatal.decode(new Uint8Array([0xff, 0xfe, 0x3d, 0xd8, 0, 0xde])), '😀'); +for (const [bytes, expected] of [ + [[0x41], '\uFFFD'], + [[0x41, 0, 0x42], 'A\uFFFD'], + [[0, 0xdc], '\uFFFD'], + [[0, 0xd8], '\uFFFD'], + [[0, 0xd8, 0x41], '\uFFFD'], + [[0, 0xd8, 0x41, 0], '\uFFFDA'], + [[0, 0xd8, 0, 0xd8, 0, 0xdc], '\uFFFD𐀀'], + [[0, 0xdc, 0, 0xdc], '\uFFFD\uFFFD'], + [[0xff, 0xfe, 0, 0xd8], '\uFFFD'], +]) { + const input = new Uint8Array(bytes); + equal(UTF16Decoder.decode(input), expected); + throws(() => fatal.decode(input), TypeError); + // Independent calls reset decoding after a fatal error. + equal(fatal.decode(new Uint8Array([0x41, 0])), 'A'); +} + +equal(UTF8Decoder.decode(new Uint8Array([0xff])), '\uFFFD'); +throws( + () => new TextDecoder('utf8', { fatal: true }).decode(new Uint8Array([0xff])), + TypeError +); +// Large inputs should not hit the argument-count limit of fromCharCode(...). +const large = new Uint8Array(140000); +for (let i = 0; i < large.length; i += 2) { + large[i] = 0x41; +} +equal(UTF16Decoder.decode(large), 'A'.repeat(70000)); diff --git a/crates/core/src/host/v8/mod.rs b/crates/core/src/host/v8/mod.rs index 2b7bd84cf5f..7a802b35d47 100644 --- a/crates/core/src/host/v8/mod.rs +++ b/crates/core/src/host/v8/mod.rs @@ -2146,6 +2146,18 @@ mod test { }) } + #[test] + fn text_decoder_dependency_construction_and_decoding() { + with_scope(|scope| { + catch_exception(scope, |scope| { + builtins::evaluate_builtins(scope)?; + eval_user_module(scope, include_str!("builtins/text_encoding.test.js"))?; + Ok(()) + }) + .expect("TextDecoder regression module should evaluate successfully"); + }); + } + #[test] fn call_call_reducer_works() { let call = |code| {