Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
85 changes: 81 additions & 4 deletions crates/core/src/host/v8/builtins/text_encoding.js
Original file line number Diff line number Diff line change
Expand Up @@ -28,10 +28,31 @@ globalThis.TextDecoder = class TextDecoder {
* @argument {any} options
*/
constructor(label = 'utf-8', options = {}) {
if (label !== 'utf-8') {
throw new RangeError('The encoding label provided is invalid');
// Encoding labels are ASCII case-insensitive, with ASCII whitespace trimmed.
// See https://encoding.spec.whatwg.org/#names-and-labels.
switch (
`${label}`.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, '').toLowerCase()
) {
case 'unicode-1-1-utf-8':
case 'unicode11utf8':
case 'unicode20utf8':
case 'utf-8':
case 'utf8':
case 'x-unicode20utf8':
this.#encoding = 'utf-8';
break;
case 'csunicode':
case 'iso-10646-ucs-2':
case 'ucs-2':
case 'unicode':
case 'unicodefeff':
case 'utf-16':
case 'utf-16le':
this.#encoding = 'utf-16le';
break;
default:
throw new RangeError('The encoding label provided is invalid');
}
this.#encoding = label;
this.#fatal = !!options.fatal;
if (options.ignoreBOM) {
throw new TypeError("Option 'ignoreBOM' not supported");
Expand All @@ -52,13 +73,69 @@ globalThis.TextDecoder = class TextDecoder {
* @argument {any} input
* @argument {any} options
*/
decode(input, options = {}) {
decode(input = new Uint8Array(), options = {}) {
if (options.stream) {
throw new TypeError("Option 'stream' not supported");
}
if (input instanceof ArrayBuffer || input instanceof SharedArrayBuffer) {
input = new Uint8Array(input);
}
if (this.#encoding === 'utf-16le') {
return utf16le_decode(input, this.#fatal);
}
return utf8_decode(input, this.#fatal);
}
};

/**
* Non-streaming UTF-16LE decoding. Do not reinterpret the input as a Uint16Array:
* views can start at odd offsets and the host's byte order need not be LE.
* @argument {ArrayBufferView} input
* @argument {boolean} fatal
*/
function utf16le_decode(input, fatal) {
if (!ArrayBuffer.isView(input)) {
throw new TypeError('argument is not an `ArrayBuffer` or a view on one');
}
const bytes = new DataView(input.buffer, input.byteOffset, input.byteLength);
let output = '';
const error = () => {
if (fatal) {
throw new TypeError('The encoded data is not valid UTF-16LE');
}
output += '\uFFFD';
};
let offset = 0;
while (offset + 1 < bytes.byteLength) {
const unit = bytes.getUint16(offset, true);
offset += 2;
// Strip only a leading BOM, on each independent decode call.
if (offset === 2 && unit === 0xfeff) {
continue;
}
if (unit >= 0xd800 && unit <= 0xdbff) {
if (offset + 1 >= bytes.byteLength) {
// A pending surrogate and an odd trailing byte are one EOF error.
error();
offset = bytes.byteLength;
break;
}
const next = bytes.getUint16(offset, true);
if (next >= 0xdc00 && next <= 0xdfff) {
output += String.fromCharCode(unit, next);
offset += 2;
} else {
// Leave the next code unit to be processed again after the error.
error();
}
} else if (unit >= 0xdc00 && unit <= 0xdfff) {
error();
} else {
output += String.fromCharCode(unit);
}
}
if (offset < bytes.byteLength) {
error();
}
return output;
}
151 changes: 151 additions & 0 deletions crates/core/src/host/v8/builtins/text_encoding.test.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,151 @@
// Evaluated as a user module by the V8 host test, after installing the builtins.
function equal(actual, expected) {
if (actual !== expected) {
throw new Error(
`Expected ${JSON.stringify(expected)}, got ${JSON.stringify(actual)}`
);
}
}

function throws(action, type) {
try {
action();
} catch (error) {
if (error instanceof type) {
return;
}
throw error;
}
throw new Error(`Expected ${type.name}`);
}

// h3-js's generated browser bundle constructs both of these eagerly during
// module evaluation, including the unused UTF16Decoder (issue #5994).
const UTF8Decoder = new TextDecoder('utf8');
const UTF16Decoder = new TextDecoder('utf-16le');
equal(UTF8Decoder.encoding, 'utf-8');
equal(UTF16Decoder.encoding, 'utf-16le');
equal(UTF8Decoder.decode(new Uint8Array([0x48, 0xc3, 0xa9])), 'Hé');
equal(UTF16Decoder.decode(new Uint8Array([0x48, 0, 0xe9, 0])), 'Hé');

for (const [encoding, labels] of [
[
'utf-8',
[
'unicode-1-1-utf-8',
'unicode11utf8',
'unicode20utf8',
'utf-8',
'utf8',
'x-unicode20utf8',
],
],
[
'utf-16le',
[
'csunicode',
'iso-10646-ucs-2',
'ucs-2',
'unicode',
'unicodefeff',
'utf-16',
'utf-16le',
],
],
]) {
for (const label of labels) {
equal(new TextDecoder(label).encoding, encoding);
equal(
new TextDecoder(` \t\n\f\r${label.toUpperCase()}\r\f\n\t `).encoding,
encoding
);
}
}
equal(new TextDecoder().encoding, 'utf-8');
equal(new TextDecoder({ toString: () => 'UTF8' }).encoding, 'utf-8');
throws(() => new TextDecoder(Symbol()), TypeError);
for (const label of [
'',
'utf-16be',
'latin1',
'replacement',
'utf_8',
'\u00a0utf8',
'utf8\u000b',
'utf8\uFEFF',
]) {
throws(() => new TextDecoder(label), RangeError);
}

for (const decoder of [UTF8Decoder, UTF16Decoder]) {
equal(decoder.fatal, false);
equal(decoder.ignoreBOM, false);
equal(decoder.decode(), '');
equal(decoder.decode(new Uint8Array()), '');
throws(() => decoder.decode(null), TypeError);
throws(() => decoder.decode([0x41, 0]), TypeError);
throws(() => decoder.decode(new Uint8Array(), { stream: true }), TypeError);
throws(
() => new TextDecoder(decoder.encoding, { ignoreBOM: true }),
TypeError
);
}

// Buffer sources must decode the bytes in the view, including odd offsets.
const buffer = new Uint8Array([0xff, 0x41, 0, 0x3d, 0xd8, 0, 0xde, 0xff]);
equal(UTF16Decoder.decode(buffer.subarray(1, 7)), 'A😀');
equal(UTF16Decoder.decode(new DataView(buffer.buffer, 1, 6)), 'A😀');
equal(UTF16Decoder.decode(buffer.slice(1, 7).buffer), 'A😀');
const words = new Uint16Array(3);
new Uint8Array(words.buffer).set([0x41, 0, 0x3d, 0xd8, 0, 0xde]);
equal(UTF16Decoder.decode(words), 'A😀');
const shared = new SharedArrayBuffer(6);
new Uint8Array(shared).set([0x41, 0, 0x3d, 0xd8, 0, 0xde]);
equal(UTF16Decoder.decode(shared), 'A😀');
equal(UTF16Decoder.decode(new DataView(shared, 2, 4)), '😀');

equal(
UTF16Decoder.decode(new Uint8Array([0xff, 0xfe, 0x41, 0, 0xff, 0xfe])),
'A\uFEFF'
);
equal(UTF16Decoder.decode(new Uint8Array([0xff, 0xfe])), '');
equal(UTF16Decoder.decode(new Uint8Array([0xff, 0xfe, 0x42, 0])), 'B');
// A big-endian BOM does not change the selected encoding.
equal(
UTF16Decoder.decode(new Uint8Array([0xfe, 0xff, 0, 0x41])),
'\uFFFE\u4100'
);
equal(UTF16Decoder.decode(new Uint8Array([0, 0, 0xff, 0xff])), '\0\uFFFF');

const fatal = new TextDecoder('utf-16le', { fatal: true });
equal(fatal.fatal, true);
equal(fatal.decode(new Uint8Array([0xff, 0xfe, 0x3d, 0xd8, 0, 0xde])), '😀');
for (const [bytes, expected] of [
[[0x41], '\uFFFD'],
[[0x41, 0, 0x42], 'A\uFFFD'],
[[0, 0xdc], '\uFFFD'],
[[0, 0xd8], '\uFFFD'],
[[0, 0xd8, 0x41], '\uFFFD'],
[[0, 0xd8, 0x41, 0], '\uFFFDA'],
[[0, 0xd8, 0, 0xd8, 0, 0xdc], '\uFFFD𐀀'],
[[0, 0xdc, 0, 0xdc], '\uFFFD\uFFFD'],
[[0xff, 0xfe, 0, 0xd8], '\uFFFD'],
]) {
const input = new Uint8Array(bytes);
equal(UTF16Decoder.decode(input), expected);
throws(() => fatal.decode(input), TypeError);
// Independent calls reset decoding after a fatal error.
equal(fatal.decode(new Uint8Array([0x41, 0])), 'A');
}

equal(UTF8Decoder.decode(new Uint8Array([0xff])), '\uFFFD');
throws(
() => new TextDecoder('utf8', { fatal: true }).decode(new Uint8Array([0xff])),
TypeError
);
// Large inputs should not hit the argument-count limit of fromCharCode(...).
const large = new Uint8Array(140000);
for (let i = 0; i < large.length; i += 2) {
large[i] = 0x41;
}
equal(UTF16Decoder.decode(large), 'A'.repeat(70000));
12 changes: 12 additions & 0 deletions crates/core/src/host/v8/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2146,6 +2146,18 @@ mod test {
})
}

#[test]
fn text_decoder_dependency_construction_and_decoding() {
with_scope(|scope| {
catch_exception(scope, |scope| {
builtins::evaluate_builtins(scope)?;
eval_user_module(scope, include_str!("builtins/text_encoding.test.js"))?;
Ok(())
})
.expect("TextDecoder regression module should evaluate successfully");
});
}

#[test]
fn call_call_reducer_works() {
let call = |code| {
Expand Down