Skip to content

Commit 9a13607

Browse files
committed
lib: implement WHATWG-spec Big5 decoder in js
TextDecoder('big5') was routed through ICU's Big5 converter, whose table includes vendor/PUA mappings the WHATWG Encoding Standard does not define. Byte sequences the standard defines as invalid (and which must decode to U+FFFD, or throw when fatal) instead decoded to those extra ICU-only characters, e.g. bytes 0x83 0x5C decoded to U+F00E instead of U+FFFD U+005C, even with `fatal: true`. This adds a small, self-contained decoder for the 'big5' label (and its aliases, which the standard maps to the same decoder) that implements the algorithm and index table from the Encoding Standard directly, mirroring how single-byte.js already reimplements the legacy single-byte encodings instead of relying on ICU for them. It does not touch the ICU-backed path used by any other encoding. The decoder module is required lazily from internal/encoding.js, only when a Big5 TextDecoder is constructed, so its large base64 index table is not pulled into the startup snapshot; internal/encoding.js itself is loaded during bootstrap. Refs: https://encoding.spec.whatwg.org/#big5-decoder Refs: #61041 Fixes: #40091 Signed-off-by: agape1225 <49804691+agape1225@users.noreply.github.com>
1 parent a1bfce9 commit 9a13607

3 files changed

Lines changed: 236 additions & 0 deletions

File tree

‎lib/internal/encoding.js‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -24,6 +24,7 @@ const {
2424
ERR_NO_ICU,
2525
} = require('internal/errors').codes;
2626
const kSingleByte = Symbol('single-byte');
27+
const kBig5 = Symbol('big5');
2728
const kHandle = Symbol('handle');
2829
const kFlags = Symbol('flags');
2930
const kEncoding = Symbol('encoding');
@@ -449,13 +450,24 @@ class TextDecoder {
449450
this[kUTF8FastPath] = false;
450451
this[kHandle] = undefined;
451452
this[kSingleByte] = undefined; // Does not care about streaming or BOM
453+
this[kBig5] = false;
452454
this[kChunk] = null; // A copy of previous streaming tail or null
453455

454456
if (enc === 'utf-8') {
455457
this[kUTF8FastPath] = true;
456458
this[kBOMSeen] = false;
457459
} else if (isSinglebyteEncoding(enc)) {
458460
this[kSingleByte] = createSinglebyteDecoder(enc, this[kFatal]);
461+
} else if (enc === 'big5') {
462+
// Not routed through ICU: ICU's own Big5 conversion table includes
463+
// vendor/PUA mappings the WHATWG Encoding Standard does not, so byte
464+
// sequences the standard defines as invalid would otherwise decode to
465+
// those extra characters instead of U+FFFD.
466+
// Loaded lazily: the module carries a large index table that should not
467+
// be pulled into the startup snapshot for the common non-Big5 case.
468+
const { createBig5Decoder } = require('internal/encoding/big5');
469+
this[kBig5] = true;
470+
this[kHandle] = createBig5Decoder(this[kFatal]);
459471
} else {
460472
this.#prepareConverter(); // Need to throw early if we don't support the encoding
461473
}
@@ -485,6 +497,12 @@ class TextDecoder {
485497
if (this[kSingleByte]) return this[kSingleByte](parseInput(input));
486498

487499
const stream = options?.stream;
500+
501+
if (this[kBig5]) {
502+
input = parseInput(input);
503+
return stream ? this[kHandle].write(input) : this[kHandle].end(input);
504+
}
505+
488506
if (this[kUTF8FastPath]) {
489507
const chunk = this[kChunk];
490508
const ignoreBom = this[kIgnoreBOM] || this[kBOMSeen];

‎lib/internal/encoding/big5.js‎

Lines changed: 116 additions & 0 deletions
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.
Lines changed: 102 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,102 @@
1+
'use strict';
2+
3+
// Big5 is decoded by a WHATWG-spec-faithful implementation (lib/internal/encoding/big5.js)
4+
// rather than through ICU, because ICU's own Big5 conversion table includes
5+
// vendor/PUA mappings the WHATWG Encoding Standard does not: byte sequences
6+
// the standard defines as invalid decode to those extra characters instead
7+
// of U+FFFD when routed through ICU.
8+
// Refs: https://github.com/nodejs/node/issues/61041
9+
// Refs: https://github.com/nodejs/node/issues/40091
10+
// Refs: https://encoding.spec.whatwg.org/#big5-decoder
11+
12+
require('../common');
13+
const assert = require('assert');
14+
15+
function codePoints(str) {
16+
return [...str].map((c) => c.codePointAt(0));
17+
}
18+
19+
for (const label of ['big5', 'Big5', 'BIG5', 'big5-hkscs', 'cn-big5', 'csbig5', 'x-x-big5']) {
20+
const decoder = new TextDecoder(label);
21+
assert.strictEqual(decoder.encoding, 'big5');
22+
}
23+
24+
// ASCII round-trips as-is.
25+
{
26+
const decoder = new TextDecoder('big5');
27+
assert.strictEqual(decoder.decode(Uint8Array.from([0x41, 0x42, 0x43])), 'ABC');
28+
}
29+
30+
// A valid 2-byte Big5 sequence decodes to the expected code point.
31+
// 0xA4 0x40 is the first Hanzi in the Big5 table: U+4E00 ("一", "one").
32+
{
33+
const decoder = new TextDecoder('big5');
34+
assert.strictEqual(decoder.decode(Uint8Array.from([0xa4, 0x40])), '一');
35+
}
36+
37+
// Regression test: an unassigned Big5 pointer must decode to U+FFFD, not to
38+
// whatever extra character ICU's own (non-spec) Big5 table maps it to.
39+
// https://github.com/nodejs/node/issues/40091
40+
{
41+
const decoder = new TextDecoder('big5');
42+
const result = decoder.decode(Uint8Array.from([0x41, 0x42, 0x83, 0x5c, 0x43, 0x44]));
43+
assert.deepStrictEqual(codePoints(result), [0x41, 0x42, 0xfffd, 0x5c, 0x43, 0x44]);
44+
}
45+
46+
// A lead byte with no trailing byte (end of input) is also an error.
47+
{
48+
const decoder = new TextDecoder('big5');
49+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0xa4]))), [0xfffd]);
50+
}
51+
52+
// `fatal: true` must throw instead of substituting U+FFFD, and must
53+
// actually recognize this sequence as invalid (unlike plain ICU, which
54+
// treats it as valid and never throws even in fatal mode).
55+
{
56+
const decoder = new TextDecoder('big5', { fatal: true });
57+
assert.throws(() => {
58+
decoder.decode(Uint8Array.from([0x83, 0x5c]));
59+
}, { name: 'TypeError', code: 'ERR_ENCODING_INVALID_ENCODED_DATA' });
60+
}
61+
62+
// The four Big5 pointers that map to two combining code points instead of
63+
// one, per the spec's special-cased steps ahead of the index table lookup.
64+
{
65+
const decoder = new TextDecoder('big5');
66+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0x62]))), [0xca, 0x0304]);
67+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0x64]))), [0xca, 0x030c]);
68+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0xa3]))), [0xea, 0x0304]);
69+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0xa5]))), [0xea, 0x030c]);
70+
}
71+
72+
// Streaming: a valid 2-byte sequence split across chunk boundaries must
73+
// still decode correctly, and must not be flushed early.
74+
{
75+
const decoder = new TextDecoder('big5');
76+
const r1 = decoder.decode(Uint8Array.from([0xa4]), { stream: true });
77+
assert.strictEqual(r1, '');
78+
const r2 = decoder.decode(Uint8Array.from([0x40]));
79+
assert.strictEqual(r2, '一');
80+
}
81+
82+
// A supplementary-plane code point (outside the BMP) must be encoded as a
83+
// surrogate pair in the resulting JS string.
84+
{
85+
// Find a Big5 pointer whose code point is astral, by scanning the same
86+
// byte ranges the decoder itself accepts.
87+
const decoder = new TextDecoder('big5');
88+
let found = false;
89+
outer:
90+
for (let lead = 0x81; lead <= 0xfe && !found; lead++) {
91+
for (let byte = 0x40; byte <= 0xfe; byte++) {
92+
if (!((byte >= 0x40 && byte <= 0x7e) || (byte >= 0xa1 && byte <= 0xfe))) continue;
93+
const result = decoder.decode(Uint8Array.from([lead, byte]));
94+
if (result.length === 2 && codePoints(result).length === 1) {
95+
assert.ok(codePoints(result)[0] > 0xffff);
96+
found = true;
97+
break outer;
98+
}
99+
}
100+
}
101+
assert.ok(found, 'expected to find at least one astral Big5 mapping');
102+
}

0 commit comments

Comments
 (0)