Skip to content

Commit b777d0a

Browse files
committed
lib: implement WHATWG-spec Big5 decoder in js
TextDecoder('big5') was routed through ICU's Big5 converter, whose table includes vendor/PUA mappings the WHATWG Encoding Standard does not define. Byte sequences the standard defines as invalid (and which must decode to U+FFFD, or throw when fatal) instead decoded to those extra ICU-only characters, e.g. bytes 0x83 0x5C decoded to U+F00E instead of U+FFFD U+005C, even with `fatal: true`. This adds a small, self-contained decoder for the 'big5' label (and its aliases, which the standard maps to the same decoder) that implements the algorithm and index table from the Encoding Standard directly, mirroring how single-byte.js already reimplements the legacy single-byte encodings instead of relying on ICU for them. It does not touch the ICU-backed path used by any other encoding. Refs: https://encoding.spec.whatwg.org/#big5-decoder Refs: #61041 Fixes: #40091 Signed-off-by: agape1225 <49804691+agape1225@users.noreply.github.com>
1 parent f7968e7 commit b777d0a

3 files changed

Lines changed: 239 additions & 0 deletions

File tree

‎lib/internal/encoding.js‎

Lines changed: 16 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -24,6 +24,7 @@ const {
2424
ERR_NO_ICU,
2525
} = require('internal/errors').codes;
2626
const kSingleByte = Symbol('single-byte');
27+
const kBig5 = Symbol('big5');
2728
const kHandle = Symbol('handle');
2829
const kFlags = Symbol('flags');
2930
const kEncoding = Symbol('encoding');
@@ -34,6 +35,7 @@ const kUTF8FastPath = Symbol('kUTF8FastPath');
3435
const kIgnoreBOM = Symbol('kIgnoreBOM');
3536

3637
const { isSinglebyteEncoding, createSinglebyteDecoder } = require('internal/encoding/single-byte');
38+
const { isBig5Encoding, createBig5Decoder } = require('internal/encoding/big5');
3739
const { unfinishedBytesUtf8, mergePrefixUtf8 } = require('internal/encoding/util');
3840

3941
const {
@@ -449,13 +451,21 @@ class TextDecoder {
449451
this[kUTF8FastPath] = false;
450452
this[kHandle] = undefined;
451453
this[kSingleByte] = undefined; // Does not care about streaming or BOM
454+
this[kBig5] = false;
452455
this[kChunk] = null; // A copy of previous streaming tail or null
453456

454457
if (enc === 'utf-8') {
455458
this[kUTF8FastPath] = true;
456459
this[kBOMSeen] = false;
457460
} else if (isSinglebyteEncoding(enc)) {
458461
this[kSingleByte] = createSinglebyteDecoder(enc, this[kFatal]);
462+
} else if (isBig5Encoding(enc)) {
463+
// Not routed through ICU: ICU's own Big5 conversion table includes
464+
// vendor/PUA mappings the WHATWG Encoding Standard does not, so byte
465+
// sequences the standard defines as invalid would otherwise decode to
466+
// those extra characters instead of U+FFFD.
467+
this[kBig5] = true;
468+
this[kHandle] = createBig5Decoder(this[kFatal]);
459469
} else {
460470
this.#prepareConverter(); // Need to throw early if we don't support the encoding
461471
}
@@ -485,6 +495,12 @@ class TextDecoder {
485495
if (this[kSingleByte]) return this[kSingleByte](parseInput(input));
486496

487497
const stream = options?.stream;
498+
499+
if (this[kBig5]) {
500+
input = parseInput(input);
501+
return stream ? this[kHandle].write(input) : this[kHandle].end(input);
502+
}
503+
488504
if (this[kUTF8FastPath]) {
489505
const chunk = this[kChunk];
490506
const ignoreBom = this[kIgnoreBOM] || this[kBOMSeen];

‎lib/internal/encoding/big5.js‎

Lines changed: 121 additions & 0 deletions
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.
Lines changed: 102 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,102 @@
1+
'use strict';
2+
3+
// Big5 is decoded by a WHATWG-spec-faithful implementation (lib/internal/encoding/big5.js)
4+
// rather than through ICU, because ICU's own Big5 conversion table includes
5+
// vendor/PUA mappings the WHATWG Encoding Standard does not: byte sequences
6+
// the standard defines as invalid decode to those extra characters instead
7+
// of U+FFFD when routed through ICU.
8+
// Refs: https://github.com/nodejs/node/issues/61041
9+
// Refs: https://github.com/nodejs/node/issues/40091
10+
// Refs: https://encoding.spec.whatwg.org/#big5-decoder
11+
12+
require('../common');
13+
const assert = require('assert');
14+
15+
function codePoints(str) {
16+
return [...str].map((c) => c.codePointAt(0));
17+
}
18+
19+
for (const label of ['big5', 'Big5', 'BIG5', 'big5-hkscs', 'cn-big5', 'csbig5', 'x-x-big5']) {
20+
const decoder = new TextDecoder(label);
21+
assert.strictEqual(decoder.encoding, 'big5');
22+
}
23+
24+
// ASCII round-trips as-is.
25+
{
26+
const decoder = new TextDecoder('big5');
27+
assert.strictEqual(decoder.decode(Uint8Array.from([0x41, 0x42, 0x43])), 'ABC');
28+
}
29+
30+
// A valid 2-byte Big5 sequence decodes to the expected code point.
31+
// 0xA4 0x40 is the first Hanzi in the Big5 table: U+4E00 ("一", "one").
32+
{
33+
const decoder = new TextDecoder('big5');
34+
assert.strictEqual(decoder.decode(Uint8Array.from([0xa4, 0x40])), '一');
35+
}
36+
37+
// Regression test: an unassigned Big5 pointer must decode to U+FFFD, not to
38+
// whatever extra character ICU's own (non-spec) Big5 table maps it to.
39+
// https://github.com/nodejs/node/issues/40091
40+
{
41+
const decoder = new TextDecoder('big5');
42+
const result = decoder.decode(Uint8Array.from([0x41, 0x42, 0x83, 0x5c, 0x43, 0x44]));
43+
assert.deepStrictEqual(codePoints(result), [0x41, 0x42, 0xfffd, 0x5c, 0x43, 0x44]);
44+
}
45+
46+
// A lead byte with no trailing byte (end of input) is also an error.
47+
{
48+
const decoder = new TextDecoder('big5');
49+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0xa4]))), [0xfffd]);
50+
}
51+
52+
// `fatal: true` must throw instead of substituting U+FFFD, and must
53+
// actually recognize this sequence as invalid (unlike plain ICU, which
54+
// treats it as valid and never throws even in fatal mode).
55+
{
56+
const decoder = new TextDecoder('big5', { fatal: true });
57+
assert.throws(() => {
58+
decoder.decode(Uint8Array.from([0x83, 0x5c]));
59+
}, { name: 'TypeError', code: 'ERR_ENCODING_INVALID_ENCODED_DATA' });
60+
}
61+
62+
// The four Big5 pointers that map to two combining code points instead of
63+
// one, per the spec's special-cased steps ahead of the index table lookup.
64+
{
65+
const decoder = new TextDecoder('big5');
66+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0x62]))), [0xca, 0x0304]);
67+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0x64]))), [0xca, 0x030c]);
68+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0xa3]))), [0xea, 0x0304]);
69+
assert.deepStrictEqual(codePoints(decoder.decode(Uint8Array.from([0x88, 0xa5]))), [0xea, 0x030c]);
70+
}
71+
72+
// Streaming: a valid 2-byte sequence split across chunk boundaries must
73+
// still decode correctly, and must not be flushed early.
74+
{
75+
const decoder = new TextDecoder('big5');
76+
const r1 = decoder.decode(Uint8Array.from([0xa4]), { stream: true });
77+
assert.strictEqual(r1, '');
78+
const r2 = decoder.decode(Uint8Array.from([0x40]));
79+
assert.strictEqual(r2, '一');
80+
}
81+
82+
// A supplementary-plane code point (outside the BMP) must be encoded as a
83+
// surrogate pair in the resulting JS string.
84+
{
85+
// Find a Big5 pointer whose code point is astral, by scanning the same
86+
// byte ranges the decoder itself accepts.
87+
const decoder = new TextDecoder('big5');
88+
let found = false;
89+
outer:
90+
for (let lead = 0x81; lead <= 0xfe && !found; lead++) {
91+
for (let byte = 0x40; byte <= 0xfe; byte++) {
92+
if (!((byte >= 0x40 && byte <= 0x7e) || (byte >= 0xa1 && byte <= 0xfe))) continue;
93+
const result = decoder.decode(Uint8Array.from([lead, byte]));
94+
if (result.length === 2 && codePoints(result).length === 1) {
95+
assert.ok(codePoints(result)[0] > 0xffff);
96+
found = true;
97+
break outer;
98+
}
99+
}
100+
}
101+
assert.ok(found, 'expected to find at least one astral Big5 mapping');
102+
}

0 commit comments

Comments
 (0)