Skip to content

Commit 00606a2

Browse files
committed
preserve BOMs in decoded strings
1 parent 2990083 commit 00606a2

76 files changed

Lines changed: 96 additions & 312 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎Plugins/BridgeJS/Sources/BridgeJSLink/BridgeJSLink.swift‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -346,7 +346,6 @@ public struct BridgeJSLink {
346346
u6 = 0,
347347
u7 = 0;
348348
while (byteIndex < count) {
349-
const start = byteIndex;
350349
const b0 = byteAt(word0, word1, byteIndex++);
351350
let codePoint;
352351
if (b0 < 128) codePoint = b0;
@@ -361,8 +360,6 @@ public struct BridgeJSLink {
361360
b3 = byteAt(word0, word1, byteIndex++);
362361
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
363362
}
364-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
365-
if (start === 0 && codePoint === 0xfeff) continue;
366363
const supplementary = codePoint > 0xffff;
367364
const codeUnit = supplementary
368365
? 0xd800 + ((codePoint - 0x10000) >>> 10)
@@ -487,7 +484,7 @@ public struct BridgeJSLink {
487484
"let \(JSGlueVariableScope.reservedDecodeString);",
488485
"let \(JSGlueVariableScope.reservedDecodeUTF8);",
489486
"const \(JSGlueVariableScope.reservedImmortalStrings) = new Map();",
490-
"const \(JSGlueVariableScope.reservedTextDecoder) = new TextDecoder(\"utf-8\");",
487+
"const \(JSGlueVariableScope.reservedTextDecoder) = new TextDecoder(\"utf-8\", { ignoreBOM: true });",
491488
"const \(JSGlueVariableScope.reservedTextEncoder) = new TextEncoder(\"utf-8\");",
492489
"let \(JSGlueVariableScope.reservedStorageToReturnString);",
493490
"let \(JSGlueVariableScope.reservedStorageToReturnBytes);",

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/Inputs/StringDecoderTests.mjs‎

Lines changed: 14 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -6,7 +6,6 @@ const source = await readFile(process.argv[2] ?? new URL('../__Snapshots__/Bridg
66
const shared = process.argv[3] === 'shared';
77
const { createInstantiator } = await import(`data:text/javascript;base64,${Buffer.from(source).toString('base64')}`);
88
const encoder = new TextEncoder();
9-
const decoder = new TextDecoder();
109
const swift = { memory: { heap: [], retain: (value) => value } };
1110
const bridge = await createInstantiator({ getImports: () => ({}) }, swift);
1211
const imports = {};
@@ -35,21 +34,27 @@ function large(value, immortal = false, ptr = 64) {
3534
}
3635
const values = ['', 'div', 'abcdefgh', 'abcdefghi', 'abcdefghij', 'abcdefé', 'abcdefgé',
3736
'abcdefghé', 'abcdefghié', 'abc€', 'abcdefg€', 'abc😄', 'abcdef😄', 'abcdefg😄',
38-
'a\0b', '\ufeff', '\ufeffx', 'x\ufeff', '\ufeff\ufeffx', '\ufeffabcdefghijk',
39-
'abc\ufeffdefghijk', '\ufeff\ufeffabcdefghijk'];
40-
for (const value of values) {
41-
const expected = decoder.decode(encoder.encode(value));
37+
'😄😄', 'éééé', 'a\0b', '\ufeff', '\ufeffx', '\ufeffabcde', '\ufeffabcdef',
38+
'x\ufeff', '\ufeff\ufeffx', '\ufeffabcdefghijk',
39+
'abc\ufeffdefghijk', '\ufeff\ufeffabcdefghijk',
40+
// UTF-8 widths, the surrogate gap, and supplementary-plane boundaries.
41+
'\u0080', '\u07ff', '\u0800', '\ud7ff', '\ue000', '\uffff', '\u{10000}', '\u{10ffff}'];
42+
for (const [index, value] of values.entries()) {
4243
if (encoder.encode(value).length <= 8) {
4344
words = small(value);
44-
assert.equal(exports.checkString(), expected);
45+
assert.equal(exports.checkString(), value);
46+
}
47+
for (const immortal of [false, true]) {
48+
// Distinct locations keep immortal cache entries faithful to Swift literals.
49+
words = large(value, immortal, 64 + index * 64);
50+
assert.equal(exports.checkString(), value);
51+
assert.equal(exports.checkString(), value);
4552
}
46-
words = large(value);
47-
assert.equal(exports.checkString(), expected);
4853
}
4954
// Alternate full and short inputs: bytes outside count must never leak into the result.
5055
for (const value of ['abcd😄', 'é', 'abcde€', '\ufeff', 'abcdefgh', '', 'x\0', '€', '\ufeff\ufeff']) {
5156
words = small(value);
52-
assert.equal(exports.checkString(), decoder.decode(encoder.encode(value)));
57+
assert.equal(exports.checkString(), value);
5358
}
5459
// Every ASCII byte, including NUL, at every small-string length.
5560
for (let byte = 0; byte < 128; byte++) {

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/Alias.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -17,7 +17,7 @@ export async function createInstantiator(options, swift) {
1717
let decodeString;
1818
let decodeUTF8;
1919
const immortalStrings = new Map();
20-
const textDecoder = new TextDecoder("utf-8");
20+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
2121
const textEncoder = new TextEncoder("utf-8");
2222
let tmpRetString;
2323
let tmpRetBytes;
@@ -706,7 +706,6 @@ export async function createInstantiator(options, swift) {
706706
u6 = 0,
707707
u7 = 0;
708708
while (byteIndex < count) {
709-
const start = byteIndex;
710709
const b0 = byteAt(word0, word1, byteIndex++);
711710
let codePoint;
712711
if (b0 < 128) codePoint = b0;
@@ -721,8 +720,6 @@ export async function createInstantiator(options, swift) {
721720
b3 = byteAt(word0, word1, byteIndex++);
722721
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
723722
}
724-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
725-
if (start === 0 && codePoint === 0xfeff) continue;
726723
const supplementary = codePoint > 0xffff;
727724
const codeUnit = supplementary
728725
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/AliasInClosure.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -11,7 +11,7 @@ export async function createInstantiator(options, swift) {
1111
let decodeString;
1212
let decodeUTF8;
1313
const immortalStrings = new Map();
14-
const textDecoder = new TextDecoder("utf-8");
14+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
1515
const textEncoder = new TextEncoder("utf-8");
1616
let tmpRetString;
1717
let tmpRetBytes;
@@ -318,7 +318,6 @@ export async function createInstantiator(options, swift) {
318318
u6 = 0,
319319
u7 = 0;
320320
while (byteIndex < count) {
321-
const start = byteIndex;
322321
const b0 = byteAt(word0, word1, byteIndex++);
323322
let codePoint;
324323
if (b0 < 128) codePoint = b0;
@@ -333,8 +332,6 @@ export async function createInstantiator(options, swift) {
333332
b3 = byteAt(word0, word1, byteIndex++);
334333
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
335334
}
336-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
337-
if (start === 0 && codePoint === 0xfeff) continue;
338335
const supplementary = codePoint > 0xffff;
339336
const codeUnit = supplementary
340337
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/ArrayTypes.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -24,7 +24,7 @@ export async function createInstantiator(options, swift) {
2424
let decodeString;
2525
let decodeUTF8;
2626
const immortalStrings = new Map();
27-
const textDecoder = new TextDecoder("utf-8");
27+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
2828
const textEncoder = new TextEncoder("utf-8");
2929
let tmpRetString;
3030
let tmpRetBytes;
@@ -768,7 +768,6 @@ export async function createInstantiator(options, swift) {
768768
u6 = 0,
769769
u7 = 0;
770770
while (byteIndex < count) {
771-
const start = byteIndex;
772771
const b0 = byteAt(word0, word1, byteIndex++);
773772
let codePoint;
774773
if (b0 < 128) codePoint = b0;
@@ -783,8 +782,6 @@ export async function createInstantiator(options, swift) {
783782
b3 = byteAt(word0, word1, byteIndex++);
784783
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
785784
}
786-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
787-
if (start === 0 && codePoint === 0xfeff) continue;
788785
const supplementary = codePoint > 0xffff;
789786
const codeUnit = supplementary
790787
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/Async.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,7 @@ export async function createInstantiator(options, swift) {
2121
let decodeString;
2222
let decodeUTF8;
2323
const immortalStrings = new Map();
24-
const textDecoder = new TextDecoder("utf-8");
24+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
2525
const textEncoder = new TextEncoder("utf-8");
2626
let tmpRetString;
2727
let tmpRetBytes;
@@ -752,7 +752,6 @@ export async function createInstantiator(options, swift) {
752752
u6 = 0,
753753
u7 = 0;
754754
while (byteIndex < count) {
755-
const start = byteIndex;
756755
const b0 = byteAt(word0, word1, byteIndex++);
757756
let codePoint;
758757
if (b0 < 128) codePoint = b0;
@@ -767,8 +766,6 @@ export async function createInstantiator(options, swift) {
767766
b3 = byteAt(word0, word1, byteIndex++);
768767
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
769768
}
770-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
771-
if (start === 0 && codePoint === 0xfeff) continue;
772769
const supplementary = codePoint > 0xffff;
773770
const codeUnit = supplementary
774771
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/AsyncAssociatedValueEnum.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -18,7 +18,7 @@ export async function createInstantiator(options, swift) {
1818
let decodeString;
1919
let decodeUTF8;
2020
const immortalStrings = new Map();
21-
const textDecoder = new TextDecoder("utf-8");
21+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
2222
const textEncoder = new TextEncoder("utf-8");
2323
let tmpRetString;
2424
let tmpRetBytes;
@@ -398,7 +398,6 @@ export async function createInstantiator(options, swift) {
398398
u6 = 0,
399399
u7 = 0;
400400
while (byteIndex < count) {
401-
const start = byteIndex;
402401
const b0 = byteAt(word0, word1, byteIndex++);
403402
let codePoint;
404403
if (b0 < 128) codePoint = b0;
@@ -413,8 +412,6 @@ export async function createInstantiator(options, swift) {
413412
b3 = byteAt(word0, word1, byteIndex++);
414413
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
415414
}
416-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
417-
if (start === 0 && codePoint === 0xfeff) continue;
418415
const supplementary = codePoint > 0xffff;
419416
const codeUnit = supplementary
420417
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/AsyncImport.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -11,7 +11,7 @@ export async function createInstantiator(options, swift) {
1111
let decodeString;
1212
let decodeUTF8;
1313
const immortalStrings = new Map();
14-
const textDecoder = new TextDecoder("utf-8");
14+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
1515
const textEncoder = new TextEncoder("utf-8");
1616
let tmpRetString;
1717
let tmpRetBytes;
@@ -530,7 +530,6 @@ export async function createInstantiator(options, swift) {
530530
u6 = 0,
531531
u7 = 0;
532532
while (byteIndex < count) {
533-
const start = byteIndex;
534533
const b0 = byteAt(word0, word1, byteIndex++);
535534
let codePoint;
536535
if (b0 < 128) codePoint = b0;
@@ -545,8 +544,6 @@ export async function createInstantiator(options, swift) {
545544
b3 = byteAt(word0, word1, byteIndex++);
546545
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
547546
}
548-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
549-
if (start === 0 && codePoint === 0xfeff) continue;
550547
const supplementary = codePoint > 0xffff;
551548
const codeUnit = supplementary
552549
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/AsyncStaticImport.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -11,7 +11,7 @@ export async function createInstantiator(options, swift) {
1111
let decodeString;
1212
let decodeUTF8;
1313
const immortalStrings = new Map();
14-
const textDecoder = new TextDecoder("utf-8");
14+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
1515
const textEncoder = new TextEncoder("utf-8");
1616
let tmpRetString;
1717
let tmpRetBytes;
@@ -426,7 +426,6 @@ export async function createInstantiator(options, swift) {
426426
u6 = 0,
427427
u7 = 0;
428428
while (byteIndex < count) {
429-
const start = byteIndex;
430429
const b0 = byteAt(word0, word1, byteIndex++);
431430
let codePoint;
432431
if (b0 < 128) codePoint = b0;
@@ -441,8 +440,6 @@ export async function createInstantiator(options, swift) {
441440
b3 = byteAt(word0, word1, byteIndex++);
442441
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
443442
}
444-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
445-
if (start === 0 && codePoint === 0xfeff) continue;
446443
const supplementary = codePoint > 0xffff;
447444
const codeUnit = supplementary
448445
? 0xd800 + ((codePoint - 0x10000) >>> 10)

‎Plugins/BridgeJS/Tests/BridgeJSToolTests/__Snapshots__/BridgeJSLinkTests/ClassWithNestedTypes.js‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -16,7 +16,7 @@ export async function createInstantiator(options, swift) {
1616
let decodeString;
1717
let decodeUTF8;
1818
const immortalStrings = new Map();
19-
const textDecoder = new TextDecoder("utf-8");
19+
const textDecoder = new TextDecoder("utf-8", { ignoreBOM: true });
2020
const textEncoder = new TextEncoder("utf-8");
2121
let tmpRetString;
2222
let tmpRetBytes;
@@ -267,7 +267,6 @@ export async function createInstantiator(options, swift) {
267267
u6 = 0,
268268
u7 = 0;
269269
while (byteIndex < count) {
270-
const start = byteIndex;
271270
const b0 = byteAt(word0, word1, byteIndex++);
272271
let codePoint;
273272
if (b0 < 128) codePoint = b0;
@@ -282,8 +281,6 @@ export async function createInstantiator(options, swift) {
282281
b3 = byteAt(word0, word1, byteIndex++);
283282
codePoint = ((b0 & 7) << 18) | ((b1 & 63) << 12) | ((b2 & 63) << 6) | (b3 & 63);
284283
}
285-
// Match TextDecoder: consume exactly one leading UTF-8 BOM.
286-
if (start === 0 && codePoint === 0xfeff) continue;
287284
const supplementary = codePoint > 0xffff;
288285
const codeUnit = supplementary
289286
? 0xd800 + ((codePoint - 0x10000) >>> 10)

0 commit comments

Comments
 (0)