mirror of
https://github.com/vercel-labs/scriptc.git
synced 2026-10-02 00:25:34 +08:00
* feat: support static non-UTF-8 TextDecoder labels - Recognize static WHATWG labels and aliases during builtin lowering. - Add feature-gated portable decoders generated against the pinned Node runtime. - Cover legacy encodings, malformed input, and unsupported dynamic labels. * fix: match Node legacy decoder recovery * fix TextDecoder edge-case parity * fix(runtime): match Node malformed text decoding * fix TextDecoder recovery edge cases
183 lines
6.0 KiB
JavaScript
183 lines
6.0 KiB
JavaScript
#!/usr/bin/env node
|
|
/* Generate the immutable lookup tables used by the static TextDecoder
|
|
* lowering. Node is the corpus oracle, so derive the legacy mappings from
|
|
* the Node 24 TextDecoder rather than from a platform iconv installation.
|
|
* The generated header is checked in; normal builds do not run this script. */
|
|
import { readFileSync, writeFileSync } from "node:fs";
|
|
import { resolve } from "node:path";
|
|
import { fileURLToPath } from "node:url";
|
|
|
|
const root = fileURLToPath(new URL("..", import.meta.url));
|
|
const output = resolve(root, "packages/runtime/src/scr_text_decoder_data.h");
|
|
|
|
const pinnedNodeVersion = readFileSync(resolve(root, ".node-version"), "utf8").trim();
|
|
if (process.versions.node !== pinnedNodeVersion) {
|
|
throw new Error(
|
|
`TextDecoder tables must be generated with the pinned Node ${pinnedNodeVersion} (got ${process.version})`,
|
|
);
|
|
}
|
|
|
|
// Keep this order synchronized with TEXT_DECODER_SINGLE_BYTE_NAMES in
|
|
// lower-builtins.ts. ISO-8859-8-I intentionally shares ISO-8859-8's table.
|
|
const singleByteNames = [
|
|
"ibm866",
|
|
"iso-8859-2",
|
|
"iso-8859-3",
|
|
"iso-8859-4",
|
|
"iso-8859-5",
|
|
"iso-8859-6",
|
|
"iso-8859-7",
|
|
"iso-8859-8",
|
|
"iso-8859-10",
|
|
"iso-8859-13",
|
|
"iso-8859-14",
|
|
"iso-8859-15",
|
|
"iso-8859-16",
|
|
"koi8-r",
|
|
"koi8-u",
|
|
"macintosh",
|
|
"windows-874",
|
|
"windows-1250",
|
|
"windows-1251",
|
|
"windows-1252",
|
|
"windows-1253",
|
|
"windows-1254",
|
|
"windows-1255",
|
|
"windows-1256",
|
|
"windows-1257",
|
|
"windows-1258",
|
|
"x-mac-cyrillic",
|
|
];
|
|
|
|
function codePoints(text) {
|
|
return Array.from(text, (char) => char.codePointAt(0));
|
|
}
|
|
|
|
function oneCodePoint(decoder, bytes) {
|
|
const points = codePoints(decoder.decode(Uint8Array.from(bytes)));
|
|
return points.length === 1 && points[0] !== 0xfffd ? points[0] : 0;
|
|
}
|
|
|
|
function hex(value, width = 4) {
|
|
return `0x${value.toString(16).padStart(width, "0")}`;
|
|
}
|
|
|
|
function emitRows(values, width = 4, perLine = 12, indent = " ") {
|
|
const lines = [];
|
|
for (let i = 0; i < values.length; i += perLine) {
|
|
lines.push(`${indent}${values.slice(i, i + perLine).map((v) => hex(v, width)).join(", ")},`);
|
|
}
|
|
return lines.join("\n");
|
|
}
|
|
|
|
function denseTable(label, count, bytesOf) {
|
|
const decoder = new TextDecoder(label);
|
|
return Array.from({ length: count }, (_, pointer) => oneCodePoint(decoder, bytesOf(pointer)));
|
|
}
|
|
|
|
const chunks = [
|
|
"/* Generated by scripts/gen-text-decoder-tables.mjs with " + process.version +
|
|
". Do not edit by hand. */",
|
|
"#ifndef SCR_TEXT_DECODER_DATA_H",
|
|
"#define SCR_TEXT_DECODER_DATA_H",
|
|
"",
|
|
"#include <stdint.h>",
|
|
"",
|
|
`#define SCR_TD_SINGLE_COUNT ${singleByteNames.length}`,
|
|
"static const uint16_t scr_td_single[SCR_TD_SINGLE_COUNT][128] = {",
|
|
];
|
|
|
|
for (const label of singleByteNames) {
|
|
const decoder = new TextDecoder(label);
|
|
const values = Array.from({ length: 128 }, (_, i) => {
|
|
const points = codePoints(decoder.decode(Uint8Array.of(i + 0x80)));
|
|
if (points.length !== 1) throw new Error(`${label} byte ${hex(i + 0x80, 2)} did not decode once`);
|
|
return points[0];
|
|
});
|
|
chunks.push(` /* ${label} */ {`, emitRows(values, 4, 12, " "), " },");
|
|
}
|
|
chunks.push("};", "");
|
|
|
|
const gb = denseTable("gb18030", 126 * 190, (pointer) => {
|
|
const lead = Math.floor(pointer / 190) + 0x81;
|
|
const trail = pointer % 190;
|
|
return [lead, trail + (trail < 0x3f ? 0x40 : 0x41)];
|
|
});
|
|
chunks.push(
|
|
"static const uint16_t scr_td_gb18030[126 * 190] = {",
|
|
emitRows(gb),
|
|
"};",
|
|
"",
|
|
);
|
|
|
|
const big5 = denseTable("big5", 126 * 157, (pointer) => {
|
|
const lead = Math.floor(pointer / 157) + 0x81;
|
|
const trail = pointer % 157;
|
|
return [lead, trail + (trail < 0x3f ? 0x40 : 0x62)];
|
|
});
|
|
chunks.push("static const uint16_t scr_td_big5[126 * 157] = {", emitRows(big5), "};", "");
|
|
|
|
const jis0208 = denseTable("euc-jp", 94 * 94, (pointer) => [
|
|
Math.floor(pointer / 94) + 0xa1,
|
|
pointer % 94 + 0xa1,
|
|
]);
|
|
chunks.push("static const uint16_t scr_td_jis0208[94 * 94] = {", emitRows(jis0208), "};", "");
|
|
|
|
const jis0212 = denseTable("euc-jp", 94 * 94, (pointer) => [
|
|
0x8f,
|
|
Math.floor(pointer / 94) + 0xa1,
|
|
pointer % 94 + 0xa1,
|
|
]);
|
|
chunks.push("static const uint16_t scr_td_jis0212[94 * 94] = {", emitRows(jis0212), "};", "");
|
|
|
|
const shiftJis = denseTable("shift_jis", 60 * 188, (pointer) => {
|
|
const lead = Math.floor(pointer / 188);
|
|
const leadByte = lead < 0x1f ? lead + 0x81 : lead + 0xc1;
|
|
const trail = pointer % 188;
|
|
return [leadByte, trail + (trail < 0x3f ? 0x40 : 0x41)];
|
|
});
|
|
chunks.push("static const uint16_t scr_td_shift_jis[60 * 188] = {", emitRows(shiftJis), "};", "");
|
|
|
|
const eucKr = denseTable("euc-kr", 94 * 94, (pointer) => [
|
|
Math.floor(pointer / 94) + 0xa1,
|
|
pointer % 94 + 0xa1,
|
|
]);
|
|
chunks.push("static const uint16_t scr_td_euc_kr[94 * 94] = {", emitRows(eucKr), "};", "");
|
|
|
|
// Four-byte gb18030 mappings form long consecutive runs. Record the runs,
|
|
// including their end pointer, so invalid holes and the post-Unicode tail do
|
|
// not accidentally interpolate to scalar values.
|
|
const gbRanges = [];
|
|
const gbDecoder = new TextDecoder("gb18030");
|
|
let run = null;
|
|
for (let pointer = 0; pointer <= 1237575; pointer++) {
|
|
let rest = pointer;
|
|
const byte4 = (rest % 10) + 0x30;
|
|
rest = Math.floor(rest / 10);
|
|
const byte3 = (rest % 126) + 0x81;
|
|
rest = Math.floor(rest / 126);
|
|
const byte2 = (rest % 10) + 0x30;
|
|
const byte1 = Math.floor(rest / 10) + 0x81;
|
|
const cp = oneCodePoint(gbDecoder, [byte1, byte2, byte3, byte4]);
|
|
if (cp !== 0 && run !== null && pointer === run.end + 1 && cp === run.cp + (pointer - run.start)) {
|
|
run.end = pointer;
|
|
continue;
|
|
}
|
|
if (run !== null) gbRanges.push(run);
|
|
run = cp === 0 ? null : { start: pointer, end: pointer, cp };
|
|
}
|
|
if (run !== null) gbRanges.push(run);
|
|
|
|
chunks.push(
|
|
"typedef struct { uint32_t start, end, code_point; } ScrTdGbRange;",
|
|
"static const ScrTdGbRange scr_td_gb_ranges[] = {",
|
|
...gbRanges.map(({ start, end, cp }) => ` { ${start}u, ${end}u, ${hex(cp, 6)}u },`),
|
|
"};",
|
|
"",
|
|
"#endif",
|
|
"",
|
|
);
|
|
|
|
writeFileSync(output, chunks.join("\n"));
|
|
console.log(`wrote ${output} (${singleByteNames.length} single-byte tables, ${gbRanges.length} gb18030 ranges)`);
|