#!/usr/bin/env node /* Generate the immutable lookup tables used by the static TextDecoder * lowering. Node is the corpus oracle, so derive the legacy mappings from * the Node 24 TextDecoder rather than from a platform iconv installation. * The generated header is checked in; normal builds do not run this script. */ import { readFileSync, writeFileSync } from "node:fs"; import { resolve } from "node:path"; import { fileURLToPath } from "node:url"; const root = fileURLToPath(new URL("..", import.meta.url)); const output = resolve(root, "packages/runtime/src/scr_text_decoder_data.h"); const pinnedNodeVersion = readFileSync(resolve(root, ".node-version"), "utf8").trim(); if (process.versions.node !== pinnedNodeVersion) { throw new Error( `TextDecoder tables must be generated with the pinned Node ${pinnedNodeVersion} (got ${process.version})`, ); } // Keep this order synchronized with TEXT_DECODER_SINGLE_BYTE_NAMES in // lower-builtins.ts. ISO-8859-8-I intentionally shares ISO-8859-8's table. const singleByteNames = [ "ibm866", "iso-8859-2", "iso-8859-3", "iso-8859-4", "iso-8859-5", "iso-8859-6", "iso-8859-7", "iso-8859-8", "iso-8859-10", "iso-8859-13", "iso-8859-14", "iso-8859-15", "iso-8859-16", "koi8-r", "koi8-u", "macintosh", "windows-874", "windows-1250", "windows-1251", "windows-1252", "windows-1253", "windows-1254", "windows-1255", "windows-1256", "windows-1257", "windows-1258", "x-mac-cyrillic", ]; function codePoints(text) { return Array.from(text, (char) => char.codePointAt(0)); } function oneCodePoint(decoder, bytes) { const points = codePoints(decoder.decode(Uint8Array.from(bytes))); return points.length === 1 && points[0] !== 0xfffd ? points[0] : 0; } function hex(value, width = 4) { return `0x${value.toString(16).padStart(width, "0")}`; } function emitRows(values, width = 4, perLine = 12, indent = " ") { const lines = []; for (let i = 0; i < values.length; i += perLine) { lines.push(`${indent}${values.slice(i, i + perLine).map((v) => hex(v, width)).join(", ")},`); } return lines.join("\n"); } function denseTable(label, count, bytesOf) { const decoder = new TextDecoder(label); return Array.from({ length: count }, (_, pointer) => oneCodePoint(decoder, bytesOf(pointer))); } const chunks = [ "/* Generated by scripts/gen-text-decoder-tables.mjs with " + process.version + ". Do not edit by hand. */", "#ifndef SCR_TEXT_DECODER_DATA_H", "#define SCR_TEXT_DECODER_DATA_H", "", "#include ", "", `#define SCR_TD_SINGLE_COUNT ${singleByteNames.length}`, "static const uint16_t scr_td_single[SCR_TD_SINGLE_COUNT][128] = {", ]; for (const label of singleByteNames) { const decoder = new TextDecoder(label); const values = Array.from({ length: 128 }, (_, i) => { const points = codePoints(decoder.decode(Uint8Array.of(i + 0x80))); if (points.length !== 1) throw new Error(`${label} byte ${hex(i + 0x80, 2)} did not decode once`); return points[0]; }); chunks.push(` /* ${label} */ {`, emitRows(values, 4, 12, " "), " },"); } chunks.push("};", ""); const gb = denseTable("gb18030", 126 * 190, (pointer) => { const lead = Math.floor(pointer / 190) + 0x81; const trail = pointer % 190; return [lead, trail + (trail < 0x3f ? 0x40 : 0x41)]; }); chunks.push( "static const uint16_t scr_td_gb18030[126 * 190] = {", emitRows(gb), "};", "", ); const big5 = denseTable("big5", 126 * 157, (pointer) => { const lead = Math.floor(pointer / 157) + 0x81; const trail = pointer % 157; return [lead, trail + (trail < 0x3f ? 0x40 : 0x62)]; }); chunks.push("static const uint16_t scr_td_big5[126 * 157] = {", emitRows(big5), "};", ""); const jis0208 = denseTable("euc-jp", 94 * 94, (pointer) => [ Math.floor(pointer / 94) + 0xa1, pointer % 94 + 0xa1, ]); chunks.push("static const uint16_t scr_td_jis0208[94 * 94] = {", emitRows(jis0208), "};", ""); const jis0212 = denseTable("euc-jp", 94 * 94, (pointer) => [ 0x8f, Math.floor(pointer / 94) + 0xa1, pointer % 94 + 0xa1, ]); chunks.push("static const uint16_t scr_td_jis0212[94 * 94] = {", emitRows(jis0212), "};", ""); const shiftJis = denseTable("shift_jis", 60 * 188, (pointer) => { const lead = Math.floor(pointer / 188); const leadByte = lead < 0x1f ? lead + 0x81 : lead + 0xc1; const trail = pointer % 188; return [leadByte, trail + (trail < 0x3f ? 0x40 : 0x41)]; }); chunks.push("static const uint16_t scr_td_shift_jis[60 * 188] = {", emitRows(shiftJis), "};", ""); const eucKr = denseTable("euc-kr", 94 * 94, (pointer) => [ Math.floor(pointer / 94) + 0xa1, pointer % 94 + 0xa1, ]); chunks.push("static const uint16_t scr_td_euc_kr[94 * 94] = {", emitRows(eucKr), "};", ""); // Four-byte gb18030 mappings form long consecutive runs. Record the runs, // including their end pointer, so invalid holes and the post-Unicode tail do // not accidentally interpolate to scalar values. const gbRanges = []; const gbDecoder = new TextDecoder("gb18030"); let run = null; for (let pointer = 0; pointer <= 1237575; pointer++) { let rest = pointer; const byte4 = (rest % 10) + 0x30; rest = Math.floor(rest / 10); const byte3 = (rest % 126) + 0x81; rest = Math.floor(rest / 126); const byte2 = (rest % 10) + 0x30; const byte1 = Math.floor(rest / 10) + 0x81; const cp = oneCodePoint(gbDecoder, [byte1, byte2, byte3, byte4]); if (cp !== 0 && run !== null && pointer === run.end + 1 && cp === run.cp + (pointer - run.start)) { run.end = pointer; continue; } if (run !== null) gbRanges.push(run); run = cp === 0 ? null : { start: pointer, end: pointer, cp }; } if (run !== null) gbRanges.push(run); chunks.push( "typedef struct { uint32_t start, end, code_point; } ScrTdGbRange;", "static const ScrTdGbRange scr_td_gb_ranges[] = {", ...gbRanges.map(({ start, end, cp }) => ` { ${start}u, ${end}u, ${hex(cp, 6)}u },`), "};", "", "#endif", "", ); writeFileSync(output, chunks.join("\n")); console.log(`wrote ${output} (${singleByteNames.length} single-byte tables, ${gbRanges.length} gb18030 ranges)`);