mirror of
https://github.com/vercel-labs/scriptc.git
synced 2026-10-02 00:25:34 +08:00
Support native runtime values used by OpenTUI
- Add shared ArrayBuffer storage and native Unicode grapheme segmentation. - Preserve builtin modules, globals, and byte views across JavaScript bindings. - Compile package initialization expressions and optional native callables.
This commit is contained in:
@@ -0,0 +1,91 @@
|
||||
#!/usr/bin/env node
|
||||
// Unicode 17 default extended grapheme clusters (UAX #29), matching the
|
||||
// pinned Node release. Normal builds use the checked-in tables and cases.
|
||||
import { createHash } from "node:crypto";
|
||||
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { join } from "node:path";
|
||||
|
||||
const root = fileURLToPath(new URL("..", import.meta.url));
|
||||
const inputs = [
|
||||
["auxiliary/GraphemeBreakProperty.txt", "d6b51d1d2ae5c33b451b7ed994b48f1f4dc62b2272a5831e7fd418514a6bae89"],
|
||||
["DerivedCoreProperties.txt", "24c7fed1195c482faaefd5c1e7eb821c5ee1fb6de07ecdbaa64b56a99da22c08"],
|
||||
["emoji/emoji-data.txt", "2cb2bb9455cda83e8481541ecf5b6dfda66a3bb89efa3fa7c5297eccf607b72b"],
|
||||
["auxiliary/GraphemeBreakTest.txt", "e2d134d2c52919bace503ebb6a551c1855fe1a1faec18478c78fff254a1793ec"],
|
||||
];
|
||||
const files = await Promise.all(inputs.map(async ([path, hash]) => {
|
||||
const bytes = process.argv[2]
|
||||
? await readFile(join(process.argv[2], path.split("/").at(-1)))
|
||||
: await (async () => {
|
||||
const response = await fetch(`https://www.unicode.org/Public/17.0.0/ucd/${path}`);
|
||||
if (!response.ok) throw new Error(`${path}: ${response.status}`);
|
||||
return Buffer.from(await response.arrayBuffer());
|
||||
})();
|
||||
if (createHash("sha256").update(bytes).digest("hex") !== hash) throw new Error(`${path}: checksum mismatch`);
|
||||
return bytes.toString("utf8");
|
||||
}));
|
||||
|
||||
const names = ["Other", "CR", "LF", "Control", "Extend", "ZWJ", "Regional_Indicator", "Prepend", "SpacingMark", "L", "V", "T", "LV", "LVT"];
|
||||
const data = new Uint8Array(0x110000);
|
||||
function properties(text, select) {
|
||||
for (const line of text.split("\n")) {
|
||||
const fields = line.split("#")[0].trim().split(";").map((x) => x.trim());
|
||||
if (fields.length < 2) continue;
|
||||
const value = select(fields);
|
||||
if (!value) continue;
|
||||
const [start, end = start] = fields[0].split("..").map((x) => parseInt(x, 16));
|
||||
for (let cp = start; cp <= end; cp++) data[cp] |= value;
|
||||
}
|
||||
}
|
||||
properties(files[0], ([, value]) => names.indexOf(value));
|
||||
properties(files[1], ([, property, value]) => property === "InCB" ? ["None", "Consonant", "Extend", "Linker"].indexOf(value) << 4 : 0);
|
||||
properties(files[2], ([, property]) => property === "Extended_Pictographic" ? 64 : 0);
|
||||
// Hangul syllables have an algorithmic LV/LVT classification.
|
||||
data.fill(0, 0xac00, 0xd7a4);
|
||||
const rows = [];
|
||||
for (let start = 0; start < data.length;) {
|
||||
let end = start + 1;
|
||||
while (end < data.length && data[end] === data[start]) end++;
|
||||
if (data[start]) rows.push(` { 0x${start.toString(16)}, 0x${(end - 1).toString(16)}, ${data[start]} },`);
|
||||
start = end;
|
||||
}
|
||||
const header = [
|
||||
"/* Generated by scripts/gen-grapheme-tables.mjs; Unicode 17.0.0.",
|
||||
" * Unicode data is licensed under vendor/unicode/LICENSE. Do not edit. */",
|
||||
"#ifndef SCR_GRAPHEME_DATA_H",
|
||||
"#define SCR_GRAPHEME_DATA_H",
|
||||
"enum { " + names.map((name, index) => `SCR_GCB_${name.toUpperCase()} = ${index}`).join(", ") + " };",
|
||||
"typedef struct { uint32_t start, end; uint8_t properties; } ScrGraphemeRange;",
|
||||
"static const ScrGraphemeRange scr_grapheme_ranges[] = {",
|
||||
...rows,
|
||||
"};",
|
||||
"#endif",
|
||||
"",
|
||||
].join("\n");
|
||||
await writeFile(join(root, "packages/runtime/src/scr_grapheme_data.h"), header);
|
||||
|
||||
const cases = [];
|
||||
for (const line of files[3].split("\n")) {
|
||||
const content = line.split("#")[0].trim();
|
||||
if (!content) continue;
|
||||
let input = "";
|
||||
const breaks = [];
|
||||
for (const token of content.split(/\s+/)) {
|
||||
if (token === "÷") breaks.push(input.length);
|
||||
else if (token !== "×") input += String.fromCodePoint(parseInt(token, 16));
|
||||
}
|
||||
cases.push([input, breaks.join(",")]);
|
||||
}
|
||||
const corpusDir = join(root, "tests/corpus/intl-grapheme-conformance");
|
||||
await mkdir(corpusDir, { recursive: true });
|
||||
// Escape non-ASCII source so control characters never corrupt the fixture.
|
||||
const quote = (text) => JSON.stringify(text).replace(/[\u007f-\uffff]/g, (ch) => `\\u${ch.charCodeAt(0).toString(16).padStart(4, "0")}`);
|
||||
await writeFile(join(corpusDir, "cases.ts"), [
|
||||
"// Generated from Unicode 17.0.0 GraphemeBreakTest.txt by scripts/gen-grapheme-tables.mjs.",
|
||||
"// Unicode data license: packages/runtime/vendor/unicode/LICENSE.",
|
||||
"export const cases: string[][] = [",
|
||||
...cases.map(([input, breaks]) => ` [${quote(input)}, ${quote(breaks)}],`),
|
||||
"];",
|
||||
"",
|
||||
].join("\n"));
|
||||
console.log(`wrote ${rows.length} property ranges and ${cases.length} grapheme conformance cases`);
|
||||
Reference in New Issue
Block a user