mirror of
https://github.com/vercel-labs/scriptc.git
synced 2026-10-02 08:35:07 +08:00
node:querystring lowers statically — parse, stringify, escape, unescape
- scr_qs.c ports Node v24's legacy codec quirk-faithfully (the scan state machine with maxKeys' pair budget, unescapeBuffer's code-unit byte truncation with U+FFFD replacement decode, encodeStringified's value rules over the DOM crossing); link-gated by moduleUsesQs, and escape rides the always-linked component encoder - parse's result is the call site's ParsedUrlQuery dictionary (the networkInterfaces verification stance; the runtime fills the overflow map and groups repeats into string[] buckets), sep/eq/maxKeys complete to Node's defaults, and custom encoder/decoder options fence by name - decode/encode lower as Node's own aliases; both backends emit the same runtime calls byte-identically - querystring joins SUPPORTED_BUILTIN_MODULES, the fallback declarations, and the regenerated surface manifest, so npm-static packages requiring it unguarded stop falling back to the island (the shim stays for island code) - corpus 2460-2464 pin the grammar corners against Node: malformed escapes, '+' vs %2B, repeated keys, maxKeys with empty segments, multi-char/multi-byte separators, astral truncation, and every CJS acquisition spelling
This commit is contained in:
@@ -1535,6 +1535,50 @@ declare module "node:string_decoder" {
|
||||
export * from "string_decoder";
|
||||
}
|
||||
|
||||
/* node:querystring — Node's legacy query-string codec (NOT
|
||||
* URLSearchParams: '+' means space on the parse side and escape encodes
|
||||
* spaces as %20). parse answers the null-prototype dictionary as a pure
|
||||
* index-signature record (repeated keys become string[] buckets;
|
||||
* @types/node's Dict adds an undefined arm the lowering tolerates);
|
||||
* stringify serializes string/number/boolean values and arrays of those
|
||||
* (Node's rules: arrays expand to repeated keys, null/undefined and
|
||||
* anything else serialize as the empty value). The maxKeys option lowers
|
||||
* (0 removes the cap, Node's rule); the custom encoder/decoder options
|
||||
* typecheck under the options-record stance and fence by name at the
|
||||
* call. decode/encode are Node's own aliases of parse/stringify. */
|
||||
declare module "querystring" {
|
||||
export interface ParseOptions {
|
||||
maxKeys?: number;
|
||||
decodeURIComponent?: (str: string) => string;
|
||||
[option: string]: unknown;
|
||||
}
|
||||
export interface StringifyOptions {
|
||||
encodeURIComponent?: (str: string) => string;
|
||||
[option: string]: unknown;
|
||||
}
|
||||
export interface ParsedUrlQuery {
|
||||
[key: string]: string | string[] | undefined;
|
||||
}
|
||||
export interface ParsedUrlQueryInput {
|
||||
[key: string]:
|
||||
| string
|
||||
| number
|
||||
| boolean
|
||||
| ReadonlyArray<string | number | boolean>
|
||||
| null
|
||||
| undefined;
|
||||
}
|
||||
export function parse(str: string, sep?: string | null, eq?: string | null, options?: ParseOptions): ParsedUrlQuery;
|
||||
export function stringify(obj?: ParsedUrlQueryInput, sep?: string | null, eq?: string | null, options?: StringifyOptions): string;
|
||||
export const decode: typeof parse;
|
||||
export const encode: typeof stringify;
|
||||
export function escape(str: string): string;
|
||||
export function unescape(str: string): string;
|
||||
}
|
||||
declare module "node:querystring" {
|
||||
export * from "querystring";
|
||||
}
|
||||
|
||||
/* node:readline — the question/close slice: createInterface over exactly
|
||||
* { input: process.stdin, output: process.stdout }, question(query, cb)
|
||||
* (the query writes to stdout, the callback gets the next line's text),
|
||||
|
||||
@@ -127,6 +127,13 @@ export interface CcOptions {
|
||||
* cross-compiles everywhere. sp-free binaries keep their exact link
|
||||
* line (scr_url.c never references the unit). */
|
||||
searchParams?: boolean;
|
||||
/** The program uses the node:querystring surface (moduleUsesQs on the
|
||||
* IR): compiles scr_qs.c into the binary — the searchParams gating
|
||||
* precedent: pure data transforms (no loop hooks, no install),
|
||||
* cross-compiles everywhere. qs-free binaries keep their exact link
|
||||
* line (escape-only programs ride the always-linked component encoder
|
||||
* and never flip this). */
|
||||
qs?: boolean;
|
||||
/** The program uses the node:stream class surface (moduleUsesStream on
|
||||
* the IR): compiles scr_stream.c into the binary — always alongside
|
||||
* scr_events_emitter.c, which moduleUsesEmitter answers true for
|
||||
@@ -957,6 +964,7 @@ export async function compileC(opts: CcOptions): Promise<void> {
|
||||
...(opts.emitter || net ? [rt(join(rtDir, "scr_dyn_handle.c"))] : []),
|
||||
...(opts.symbol ? [rt(join(rtDir, "scr_symbol.c"))] : []),
|
||||
...(opts.searchParams ? [rt(join(rtDir, "scr_url_params.c"))] : []),
|
||||
...(opts.qs ? [rt(join(rtDir, "scr_qs.c"))] : []),
|
||||
...(opts.stream ? [rt(join(rtDir, "scr_stream.c"))] : []),
|
||||
// The readiness-poller backends (scr_platform.h): kqueue on macOS/BSD,
|
||||
// epoll on Linux, WSAPoll on Windows — each TU is empty off its
|
||||
|
||||
@@ -2578,6 +2578,38 @@ export function emitExpr(E: CEmitter, e: IrExpr): Temp {
|
||||
return finish(`scr_sp_key_at(${arg(0)}, ${arg(1)})`);
|
||||
case "sp.valAt":
|
||||
return finish(`scr_sp_val_at(${arg(0)}, ${arg(1)})`);
|
||||
// node:querystring (scr_qs.c — linked exactly when parse/
|
||||
// stringify/unescape appear, moduleUsesQs). escape IS the
|
||||
// component encoder (Node's qsEscape set equals
|
||||
// encodeURIComponent's), so it emits the always-linked codec
|
||||
// and never pulls the unit. Borrow; string results +1; no throw.
|
||||
case "qs.escape":
|
||||
return finish(`scr_str_encode_uri_component(${arg(0)})`);
|
||||
case "qs.unescape":
|
||||
return finish(`scr_qs_unescape(${arg(0)})`);
|
||||
case "qs.stringify":
|
||||
return finish(`scr_qs_stringify(${arg(0)}, ${arg(1)}, ${arg(2)})`);
|
||||
case "qs.parse": {
|
||||
// The ParsedUrlQuery dictionary: a fresh pure-index-signature
|
||||
// record whose overflow map the runtime scan fills
|
||||
// (scr_qs_parse_into groups repeats into string[] buckets).
|
||||
// The frontend verified the shape (lowerQuerystringParseCall);
|
||||
// lookups here only guard emitter bugs. Args: qs, sep, eq,
|
||||
// maxKeys.
|
||||
if (e.type.kind !== "record") throw new Error("emitter bug: qs.parse result is not a record");
|
||||
const dictShape = E.recordsById.get(e.type.shapeId);
|
||||
const iv = dictShape?.indexValue;
|
||||
if (!dictShape || iv?.kind !== "union") throw new Error("emitter bug: qs.parse dict shape");
|
||||
const ivDef = E.unionsById.get(iv.unionId);
|
||||
const strTag = ivDef?.arms.findIndex((a) => a.kind === "string") ?? -1;
|
||||
const arrTag = ivDef?.arms.findIndex((a) => a.kind === "array") ?? -1;
|
||||
if (strTag < 0 || arrTag < 0) throw new Error("emitter bug: qs.parse index union lacks its arms");
|
||||
const dict = E.newTemp(e.type, `${mangleRecordNew(e.type.shapeId)}()`);
|
||||
E.line(
|
||||
`scr_qs_parse_into(${dict.name}->${OVERFLOW_MEMBER}, ${arg(0)}, ${arg(1)}, ${arg(2)}, ${arg(3)}, ${strTag}, ${arrTag});${E.srcComment(e.loc)}`,
|
||||
);
|
||||
return dict;
|
||||
}
|
||||
// Stats (scr_lib.c): statSync throws like the other sync fs
|
||||
// calls; the getters are pure reads.
|
||||
case "fs.openSync":
|
||||
|
||||
@@ -384,6 +384,14 @@ const LIB_FN_SYMS: Record<string, string> = {
|
||||
"sp.toString": "scr_sp_to_string",
|
||||
"sp.keyAt": "scr_sp_key_at",
|
||||
"sp.valAt": "scr_sp_val_at",
|
||||
// node:querystring (scr_qs.c): unescape/stringify are plain generic
|
||||
// calls (never throw; string results +1), and escape IS the component
|
||||
// encoder (Node's qsEscape set equals encodeURIComponent's) so it emits
|
||||
// the always-linked codec. qs.parse is special-cased in emitLibCall
|
||||
// (the result dictionary's construction).
|
||||
"qs.escape": "scr_str_encode_uri_component",
|
||||
"qs.unescape": "scr_qs_unescape",
|
||||
"qs.stringify": "scr_qs_stringify",
|
||||
// Timer handle bookkeeping (never throws; the comma-shaped chaining
|
||||
// forms are special-cased in emitLibCall).
|
||||
"timers.hasRef": "scr_timer_has_ref",
|
||||
@@ -9319,6 +9327,31 @@ class LlEmitter {
|
||||
for (const a of e.args) this.emitExpr(a);
|
||||
return { name: "", type: e.type };
|
||||
}
|
||||
if (e.fn === "qs.parse") {
|
||||
// The ParsedUrlQuery dictionary: a fresh pure-index-signature
|
||||
// record whose overflow map the runtime scan fills
|
||||
// (scr_qs_parse_into groups repeats into string[] buckets) — the
|
||||
// C emitter's shape exactly. The frontend verified the structure;
|
||||
// lookups here only guard emitter bugs. Args: qs, sep, eq, maxKeys.
|
||||
if (e.type.kind !== "record") throw new Error("llvm emitter bug: qs.parse result is not a record");
|
||||
const dictShape = this.recordsById.get(e.type.shapeId);
|
||||
const iv = dictShape?.indexValue;
|
||||
if (!dictShape || iv?.kind !== "union") throw new Error("llvm emitter bug: qs.parse dict shape");
|
||||
const ivDef = this.unionsById.get(iv.unionId);
|
||||
const strTag = ivDef?.arms.findIndex((a) => a.kind === "string") ?? -1;
|
||||
const arrTag = ivDef?.arms.findIndex((a) => a.kind === "array") ?? -1;
|
||||
if (strTag < 0 || arrTag < 0) throw new Error("llvm emitter bug: qs.parse index union lacks its arms");
|
||||
const args = e.args.map((a) => this.emitExpr(a));
|
||||
this.declare(`declare void @scr_qs_parse_into(ptr, ptr, ptr, ptr, double, i32, i32)`);
|
||||
const dict = B.tmp();
|
||||
B.line(`${dict} = call ptr @${mangleRecordNew(e.type.shapeId)}()`);
|
||||
const out = this.own({ name: dict, type: e.type });
|
||||
const ovf = this.recordOvfPtr(dict, e.type.shapeId);
|
||||
B.line(
|
||||
`call void @scr_qs_parse_into(ptr ${ovf}, ptr ${args[0]!.name}, ptr ${args[1]!.name}, ptr ${args[2]!.name}, double ${args[3]!.name}, i32 ${strTag}, i32 ${arrTag})`,
|
||||
);
|
||||
return out;
|
||||
}
|
||||
if (e.fn === "os.networkInterfaces") {
|
||||
// The Dict<NetworkInterfaceInfo[]> record, built inline from a
|
||||
// getifaddrs(3) snapshot — emit-exprs.ts's builder, block-lowered.
|
||||
|
||||
@@ -13,6 +13,8 @@ import {
|
||||
FS_READDIR_DOCUMENTED_OPTIONS,
|
||||
FS_WATCH_DOCUMENTED_OPTIONS,
|
||||
FS_WRITE_FILE_DOCUMENTED_OPTIONS,
|
||||
QS_PARSE_DOCUMENTED_OPTIONS,
|
||||
QS_STRINGIFY_DOCUMENTED_OPTIONS,
|
||||
READLINE_DOCUMENTED_OPTIONS,
|
||||
builtinConstLit,
|
||||
fenceOrDropOptionKey,
|
||||
@@ -404,6 +406,20 @@ function optionMember(p: ts.ObjectLiteralElementLike): { name: string; value: ts
|
||||
if (bi.module === "os" && bi.member === "userInfo") {
|
||||
return lowerOsUserInfoCall(L, expr, loc);
|
||||
}
|
||||
// node:querystring — parse/stringify are entirely special-cased (the
|
||||
// sep/eq/options completions, parse's call-site-shaped dictionary
|
||||
// result, stringify's DOM-crossing object argument); decode/encode
|
||||
// are Node's own aliases of the pair (`const decode = parse` in the
|
||||
// module source) and take the same lowerings. escape/unescape ride
|
||||
// the generic table tail below.
|
||||
if (bi.module === "querystring") {
|
||||
if (bi.member === "parse" || bi.member === "decode") {
|
||||
return lowerQuerystringParseCall(L, expr, loc);
|
||||
}
|
||||
if (bi.member === "stringify" || bi.member === "encode") {
|
||||
return lowerQuerystringStringifyCall(L, expr, loc);
|
||||
}
|
||||
}
|
||||
// node:timers/promises — setTimeout([delay]) and setImmediate(): void
|
||||
// promises the shared timer heap settles. The omitted delay completes
|
||||
// to Node's 1ms floor (scr_timer_coerce_ms clamps anyway; the literal
|
||||
@@ -4502,6 +4518,187 @@ function optionMember(p: ts.ObjectLiteralElementLike): { name: string; value: ts
|
||||
return { kind: "recordLit", fields, type: result, loc };
|
||||
}
|
||||
|
||||
/** querystring's sep/eq arguments: an omitted argument, the literal
|
||||
* null, and the literal undefined all mean the default (Node's falsy
|
||||
* rule — parse(s, null, null, opts) is the canonical maxKeys spelling);
|
||||
* a string expression passes through (the runtime applies the same
|
||||
* falsy rule to '' at runtime). Everything else fences. */
|
||||
function qsSepEqArg(L: Lowerer, node: ts.Expression | undefined, dflt: string,
|
||||
what: string, loc: SrcLoc,): IrExpr {
|
||||
if (
|
||||
!node ||
|
||||
node.kind === ts.SyntaxKind.NullKeyword ||
|
||||
(ts.isIdentifier(node) && node.text === "undefined")
|
||||
) {
|
||||
return { kind: "strLit", value: dflt, type: STRING, loc };
|
||||
}
|
||||
const v = L.lowerExpr(node);
|
||||
if (v.type.kind !== "string") {
|
||||
L.noLowering(
|
||||
`${what} with a '${L.fmt(v.type)}' separator`,
|
||||
node,
|
||||
"pass a string, or null/undefined for the default (narrow unions first)",
|
||||
);
|
||||
}
|
||||
return v;
|
||||
}
|
||||
|
||||
/** querystring.parse / querystring.decode: the scan runs in the runtime
|
||||
* (scr_qs_parse_into fills the result dictionary's overflow map), so the
|
||||
* frontend completes sep/eq/maxKeys to Node's defaults and verifies the
|
||||
* call site's mapped result IS the ParsedUrlQuery dictionary — a pure
|
||||
* index-signature record over `string | string[]` (an undefined arm
|
||||
* tolerated: @types/node's Dict) — the networkInterfaces verification
|
||||
* stance. The default decoder is the one lowered decoder; a custom
|
||||
* decodeURIComponent option fences by name. */
|
||||
export function lowerQuerystringParseCall(L: Lowerer, call: ts.CallExpression, loc: SrcLoc): IrExpr {
|
||||
if (call.arguments.length > 4 || call.arguments.some(ts.isSpreadElement)) {
|
||||
L.noLowering(`querystring.parse with ${call.arguments.length} arguments`, call);
|
||||
}
|
||||
const fence: () => never = () =>
|
||||
L.noLowering(
|
||||
"querystring.parse where the result is not the ParsedUrlQuery dictionary",
|
||||
call,
|
||||
"the `{ [key: string]: string | string[] }` shape is the supported result",
|
||||
);
|
||||
const result = L.mapTypeOf(L.typeOf(call));
|
||||
if (result?.kind !== "record") fence();
|
||||
const dictShape = L.shapes.get(result.shapeId);
|
||||
if (!dictShape || dictShape.tuple || dictShape.fields.length > 0 || !dictShape.indexValue) fence();
|
||||
const iv = dictShape.indexValue;
|
||||
if (iv.kind !== "union") fence();
|
||||
const ivDef = L.unions.get(iv.unionId);
|
||||
if (!ivDef) fence();
|
||||
let sawStr = false;
|
||||
let sawArr = false;
|
||||
for (const arm of ivDef.arms) {
|
||||
if (arm.kind === "string") sawStr = true;
|
||||
else if (arm.kind === "array" && arm.elem.kind === "string") sawArr = true;
|
||||
// undefined rides @types/node's Dict; the f64 arm is the
|
||||
// header-family canonicalization (types.ts interns every
|
||||
// `string | string[]`-slotted dictionary as the one canonical
|
||||
// header shape, whose slot adds number type-level only — parse
|
||||
// never stores one).
|
||||
else if (arm.kind !== "undefinedT" && arm.kind !== "f64") fence();
|
||||
}
|
||||
if (!sawStr || !sawArr) fence();
|
||||
const str = call.arguments[0]
|
||||
? L.lowerExprExpecting(call.arguments[0], STRING)
|
||||
: L.noLowering("querystring.parse without a query string", call);
|
||||
const sep = qsSepEqArg(L, call.arguments[1], "&", "querystring.parse", loc);
|
||||
const eq = qsSepEqArg(L, call.arguments[2], "=", "querystring.parse", loc);
|
||||
// The options walk: maxKeys lowers (Node's rule — > 0 caps the pair
|
||||
// count, 0 and negatives mean unlimited — lives in the runtime, so
|
||||
// any number expression works); a custom decodeURIComponent changes
|
||||
// every decoded byte and fences by name.
|
||||
let maxKeys: IrExpr = { kind: "numLit", value: 1000, type: F64, loc };
|
||||
const optsNode = call.arguments[3];
|
||||
if (optsNode) {
|
||||
if (!ts.isObjectLiteralExpression(optsNode)) {
|
||||
L.noLowering(
|
||||
"querystring.parse with a non-literal options argument",
|
||||
optsNode,
|
||||
"the supported form spells the options inline: parse(s, sep, eq, { maxKeys: n })",
|
||||
);
|
||||
}
|
||||
for (const p of optsNode.properties) {
|
||||
const m = optionMember(p);
|
||||
if (!m) {
|
||||
L.noLowering(
|
||||
"querystring.parse with this options shape",
|
||||
p,
|
||||
"spreads and computed keys have no lowering — write each member inline",
|
||||
);
|
||||
}
|
||||
if (m.name === "maxKeys") {
|
||||
maxKeys = L.lowerExprExpecting(m.value, F64);
|
||||
} else if (m.name === "decodeURIComponent") {
|
||||
L.noLowering(
|
||||
"querystring.parse with a custom decodeURIComponent",
|
||||
p,
|
||||
"the default decoder is the lowered surface (strict decodeURIComponent with Node's lenient fallback)",
|
||||
);
|
||||
} else {
|
||||
fenceOrDropOptionKey(
|
||||
L, p, m.name, "querystring.parse", QS_PARSE_DOCUMENTED_OPTIONS,
|
||||
"maxKeys is the supported option",
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
return { kind: "libCall", fn: "qs.parse", args: [str, sep, eq, maxKeys], type: result, loc };
|
||||
}
|
||||
|
||||
/** querystring.stringify / querystring.encode: the object crosses as a
|
||||
* DOM value (dynFrom — JSON-safe records and, in JS sources, dyn values
|
||||
* directly) and Node's encodeStringified rules run in the runtime
|
||||
* (scr_qs_stringify), so arrays expand to repeated keys and
|
||||
* null/undefined values are empty. The default encoder is the one
|
||||
* lowered encoder; a custom encodeURIComponent option fences by name. */
|
||||
export function lowerQuerystringStringifyCall(L: Lowerer, call: ts.CallExpression, loc: SrcLoc): IrExpr {
|
||||
if (call.arguments.length > 4 || call.arguments.some(ts.isSpreadElement)) {
|
||||
L.noLowering(`querystring.stringify with ${call.arguments.length} arguments`, call);
|
||||
}
|
||||
const objNode = call.arguments[0];
|
||||
// stringify() / stringify(undefined) / stringify(null): Node answers
|
||||
// '' for every non-object — the constant folds.
|
||||
if (
|
||||
!objNode ||
|
||||
objNode.kind === ts.SyntaxKind.NullKeyword ||
|
||||
(ts.isIdentifier(objNode) && objNode.text === "undefined")
|
||||
) {
|
||||
return { kind: "strLit", value: "", type: STRING, loc };
|
||||
}
|
||||
const objV = L.lowerExpr(objNode);
|
||||
let obj: IrExpr;
|
||||
if (objV.type.kind === "dyn") {
|
||||
obj = objV;
|
||||
} else if (objV.type.kind === "record" && L.dynConvertible(objV.type)) {
|
||||
obj = { kind: "dynFrom", value: objV, type: DYN, loc };
|
||||
} else {
|
||||
L.noLowering(
|
||||
`querystring.stringify of '${L.fmt(objV.type)}' values`,
|
||||
objNode,
|
||||
"pass a record of string/number/boolean values (arrays of those expand to repeated keys; null/undefined values serialize empty)",
|
||||
);
|
||||
}
|
||||
const sep = qsSepEqArg(L, call.arguments[1], "&", "querystring.stringify", loc);
|
||||
const eq = qsSepEqArg(L, call.arguments[2], "=", "querystring.stringify", loc);
|
||||
const optsNode = call.arguments[3];
|
||||
if (optsNode) {
|
||||
if (!ts.isObjectLiteralExpression(optsNode)) {
|
||||
L.noLowering(
|
||||
"querystring.stringify with a non-literal options argument",
|
||||
optsNode,
|
||||
"the supported form spells the options inline (and the only documented option, encodeURIComponent, has no lowering)",
|
||||
);
|
||||
}
|
||||
for (const p of optsNode.properties) {
|
||||
const m = optionMember(p);
|
||||
if (!m) {
|
||||
L.noLowering(
|
||||
"querystring.stringify with this options shape",
|
||||
p,
|
||||
"spreads and computed keys have no lowering — write each member inline",
|
||||
);
|
||||
}
|
||||
if (m.name === "encodeURIComponent") {
|
||||
L.noLowering(
|
||||
"querystring.stringify with a custom encodeURIComponent",
|
||||
p,
|
||||
"the default encoder (querystring.escape's component set) is the lowered surface",
|
||||
);
|
||||
} else {
|
||||
fenceOrDropOptionKey(
|
||||
L, p, m.name, "querystring.stringify", QS_STRINGIFY_DOCUMENTED_OPTIONS,
|
||||
"no stringify options have a lowering",
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
return { kind: "libCall", fn: "qs.stringify", args: [obj, sep, eq], type: STRING, loc };
|
||||
}
|
||||
|
||||
export function lowerOsNetworkInterfacesCall(L: Lowerer, call: ts.CallExpression, loc: SrcLoc): IrExpr {
|
||||
if (call.arguments.length !== 0) {
|
||||
L.noLowering(`networkInterfaces with ${call.arguments.length} arguments`, call, "networkInterfaces() takes no arguments");
|
||||
|
||||
@@ -230,6 +230,16 @@ export const DNS_LOOKUP_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
|
||||
"family", "hints", "all", "order", "verbatim",
|
||||
]);
|
||||
|
||||
/** querystring.parse's documented option keys (Node v24). */
|
||||
export const QS_PARSE_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
|
||||
"maxKeys", "decodeURIComponent",
|
||||
]);
|
||||
|
||||
/** querystring.stringify's documented option keys (Node v24). */
|
||||
export const QS_STRINGIFY_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
|
||||
"encodeURIComponent",
|
||||
]);
|
||||
|
||||
/** readline.createInterface's documented option keys. */
|
||||
export const READLINE_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
|
||||
"input", "output", "completer", "terminal", "history", "historySize",
|
||||
@@ -654,6 +664,23 @@ export const BUILTIN_MODULE_FNS: Record<string, Record<string, BuiltinModuleFn |
|
||||
// end are special-cased (lowerNew + lowerStringDecoderMethodCall); no
|
||||
// function members exist to table.
|
||||
string_decoder: {},
|
||||
// node:querystring — the legacy query-string codec (NOT URLSearchParams;
|
||||
// the escaping and '+' rules differ). escape/unescape ride the table
|
||||
// path directly; parse and stringify are special-cased in
|
||||
// lowerBuiltinModuleCall (parse's result is the call site's mapped
|
||||
// ParsedUrlQuery dictionary — the networkInterfaces verification stance
|
||||
// — and its sep/eq/options complete there; stringify's object argument
|
||||
// crosses as a DOM value). decode/encode are Node's own aliases of
|
||||
// parse/stringify (`const decode = parse` in lib/querystring.js) and
|
||||
// route to the same special cases; the entries carry canonical shapes.
|
||||
querystring: {
|
||||
parse: { fn: "qs.parse", params: [STRING], result: VOID },
|
||||
decode: { fn: "qs.parse", params: [STRING], result: VOID },
|
||||
stringify: { fn: "qs.stringify", params: [DYN], result: STRING },
|
||||
encode: { fn: "qs.stringify", params: [DYN], result: STRING },
|
||||
escape: { fn: "qs.escape", params: [STRING], result: STRING },
|
||||
unescape: { fn: "qs.unescape", params: [STRING], result: STRING },
|
||||
},
|
||||
// node:readline: createInterface's options are entirely special-cased
|
||||
// (exactly { input: process.stdin, output?: process.stdout } — see
|
||||
// lowerBuiltinModuleCall); the entry carries the canonical shape. The
|
||||
|
||||
@@ -50,7 +50,7 @@ export function isNodeTypesPath(file: string): boolean {
|
||||
* this is exactly the set of `declare module` names in that file; when
|
||||
* @types/node stands in (which declares ALL node builtins) the supported
|
||||
* surface must not widen, so preflight allowlists this same fixed set. */
|
||||
export const SUPPORTED_BUILTIN_MODULES = ["fs", "path", "path/posix", "path/win32", "os", "url", "fs/promises", "crypto", "zlib", "child_process", "net", "http", "tls", "https", "dgram", "dns", "util", "util/types", "string_decoder", "readline", "http2", "assert", "assert/strict", "worker_threads", "buffer", "cluster", "tty", "async_hooks", "events", "stream", "test", "timers", "timers/promises", "diagnostics_channel", "perf_hooks", "module"] as const;
|
||||
export const SUPPORTED_BUILTIN_MODULES = ["fs", "path", "path/posix", "path/win32", "os", "url", "fs/promises", "crypto", "zlib", "child_process", "net", "http", "tls", "https", "dgram", "dns", "util", "util/types", "string_decoder", "querystring", "readline", "http2", "assert", "assert/strict", "worker_threads", "buffer", "cluster", "tty", "async_hooks", "events", "stream", "test", "timers", "timers/promises", "diagnostics_channel", "perf_hooks", "module"] as const;
|
||||
|
||||
/** Builtins Node itself serves ONLY under the node: prefix —
|
||||
* require("test") is MODULE_NOT_FOUND in Node, so the bare name stays a
|
||||
|
||||
@@ -4,7 +4,7 @@ import { compileC, resolveCc, targetPlatform } from "./backend/cc.js";
|
||||
import { emitModule } from "./backend/emission/emitter.js";
|
||||
import { emitLlvmModule, LlvmUnsupportedError } from "./backend/llvm/emitter.js";
|
||||
import { checkerPanicDiag, iceDiag, isCheckerPanic, type ScrDiagnostic } from "./diagnostics/diagnostic.js";
|
||||
import { moduleEmbedsBuiltin, moduleEmbedsCompressedNpm, moduleUsesAssert, moduleUsesDc, moduleUsesDgram, moduleUsesDynAsync, moduleUsesDynInvoke, moduleUsesEmitter, moduleUsesFetch, moduleUsesFsWatch, moduleUsesHttp2, moduleUsesHttpServer, moduleUsesInspect, moduleUsesNet, moduleUsesNodeTest, moduleUsesProcessEvents, moduleUsesRegex, moduleUsesSearchParams, moduleUsesStream, moduleUsesSymbol, moduleUsesTls, moduleUsesZlib } from "./ir/nodes.js";
|
||||
import { moduleEmbedsBuiltin, moduleEmbedsCompressedNpm, moduleUsesAssert, moduleUsesDc, moduleUsesDgram, moduleUsesDynAsync, moduleUsesDynInvoke, moduleUsesEmitter, moduleUsesFetch, moduleUsesFsWatch, moduleUsesHttp2, moduleUsesHttpServer, moduleUsesInspect, moduleUsesNet, moduleUsesNodeTest, moduleUsesProcessEvents, moduleUsesQs, moduleUsesRegex, moduleUsesSearchParams, moduleUsesStream, moduleUsesSymbol, moduleUsesTls, moduleUsesZlib } from "./ir/nodes.js";
|
||||
import { serializeModule } from "./ir/serialize.js";
|
||||
import { validateModule } from "./ir/validate.js";
|
||||
import { checkPreflight, isNodeTypesPath, loadProgram, resolveNpmImport, type LoadResult } from "./frontend/program.js";
|
||||
@@ -494,6 +494,9 @@ export async function compile(entryPath: string, opts: CompileOptions): Promise<
|
||||
// The link switch for scr_url_params.c: sp.* libCalls, the
|
||||
// url.searchParams getter, or a searchParams-kind type on the IR.
|
||||
searchParams: moduleUsesSearchParams(lowered.module),
|
||||
// The link switch for scr_qs.c: the qs.* libCalls that live there
|
||||
// (parse/stringify/unescape; escape rides the always-linked encoder).
|
||||
qs: moduleUsesQs(lowered.module),
|
||||
// The link switch for scr_stream.c: the node:stream class surface on
|
||||
// the IR (stream libCalls or the %Readable-family class defs).
|
||||
stream: moduleUsesStream(lowered.module),
|
||||
|
||||
@@ -1715,6 +1715,31 @@ export type IrLibFn =
|
||||
| "sp.toString"
|
||||
| "sp.keyAt"
|
||||
| "sp.valAt"
|
||||
/** node:querystring (scr_qs.c — link-gated by moduleUsesQs; NOT
|
||||
* URLSearchParams: the legacy codec's escaping and '+' rules differ).
|
||||
* qs.escape is Node's qsEscape, which encodes exactly the component
|
||||
* unreserved set — it emits the always-linked
|
||||
* scr_str_encode_uri_component, so escape-only programs never pull the
|
||||
* unit. qs.unescape is Node's qsUnescape: strict decodeURIComponent
|
||||
* first, the lenient legacy unescapeBuffer fallback on failure (never
|
||||
* throws). qs.parse takes (str, sep, eq, maxKeys) — the frontend
|
||||
* completes omitted/null sep/eq to "&"/"=" and the omitted maxKeys
|
||||
* option to Node's 1000 (0 and negatives mean unlimited, Node's rule) —
|
||||
* and its result type is the CALL SITE's mapped ParsedUrlQuery shape (a
|
||||
* pure index-signature record over `string | string[]`, an undefined
|
||||
* arm tolerated — @types/node's Dict), verified structurally by the
|
||||
* frontend (lowerQuerystringParseCall); the emitters construct the
|
||||
* record and hand its overflow map to scr_qs_parse_into with the two
|
||||
* union tags. qs.stringify takes (obj, sep, eq) with obj a DOM value
|
||||
* (the frontend dynFroms the typed record; JS-world dyn values pass
|
||||
* through) — Node's encodeStringified rules run in the runtime, so
|
||||
* arrays expand to repeated keys and null/undefined/nested objects are
|
||||
* empty values. Custom encoder/decoder options fence at compile time.
|
||||
* All borrow; string results +1; none throw. */
|
||||
| "qs.parse"
|
||||
| "qs.stringify"
|
||||
| "qs.escape"
|
||||
| "qs.unescape"
|
||||
/** ES Symbol values (scr_symbol.c — link-gated by moduleUsesSymbol).
|
||||
* sym.new: `Symbol(desc)` — a fresh runtime-unique identity (+1) whose
|
||||
* one arg is the description string (borrowed); sym.newAnon is the
|
||||
@@ -5311,6 +5336,33 @@ export function moduleUsesSearchParams(mod: IrModule): boolean {
|
||||
return found;
|
||||
}
|
||||
|
||||
/** True when the module uses the node:querystring surface — the qs.*
|
||||
* libCalls that live in scr_qs.c (parse/stringify/unescape; qs.escape
|
||||
* emits the always-linked component encoder and deliberately does NOT
|
||||
* flip this switch) — the link switch that pulls scr_qs.c into the
|
||||
* binary (the moduleUsesSearchParams precedent: pure data transforms, no
|
||||
* loop hooks, cross-compiles everywhere). qs-free programs keep their
|
||||
* exact link line. Same generic-walk shape as moduleUsesZlib. */
|
||||
export function moduleUsesQs(mod: IrModule): boolean {
|
||||
let found = false;
|
||||
const visit = (v: unknown): void => {
|
||||
if (found || v === null || typeof v !== "object") return;
|
||||
if (Array.isArray(v)) {
|
||||
for (const item of v) visit(item);
|
||||
return;
|
||||
}
|
||||
const node = v as { kind?: unknown; fn?: unknown };
|
||||
if (node.kind === "libCall" && typeof node.fn === "string" &&
|
||||
(node.fn === "qs.parse" || node.fn === "qs.stringify" || node.fn === "qs.unescape")) {
|
||||
found = true;
|
||||
return;
|
||||
}
|
||||
for (const key of Object.keys(v)) visit((v as Record<string, unknown>)[key]);
|
||||
};
|
||||
visit(mod);
|
||||
return found;
|
||||
}
|
||||
|
||||
/** True when the module contains any fs.watch/watcher.* libCall — the
|
||||
* link switch that pulls scr_watch.c into the binary and has the emitted
|
||||
* main call scr_watch_install (cc.ts + emitter; the scr_net gating
|
||||
|
||||
@@ -251,6 +251,14 @@ export const LIB_FN_SIGS: Record<IrLibFn, { argTypes: (IrType | null)[]; result:
|
||||
"sp.toString": { argTypes: [SEARCH_PARAMS_T], result: STRING },
|
||||
"sp.keyAt": { argTypes: [SEARCH_PARAMS_T, F64], result: STRING },
|
||||
"sp.valAt": { argTypes: [SEARCH_PARAMS_T, F64], result: STRING },
|
||||
// node:querystring. qs.parse's result is the call site's ParsedUrlQuery
|
||||
// dictionary record (VOID is the networkInterfaces sentinel — the
|
||||
// libCall case checks the structure); qs.stringify's object argument
|
||||
// is a DOM value (the frontend dynFroms typed records).
|
||||
"qs.parse": { argTypes: [STRING, STRING, STRING, F64], result: VOID },
|
||||
"qs.stringify": { argTypes: [DYN, STRING, STRING], result: STRING },
|
||||
"qs.escape": { argTypes: [STRING], result: STRING },
|
||||
"qs.unescape": { argTypes: [STRING], result: STRING },
|
||||
"fs.statSync": { argTypes: [STRING], result: STATS_T },
|
||||
"fs.lstatSync": { argTypes: [STRING], result: STATS_T },
|
||||
"fs.openSync": { argTypes: [STRING, STRING], result: F64 },
|
||||
@@ -3534,6 +3542,25 @@ function validateFunction(
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (e.fn === "qs.parse") {
|
||||
// Result: a pure index-signature record whose value union
|
||||
// carries a string arm and a string[] arm (undefined tolerated
|
||||
// — @types/node's Dict — and f64 too: the header-family
|
||||
// canonicalization interns every such dictionary with the
|
||||
// number arm, type-level only) — the structure
|
||||
// lowerQuerystringParseCall pinned.
|
||||
const shape = e.type.kind === "record" ? records.get(e.type.shapeId) : undefined;
|
||||
let ok = shape !== undefined && !shape.tuple && shape.fields.length === 0 && shape.indexValue !== undefined;
|
||||
const ivDef = ok && shape!.indexValue!.kind === "union" ? unions.get(shape!.indexValue!.unionId) : undefined;
|
||||
ok = ok && ivDef !== undefined &&
|
||||
ivDef.arms.some((a) => a.kind === "string") &&
|
||||
ivDef.arms.some((a) => a.kind === "array" && a.elem.kind === "string") &&
|
||||
ivDef.arms.every((a) => a.kind === "string" || a.kind === "array" || a.kind === "undefinedT" || a.kind === "f64");
|
||||
if (!ok) {
|
||||
err(`libCall qs.parse must return the ParsedUrlQuery dictionary record`, e.loc);
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (e.fn === "child.onExit" || e.fn === "child.onError") {
|
||||
// The listener: a closure with no params, or exactly the
|
||||
// supported parameter shapes per event — exit takes (code:
|
||||
|
||||
@@ -1113,6 +1113,49 @@
|
||||
"status": "static",
|
||||
"note": "recognized module (bare and node:-prefixed specifiers)"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring",
|
||||
"status": "static",
|
||||
"note": "recognized module (bare and node:-prefixed specifiers)"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring.decode",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring.decode",
|
||||
"status": "static"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring.encode",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring.encode",
|
||||
"status": "static"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring.escape",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring.escape",
|
||||
"status": "static"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring.parse",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring.parse",
|
||||
"status": "static"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring.stringify",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring.stringify",
|
||||
"status": "static"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.querystring.unescape",
|
||||
"kind": "node-builtin",
|
||||
"name": "querystring.unescape",
|
||||
"status": "static"
|
||||
},
|
||||
{
|
||||
"id": "node-builtin.readline",
|
||||
"kind": "node-builtin",
|
||||
|
||||
@@ -0,0 +1,496 @@
|
||||
/* node:querystring (scr_runtime.h has the API contract): Node v24's
|
||||
* legacy query-string codec — NOT URLSearchParams (the escaping and '+'
|
||||
* rules differ; that surface lives in scr_url_params.c). LINK-GATED:
|
||||
* compiled into the binary only when the IR uses the qs surface
|
||||
* (moduleUsesQs — the scr_url_params gating precedent), so qs-free
|
||||
* programs pay zero bytes. querystring.escape never reaches this unit at
|
||||
* all: Node's qsEscape encodes exactly the component unreserved set, so
|
||||
* the frontend lowers it to the always-linked
|
||||
* scr_str_encode_uri_component.
|
||||
*
|
||||
* Everything here is a quirk-faithful port of lib/querystring.js (Node is
|
||||
* the oracle): parse's interleaved sep/eq scan with its naive
|
||||
* partial-match resets and the encodeCheck fast path, unescapeBuffer's
|
||||
* UTF-16-code-unit byte truncation, and stringify's encodeStringified
|
||||
* value rules. The scans run byte-wise over the runtime's UTF-8 storage —
|
||||
* equivalent to Node's code-unit scans for well-formed input, because the
|
||||
* machine only assigns meaning to ASCII ('%', '+', hex, and the sep/eq
|
||||
* sequences positionally) and UTF-8 multibyte alignment mirrors UTF-16
|
||||
* unit alignment. */
|
||||
#include "scr_runtime.h"
|
||||
|
||||
#include <math.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
/* ── a tiny growable byte buffer (scr_url_params.c's, redeclared) ────── */
|
||||
|
||||
typedef struct {
|
||||
char *data;
|
||||
size_t len;
|
||||
size_t cap;
|
||||
} QsBuf;
|
||||
|
||||
static void qb_init(QsBuf *b) {
|
||||
b->cap = 64;
|
||||
b->len = 0;
|
||||
b->data = malloc(b->cap);
|
||||
if (!b->data) {
|
||||
fputs("scriptc: out of memory\n", stderr);
|
||||
abort();
|
||||
}
|
||||
}
|
||||
|
||||
static void qb_append(QsBuf *b, const char *bytes, size_t n) {
|
||||
if (b->len + n > b->cap) {
|
||||
while (b->len + n > b->cap) b->cap *= 2;
|
||||
char *grown = realloc(b->data, b->cap);
|
||||
if (!grown) {
|
||||
fputs("scriptc: out of memory\n", stderr);
|
||||
abort();
|
||||
}
|
||||
b->data = grown;
|
||||
}
|
||||
memcpy(b->data + b->len, bytes, n);
|
||||
b->len += n;
|
||||
}
|
||||
|
||||
static void qb_push(QsBuf *b, char c) { qb_append(b, &c, 1); }
|
||||
|
||||
static void qb_append_str(QsBuf *b, const ScrStr *s) {
|
||||
qb_append(b, s->data, s->len);
|
||||
}
|
||||
|
||||
static ScrStr *qb_take(QsBuf *b) {
|
||||
ScrStr *s = scr_str_new(b->data, b->len);
|
||||
free(b->data);
|
||||
return s;
|
||||
}
|
||||
|
||||
/* ── shared scan helpers ─────────────────────────────────────────────── */
|
||||
|
||||
static int qs_unhex(unsigned c) {
|
||||
if (c >= '0' && c <= '9') return (int)(c - '0');
|
||||
if (c >= 'A' && c <= 'F') return (int)(c - 'A' + 10);
|
||||
if (c >= 'a' && c <= 'f') return (int)(c - 'a' + 10);
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* One UTF-8 code point at p (well-formed by the runtime invariant);
|
||||
* *adv gets the byte length. */
|
||||
static uint32_t qs_utf8_decode(const char *p, size_t *adv) {
|
||||
unsigned char b0 = (unsigned char)p[0];
|
||||
if (b0 < 0x80) {
|
||||
*adv = 1;
|
||||
return b0;
|
||||
}
|
||||
if (b0 < 0xE0) {
|
||||
*adv = 2;
|
||||
return (uint32_t)((b0 & 0x1Fu) << 6) | ((unsigned char)p[1] & 0x3Fu);
|
||||
}
|
||||
if (b0 < 0xF0) {
|
||||
*adv = 3;
|
||||
return (uint32_t)((b0 & 0x0Fu) << 12) |
|
||||
(uint32_t)(((unsigned char)p[1] & 0x3Fu) << 6) |
|
||||
((unsigned char)p[2] & 0x3Fu);
|
||||
}
|
||||
*adv = 4;
|
||||
return (uint32_t)((b0 & 0x07u) << 18) |
|
||||
(uint32_t)(((unsigned char)p[1] & 0x3Fu) << 12) |
|
||||
(uint32_t)(((unsigned char)p[2] & 0x3Fu) << 6) |
|
||||
((unsigned char)p[3] & 0x3Fu);
|
||||
}
|
||||
|
||||
static void qs_utf8_encode(QsBuf *b, uint32_t cp) {
|
||||
char tmp[4];
|
||||
if (cp < 0x80) {
|
||||
tmp[0] = (char)cp;
|
||||
qb_append(b, tmp, 1);
|
||||
} else if (cp < 0x800) {
|
||||
tmp[0] = (char)(0xC0 | (cp >> 6));
|
||||
tmp[1] = (char)(0x80 | (cp & 0x3F));
|
||||
qb_append(b, tmp, 2);
|
||||
} else if (cp < 0x10000) {
|
||||
tmp[0] = (char)(0xE0 | (cp >> 12));
|
||||
tmp[1] = (char)(0x80 | ((cp >> 6) & 0x3F));
|
||||
tmp[2] = (char)(0x80 | (cp & 0x3F));
|
||||
qb_append(b, tmp, 3);
|
||||
} else {
|
||||
tmp[0] = (char)(0xF0 | (cp >> 18));
|
||||
tmp[1] = (char)(0x80 | ((cp >> 12) & 0x3F));
|
||||
tmp[2] = (char)(0x80 | ((cp >> 6) & 0x3F));
|
||||
tmp[3] = (char)(0x80 | (cp & 0x3F));
|
||||
qb_append(b, tmp, 4);
|
||||
}
|
||||
}
|
||||
|
||||
/* ── unescape (Node's qsUnescape) ────────────────────────────────────── */
|
||||
|
||||
/* The string's UTF-16 code units (astral chars split into surrogate
|
||||
* pairs) — unescapeBuffer's scan is defined over these, byte-truncation
|
||||
* quirk included, so the fallback builds them first. Caller frees. */
|
||||
static uint16_t *qs_utf16_units(const ScrStr *s, size_t *out_n) {
|
||||
/* One unit per byte is a safe cap (ASCII 1:1; multibyte shrinks). */
|
||||
uint16_t *units = malloc(sizeof(uint16_t) * (s->len ? s->len : 1));
|
||||
if (!units) {
|
||||
fputs("scriptc: out of memory\n", stderr);
|
||||
abort();
|
||||
}
|
||||
size_t n = 0, i = 0;
|
||||
while (i < s->len) {
|
||||
size_t adv;
|
||||
uint32_t cp = qs_utf8_decode(s->data + i, &adv);
|
||||
i += adv;
|
||||
if (cp >= 0x10000) {
|
||||
cp -= 0x10000;
|
||||
units[n++] = (uint16_t)(0xD800 + (cp >> 10));
|
||||
units[n++] = (uint16_t)(0xDC00 + (cp & 0x3FF));
|
||||
} else {
|
||||
units[n++] = (uint16_t)cp;
|
||||
}
|
||||
}
|
||||
*out_n = n;
|
||||
return units;
|
||||
}
|
||||
|
||||
/* WHATWG UTF-8 decode with replacement (maximal subpart) — the semantics
|
||||
* of Buffer.prototype.toString('utf8') the fallback path ends with. */
|
||||
static ScrStr *qs_utf8_replace_decode(const unsigned char *p, size_t n) {
|
||||
QsBuf out;
|
||||
qb_init(&out);
|
||||
size_t i = 0;
|
||||
while (i < n) {
|
||||
unsigned char b0 = p[i];
|
||||
if (b0 < 0x80) {
|
||||
qb_push(&out, (char)b0);
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
int need;
|
||||
unsigned lo = 0x80, hi = 0xBF;
|
||||
if (b0 >= 0xC2 && b0 <= 0xDF) need = 1;
|
||||
else if (b0 == 0xE0) { need = 2; lo = 0xA0; }
|
||||
else if (b0 == 0xED) { need = 2; hi = 0x9F; }
|
||||
else if (b0 >= 0xE1 && b0 <= 0xEF) need = 2;
|
||||
else if (b0 == 0xF0) { need = 3; lo = 0x90; }
|
||||
else if (b0 == 0xF4) { need = 3; hi = 0x8F; }
|
||||
else if (b0 >= 0xF1 && b0 <= 0xF3) need = 3;
|
||||
else { /* 0x80..0xC1, 0xF5..0xFF: never a leading byte */
|
||||
qs_utf8_encode(&out, 0xFFFD);
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
/* Consume as many valid continuations as exist (maximal subpart):
|
||||
* an invalid or missing continuation replaces the consumed prefix
|
||||
* with ONE U+FFFD and rescans at the offending byte. */
|
||||
uint32_t cp = b0 & (uint32_t)(0xFF >> (need + 2));
|
||||
size_t j = i + 1;
|
||||
int k = 0;
|
||||
bool ok = true;
|
||||
for (; k < need; k++, j++) {
|
||||
if (j >= n) { ok = false; break; }
|
||||
unsigned c = p[j];
|
||||
unsigned clo = (k == 0) ? lo : 0x80, chi = (k == 0) ? hi : 0xBF;
|
||||
if (c < clo || c > chi) { ok = false; break; }
|
||||
cp = (cp << 6) | (c & 0x3F);
|
||||
}
|
||||
if (!ok) {
|
||||
qs_utf8_encode(&out, 0xFFFD);
|
||||
i = j; /* prefix consumed; rescan at the offending byte */
|
||||
continue;
|
||||
}
|
||||
qs_utf8_encode(&out, cp);
|
||||
i = j;
|
||||
}
|
||||
return qb_take(&out);
|
||||
}
|
||||
|
||||
/* Node's unescapeBuffer(s) (decodeSpaces false — neither qsUnescape nor
|
||||
* the exported unescape passes it) over the UTF-16 units, then the
|
||||
* replacement decode of the byte buffer. */
|
||||
static ScrStr *qs_unescape_buffer(const ScrStr *s) {
|
||||
size_t n;
|
||||
uint16_t *units = qs_utf16_units(s, &n);
|
||||
unsigned char *out = malloc(n ? n : 1);
|
||||
if (!out) {
|
||||
fputs("scriptc: out of memory\n", stderr);
|
||||
abort();
|
||||
}
|
||||
size_t w = 0, index = 0;
|
||||
/* maxLength = s.length - 2 as a signed comparison (n < 2 must refuse
|
||||
* every escape, matching Node's `index < maxLength` on negatives). */
|
||||
while (index < n) {
|
||||
unsigned cur = units[index];
|
||||
if (cur == '%' && n >= 3 && index < n - 2) {
|
||||
unsigned c1 = units[index + 1];
|
||||
int hexHigh = (c1 < 256) ? qs_unhex(c1) : -1;
|
||||
if (hexHigh < 0) {
|
||||
out[w++] = '%';
|
||||
index++;
|
||||
continue;
|
||||
}
|
||||
unsigned c2 = units[index + 2];
|
||||
int hexLow = (c2 < 256) ? qs_unhex(c2) : -1;
|
||||
if (hexLow < 0) {
|
||||
out[w++] = '%';
|
||||
index++; /* Node: ++index then index-- — net one step */
|
||||
continue;
|
||||
}
|
||||
cur = (unsigned)(hexHigh * 16 + hexLow);
|
||||
index += 2;
|
||||
}
|
||||
out[w++] = (unsigned char)(cur & 0xFF); /* Buffer write truncates */
|
||||
index++;
|
||||
}
|
||||
free(units);
|
||||
ScrStr *res = qs_utf8_replace_decode(out, w);
|
||||
free(out);
|
||||
return res;
|
||||
}
|
||||
|
||||
ScrStr *scr_qs_unescape(const ScrStr *s) {
|
||||
ScrStr *strict = scr_str_decode_uri_component_try((ScrStr *)s);
|
||||
if (strict) return strict;
|
||||
return qs_unescape_buffer(s);
|
||||
}
|
||||
|
||||
/* ── parse ───────────────────────────────────────────────────────────── */
|
||||
|
||||
/* addKeyVal's tail: decode-if-encoded, then group into the overflow map
|
||||
* (first value = the str_tag arm; a repeat REPLACES it with a two-element
|
||||
* string[] in the arr_tag arm; later repeats push). key/value move in. */
|
||||
static void qs_add_key_val(ScrMap *out, ScrStr *key, ScrStr *value,
|
||||
bool key_encoded, bool val_encoded,
|
||||
uint32_t str_tag, uint32_t arr_tag) {
|
||||
if (key->len > 0 && key_encoded) {
|
||||
ScrStr *dec = scr_qs_unescape(key);
|
||||
scr_str_release(key);
|
||||
key = dec;
|
||||
}
|
||||
if (value->len > 0 && val_encoded) {
|
||||
ScrStr *dec = scr_qs_unescape(value);
|
||||
scr_str_release(value);
|
||||
value = dec;
|
||||
}
|
||||
ScrUnion *cell = scr_map_get_str_ref(out, key);
|
||||
if (!cell) {
|
||||
scr_map_set_str_ref(out, key,
|
||||
scr_union_new_ref(str_tag, value, &scr_str_retain_v,
|
||||
&scr_str_release_v, NULL));
|
||||
scr_str_release(key);
|
||||
return;
|
||||
}
|
||||
if (cell->tag == str_tag) {
|
||||
/* obj[key] = [curValue, value] */
|
||||
ScrArr *rows = scr_arr_new(SCR_ELEM_STR, 2);
|
||||
scr_arr_push_ref(rows, scr_str_retain_v(scr_union_peek(cell)));
|
||||
scr_arr_push_ref(rows, value);
|
||||
scr_map_set_str_ref(out, key,
|
||||
scr_union_new_ref(arr_tag, rows, &scr_arr_retain_v,
|
||||
&scr_arr_release_v, NULL));
|
||||
} else {
|
||||
scr_arr_push_ref((ScrArr *)scr_union_peek(cell), value);
|
||||
}
|
||||
scr_union_release(cell);
|
||||
scr_str_release(key);
|
||||
}
|
||||
|
||||
void scr_qs_parse_into(ScrMap *out, const ScrStr *qs, const ScrStr *sep,
|
||||
const ScrStr *eq, double max_keys, uint32_t str_tag,
|
||||
uint32_t arr_tag) {
|
||||
if (qs->len == 0) return;
|
||||
/* Node's falsy rule: undefined/null/'' all mean the default. */
|
||||
const char *sep_b = (sep && sep->len) ? sep->data : "&";
|
||||
size_t sep_len = (sep && sep->len) ? sep->len : 1;
|
||||
const char *eq_b = (eq && eq->len) ? eq->data : "=";
|
||||
size_t eq_len = (eq && eq->len) ? eq->len : 1;
|
||||
|
||||
/* pairs: Node's `maxKeys > 0 ? maxKeys : -1` as a double so Infinity
|
||||
* decrements forever, exactly like Node's -1 sentinel. */
|
||||
double pairs = max_keys > 0 ? max_keys : -1;
|
||||
|
||||
const char *p = qs->data;
|
||||
size_t n = qs->len;
|
||||
QsBuf key, value;
|
||||
qb_init(&key);
|
||||
qb_init(&value);
|
||||
size_t last_pos = 0, sep_idx = 0, eq_idx = 0;
|
||||
bool key_encoded = false, val_encoded = false;
|
||||
int encode_check = 0;
|
||||
bool returned = false;
|
||||
|
||||
for (size_t i = 0; i < n; ++i) {
|
||||
unsigned char code = (unsigned char)p[i];
|
||||
|
||||
/* Try matching the pair separator (e.g. '&'). */
|
||||
if (code == (unsigned char)sep_b[sep_idx]) {
|
||||
if (++sep_idx == sep_len) {
|
||||
/* Key/value pair separator match. */
|
||||
size_t end = i - sep_idx + 1;
|
||||
if (eq_idx < eq_len) {
|
||||
/* No (entire) key/value separator seen. */
|
||||
if (last_pos < end) {
|
||||
qb_append(&key, p + last_pos, end - last_pos);
|
||||
} else if (key.len == 0) {
|
||||
/* An empty substring between separators. */
|
||||
if (--pairs == 0) { returned = true; break; }
|
||||
last_pos = i + 1;
|
||||
sep_idx = eq_idx = 0;
|
||||
continue;
|
||||
}
|
||||
} else if (last_pos < end) {
|
||||
qb_append(&value, p + last_pos, end - last_pos);
|
||||
}
|
||||
|
||||
qs_add_key_val(out, scr_str_new(key.data, key.len),
|
||||
scr_str_new(value.data, value.len), key_encoded,
|
||||
val_encoded, str_tag, arr_tag);
|
||||
|
||||
if (--pairs == 0) { returned = true; break; }
|
||||
key_encoded = val_encoded = false;
|
||||
key.len = value.len = 0;
|
||||
encode_check = 0;
|
||||
last_pos = i + 1;
|
||||
sep_idx = eq_idx = 0;
|
||||
}
|
||||
} else {
|
||||
sep_idx = 0;
|
||||
/* Try matching the key/value separator (e.g. '=') if we haven't. */
|
||||
if (eq_idx < eq_len) {
|
||||
if (code == (unsigned char)eq_b[eq_idx]) {
|
||||
if (++eq_idx == eq_len) {
|
||||
/* Key/value separator match. */
|
||||
size_t end = i - eq_idx + 1;
|
||||
if (last_pos < end) qb_append(&key, p + last_pos, end - last_pos);
|
||||
encode_check = 0;
|
||||
last_pos = i + 1;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
eq_idx = 0;
|
||||
if (!key_encoded) {
|
||||
/* Match a valid encoded byte once, to minimize decode calls. */
|
||||
if (code == '%') {
|
||||
encode_check = 1;
|
||||
continue;
|
||||
} else if (encode_check > 0) {
|
||||
if (qs_unhex(code) >= 0) {
|
||||
if (++encode_check == 3) key_encoded = true;
|
||||
continue;
|
||||
} else {
|
||||
encode_check = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (code == '+') {
|
||||
if (last_pos < i) qb_append(&key, p + last_pos, i - last_pos);
|
||||
qb_push(&key, ' ');
|
||||
last_pos = i + 1;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (code == '+') {
|
||||
if (last_pos < i) qb_append(&value, p + last_pos, i - last_pos);
|
||||
qb_push(&value, ' ');
|
||||
last_pos = i + 1;
|
||||
} else if (!val_encoded) {
|
||||
if (code == '%') {
|
||||
encode_check = 1;
|
||||
} else if (encode_check > 0) {
|
||||
if (qs_unhex(code) >= 0) {
|
||||
if (++encode_check == 3) val_encoded = true;
|
||||
} else {
|
||||
encode_check = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!returned) {
|
||||
/* Leftover key or value data. */
|
||||
bool ended_empty = false;
|
||||
if (last_pos < n) {
|
||||
if (eq_idx < eq_len) qb_append(&key, p + last_pos, n - last_pos);
|
||||
else if (sep_idx < sep_len) qb_append(&value, p + last_pos, n - last_pos);
|
||||
} else if (eq_idx == 0 && key.len == 0) {
|
||||
ended_empty = true; /* ended on an empty substring */
|
||||
}
|
||||
if (!ended_empty) {
|
||||
qs_add_key_val(out, scr_str_new(key.data, key.len),
|
||||
scr_str_new(value.data, value.len), key_encoded,
|
||||
val_encoded, str_tag, arr_tag);
|
||||
}
|
||||
}
|
||||
free(key.data);
|
||||
free(value.data);
|
||||
}
|
||||
|
||||
/* ── stringify ───────────────────────────────────────────────────────── */
|
||||
|
||||
/* Node's encodeStringified over one DOM value: strings escape, finite
|
||||
* numbers render then escape, booleans are bare, everything else is the
|
||||
* empty value. Appends to b. */
|
||||
static void qs_stringify_value(QsBuf *b, const ScrDyn *v) {
|
||||
switch (v->kind) {
|
||||
case SCR_DYN_STR: {
|
||||
ScrStr *enc = scr_str_encode_uri_component(v->v.str);
|
||||
qb_append_str(b, enc);
|
||||
scr_str_release(enc);
|
||||
return;
|
||||
}
|
||||
case SCR_DYN_NUM: {
|
||||
if (!isfinite(v->v.num)) return;
|
||||
ScrStr *num = scr_f64_to_scrstr(v->v.num);
|
||||
ScrStr *enc = scr_str_encode_uri_component(num);
|
||||
qb_append_str(b, enc);
|
||||
scr_str_release(enc);
|
||||
scr_str_release(num);
|
||||
return;
|
||||
}
|
||||
case SCR_DYN_BOOL:
|
||||
if (v->v.b) qb_append(b, "true", 4);
|
||||
else qb_append(b, "false", 5);
|
||||
return;
|
||||
default:
|
||||
return; /* null/undefined/objects/arrays/functions → '' */
|
||||
}
|
||||
}
|
||||
|
||||
ScrStr *scr_qs_stringify(const ScrDyn *obj, const ScrStr *sep,
|
||||
const ScrStr *eq) {
|
||||
const char *sep_b = (sep && sep->len) ? sep->data : "&";
|
||||
size_t sep_len = (sep && sep->len) ? sep->len : 1;
|
||||
const char *eq_b = (eq && eq->len) ? eq->data : "=";
|
||||
size_t eq_len = (eq && eq->len) ? eq->len : 1;
|
||||
if (!obj || obj->kind != SCR_DYN_OBJ) return scr_str_new("", 0);
|
||||
|
||||
QsBuf fields;
|
||||
qb_init(&fields);
|
||||
/* JS own-key order (array-index keys ascending first) — ObjectKeys. */
|
||||
ScrDyn *keys = scr_dyn_obj_keys((ScrDyn *)obj);
|
||||
for (size_t i = 0; i < keys->v.arr.len; i++) {
|
||||
const ScrDyn *kd = keys->v.arr.items[i];
|
||||
const ScrStr *k = kd->v.str;
|
||||
const ScrDyn *v = scr_dyn_obj_get(obj, k->data, k->len);
|
||||
if (!v) continue; /* unreachable: keys came from the object */
|
||||
ScrStr *ks = scr_str_encode_uri_component((ScrStr *)k);
|
||||
if (v->kind == SCR_DYN_ARR) {
|
||||
for (size_t j = 0; j < v->v.arr.len; j++) {
|
||||
if (fields.len) qb_append(&fields, sep_b, sep_len);
|
||||
qb_append_str(&fields, ks);
|
||||
qb_append(&fields, eq_b, eq_len);
|
||||
qs_stringify_value(&fields, v->v.arr.items[j]);
|
||||
}
|
||||
} else {
|
||||
if (fields.len) qb_append(&fields, sep_b, sep_len);
|
||||
qb_append_str(&fields, ks);
|
||||
qb_append(&fields, eq_b, eq_len);
|
||||
qs_stringify_value(&fields, v);
|
||||
}
|
||||
scr_str_release(ks);
|
||||
}
|
||||
scr_dyn_release(keys);
|
||||
return qb_take(&fields);
|
||||
}
|
||||
@@ -484,6 +484,64 @@ ScrStr *scr_str_encode_uri_component(ScrStr *s);
|
||||
* malformed") and returns NULL. Borrows s; result +1. */
|
||||
ScrStr *scr_str_decode_uri_component(ScrStr *s);
|
||||
|
||||
/* The non-throwing core of decodeURIComponent: NULL on malformed input
|
||||
* instead of the URIError — the try/catch shape node:querystring's
|
||||
* unescape wraps around decodeURIComponent (scr_qs.c's strict pass). */
|
||||
ScrStr *scr_str_decode_uri_component_try(ScrStr *s);
|
||||
|
||||
/* ── node:querystring (scr_qs.c — LINK-GATED by moduleUsesQs; the
|
||||
* scr_url_params.c precedent: pure data transforms, no loop hooks).
|
||||
* querystring.escape needs no entry here: Node's qsEscape encodes exactly
|
||||
* the component unreserved set, so the frontend lowers it to
|
||||
* scr_str_encode_uri_component (always linked; a program using only
|
||||
* escape never pulls this unit). */
|
||||
|
||||
/* querystring.unescape — Node's qsUnescape: strict decodeURIComponent
|
||||
* first, and on failure the lenient legacy unescapeBuffer(s).toString()
|
||||
* (valid %XX escapes decode to their byte, malformed escapes copy
|
||||
* literally, every non-escape UTF-16 CODE UNIT truncates to its low byte
|
||||
* — Node's Buffer element write — and the byte buffer decodes as UTF-8
|
||||
* with U+FFFD replacement per maximal subpart, Buffer.toString's rule).
|
||||
* Borrows s; result +1; never throws. */
|
||||
ScrStr *scr_qs_unescape(const ScrStr *s);
|
||||
|
||||
/* querystring.parse — Node v24's scan state machine, byte-wise over the
|
||||
* UTF-8 storage (equivalent to the code-unit scan for well-formed input:
|
||||
* the machine compares sep/eq sequences positionally and treats '%'/'+'/
|
||||
* hex as ASCII, and multibyte alignment is preserved). Decoded pairs land
|
||||
* in `out` — the PURE-index-signature Dict record's overflow map (the
|
||||
* emitters pass rec->sc_ovf) — grouped like Node's addKeyVal: a first
|
||||
* value stores the `str_tag` union arm, a repeat REPLACES it with a
|
||||
* two-element string[] wrapped in `arr_tag`, later repeats push. sep/eq
|
||||
* fall back to "&"/"=" when empty (Node's falsy rule; the frontend
|
||||
* completes omitted/null arguments to the defaults). max_keys is Node's
|
||||
* rule exactly: > 0 caps the PAIR count (empty skipped segments count,
|
||||
* like Node's --pairs), anything else (0, negatives, NaN) is unlimited;
|
||||
* the frontend completes the omitted option to 1000. Only the DEFAULT
|
||||
* decoder runs (custom decodeURIComponent options fence at compile time),
|
||||
* so '+' means ' ' and segments decode with scr_qs_unescape's semantics
|
||||
* only when they carry a full valid %XX triple (Node's encodeCheck).
|
||||
* Borrows everything; never throws. */
|
||||
typedef struct ScrMap ScrMap; /* full definition below (C11 repeat) */
|
||||
void scr_qs_parse_into(ScrMap *out, const ScrStr *qs, const ScrStr *sep,
|
||||
const ScrStr *eq, double max_keys, uint32_t str_tag,
|
||||
uint32_t arr_tag);
|
||||
|
||||
/* querystring.stringify — Node's stringify over a borrowed DOM value (the
|
||||
* frontend dynFroms the typed record; JS-world dyn values pass straight
|
||||
* through). Non-object DOMs answer "" like Node; object keys iterate in
|
||||
* JS own-key order (scr_dyn_obj_keys — array-index keys ascending first,
|
||||
* then insertion order, Node's ObjectKeys). Values serialize per Node's
|
||||
* encodeStringified: strings escape, finite numbers render shortest-
|
||||
* roundtrip then escape ('1e+21' → '1e%2B21'), booleans are bare
|
||||
* true/false, arrays expand to repeated keys (empty arrays emit NOTHING,
|
||||
* key included), and everything else (null, undefined, nested objects/
|
||||
* arrays, functions) is the empty value — key and eq still emitted.
|
||||
* sep/eq fall back to "&"/"=" when empty (Node's `sep ||= '&'`). Result
|
||||
* +1; never throws. */
|
||||
ScrStr *scr_qs_stringify(const struct ScrDyn *obj, const ScrStr *sep,
|
||||
const ScrStr *eq);
|
||||
|
||||
/* encodeURI — the same ECMA-262 Encode() with the reserved set and '#'
|
||||
* kept unescaped (scr_encode_uri_impl's keep_reserved arm). Total by the
|
||||
* well-formed-UTF-8 invariant. Borrows s; result +1. */
|
||||
|
||||
@@ -1212,7 +1212,12 @@ static int scr_uri_hex_byte(const char *p, size_t rem) {
|
||||
return (hi << 4) | lo;
|
||||
}
|
||||
|
||||
ScrStr *scr_str_decode_uri_component(ScrStr *s) {
|
||||
/* The non-throwing core of decodeURIComponent: NULL on malformed input
|
||||
* (bad hex, invalid UTF-8 octets) instead of the URIError — the
|
||||
* querystring unit's unescape needs exactly the try/catch shape Node's
|
||||
* qsUnescape wraps around decodeURIComponent (scr_qs.c), and the throwing
|
||||
* entry point below stays byte-identical by rethrowing over NULL. */
|
||||
ScrStr *scr_str_decode_uri_component_try(ScrStr *s) {
|
||||
/* Decoding only ever shrinks (%XX → 1 byte), so len is a safe cap. */
|
||||
ScrStr *out = scr_str_alloc_raw(0, s->len);
|
||||
size_t w = 0;
|
||||
@@ -1258,7 +1263,15 @@ ScrStr *scr_str_decode_uri_component(ScrStr *s) {
|
||||
return out;
|
||||
malformed:
|
||||
scr_str_release(out);
|
||||
scr_uri_malformed();
|
||||
return NULL;
|
||||
}
|
||||
|
||||
ScrStr *scr_str_decode_uri_component(ScrStr *s) {
|
||||
ScrStr *out = scr_str_decode_uri_component_try(s);
|
||||
if (!out) {
|
||||
scr_uri_malformed();
|
||||
return NULL;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
// node:querystring.parse — the grammar corners of Node's legacy scan
|
||||
// (scr_qs.c's quirk-faithful port; Node is the oracle), generated as an
|
||||
// input sweep: empty/keyless/valueless segments, repeated keys becoming
|
||||
// arrays, '=' inside values, '+' meaning space (but '%2B' staying '+'),
|
||||
// malformed percent-escapes (bad hex copies literally; valid escapes
|
||||
// that decode to invalid UTF-8 take the lenient unescapeBuffer fallback
|
||||
// with U+FFFD replacement — lone-surrogate escapes included), raw
|
||||
// non-ASCII and astral input, and the encodeCheck fast path (a segment
|
||||
// only decodes when it carries a full valid %XX triple).
|
||||
import { parse } from "node:querystring";
|
||||
|
||||
const cases: string[] = [
|
||||
"",
|
||||
"a",
|
||||
"a=",
|
||||
"=a",
|
||||
"=",
|
||||
"&",
|
||||
"&&",
|
||||
"a&&b",
|
||||
"a=1&",
|
||||
"&a=1",
|
||||
"a=1&&b=2",
|
||||
"a=1&a=2&a=3",
|
||||
"k=v&k=w&k=x&single=s",
|
||||
"a=b=c&==x",
|
||||
"a==b",
|
||||
"a%3Db=c",
|
||||
"key only",
|
||||
"sp ace=v al",
|
||||
"a+b=c+d&%20=+",
|
||||
"a=+%2B+",
|
||||
"a=%2B+b",
|
||||
"%zz",
|
||||
"%zz%25=%25zz",
|
||||
"%25%25=%25",
|
||||
"a=%E2%98%83&b=%FF&c=%zz&d=%1",
|
||||
"foo=%F0%9F%98%80&bar=%ED%A0%80",
|
||||
"%E2%98%83=%E2%98%83",
|
||||
"☃=☃&é=é",
|
||||
"😀=🌍",
|
||||
"%",
|
||||
"%2",
|
||||
"a=%",
|
||||
"a=%F",
|
||||
"a=%FG",
|
||||
"%GG=1",
|
||||
"1=a&0=b&x=c",
|
||||
];
|
||||
for (const c of cases) {
|
||||
console.log(JSON.stringify(c), JSON.stringify(parse(c)));
|
||||
}
|
||||
|
||||
// The result dictionary reads like any index-signature record: repeated
|
||||
// keys narrow to their array bucket, singles to the string arm, absent
|
||||
// keys to undefined.
|
||||
const r = parse("k=v&k=w&single=s");
|
||||
const k = r["k"];
|
||||
if (Array.isArray(k)) console.log("bucket", k.length, k.join("|"));
|
||||
const s = r["single"];
|
||||
if (typeof s === "string") console.log("single", s);
|
||||
console.log("missing", r["missing"] === undefined);
|
||||
console.log("keys", Object.keys(parse("1=a&0=b&x=c")).join(","));
|
||||
@@ -0,0 +1,35 @@
|
||||
// node:querystring.parse — separators and maxKeys (Node is the oracle).
|
||||
// Custom sep/eq including multi-character and multi-byte sequences (the
|
||||
// scan's naive partial-match resets are quirk-faithful: an overlapping
|
||||
// partial match is NOT re-examined, Node's own behavior), the falsy rule
|
||||
// (null/undefined/'' all mean the defaults), and maxKeys' pair budget —
|
||||
// which empty skipped segments consume too, and which 0 and negatives
|
||||
// remove entirely (Node's `maxKeys > 0 ? maxKeys : -1`).
|
||||
import { parse } from "node:querystring";
|
||||
|
||||
// Custom separators.
|
||||
console.log("S1", JSON.stringify(parse("a:1;b:2", ";", ":")));
|
||||
console.log("S2", JSON.stringify(parse("a::1;;b::2", ";;", "::")));
|
||||
console.log("S3", JSON.stringify(parse("aXYb=1XYXc=2", "XYX")));
|
||||
console.log("S4", JSON.stringify(parse("a==b=c===d", undefined, "==")));
|
||||
console.log("S5", JSON.stringify(parse("a=1&b=2", "", "")));
|
||||
console.log("S6", JSON.stringify(parse("a☃1;b☃2", ";", "☃")));
|
||||
console.log("S7", JSON.stringify(parse("aabX=1", "ab")));
|
||||
console.log("S8", JSON.stringify(parse("a=1&b=2", null, null)));
|
||||
console.log("S9", JSON.stringify(parse("x🌍y=1&z=2", null, "🌍")));
|
||||
|
||||
// maxKeys: the pair budget.
|
||||
console.log("M1", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: 2 })));
|
||||
console.log("M2", JSON.stringify(parse("a=1&b=2", null, null, { maxKeys: 0 })));
|
||||
console.log("M3", JSON.stringify(parse("&&&a=1&b=2", null, null, { maxKeys: 2 })));
|
||||
console.log("M4", JSON.stringify(parse("a=1&&b=2&c=3", null, null, { maxKeys: 2 })));
|
||||
console.log("M5", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: -5 })));
|
||||
console.log("M6", JSON.stringify(parse("a=1&a=2&a=3&b=4", null, null, { maxKeys: 3 })));
|
||||
console.log("M7", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: Infinity })));
|
||||
console.log("M8", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: 1000 })));
|
||||
console.log("M9", JSON.stringify(parse("a:1;b:2;c:3", ";", ":", { maxKeys: 2 })));
|
||||
|
||||
// A runtime maxKeys expression (the budget lives in the runtime).
|
||||
let budget = 1;
|
||||
budget += 1;
|
||||
console.log("M10", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: budget })));
|
||||
@@ -0,0 +1,40 @@
|
||||
// node:querystring.stringify — Node's encodeStringified value rules
|
||||
// (scr_qs.c over the DOM crossing): strings escape with the querystring
|
||||
// set (spaces are %20, never '+'), finite numbers render shortest-
|
||||
// roundtrip then escape ('1e+21' → '1e%2B21'), non-finite numbers and
|
||||
// null/undefined are empty VALUES (key and eq still emitted), booleans
|
||||
// are bare, arrays expand to repeated keys (empty arrays emit nothing at
|
||||
// all), custom sep/eq apply verbatim with the falsy default rule, and
|
||||
// keys escape like values. encode is Node's own alias.
|
||||
import { stringify, encode } from "node:querystring";
|
||||
|
||||
console.log("T1", stringify({ a: 1, b: "x y", c: ["1", "2"], d: true, e: "" }));
|
||||
console.log("T2", stringify({ a: Infinity, b: NaN, c: -0, d: 1e21, e: 0.1, f: -1.5 }));
|
||||
console.log("T3", stringify({ "a b": "c", "☃": "+", "": "empty", k: "" }));
|
||||
console.log("T4", JSON.stringify(stringify({})));
|
||||
console.log("T5", stringify({ a: [] as string[], b: "x" }));
|
||||
console.log("T6", stringify({ a: ["x", "y"] }, ";", ":"));
|
||||
console.log("T7", stringify({ a: "1", b: "2" }, "", ""));
|
||||
console.log("T8", stringify({ u: undefined, n: null, s: "v" } as Record<string, string | null | undefined>));
|
||||
console.log("T9", stringify({ a: [1, 2.5, true] as (number | boolean)[] }));
|
||||
console.log("T10", stringify({ "é☃": "é☃ 😀" }));
|
||||
console.log("T11", stringify({ a: "x" }, "☃", "🌍"));
|
||||
console.log("T12", encode({ x: [1, 2] }));
|
||||
|
||||
// Declared shapes with union-valued fields serialize by the value each
|
||||
// field HOLDS (an explicitly-undefined optional field keeps its key with
|
||||
// the empty value, exactly Node).
|
||||
const o: { u?: string; s: string; n: number | null } = { u: undefined, s: "v", n: null };
|
||||
console.log("T13", stringify(o));
|
||||
|
||||
// Index-signature records (the runtime-keyed build).
|
||||
const r: Record<string, string | string[]> = {};
|
||||
r["first"] = "1";
|
||||
r["multi"] = ["a", "b"];
|
||||
r["last"] = "z";
|
||||
console.log("T14", stringify(r));
|
||||
|
||||
// The round trip: stringify(parse(qs)) is stable for canonical input.
|
||||
import { parse } from "node:querystring";
|
||||
const round = "a=1&a=2&b=x%20y&c=";
|
||||
console.log("T15", stringify(parse(round)));
|
||||
@@ -0,0 +1,47 @@
|
||||
// node:querystring.escape/unescape (Node is the oracle). escape encodes
|
||||
// exactly the component unreserved set (ALPHA/DIGIT/- _ . ! ~ * ' ( )) as
|
||||
// uppercase %XX — the full printable-ASCII sweep pins the set character
|
||||
// by character, non-ASCII encodes its UTF-8 bytes. unescape is the
|
||||
// strict-then-lenient pair: decodeURIComponent when the whole string
|
||||
// decodes, else the legacy unescapeBuffer — valid %XX escapes decode to
|
||||
// their byte, malformed escapes copy literally, every non-escape UTF-16
|
||||
// code unit truncates to its LOW BYTE (Node's Buffer element write — the
|
||||
// '☃%E9' and astral corners below), and the byte buffer decodes as UTF-8
|
||||
// with U+FFFD replacement per maximal subpart.
|
||||
import { escape as esc, unescape as unesc } from "node:querystring";
|
||||
|
||||
// The printable-ASCII sweep, one character at a time.
|
||||
const parts: string[] = [];
|
||||
for (let i = 32; i < 127; i++) parts.push(String.fromCharCode(i));
|
||||
const ascii = parts.join("");
|
||||
console.log("E1", esc(ascii));
|
||||
console.log("E2", esc("héllo ☃ 😀"));
|
||||
console.log("E3", esc(""));
|
||||
console.log("E4", esc("a+b c=d&e"));
|
||||
|
||||
// Strict decodes.
|
||||
console.log("U1", JSON.stringify(unesc("%41%20%42")));
|
||||
console.log("U2", JSON.stringify(unesc("%E2%98%83")));
|
||||
console.log("U3", JSON.stringify(unesc("%F0%9F%98%80")));
|
||||
console.log("U4", JSON.stringify(unesc("abc")));
|
||||
console.log("U5", JSON.stringify(unesc("%25")));
|
||||
console.log("U6", JSON.stringify(unesc("a+b%20c")));
|
||||
|
||||
// Lenient fallbacks: malformed hex, invalid UTF-8 octets, lone-surrogate
|
||||
// escapes, trailing '%', truncated sequences, code-unit truncation.
|
||||
console.log("F1", JSON.stringify(unesc("a+b%20c%E2%98%83%zz%")));
|
||||
console.log("F2", JSON.stringify(unesc("%ED%A0%80")));
|
||||
console.log("F3", JSON.stringify(unesc("%FF%fe")));
|
||||
console.log("F4", JSON.stringify(unesc("é%E9")));
|
||||
console.log("F5", JSON.stringify(unesc("☃%E9")));
|
||||
console.log("F6", JSON.stringify(unesc("😀%E9")));
|
||||
console.log("F7", JSON.stringify(unesc("%")));
|
||||
console.log("F8", JSON.stringify(unesc("%2")));
|
||||
console.log("F9", JSON.stringify(unesc("100%")));
|
||||
console.log("F10", JSON.stringify(unesc("%E2%98")));
|
||||
console.log("F11", JSON.stringify(unesc("%GG%41")));
|
||||
console.log("F12", JSON.stringify(unesc("%c3%a9%ff")));
|
||||
|
||||
// escape/unescape compose with parse/stringify's own paths — the same
|
||||
// codec observed directly.
|
||||
console.log("C1", esc(unesc("%E2%98%83")) === "%E2%98%83");
|
||||
@@ -0,0 +1,31 @@
|
||||
// node:querystring through every CommonJS acquisition spelling a JS
|
||||
// package uses (the HTTP-client dependency chains that import it
|
||||
// unguarded are CJS): the whole-module require binding, the destructured
|
||||
// require, the member-binding require, the inline require member call,
|
||||
// and the decode/encode aliases — all keying the same lowering tables.
|
||||
"use strict";
|
||||
|
||||
const querystring = require("node:querystring");
|
||||
const { parse, stringify } = require("querystring");
|
||||
const esc = require("querystring").escape;
|
||||
|
||||
console.log("Q1", JSON.stringify(querystring.parse("a=1&a=2&b=x%20y+z&c")));
|
||||
console.log("Q2", querystring.stringify({ a: 1, arr: [1, "x", true], u: "héllo ☃", s: "a b", e: "" }));
|
||||
console.log("Q3", querystring.escape("héllo ☃ a+b!'()*~"));
|
||||
console.log("Q4", querystring.unescape("a+b%20c%E2%98%83%zz"));
|
||||
console.log("Q5", JSON.stringify(querystring.decode("x=1&x=2")));
|
||||
console.log("Q6", querystring.encode({ x: [1, 2] }));
|
||||
|
||||
console.log("D1", JSON.stringify(parse("a:1;b:2", ";", ":")));
|
||||
console.log("D2", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: 2 })));
|
||||
console.log("D3", stringify({ k: ["v", "w"] }, ";", ":"));
|
||||
|
||||
console.log("M1", esc("a b+c"));
|
||||
console.log("M2", require("querystring").unescape("%E2%98%83"));
|
||||
|
||||
// Results feed ordinary JS flows: property reads, Array.isArray splits.
|
||||
const parsed = parse("tag=a&tag=b&page=2");
|
||||
const tags = parsed.tag;
|
||||
if (Array.isArray(tags)) console.log("R1", tags.join(","));
|
||||
const page = parsed.page;
|
||||
if (typeof page === "string") console.log("R2", page);
|
||||
Reference in New Issue
Block a user