node:querystring lowers statically — parse, stringify, escape, unescape

- scr_qs.c ports Node v24's legacy codec quirk-faithfully (the scan state machine with maxKeys' pair budget, unescapeBuffer's code-unit byte truncation with U+FFFD replacement decode, encodeStringified's value rules over the DOM crossing); link-gated by moduleUsesQs, and escape rides the always-linked component encoder
- parse's result is the call site's ParsedUrlQuery dictionary (the networkInterfaces verification stance; the runtime fills the overflow map and groups repeats into string[] buckets), sep/eq/maxKeys complete to Node's defaults, and custom encoder/decoder options fence by name
- decode/encode lower as Node's own aliases; both backends emit the same runtime calls byte-identically
- querystring joins SUPPORTED_BUILTIN_MODULES, the fallback declarations, and the regenerated surface manifest, so npm-static packages requiring it unguarded stop falling back to the island (the shim stays for island code)
- corpus 2460-2464 pin the grammar corners against Node: malformed escapes, '+' vs %2B, repeated keys, maxKeys with empty segments, multi-char/multi-byte separators, astral truncation, and every CJS acquisition spelling
This commit is contained in:
Chris Tate
2026-07-23 02:38:44 -05:00
parent db6f0eb2f2
commit 7bdd40453a
19 changed files with 1253 additions and 4 deletions
+44
View File
@@ -1535,6 +1535,50 @@ declare module "node:string_decoder" {
export * from "string_decoder";
}
/* node:querystring — Node's legacy query-string codec (NOT
* URLSearchParams: '+' means space on the parse side and escape encodes
* spaces as %20). parse answers the null-prototype dictionary as a pure
* index-signature record (repeated keys become string[] buckets;
* @types/node's Dict adds an undefined arm the lowering tolerates);
* stringify serializes string/number/boolean values and arrays of those
* (Node's rules: arrays expand to repeated keys, null/undefined and
* anything else serialize as the empty value). The maxKeys option lowers
* (0 removes the cap, Node's rule); the custom encoder/decoder options
* typecheck under the options-record stance and fence by name at the
* call. decode/encode are Node's own aliases of parse/stringify. */
declare module "querystring" {
export interface ParseOptions {
maxKeys?: number;
decodeURIComponent?: (str: string) => string;
[option: string]: unknown;
}
export interface StringifyOptions {
encodeURIComponent?: (str: string) => string;
[option: string]: unknown;
}
export interface ParsedUrlQuery {
[key: string]: string | string[] | undefined;
}
export interface ParsedUrlQueryInput {
[key: string]:
| string
| number
| boolean
| ReadonlyArray<string | number | boolean>
| null
| undefined;
}
export function parse(str: string, sep?: string | null, eq?: string | null, options?: ParseOptions): ParsedUrlQuery;
export function stringify(obj?: ParsedUrlQueryInput, sep?: string | null, eq?: string | null, options?: StringifyOptions): string;
export const decode: typeof parse;
export const encode: typeof stringify;
export function escape(str: string): string;
export function unescape(str: string): string;
}
declare module "node:querystring" {
export * from "querystring";
}
/* node:readline — the question/close slice: createInterface over exactly
* { input: process.stdin, output: process.stdout }, question(query, cb)
* (the query writes to stdout, the callback gets the next line's text),
+8
View File
@@ -127,6 +127,13 @@ export interface CcOptions {
* cross-compiles everywhere. sp-free binaries keep their exact link
* line (scr_url.c never references the unit). */
searchParams?: boolean;
/** The program uses the node:querystring surface (moduleUsesQs on the
* IR): compiles scr_qs.c into the binary — the searchParams gating
* precedent: pure data transforms (no loop hooks, no install),
* cross-compiles everywhere. qs-free binaries keep their exact link
* line (escape-only programs ride the always-linked component encoder
* and never flip this). */
qs?: boolean;
/** The program uses the node:stream class surface (moduleUsesStream on
* the IR): compiles scr_stream.c into the binary — always alongside
* scr_events_emitter.c, which moduleUsesEmitter answers true for
@@ -957,6 +964,7 @@ export async function compileC(opts: CcOptions): Promise<void> {
...(opts.emitter || net ? [rt(join(rtDir, "scr_dyn_handle.c"))] : []),
...(opts.symbol ? [rt(join(rtDir, "scr_symbol.c"))] : []),
...(opts.searchParams ? [rt(join(rtDir, "scr_url_params.c"))] : []),
...(opts.qs ? [rt(join(rtDir, "scr_qs.c"))] : []),
...(opts.stream ? [rt(join(rtDir, "scr_stream.c"))] : []),
// The readiness-poller backends (scr_platform.h): kqueue on macOS/BSD,
// epoll on Linux, WSAPoll on Windows — each TU is empty off its
@@ -2578,6 +2578,38 @@ export function emitExpr(E: CEmitter, e: IrExpr): Temp {
return finish(`scr_sp_key_at(${arg(0)}, ${arg(1)})`);
case "sp.valAt":
return finish(`scr_sp_val_at(${arg(0)}, ${arg(1)})`);
// node:querystring (scr_qs.c — linked exactly when parse/
// stringify/unescape appear, moduleUsesQs). escape IS the
// component encoder (Node's qsEscape set equals
// encodeURIComponent's), so it emits the always-linked codec
// and never pulls the unit. Borrow; string results +1; no throw.
case "qs.escape":
return finish(`scr_str_encode_uri_component(${arg(0)})`);
case "qs.unescape":
return finish(`scr_qs_unescape(${arg(0)})`);
case "qs.stringify":
return finish(`scr_qs_stringify(${arg(0)}, ${arg(1)}, ${arg(2)})`);
case "qs.parse": {
// The ParsedUrlQuery dictionary: a fresh pure-index-signature
// record whose overflow map the runtime scan fills
// (scr_qs_parse_into groups repeats into string[] buckets).
// The frontend verified the shape (lowerQuerystringParseCall);
// lookups here only guard emitter bugs. Args: qs, sep, eq,
// maxKeys.
if (e.type.kind !== "record") throw new Error("emitter bug: qs.parse result is not a record");
const dictShape = E.recordsById.get(e.type.shapeId);
const iv = dictShape?.indexValue;
if (!dictShape || iv?.kind !== "union") throw new Error("emitter bug: qs.parse dict shape");
const ivDef = E.unionsById.get(iv.unionId);
const strTag = ivDef?.arms.findIndex((a) => a.kind === "string") ?? -1;
const arrTag = ivDef?.arms.findIndex((a) => a.kind === "array") ?? -1;
if (strTag < 0 || arrTag < 0) throw new Error("emitter bug: qs.parse index union lacks its arms");
const dict = E.newTemp(e.type, `${mangleRecordNew(e.type.shapeId)}()`);
E.line(
`scr_qs_parse_into(${dict.name}->${OVERFLOW_MEMBER}, ${arg(0)}, ${arg(1)}, ${arg(2)}, ${arg(3)}, ${strTag}, ${arrTag});${E.srcComment(e.loc)}`,
);
return dict;
}
// Stats (scr_lib.c): statSync throws like the other sync fs
// calls; the getters are pure reads.
case "fs.openSync":
@@ -384,6 +384,14 @@ const LIB_FN_SYMS: Record<string, string> = {
"sp.toString": "scr_sp_to_string",
"sp.keyAt": "scr_sp_key_at",
"sp.valAt": "scr_sp_val_at",
// node:querystring (scr_qs.c): unescape/stringify are plain generic
// calls (never throw; string results +1), and escape IS the component
// encoder (Node's qsEscape set equals encodeURIComponent's) so it emits
// the always-linked codec. qs.parse is special-cased in emitLibCall
// (the result dictionary's construction).
"qs.escape": "scr_str_encode_uri_component",
"qs.unescape": "scr_qs_unescape",
"qs.stringify": "scr_qs_stringify",
// Timer handle bookkeeping (never throws; the comma-shaped chaining
// forms are special-cased in emitLibCall).
"timers.hasRef": "scr_timer_has_ref",
@@ -9319,6 +9327,31 @@ class LlEmitter {
for (const a of e.args) this.emitExpr(a);
return { name: "", type: e.type };
}
if (e.fn === "qs.parse") {
// The ParsedUrlQuery dictionary: a fresh pure-index-signature
// record whose overflow map the runtime scan fills
// (scr_qs_parse_into groups repeats into string[] buckets) — the
// C emitter's shape exactly. The frontend verified the structure;
// lookups here only guard emitter bugs. Args: qs, sep, eq, maxKeys.
if (e.type.kind !== "record") throw new Error("llvm emitter bug: qs.parse result is not a record");
const dictShape = this.recordsById.get(e.type.shapeId);
const iv = dictShape?.indexValue;
if (!dictShape || iv?.kind !== "union") throw new Error("llvm emitter bug: qs.parse dict shape");
const ivDef = this.unionsById.get(iv.unionId);
const strTag = ivDef?.arms.findIndex((a) => a.kind === "string") ?? -1;
const arrTag = ivDef?.arms.findIndex((a) => a.kind === "array") ?? -1;
if (strTag < 0 || arrTag < 0) throw new Error("llvm emitter bug: qs.parse index union lacks its arms");
const args = e.args.map((a) => this.emitExpr(a));
this.declare(`declare void @scr_qs_parse_into(ptr, ptr, ptr, ptr, double, i32, i32)`);
const dict = B.tmp();
B.line(`${dict} = call ptr @${mangleRecordNew(e.type.shapeId)}()`);
const out = this.own({ name: dict, type: e.type });
const ovf = this.recordOvfPtr(dict, e.type.shapeId);
B.line(
`call void @scr_qs_parse_into(ptr ${ovf}, ptr ${args[0]!.name}, ptr ${args[1]!.name}, ptr ${args[2]!.name}, double ${args[3]!.name}, i32 ${strTag}, i32 ${arrTag})`,
);
return out;
}
if (e.fn === "os.networkInterfaces") {
// The Dict<NetworkInterfaceInfo[]> record, built inline from a
// getifaddrs(3) snapshot — emit-exprs.ts's builder, block-lowered.
@@ -13,6 +13,8 @@ import {
FS_READDIR_DOCUMENTED_OPTIONS,
FS_WATCH_DOCUMENTED_OPTIONS,
FS_WRITE_FILE_DOCUMENTED_OPTIONS,
QS_PARSE_DOCUMENTED_OPTIONS,
QS_STRINGIFY_DOCUMENTED_OPTIONS,
READLINE_DOCUMENTED_OPTIONS,
builtinConstLit,
fenceOrDropOptionKey,
@@ -404,6 +406,20 @@ function optionMember(p: ts.ObjectLiteralElementLike): { name: string; value: ts
if (bi.module === "os" && bi.member === "userInfo") {
return lowerOsUserInfoCall(L, expr, loc);
}
// node:querystring — parse/stringify are entirely special-cased (the
// sep/eq/options completions, parse's call-site-shaped dictionary
// result, stringify's DOM-crossing object argument); decode/encode
// are Node's own aliases of the pair (`const decode = parse` in the
// module source) and take the same lowerings. escape/unescape ride
// the generic table tail below.
if (bi.module === "querystring") {
if (bi.member === "parse" || bi.member === "decode") {
return lowerQuerystringParseCall(L, expr, loc);
}
if (bi.member === "stringify" || bi.member === "encode") {
return lowerQuerystringStringifyCall(L, expr, loc);
}
}
// node:timers/promises — setTimeout([delay]) and setImmediate(): void
// promises the shared timer heap settles. The omitted delay completes
// to Node's 1ms floor (scr_timer_coerce_ms clamps anyway; the literal
@@ -4502,6 +4518,187 @@ function optionMember(p: ts.ObjectLiteralElementLike): { name: string; value: ts
return { kind: "recordLit", fields, type: result, loc };
}
/** querystring's sep/eq arguments: an omitted argument, the literal
* null, and the literal undefined all mean the default (Node's falsy
* rule — parse(s, null, null, opts) is the canonical maxKeys spelling);
* a string expression passes through (the runtime applies the same
* falsy rule to '' at runtime). Everything else fences. */
function qsSepEqArg(L: Lowerer, node: ts.Expression | undefined, dflt: string,
what: string, loc: SrcLoc,): IrExpr {
if (
!node ||
node.kind === ts.SyntaxKind.NullKeyword ||
(ts.isIdentifier(node) && node.text === "undefined")
) {
return { kind: "strLit", value: dflt, type: STRING, loc };
}
const v = L.lowerExpr(node);
if (v.type.kind !== "string") {
L.noLowering(
`${what} with a '${L.fmt(v.type)}' separator`,
node,
"pass a string, or null/undefined for the default (narrow unions first)",
);
}
return v;
}
/** querystring.parse / querystring.decode: the scan runs in the runtime
* (scr_qs_parse_into fills the result dictionary's overflow map), so the
* frontend completes sep/eq/maxKeys to Node's defaults and verifies the
* call site's mapped result IS the ParsedUrlQuery dictionary — a pure
* index-signature record over `string | string[]` (an undefined arm
* tolerated: @types/node's Dict) — the networkInterfaces verification
* stance. The default decoder is the one lowered decoder; a custom
* decodeURIComponent option fences by name. */
export function lowerQuerystringParseCall(L: Lowerer, call: ts.CallExpression, loc: SrcLoc): IrExpr {
if (call.arguments.length > 4 || call.arguments.some(ts.isSpreadElement)) {
L.noLowering(`querystring.parse with ${call.arguments.length} arguments`, call);
}
const fence: () => never = () =>
L.noLowering(
"querystring.parse where the result is not the ParsedUrlQuery dictionary",
call,
"the `{ [key: string]: string | string[] }` shape is the supported result",
);
const result = L.mapTypeOf(L.typeOf(call));
if (result?.kind !== "record") fence();
const dictShape = L.shapes.get(result.shapeId);
if (!dictShape || dictShape.tuple || dictShape.fields.length > 0 || !dictShape.indexValue) fence();
const iv = dictShape.indexValue;
if (iv.kind !== "union") fence();
const ivDef = L.unions.get(iv.unionId);
if (!ivDef) fence();
let sawStr = false;
let sawArr = false;
for (const arm of ivDef.arms) {
if (arm.kind === "string") sawStr = true;
else if (arm.kind === "array" && arm.elem.kind === "string") sawArr = true;
// undefined rides @types/node's Dict; the f64 arm is the
// header-family canonicalization (types.ts interns every
// `string | string[]`-slotted dictionary as the one canonical
// header shape, whose slot adds number type-level only — parse
// never stores one).
else if (arm.kind !== "undefinedT" && arm.kind !== "f64") fence();
}
if (!sawStr || !sawArr) fence();
const str = call.arguments[0]
? L.lowerExprExpecting(call.arguments[0], STRING)
: L.noLowering("querystring.parse without a query string", call);
const sep = qsSepEqArg(L, call.arguments[1], "&", "querystring.parse", loc);
const eq = qsSepEqArg(L, call.arguments[2], "=", "querystring.parse", loc);
// The options walk: maxKeys lowers (Node's rule — > 0 caps the pair
// count, 0 and negatives mean unlimited — lives in the runtime, so
// any number expression works); a custom decodeURIComponent changes
// every decoded byte and fences by name.
let maxKeys: IrExpr = { kind: "numLit", value: 1000, type: F64, loc };
const optsNode = call.arguments[3];
if (optsNode) {
if (!ts.isObjectLiteralExpression(optsNode)) {
L.noLowering(
"querystring.parse with a non-literal options argument",
optsNode,
"the supported form spells the options inline: parse(s, sep, eq, { maxKeys: n })",
);
}
for (const p of optsNode.properties) {
const m = optionMember(p);
if (!m) {
L.noLowering(
"querystring.parse with this options shape",
p,
"spreads and computed keys have no lowering — write each member inline",
);
}
if (m.name === "maxKeys") {
maxKeys = L.lowerExprExpecting(m.value, F64);
} else if (m.name === "decodeURIComponent") {
L.noLowering(
"querystring.parse with a custom decodeURIComponent",
p,
"the default decoder is the lowered surface (strict decodeURIComponent with Node's lenient fallback)",
);
} else {
fenceOrDropOptionKey(
L, p, m.name, "querystring.parse", QS_PARSE_DOCUMENTED_OPTIONS,
"maxKeys is the supported option",
);
}
}
}
return { kind: "libCall", fn: "qs.parse", args: [str, sep, eq, maxKeys], type: result, loc };
}
/** querystring.stringify / querystring.encode: the object crosses as a
* DOM value (dynFrom — JSON-safe records and, in JS sources, dyn values
* directly) and Node's encodeStringified rules run in the runtime
* (scr_qs_stringify), so arrays expand to repeated keys and
* null/undefined values are empty. The default encoder is the one
* lowered encoder; a custom encodeURIComponent option fences by name. */
export function lowerQuerystringStringifyCall(L: Lowerer, call: ts.CallExpression, loc: SrcLoc): IrExpr {
if (call.arguments.length > 4 || call.arguments.some(ts.isSpreadElement)) {
L.noLowering(`querystring.stringify with ${call.arguments.length} arguments`, call);
}
const objNode = call.arguments[0];
// stringify() / stringify(undefined) / stringify(null): Node answers
// '' for every non-object — the constant folds.
if (
!objNode ||
objNode.kind === ts.SyntaxKind.NullKeyword ||
(ts.isIdentifier(objNode) && objNode.text === "undefined")
) {
return { kind: "strLit", value: "", type: STRING, loc };
}
const objV = L.lowerExpr(objNode);
let obj: IrExpr;
if (objV.type.kind === "dyn") {
obj = objV;
} else if (objV.type.kind === "record" && L.dynConvertible(objV.type)) {
obj = { kind: "dynFrom", value: objV, type: DYN, loc };
} else {
L.noLowering(
`querystring.stringify of '${L.fmt(objV.type)}' values`,
objNode,
"pass a record of string/number/boolean values (arrays of those expand to repeated keys; null/undefined values serialize empty)",
);
}
const sep = qsSepEqArg(L, call.arguments[1], "&", "querystring.stringify", loc);
const eq = qsSepEqArg(L, call.arguments[2], "=", "querystring.stringify", loc);
const optsNode = call.arguments[3];
if (optsNode) {
if (!ts.isObjectLiteralExpression(optsNode)) {
L.noLowering(
"querystring.stringify with a non-literal options argument",
optsNode,
"the supported form spells the options inline (and the only documented option, encodeURIComponent, has no lowering)",
);
}
for (const p of optsNode.properties) {
const m = optionMember(p);
if (!m) {
L.noLowering(
"querystring.stringify with this options shape",
p,
"spreads and computed keys have no lowering — write each member inline",
);
}
if (m.name === "encodeURIComponent") {
L.noLowering(
"querystring.stringify with a custom encodeURIComponent",
p,
"the default encoder (querystring.escape's component set) is the lowered surface",
);
} else {
fenceOrDropOptionKey(
L, p, m.name, "querystring.stringify", QS_STRINGIFY_DOCUMENTED_OPTIONS,
"no stringify options have a lowering",
);
}
}
}
return { kind: "libCall", fn: "qs.stringify", args: [obj, sep, eq], type: STRING, loc };
}
export function lowerOsNetworkInterfacesCall(L: Lowerer, call: ts.CallExpression, loc: SrcLoc): IrExpr {
if (call.arguments.length !== 0) {
L.noLowering(`networkInterfaces with ${call.arguments.length} arguments`, call, "networkInterfaces() takes no arguments");
@@ -230,6 +230,16 @@ export const DNS_LOOKUP_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
"family", "hints", "all", "order", "verbatim",
]);
/** querystring.parse's documented option keys (Node v24). */
export const QS_PARSE_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
"maxKeys", "decodeURIComponent",
]);
/** querystring.stringify's documented option keys (Node v24). */
export const QS_STRINGIFY_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
"encodeURIComponent",
]);
/** readline.createInterface's documented option keys. */
export const READLINE_DOCUMENTED_OPTIONS: ReadonlySet<string> = new Set([
"input", "output", "completer", "terminal", "history", "historySize",
@@ -654,6 +664,23 @@ export const BUILTIN_MODULE_FNS: Record<string, Record<string, BuiltinModuleFn |
// end are special-cased (lowerNew + lowerStringDecoderMethodCall); no
// function members exist to table.
string_decoder: {},
// node:querystring — the legacy query-string codec (NOT URLSearchParams;
// the escaping and '+' rules differ). escape/unescape ride the table
// path directly; parse and stringify are special-cased in
// lowerBuiltinModuleCall (parse's result is the call site's mapped
// ParsedUrlQuery dictionary — the networkInterfaces verification stance
// — and its sep/eq/options complete there; stringify's object argument
// crosses as a DOM value). decode/encode are Node's own aliases of
// parse/stringify (`const decode = parse` in lib/querystring.js) and
// route to the same special cases; the entries carry canonical shapes.
querystring: {
parse: { fn: "qs.parse", params: [STRING], result: VOID },
decode: { fn: "qs.parse", params: [STRING], result: VOID },
stringify: { fn: "qs.stringify", params: [DYN], result: STRING },
encode: { fn: "qs.stringify", params: [DYN], result: STRING },
escape: { fn: "qs.escape", params: [STRING], result: STRING },
unescape: { fn: "qs.unescape", params: [STRING], result: STRING },
},
// node:readline: createInterface's options are entirely special-cased
// (exactly { input: process.stdin, output?: process.stdout } — see
// lowerBuiltinModuleCall); the entry carries the canonical shape. The
+1 -1
View File
@@ -50,7 +50,7 @@ export function isNodeTypesPath(file: string): boolean {
* this is exactly the set of `declare module` names in that file; when
* @types/node stands in (which declares ALL node builtins) the supported
* surface must not widen, so preflight allowlists this same fixed set. */
export const SUPPORTED_BUILTIN_MODULES = ["fs", "path", "path/posix", "path/win32", "os", "url", "fs/promises", "crypto", "zlib", "child_process", "net", "http", "tls", "https", "dgram", "dns", "util", "util/types", "string_decoder", "readline", "http2", "assert", "assert/strict", "worker_threads", "buffer", "cluster", "tty", "async_hooks", "events", "stream", "test", "timers", "timers/promises", "diagnostics_channel", "perf_hooks", "module"] as const;
export const SUPPORTED_BUILTIN_MODULES = ["fs", "path", "path/posix", "path/win32", "os", "url", "fs/promises", "crypto", "zlib", "child_process", "net", "http", "tls", "https", "dgram", "dns", "util", "util/types", "string_decoder", "querystring", "readline", "http2", "assert", "assert/strict", "worker_threads", "buffer", "cluster", "tty", "async_hooks", "events", "stream", "test", "timers", "timers/promises", "diagnostics_channel", "perf_hooks", "module"] as const;
/** Builtins Node itself serves ONLY under the node: prefix —
* require("test") is MODULE_NOT_FOUND in Node, so the bare name stays a
+4 -1
View File
@@ -4,7 +4,7 @@ import { compileC, resolveCc, targetPlatform } from "./backend/cc.js";
import { emitModule } from "./backend/emission/emitter.js";
import { emitLlvmModule, LlvmUnsupportedError } from "./backend/llvm/emitter.js";
import { checkerPanicDiag, iceDiag, isCheckerPanic, type ScrDiagnostic } from "./diagnostics/diagnostic.js";
import { moduleEmbedsBuiltin, moduleEmbedsCompressedNpm, moduleUsesAssert, moduleUsesDc, moduleUsesDgram, moduleUsesDynAsync, moduleUsesDynInvoke, moduleUsesEmitter, moduleUsesFetch, moduleUsesFsWatch, moduleUsesHttp2, moduleUsesHttpServer, moduleUsesInspect, moduleUsesNet, moduleUsesNodeTest, moduleUsesProcessEvents, moduleUsesRegex, moduleUsesSearchParams, moduleUsesStream, moduleUsesSymbol, moduleUsesTls, moduleUsesZlib } from "./ir/nodes.js";
import { moduleEmbedsBuiltin, moduleEmbedsCompressedNpm, moduleUsesAssert, moduleUsesDc, moduleUsesDgram, moduleUsesDynAsync, moduleUsesDynInvoke, moduleUsesEmitter, moduleUsesFetch, moduleUsesFsWatch, moduleUsesHttp2, moduleUsesHttpServer, moduleUsesInspect, moduleUsesNet, moduleUsesNodeTest, moduleUsesProcessEvents, moduleUsesQs, moduleUsesRegex, moduleUsesSearchParams, moduleUsesStream, moduleUsesSymbol, moduleUsesTls, moduleUsesZlib } from "./ir/nodes.js";
import { serializeModule } from "./ir/serialize.js";
import { validateModule } from "./ir/validate.js";
import { checkPreflight, isNodeTypesPath, loadProgram, resolveNpmImport, type LoadResult } from "./frontend/program.js";
@@ -494,6 +494,9 @@ export async function compile(entryPath: string, opts: CompileOptions): Promise<
// The link switch for scr_url_params.c: sp.* libCalls, the
// url.searchParams getter, or a searchParams-kind type on the IR.
searchParams: moduleUsesSearchParams(lowered.module),
// The link switch for scr_qs.c: the qs.* libCalls that live there
// (parse/stringify/unescape; escape rides the always-linked encoder).
qs: moduleUsesQs(lowered.module),
// The link switch for scr_stream.c: the node:stream class surface on
// the IR (stream libCalls or the %Readable-family class defs).
stream: moduleUsesStream(lowered.module),
+52
View File
@@ -1715,6 +1715,31 @@ export type IrLibFn =
| "sp.toString"
| "sp.keyAt"
| "sp.valAt"
/** node:querystring (scr_qs.c — link-gated by moduleUsesQs; NOT
* URLSearchParams: the legacy codec's escaping and '+' rules differ).
* qs.escape is Node's qsEscape, which encodes exactly the component
* unreserved set — it emits the always-linked
* scr_str_encode_uri_component, so escape-only programs never pull the
* unit. qs.unescape is Node's qsUnescape: strict decodeURIComponent
* first, the lenient legacy unescapeBuffer fallback on failure (never
* throws). qs.parse takes (str, sep, eq, maxKeys) — the frontend
* completes omitted/null sep/eq to "&"/"=" and the omitted maxKeys
* option to Node's 1000 (0 and negatives mean unlimited, Node's rule) —
* and its result type is the CALL SITE's mapped ParsedUrlQuery shape (a
* pure index-signature record over `string | string[]`, an undefined
* arm tolerated — @types/node's Dict), verified structurally by the
* frontend (lowerQuerystringParseCall); the emitters construct the
* record and hand its overflow map to scr_qs_parse_into with the two
* union tags. qs.stringify takes (obj, sep, eq) with obj a DOM value
* (the frontend dynFroms the typed record; JS-world dyn values pass
* through) — Node's encodeStringified rules run in the runtime, so
* arrays expand to repeated keys and null/undefined/nested objects are
* empty values. Custom encoder/decoder options fence at compile time.
* All borrow; string results +1; none throw. */
| "qs.parse"
| "qs.stringify"
| "qs.escape"
| "qs.unescape"
/** ES Symbol values (scr_symbol.c — link-gated by moduleUsesSymbol).
* sym.new: `Symbol(desc)` — a fresh runtime-unique identity (+1) whose
* one arg is the description string (borrowed); sym.newAnon is the
@@ -5311,6 +5336,33 @@ export function moduleUsesSearchParams(mod: IrModule): boolean {
return found;
}
/** True when the module uses the node:querystring surface — the qs.*
* libCalls that live in scr_qs.c (parse/stringify/unescape; qs.escape
* emits the always-linked component encoder and deliberately does NOT
* flip this switch) — the link switch that pulls scr_qs.c into the
* binary (the moduleUsesSearchParams precedent: pure data transforms, no
* loop hooks, cross-compiles everywhere). qs-free programs keep their
* exact link line. Same generic-walk shape as moduleUsesZlib. */
export function moduleUsesQs(mod: IrModule): boolean {
let found = false;
const visit = (v: unknown): void => {
if (found || v === null || typeof v !== "object") return;
if (Array.isArray(v)) {
for (const item of v) visit(item);
return;
}
const node = v as { kind?: unknown; fn?: unknown };
if (node.kind === "libCall" && typeof node.fn === "string" &&
(node.fn === "qs.parse" || node.fn === "qs.stringify" || node.fn === "qs.unescape")) {
found = true;
return;
}
for (const key of Object.keys(v)) visit((v as Record<string, unknown>)[key]);
};
visit(mod);
return found;
}
/** True when the module contains any fs.watch/watcher.* libCall — the
* link switch that pulls scr_watch.c into the binary and has the emitted
* main call scr_watch_install (cc.ts + emitter; the scr_net gating
+27
View File
@@ -251,6 +251,14 @@ export const LIB_FN_SIGS: Record<IrLibFn, { argTypes: (IrType | null)[]; result:
"sp.toString": { argTypes: [SEARCH_PARAMS_T], result: STRING },
"sp.keyAt": { argTypes: [SEARCH_PARAMS_T, F64], result: STRING },
"sp.valAt": { argTypes: [SEARCH_PARAMS_T, F64], result: STRING },
// node:querystring. qs.parse's result is the call site's ParsedUrlQuery
// dictionary record (VOID is the networkInterfaces sentinel — the
// libCall case checks the structure); qs.stringify's object argument
// is a DOM value (the frontend dynFroms typed records).
"qs.parse": { argTypes: [STRING, STRING, STRING, F64], result: VOID },
"qs.stringify": { argTypes: [DYN, STRING, STRING], result: STRING },
"qs.escape": { argTypes: [STRING], result: STRING },
"qs.unescape": { argTypes: [STRING], result: STRING },
"fs.statSync": { argTypes: [STRING], result: STATS_T },
"fs.lstatSync": { argTypes: [STRING], result: STATS_T },
"fs.openSync": { argTypes: [STRING, STRING], result: F64 },
@@ -3534,6 +3542,25 @@ function validateFunction(
}
break;
}
if (e.fn === "qs.parse") {
// Result: a pure index-signature record whose value union
// carries a string arm and a string[] arm (undefined tolerated
// — @types/node's Dict — and f64 too: the header-family
// canonicalization interns every such dictionary with the
// number arm, type-level only) — the structure
// lowerQuerystringParseCall pinned.
const shape = e.type.kind === "record" ? records.get(e.type.shapeId) : undefined;
let ok = shape !== undefined && !shape.tuple && shape.fields.length === 0 && shape.indexValue !== undefined;
const ivDef = ok && shape!.indexValue!.kind === "union" ? unions.get(shape!.indexValue!.unionId) : undefined;
ok = ok && ivDef !== undefined &&
ivDef.arms.some((a) => a.kind === "string") &&
ivDef.arms.some((a) => a.kind === "array" && a.elem.kind === "string") &&
ivDef.arms.every((a) => a.kind === "string" || a.kind === "array" || a.kind === "undefinedT" || a.kind === "f64");
if (!ok) {
err(`libCall qs.parse must return the ParsedUrlQuery dictionary record`, e.loc);
}
break;
}
if (e.fn === "child.onExit" || e.fn === "child.onError") {
// The listener: a closure with no params, or exactly the
// supported parameter shapes per event — exit takes (code:
+43
View File
@@ -1113,6 +1113,49 @@
"status": "static",
"note": "recognized module (bare and node:-prefixed specifiers)"
},
{
"id": "node-builtin.querystring",
"kind": "node-builtin",
"name": "querystring",
"status": "static",
"note": "recognized module (bare and node:-prefixed specifiers)"
},
{
"id": "node-builtin.querystring.decode",
"kind": "node-builtin",
"name": "querystring.decode",
"status": "static"
},
{
"id": "node-builtin.querystring.encode",
"kind": "node-builtin",
"name": "querystring.encode",
"status": "static"
},
{
"id": "node-builtin.querystring.escape",
"kind": "node-builtin",
"name": "querystring.escape",
"status": "static"
},
{
"id": "node-builtin.querystring.parse",
"kind": "node-builtin",
"name": "querystring.parse",
"status": "static"
},
{
"id": "node-builtin.querystring.stringify",
"kind": "node-builtin",
"name": "querystring.stringify",
"status": "static"
},
{
"id": "node-builtin.querystring.unescape",
"kind": "node-builtin",
"name": "querystring.unescape",
"status": "static"
},
{
"id": "node-builtin.readline",
"kind": "node-builtin",
+496
View File
@@ -0,0 +1,496 @@
/* node:querystring (scr_runtime.h has the API contract): Node v24's
* legacy query-string codec — NOT URLSearchParams (the escaping and '+'
* rules differ; that surface lives in scr_url_params.c). LINK-GATED:
* compiled into the binary only when the IR uses the qs surface
* (moduleUsesQs — the scr_url_params gating precedent), so qs-free
* programs pay zero bytes. querystring.escape never reaches this unit at
* all: Node's qsEscape encodes exactly the component unreserved set, so
* the frontend lowers it to the always-linked
* scr_str_encode_uri_component.
*
* Everything here is a quirk-faithful port of lib/querystring.js (Node is
* the oracle): parse's interleaved sep/eq scan with its naive
* partial-match resets and the encodeCheck fast path, unescapeBuffer's
* UTF-16-code-unit byte truncation, and stringify's encodeStringified
* value rules. The scans run byte-wise over the runtime's UTF-8 storage —
* equivalent to Node's code-unit scans for well-formed input, because the
* machine only assigns meaning to ASCII ('%', '+', hex, and the sep/eq
* sequences positionally) and UTF-8 multibyte alignment mirrors UTF-16
* unit alignment. */
#include "scr_runtime.h"
#include <math.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* ── a tiny growable byte buffer (scr_url_params.c's, redeclared) ────── */
typedef struct {
char *data;
size_t len;
size_t cap;
} QsBuf;
static void qb_init(QsBuf *b) {
b->cap = 64;
b->len = 0;
b->data = malloc(b->cap);
if (!b->data) {
fputs("scriptc: out of memory\n", stderr);
abort();
}
}
static void qb_append(QsBuf *b, const char *bytes, size_t n) {
if (b->len + n > b->cap) {
while (b->len + n > b->cap) b->cap *= 2;
char *grown = realloc(b->data, b->cap);
if (!grown) {
fputs("scriptc: out of memory\n", stderr);
abort();
}
b->data = grown;
}
memcpy(b->data + b->len, bytes, n);
b->len += n;
}
static void qb_push(QsBuf *b, char c) { qb_append(b, &c, 1); }
static void qb_append_str(QsBuf *b, const ScrStr *s) {
qb_append(b, s->data, s->len);
}
static ScrStr *qb_take(QsBuf *b) {
ScrStr *s = scr_str_new(b->data, b->len);
free(b->data);
return s;
}
/* ── shared scan helpers ─────────────────────────────────────────────── */
static int qs_unhex(unsigned c) {
if (c >= '0' && c <= '9') return (int)(c - '0');
if (c >= 'A' && c <= 'F') return (int)(c - 'A' + 10);
if (c >= 'a' && c <= 'f') return (int)(c - 'a' + 10);
return -1;
}
/* One UTF-8 code point at p (well-formed by the runtime invariant);
* *adv gets the byte length. */
static uint32_t qs_utf8_decode(const char *p, size_t *adv) {
unsigned char b0 = (unsigned char)p[0];
if (b0 < 0x80) {
*adv = 1;
return b0;
}
if (b0 < 0xE0) {
*adv = 2;
return (uint32_t)((b0 & 0x1Fu) << 6) | ((unsigned char)p[1] & 0x3Fu);
}
if (b0 < 0xF0) {
*adv = 3;
return (uint32_t)((b0 & 0x0Fu) << 12) |
(uint32_t)(((unsigned char)p[1] & 0x3Fu) << 6) |
((unsigned char)p[2] & 0x3Fu);
}
*adv = 4;
return (uint32_t)((b0 & 0x07u) << 18) |
(uint32_t)(((unsigned char)p[1] & 0x3Fu) << 12) |
(uint32_t)(((unsigned char)p[2] & 0x3Fu) << 6) |
((unsigned char)p[3] & 0x3Fu);
}
static void qs_utf8_encode(QsBuf *b, uint32_t cp) {
char tmp[4];
if (cp < 0x80) {
tmp[0] = (char)cp;
qb_append(b, tmp, 1);
} else if (cp < 0x800) {
tmp[0] = (char)(0xC0 | (cp >> 6));
tmp[1] = (char)(0x80 | (cp & 0x3F));
qb_append(b, tmp, 2);
} else if (cp < 0x10000) {
tmp[0] = (char)(0xE0 | (cp >> 12));
tmp[1] = (char)(0x80 | ((cp >> 6) & 0x3F));
tmp[2] = (char)(0x80 | (cp & 0x3F));
qb_append(b, tmp, 3);
} else {
tmp[0] = (char)(0xF0 | (cp >> 18));
tmp[1] = (char)(0x80 | ((cp >> 12) & 0x3F));
tmp[2] = (char)(0x80 | ((cp >> 6) & 0x3F));
tmp[3] = (char)(0x80 | (cp & 0x3F));
qb_append(b, tmp, 4);
}
}
/* ── unescape (Node's qsUnescape) ────────────────────────────────────── */
/* The string's UTF-16 code units (astral chars split into surrogate
* pairs) — unescapeBuffer's scan is defined over these, byte-truncation
* quirk included, so the fallback builds them first. Caller frees. */
static uint16_t *qs_utf16_units(const ScrStr *s, size_t *out_n) {
/* One unit per byte is a safe cap (ASCII 1:1; multibyte shrinks). */
uint16_t *units = malloc(sizeof(uint16_t) * (s->len ? s->len : 1));
if (!units) {
fputs("scriptc: out of memory\n", stderr);
abort();
}
size_t n = 0, i = 0;
while (i < s->len) {
size_t adv;
uint32_t cp = qs_utf8_decode(s->data + i, &adv);
i += adv;
if (cp >= 0x10000) {
cp -= 0x10000;
units[n++] = (uint16_t)(0xD800 + (cp >> 10));
units[n++] = (uint16_t)(0xDC00 + (cp & 0x3FF));
} else {
units[n++] = (uint16_t)cp;
}
}
*out_n = n;
return units;
}
/* WHATWG UTF-8 decode with replacement (maximal subpart) — the semantics
* of Buffer.prototype.toString('utf8') the fallback path ends with. */
static ScrStr *qs_utf8_replace_decode(const unsigned char *p, size_t n) {
QsBuf out;
qb_init(&out);
size_t i = 0;
while (i < n) {
unsigned char b0 = p[i];
if (b0 < 0x80) {
qb_push(&out, (char)b0);
i++;
continue;
}
int need;
unsigned lo = 0x80, hi = 0xBF;
if (b0 >= 0xC2 && b0 <= 0xDF) need = 1;
else if (b0 == 0xE0) { need = 2; lo = 0xA0; }
else if (b0 == 0xED) { need = 2; hi = 0x9F; }
else if (b0 >= 0xE1 && b0 <= 0xEF) need = 2;
else if (b0 == 0xF0) { need = 3; lo = 0x90; }
else if (b0 == 0xF4) { need = 3; hi = 0x8F; }
else if (b0 >= 0xF1 && b0 <= 0xF3) need = 3;
else { /* 0x80..0xC1, 0xF5..0xFF: never a leading byte */
qs_utf8_encode(&out, 0xFFFD);
i++;
continue;
}
/* Consume as many valid continuations as exist (maximal subpart):
* an invalid or missing continuation replaces the consumed prefix
* with ONE U+FFFD and rescans at the offending byte. */
uint32_t cp = b0 & (uint32_t)(0xFF >> (need + 2));
size_t j = i + 1;
int k = 0;
bool ok = true;
for (; k < need; k++, j++) {
if (j >= n) { ok = false; break; }
unsigned c = p[j];
unsigned clo = (k == 0) ? lo : 0x80, chi = (k == 0) ? hi : 0xBF;
if (c < clo || c > chi) { ok = false; break; }
cp = (cp << 6) | (c & 0x3F);
}
if (!ok) {
qs_utf8_encode(&out, 0xFFFD);
i = j; /* prefix consumed; rescan at the offending byte */
continue;
}
qs_utf8_encode(&out, cp);
i = j;
}
return qb_take(&out);
}
/* Node's unescapeBuffer(s) (decodeSpaces false — neither qsUnescape nor
* the exported unescape passes it) over the UTF-16 units, then the
* replacement decode of the byte buffer. */
static ScrStr *qs_unescape_buffer(const ScrStr *s) {
size_t n;
uint16_t *units = qs_utf16_units(s, &n);
unsigned char *out = malloc(n ? n : 1);
if (!out) {
fputs("scriptc: out of memory\n", stderr);
abort();
}
size_t w = 0, index = 0;
/* maxLength = s.length - 2 as a signed comparison (n < 2 must refuse
* every escape, matching Node's `index < maxLength` on negatives). */
while (index < n) {
unsigned cur = units[index];
if (cur == '%' && n >= 3 && index < n - 2) {
unsigned c1 = units[index + 1];
int hexHigh = (c1 < 256) ? qs_unhex(c1) : -1;
if (hexHigh < 0) {
out[w++] = '%';
index++;
continue;
}
unsigned c2 = units[index + 2];
int hexLow = (c2 < 256) ? qs_unhex(c2) : -1;
if (hexLow < 0) {
out[w++] = '%';
index++; /* Node: ++index then index-- — net one step */
continue;
}
cur = (unsigned)(hexHigh * 16 + hexLow);
index += 2;
}
out[w++] = (unsigned char)(cur & 0xFF); /* Buffer write truncates */
index++;
}
free(units);
ScrStr *res = qs_utf8_replace_decode(out, w);
free(out);
return res;
}
ScrStr *scr_qs_unescape(const ScrStr *s) {
ScrStr *strict = scr_str_decode_uri_component_try((ScrStr *)s);
if (strict) return strict;
return qs_unescape_buffer(s);
}
/* ── parse ───────────────────────────────────────────────────────────── */
/* addKeyVal's tail: decode-if-encoded, then group into the overflow map
* (first value = the str_tag arm; a repeat REPLACES it with a two-element
* string[] in the arr_tag arm; later repeats push). key/value move in. */
static void qs_add_key_val(ScrMap *out, ScrStr *key, ScrStr *value,
bool key_encoded, bool val_encoded,
uint32_t str_tag, uint32_t arr_tag) {
if (key->len > 0 && key_encoded) {
ScrStr *dec = scr_qs_unescape(key);
scr_str_release(key);
key = dec;
}
if (value->len > 0 && val_encoded) {
ScrStr *dec = scr_qs_unescape(value);
scr_str_release(value);
value = dec;
}
ScrUnion *cell = scr_map_get_str_ref(out, key);
if (!cell) {
scr_map_set_str_ref(out, key,
scr_union_new_ref(str_tag, value, &scr_str_retain_v,
&scr_str_release_v, NULL));
scr_str_release(key);
return;
}
if (cell->tag == str_tag) {
/* obj[key] = [curValue, value] */
ScrArr *rows = scr_arr_new(SCR_ELEM_STR, 2);
scr_arr_push_ref(rows, scr_str_retain_v(scr_union_peek(cell)));
scr_arr_push_ref(rows, value);
scr_map_set_str_ref(out, key,
scr_union_new_ref(arr_tag, rows, &scr_arr_retain_v,
&scr_arr_release_v, NULL));
} else {
scr_arr_push_ref((ScrArr *)scr_union_peek(cell), value);
}
scr_union_release(cell);
scr_str_release(key);
}
void scr_qs_parse_into(ScrMap *out, const ScrStr *qs, const ScrStr *sep,
const ScrStr *eq, double max_keys, uint32_t str_tag,
uint32_t arr_tag) {
if (qs->len == 0) return;
/* Node's falsy rule: undefined/null/'' all mean the default. */
const char *sep_b = (sep && sep->len) ? sep->data : "&";
size_t sep_len = (sep && sep->len) ? sep->len : 1;
const char *eq_b = (eq && eq->len) ? eq->data : "=";
size_t eq_len = (eq && eq->len) ? eq->len : 1;
/* pairs: Node's `maxKeys > 0 ? maxKeys : -1` as a double so Infinity
* decrements forever, exactly like Node's -1 sentinel. */
double pairs = max_keys > 0 ? max_keys : -1;
const char *p = qs->data;
size_t n = qs->len;
QsBuf key, value;
qb_init(&key);
qb_init(&value);
size_t last_pos = 0, sep_idx = 0, eq_idx = 0;
bool key_encoded = false, val_encoded = false;
int encode_check = 0;
bool returned = false;
for (size_t i = 0; i < n; ++i) {
unsigned char code = (unsigned char)p[i];
/* Try matching the pair separator (e.g. '&'). */
if (code == (unsigned char)sep_b[sep_idx]) {
if (++sep_idx == sep_len) {
/* Key/value pair separator match. */
size_t end = i - sep_idx + 1;
if (eq_idx < eq_len) {
/* No (entire) key/value separator seen. */
if (last_pos < end) {
qb_append(&key, p + last_pos, end - last_pos);
} else if (key.len == 0) {
/* An empty substring between separators. */
if (--pairs == 0) { returned = true; break; }
last_pos = i + 1;
sep_idx = eq_idx = 0;
continue;
}
} else if (last_pos < end) {
qb_append(&value, p + last_pos, end - last_pos);
}
qs_add_key_val(out, scr_str_new(key.data, key.len),
scr_str_new(value.data, value.len), key_encoded,
val_encoded, str_tag, arr_tag);
if (--pairs == 0) { returned = true; break; }
key_encoded = val_encoded = false;
key.len = value.len = 0;
encode_check = 0;
last_pos = i + 1;
sep_idx = eq_idx = 0;
}
} else {
sep_idx = 0;
/* Try matching the key/value separator (e.g. '=') if we haven't. */
if (eq_idx < eq_len) {
if (code == (unsigned char)eq_b[eq_idx]) {
if (++eq_idx == eq_len) {
/* Key/value separator match. */
size_t end = i - eq_idx + 1;
if (last_pos < end) qb_append(&key, p + last_pos, end - last_pos);
encode_check = 0;
last_pos = i + 1;
}
continue;
}
eq_idx = 0;
if (!key_encoded) {
/* Match a valid encoded byte once, to minimize decode calls. */
if (code == '%') {
encode_check = 1;
continue;
} else if (encode_check > 0) {
if (qs_unhex(code) >= 0) {
if (++encode_check == 3) key_encoded = true;
continue;
} else {
encode_check = 0;
}
}
}
if (code == '+') {
if (last_pos < i) qb_append(&key, p + last_pos, i - last_pos);
qb_push(&key, ' ');
last_pos = i + 1;
continue;
}
}
if (code == '+') {
if (last_pos < i) qb_append(&value, p + last_pos, i - last_pos);
qb_push(&value, ' ');
last_pos = i + 1;
} else if (!val_encoded) {
if (code == '%') {
encode_check = 1;
} else if (encode_check > 0) {
if (qs_unhex(code) >= 0) {
if (++encode_check == 3) val_encoded = true;
} else {
encode_check = 0;
}
}
}
}
}
if (!returned) {
/* Leftover key or value data. */
bool ended_empty = false;
if (last_pos < n) {
if (eq_idx < eq_len) qb_append(&key, p + last_pos, n - last_pos);
else if (sep_idx < sep_len) qb_append(&value, p + last_pos, n - last_pos);
} else if (eq_idx == 0 && key.len == 0) {
ended_empty = true; /* ended on an empty substring */
}
if (!ended_empty) {
qs_add_key_val(out, scr_str_new(key.data, key.len),
scr_str_new(value.data, value.len), key_encoded,
val_encoded, str_tag, arr_tag);
}
}
free(key.data);
free(value.data);
}
/* ── stringify ───────────────────────────────────────────────────────── */
/* Node's encodeStringified over one DOM value: strings escape, finite
* numbers render then escape, booleans are bare, everything else is the
* empty value. Appends to b. */
static void qs_stringify_value(QsBuf *b, const ScrDyn *v) {
switch (v->kind) {
case SCR_DYN_STR: {
ScrStr *enc = scr_str_encode_uri_component(v->v.str);
qb_append_str(b, enc);
scr_str_release(enc);
return;
}
case SCR_DYN_NUM: {
if (!isfinite(v->v.num)) return;
ScrStr *num = scr_f64_to_scrstr(v->v.num);
ScrStr *enc = scr_str_encode_uri_component(num);
qb_append_str(b, enc);
scr_str_release(enc);
scr_str_release(num);
return;
}
case SCR_DYN_BOOL:
if (v->v.b) qb_append(b, "true", 4);
else qb_append(b, "false", 5);
return;
default:
return; /* null/undefined/objects/arrays/functions → '' */
}
}
ScrStr *scr_qs_stringify(const ScrDyn *obj, const ScrStr *sep,
const ScrStr *eq) {
const char *sep_b = (sep && sep->len) ? sep->data : "&";
size_t sep_len = (sep && sep->len) ? sep->len : 1;
const char *eq_b = (eq && eq->len) ? eq->data : "=";
size_t eq_len = (eq && eq->len) ? eq->len : 1;
if (!obj || obj->kind != SCR_DYN_OBJ) return scr_str_new("", 0);
QsBuf fields;
qb_init(&fields);
/* JS own-key order (array-index keys ascending first) — ObjectKeys. */
ScrDyn *keys = scr_dyn_obj_keys((ScrDyn *)obj);
for (size_t i = 0; i < keys->v.arr.len; i++) {
const ScrDyn *kd = keys->v.arr.items[i];
const ScrStr *k = kd->v.str;
const ScrDyn *v = scr_dyn_obj_get(obj, k->data, k->len);
if (!v) continue; /* unreachable: keys came from the object */
ScrStr *ks = scr_str_encode_uri_component((ScrStr *)k);
if (v->kind == SCR_DYN_ARR) {
for (size_t j = 0; j < v->v.arr.len; j++) {
if (fields.len) qb_append(&fields, sep_b, sep_len);
qb_append_str(&fields, ks);
qb_append(&fields, eq_b, eq_len);
qs_stringify_value(&fields, v->v.arr.items[j]);
}
} else {
if (fields.len) qb_append(&fields, sep_b, sep_len);
qb_append_str(&fields, ks);
qb_append(&fields, eq_b, eq_len);
qs_stringify_value(&fields, v);
}
scr_str_release(ks);
}
scr_dyn_release(keys);
return qb_take(&fields);
}
+58
View File
@@ -484,6 +484,64 @@ ScrStr *scr_str_encode_uri_component(ScrStr *s);
* malformed") and returns NULL. Borrows s; result +1. */
ScrStr *scr_str_decode_uri_component(ScrStr *s);
/* The non-throwing core of decodeURIComponent: NULL on malformed input
* instead of the URIError — the try/catch shape node:querystring's
* unescape wraps around decodeURIComponent (scr_qs.c's strict pass). */
ScrStr *scr_str_decode_uri_component_try(ScrStr *s);
/* ── node:querystring (scr_qs.c — LINK-GATED by moduleUsesQs; the
* scr_url_params.c precedent: pure data transforms, no loop hooks).
* querystring.escape needs no entry here: Node's qsEscape encodes exactly
* the component unreserved set, so the frontend lowers it to
* scr_str_encode_uri_component (always linked; a program using only
* escape never pulls this unit). */
/* querystring.unescape — Node's qsUnescape: strict decodeURIComponent
* first, and on failure the lenient legacy unescapeBuffer(s).toString()
* (valid %XX escapes decode to their byte, malformed escapes copy
* literally, every non-escape UTF-16 CODE UNIT truncates to its low byte
* — Node's Buffer element write — and the byte buffer decodes as UTF-8
* with U+FFFD replacement per maximal subpart, Buffer.toString's rule).
* Borrows s; result +1; never throws. */
ScrStr *scr_qs_unescape(const ScrStr *s);
/* querystring.parse — Node v24's scan state machine, byte-wise over the
* UTF-8 storage (equivalent to the code-unit scan for well-formed input:
* the machine compares sep/eq sequences positionally and treats '%'/'+'/
* hex as ASCII, and multibyte alignment is preserved). Decoded pairs land
* in `out` — the PURE-index-signature Dict record's overflow map (the
* emitters pass rec->sc_ovf) — grouped like Node's addKeyVal: a first
* value stores the `str_tag` union arm, a repeat REPLACES it with a
* two-element string[] wrapped in `arr_tag`, later repeats push. sep/eq
* fall back to "&"/"=" when empty (Node's falsy rule; the frontend
* completes omitted/null arguments to the defaults). max_keys is Node's
* rule exactly: > 0 caps the PAIR count (empty skipped segments count,
* like Node's --pairs), anything else (0, negatives, NaN) is unlimited;
* the frontend completes the omitted option to 1000. Only the DEFAULT
* decoder runs (custom decodeURIComponent options fence at compile time),
* so '+' means ' ' and segments decode with scr_qs_unescape's semantics
* only when they carry a full valid %XX triple (Node's encodeCheck).
* Borrows everything; never throws. */
typedef struct ScrMap ScrMap; /* full definition below (C11 repeat) */
void scr_qs_parse_into(ScrMap *out, const ScrStr *qs, const ScrStr *sep,
const ScrStr *eq, double max_keys, uint32_t str_tag,
uint32_t arr_tag);
/* querystring.stringify — Node's stringify over a borrowed DOM value (the
* frontend dynFroms the typed record; JS-world dyn values pass straight
* through). Non-object DOMs answer "" like Node; object keys iterate in
* JS own-key order (scr_dyn_obj_keys — array-index keys ascending first,
* then insertion order, Node's ObjectKeys). Values serialize per Node's
* encodeStringified: strings escape, finite numbers render shortest-
* roundtrip then escape ('1e+21' → '1e%2B21'), booleans are bare
* true/false, arrays expand to repeated keys (empty arrays emit NOTHING,
* key included), and everything else (null, undefined, nested objects/
* arrays, functions) is the empty value — key and eq still emitted.
* sep/eq fall back to "&"/"=" when empty (Node's `sep ||= '&'`). Result
* +1; never throws. */
ScrStr *scr_qs_stringify(const struct ScrDyn *obj, const ScrStr *sep,
const ScrStr *eq);
/* encodeURI — the same ECMA-262 Encode() with the reserved set and '#'
* kept unescaped (scr_encode_uri_impl's keep_reserved arm). Total by the
* well-formed-UTF-8 invariant. Borrows s; result +1. */
+15 -2
View File
@@ -1212,7 +1212,12 @@ static int scr_uri_hex_byte(const char *p, size_t rem) {
return (hi << 4) | lo;
}
ScrStr *scr_str_decode_uri_component(ScrStr *s) {
/* The non-throwing core of decodeURIComponent: NULL on malformed input
* (bad hex, invalid UTF-8 octets) instead of the URIError — the
* querystring unit's unescape needs exactly the try/catch shape Node's
* qsUnescape wraps around decodeURIComponent (scr_qs.c), and the throwing
* entry point below stays byte-identical by rethrowing over NULL. */
ScrStr *scr_str_decode_uri_component_try(ScrStr *s) {
/* Decoding only ever shrinks (%XX → 1 byte), so len is a safe cap. */
ScrStr *out = scr_str_alloc_raw(0, s->len);
size_t w = 0;
@@ -1258,7 +1263,15 @@ ScrStr *scr_str_decode_uri_component(ScrStr *s) {
return out;
malformed:
scr_str_release(out);
scr_uri_malformed();
return NULL;
}
ScrStr *scr_str_decode_uri_component(ScrStr *s) {
ScrStr *out = scr_str_decode_uri_component_try(s);
if (!out) {
scr_uri_malformed();
return NULL;
}
return out;
}
+63
View File
@@ -0,0 +1,63 @@
// node:querystring.parse — the grammar corners of Node's legacy scan
// (scr_qs.c's quirk-faithful port; Node is the oracle), generated as an
// input sweep: empty/keyless/valueless segments, repeated keys becoming
// arrays, '=' inside values, '+' meaning space (but '%2B' staying '+'),
// malformed percent-escapes (bad hex copies literally; valid escapes
// that decode to invalid UTF-8 take the lenient unescapeBuffer fallback
// with U+FFFD replacement — lone-surrogate escapes included), raw
// non-ASCII and astral input, and the encodeCheck fast path (a segment
// only decodes when it carries a full valid %XX triple).
import { parse } from "node:querystring";
const cases: string[] = [
"",
"a",
"a=",
"=a",
"=",
"&",
"&&",
"a&&b",
"a=1&",
"&a=1",
"a=1&&b=2",
"a=1&a=2&a=3",
"k=v&k=w&k=x&single=s",
"a=b=c&==x",
"a==b",
"a%3Db=c",
"key only",
"sp ace=v al",
"a+b=c+d&%20=+",
"a=+%2B+",
"a=%2B+b",
"%zz",
"%zz%25=%25zz",
"%25%25=%25",
"a=%E2%98%83&b=%FF&c=%zz&d=%1",
"foo=%F0%9F%98%80&bar=%ED%A0%80",
"%E2%98%83=%E2%98%83",
"☃=☃&é=é",
"😀=🌍",
"%",
"%2",
"a=%",
"a=%F",
"a=%FG",
"%GG=1",
"1=a&0=b&x=c",
];
for (const c of cases) {
console.log(JSON.stringify(c), JSON.stringify(parse(c)));
}
// The result dictionary reads like any index-signature record: repeated
// keys narrow to their array bucket, singles to the string arm, absent
// keys to undefined.
const r = parse("k=v&k=w&single=s");
const k = r["k"];
if (Array.isArray(k)) console.log("bucket", k.length, k.join("|"));
const s = r["single"];
if (typeof s === "string") console.log("single", s);
console.log("missing", r["missing"] === undefined);
console.log("keys", Object.keys(parse("1=a&0=b&x=c")).join(","));
+35
View File
@@ -0,0 +1,35 @@
// node:querystring.parse — separators and maxKeys (Node is the oracle).
// Custom sep/eq including multi-character and multi-byte sequences (the
// scan's naive partial-match resets are quirk-faithful: an overlapping
// partial match is NOT re-examined, Node's own behavior), the falsy rule
// (null/undefined/'' all mean the defaults), and maxKeys' pair budget —
// which empty skipped segments consume too, and which 0 and negatives
// remove entirely (Node's `maxKeys > 0 ? maxKeys : -1`).
import { parse } from "node:querystring";
// Custom separators.
console.log("S1", JSON.stringify(parse("a:1;b:2", ";", ":")));
console.log("S2", JSON.stringify(parse("a::1;;b::2", ";;", "::")));
console.log("S3", JSON.stringify(parse("aXYb=1XYXc=2", "XYX")));
console.log("S4", JSON.stringify(parse("a==b=c===d", undefined, "==")));
console.log("S5", JSON.stringify(parse("a=1&b=2", "", "")));
console.log("S6", JSON.stringify(parse("a☃1;b☃2", ";", "☃")));
console.log("S7", JSON.stringify(parse("aabX=1", "ab")));
console.log("S8", JSON.stringify(parse("a=1&b=2", null, null)));
console.log("S9", JSON.stringify(parse("x🌍y=1&z=2", null, "🌍")));
// maxKeys: the pair budget.
console.log("M1", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: 2 })));
console.log("M2", JSON.stringify(parse("a=1&b=2", null, null, { maxKeys: 0 })));
console.log("M3", JSON.stringify(parse("&&&a=1&b=2", null, null, { maxKeys: 2 })));
console.log("M4", JSON.stringify(parse("a=1&&b=2&c=3", null, null, { maxKeys: 2 })));
console.log("M5", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: -5 })));
console.log("M6", JSON.stringify(parse("a=1&a=2&a=3&b=4", null, null, { maxKeys: 3 })));
console.log("M7", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: Infinity })));
console.log("M8", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: 1000 })));
console.log("M9", JSON.stringify(parse("a:1;b:2;c:3", ";", ":", { maxKeys: 2 })));
// A runtime maxKeys expression (the budget lives in the runtime).
let budget = 1;
budget += 1;
console.log("M10", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: budget })));
+40
View File
@@ -0,0 +1,40 @@
// node:querystring.stringify — Node's encodeStringified value rules
// (scr_qs.c over the DOM crossing): strings escape with the querystring
// set (spaces are %20, never '+'), finite numbers render shortest-
// roundtrip then escape ('1e+21' → '1e%2B21'), non-finite numbers and
// null/undefined are empty VALUES (key and eq still emitted), booleans
// are bare, arrays expand to repeated keys (empty arrays emit nothing at
// all), custom sep/eq apply verbatim with the falsy default rule, and
// keys escape like values. encode is Node's own alias.
import { stringify, encode } from "node:querystring";
console.log("T1", stringify({ a: 1, b: "x y", c: ["1", "2"], d: true, e: "" }));
console.log("T2", stringify({ a: Infinity, b: NaN, c: -0, d: 1e21, e: 0.1, f: -1.5 }));
console.log("T3", stringify({ "a b": "c", "☃": "+", "": "empty", k: "" }));
console.log("T4", JSON.stringify(stringify({})));
console.log("T5", stringify({ a: [] as string[], b: "x" }));
console.log("T6", stringify({ a: ["x", "y"] }, ";", ":"));
console.log("T7", stringify({ a: "1", b: "2" }, "", ""));
console.log("T8", stringify({ u: undefined, n: null, s: "v" } as Record<string, string | null | undefined>));
console.log("T9", stringify({ a: [1, 2.5, true] as (number | boolean)[] }));
console.log("T10", stringify({ "é☃": "é☃ 😀" }));
console.log("T11", stringify({ a: "x" }, "☃", "🌍"));
console.log("T12", encode({ x: [1, 2] }));
// Declared shapes with union-valued fields serialize by the value each
// field HOLDS (an explicitly-undefined optional field keeps its key with
// the empty value, exactly Node).
const o: { u?: string; s: string; n: number | null } = { u: undefined, s: "v", n: null };
console.log("T13", stringify(o));
// Index-signature records (the runtime-keyed build).
const r: Record<string, string | string[]> = {};
r["first"] = "1";
r["multi"] = ["a", "b"];
r["last"] = "z";
console.log("T14", stringify(r));
// The round trip: stringify(parse(qs)) is stable for canonical input.
import { parse } from "node:querystring";
const round = "a=1&a=2&b=x%20y&c=";
console.log("T15", stringify(parse(round)));
+47
View File
@@ -0,0 +1,47 @@
// node:querystring.escape/unescape (Node is the oracle). escape encodes
// exactly the component unreserved set (ALPHA/DIGIT/- _ . ! ~ * ' ( )) as
// uppercase %XX — the full printable-ASCII sweep pins the set character
// by character, non-ASCII encodes its UTF-8 bytes. unescape is the
// strict-then-lenient pair: decodeURIComponent when the whole string
// decodes, else the legacy unescapeBuffer — valid %XX escapes decode to
// their byte, malformed escapes copy literally, every non-escape UTF-16
// code unit truncates to its LOW BYTE (Node's Buffer element write — the
// '☃%E9' and astral corners below), and the byte buffer decodes as UTF-8
// with U+FFFD replacement per maximal subpart.
import { escape as esc, unescape as unesc } from "node:querystring";
// The printable-ASCII sweep, one character at a time.
const parts: string[] = [];
for (let i = 32; i < 127; i++) parts.push(String.fromCharCode(i));
const ascii = parts.join("");
console.log("E1", esc(ascii));
console.log("E2", esc("héllo ☃ 😀"));
console.log("E3", esc(""));
console.log("E4", esc("a+b c=d&e"));
// Strict decodes.
console.log("U1", JSON.stringify(unesc("%41%20%42")));
console.log("U2", JSON.stringify(unesc("%E2%98%83")));
console.log("U3", JSON.stringify(unesc("%F0%9F%98%80")));
console.log("U4", JSON.stringify(unesc("abc")));
console.log("U5", JSON.stringify(unesc("%25")));
console.log("U6", JSON.stringify(unesc("a+b%20c")));
// Lenient fallbacks: malformed hex, invalid UTF-8 octets, lone-surrogate
// escapes, trailing '%', truncated sequences, code-unit truncation.
console.log("F1", JSON.stringify(unesc("a+b%20c%E2%98%83%zz%")));
console.log("F2", JSON.stringify(unesc("%ED%A0%80")));
console.log("F3", JSON.stringify(unesc("%FF%fe")));
console.log("F4", JSON.stringify(unesc("é%E9")));
console.log("F5", JSON.stringify(unesc("☃%E9")));
console.log("F6", JSON.stringify(unesc("😀%E9")));
console.log("F7", JSON.stringify(unesc("%")));
console.log("F8", JSON.stringify(unesc("%2")));
console.log("F9", JSON.stringify(unesc("100%")));
console.log("F10", JSON.stringify(unesc("%E2%98")));
console.log("F11", JSON.stringify(unesc("%GG%41")));
console.log("F12", JSON.stringify(unesc("%c3%a9%ff")));
// escape/unescape compose with parse/stringify's own paths — the same
// codec observed directly.
console.log("C1", esc(unesc("%E2%98%83")) === "%E2%98%83");
+31
View File
@@ -0,0 +1,31 @@
// node:querystring through every CommonJS acquisition spelling a JS
// package uses (the HTTP-client dependency chains that import it
// unguarded are CJS): the whole-module require binding, the destructured
// require, the member-binding require, the inline require member call,
// and the decode/encode aliases — all keying the same lowering tables.
"use strict";
const querystring = require("node:querystring");
const { parse, stringify } = require("querystring");
const esc = require("querystring").escape;
console.log("Q1", JSON.stringify(querystring.parse("a=1&a=2&b=x%20y+z&c")));
console.log("Q2", querystring.stringify({ a: 1, arr: [1, "x", true], u: "héllo ☃", s: "a b", e: "" }));
console.log("Q3", querystring.escape("héllo ☃ a+b!'()*~"));
console.log("Q4", querystring.unescape("a+b%20c%E2%98%83%zz"));
console.log("Q5", JSON.stringify(querystring.decode("x=1&x=2")));
console.log("Q6", querystring.encode({ x: [1, 2] }));
console.log("D1", JSON.stringify(parse("a:1;b:2", ";", ":")));
console.log("D2", JSON.stringify(parse("a=1&b=2&c=3", null, null, { maxKeys: 2 })));
console.log("D3", stringify({ k: ["v", "w"] }, ";", ":"));
console.log("M1", esc("a b+c"));
console.log("M2", require("querystring").unescape("%E2%98%83"));
// Results feed ordinary JS flows: property reads, Array.isArray splits.
const parsed = parse("tag=a&tag=b&page=2");
const tags = parsed.tag;
if (Array.isArray(tags)) console.log("R1", tags.join(","));
const page = parsed.page;
if (typeof page === "string") console.log("R2", page);