fix(v1.0.133): hotfix v1.0.132 stale bundles + classifyIp + stats bar

v1.0.132 had three cascading bugs:
- CI postinstall hard-fail (#564) blocked its own Linux+Node20 runners → bundle rebuild failed → npm package shipped with v1.0.131-era bundles (source had fixes, runtime didn't)
- ctx_fetch_and_index threw ReferenceError: classifyIp is not defined (esbuild renamed the symbol; embedded SSRF guard called the canonical name)
- ctx_stats per-conversation bar read session_events.LENGTH(data) only (~200 bytes); never queried content DB chunks despite Slice 1 FK plumbing

Fixes:
- .github/workflows/bundle.yml + ci.yml node-version 20 → 22.5 (so CI can install + rebuild bundles)
- src/server.ts buildFetchCode: emit named function expression + var classifyIp = <name> alias so canonical identifier survives bundler rename (root cause at L1940)
- src/session/analytics.ts: new getContentBytesForSession() helper + getRealBytesStats accepts contentDbPath + ctx_stats handler in src/server.ts:2926 passes contentDbPath; per-session bar now sums chunk content bytes via session_id FK (architect-approved render-time read-only join, no time-window backfill)

Bundles included in this commit (CI blocked last time, ensuring npm tarball gets working code).

Tests: 3317 pass / 10 fail (8 OpenCode + 2 lifecycle env) / 29 skipped. Targeted modified-file tests: 540/540 pass.
This commit is contained in:
Mert Koseoglu
2026-05-14 21:20:40 +03:00
parent 7e4acb9d51
commit 1b7d7a727f
9 changed files with 557 additions and 284 deletions
+1 -1
View File
@@ -20,7 +20,7 @@ jobs:
- uses: actions/setup-node@v4
with:
node-version: 20
node-version: "22.5"
- run: npm install
+1 -1
View File
@@ -21,7 +21,7 @@ jobs:
- uses: actions/setup-node@v4
with:
node-version: 20
node-version: "22.5"
- uses: actions/setup-python@v5
with:
+168 -159
View File
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+119 -117
View File
File diff suppressed because one or more lines are too long
+27 -2
View File
@@ -1937,7 +1937,25 @@ export function buildFetchCode(url: string, outputPath: string): string {
// can serve a public IP for the parent's pre-flight ssrfGuard lookup and
// then a blocked IP (e.g. 169.254.169.254 IMDS) for the subprocess fetch's
// own lookup — classic DNS rebinding across the parent/child boundary.
const classifyIpSrc = classifyIp.toString();
//
// CRITICAL: bundlers (esbuild) rename top-level identifiers — `classifyIp`
// becomes e.g. `_h` in server.bundle.mjs. `classifyIp.toString()` returns
// the renamed source `function _h(t){...}`, but the embedded subprocess
// template references the literal name `classifyIp` (and the function's
// own internal recursion is also `_h(...)`). Result: the subprocess sees
// `function _h(t){...; return _h(...)}` injected, then references to
// `classifyIp` blow up with `ReferenceError: classifyIp is not defined`.
//
// Fix: emit `var <fnName> = <fn-expr>; var classifyIp = <fnName>;`. The
// named function expression preserves recursion under whatever name the
// bundler chose, and the alias re-exposes the canonical `classifyIp`
// identifier the rest of the embedded script depends on.
const classifyIpInner = classifyIp.toString();
const classifyIpFnName = classifyIp.name || "classifyIp";
const classifyIpSrc =
classifyIpFnName === "classifyIp"
? `var classifyIp = ${classifyIpInner};`
: `var ${classifyIpFnName} = ${classifyIpInner};\nvar classifyIp = ${classifyIpFnName};`;
const strictMode = process.env.CTX_FETCH_STRICT === "1";
return `
const TurndownService = require(${turndownPath});
@@ -2900,7 +2918,14 @@ server.registerTool(
}
if (sid) {
conversation = getConversationStats({ sessionId: sid, sessionsDir: getSessionDir(), worktreeHash: dbHash });
const convReal = getRealBytesStats({ sessionId: sid, sessionsDir: getSessionDir(), worktreeHash: dbHash });
// v1.0.133 Slice 3: pass contentDbPath so getRealBytesStats can
// join chunks WHERE session_id = sid and fold the indexed
// content bytes into the per-conversation bar. Without this,
// Mert's session showed ~200B (event metadata only) even with
// 49 MB of indexed content sitting in the content DB.
// Render-time read-only — no DB mutation, no backfill.
const contentDbPath = getStorePath();
const convReal = getRealBytesStats({ sessionId: sid, sessionsDir: getSessionDir(), worktreeHash: dbHash, contentDbPath });
const lifeReal = getRealBytesStats({ sessionsDir: getSessionDir() });
realBytes = { conversation: convReal, lifetime: lifeReal };
}
+86 -1
View File
@@ -1014,9 +1014,69 @@ export interface RealBytesStats {
bytesAvoided: number;
bytesReturned: number;
snapshotBytes: number;
/**
* v1.0.133 Slice 3: bytes attributed to this session in the FTS5 content
* DB — `SUM(LENGTH(title) + LENGTH(content)) FROM chunks WHERE session_id = ?`.
*
* Read-only, render-time computation. Populated only when
* `getRealBytesStats` is called with both `sessionId` AND `contentDbPath`
* (i.e. the conversation tier from ctx_stats). Lifetime / project tiers
* leave this at 0 — aggregating across every adapter's content DB is a
* separate concern.
*
* Legacy chunks with empty `session_id` (pre-Slice-1) are NOT backfilled:
* the architect rejected the time-window join as unsafe. Old conversations
* stay low; new conversations populate honestly.
*/
contentBytes: number;
totalSavedTokens: number;
}
/**
* v1.0.133 Slice 3: Sum the bytes attributed to one session in the FTS5
* content DB.
*
* Returns `LENGTH(title) + LENGTH(content)` summed across every chunk
* whose `session_id` column matches `sessionId`. Best-effort — returns 0
* when the DB file is missing, the schema lacks the `session_id` column
* (pre-Slice-1 content DBs), or the query fails. Never throws.
*
* Render-time only. Does NOT mutate the content DB. Architect-approved
* because the read-only join carries no risk of cross-session attribution
* (the FK was set at chunk insert time by Slice 1).
*/
export function getContentBytesForSession(
sessionId: string,
contentDbPath: string,
opts?: { loadDatabase?: () => unknown },
): number {
if (!sessionId || !contentDbPath) return 0;
if (!existsSync(contentDbPath)) return 0;
let DatabaseCtor: ReturnType<typeof loadDatabaseImpl> | null = null;
try {
DatabaseCtor = opts?.loadDatabase
? (opts.loadDatabase() as ReturnType<typeof loadDatabaseImpl>)
: loadDatabaseImpl();
} catch { return 0; }
if (!DatabaseCtor) return 0;
try {
const db = new DatabaseCtor(contentDbPath, { readonly: true });
try {
const row = db.prepare(
`SELECT COALESCE(SUM(LENGTH(content) + LENGTH(title)), 0) AS bytes
FROM chunks WHERE session_id = ?`,
).get(sessionId) as { bytes: number } | undefined;
return Number(row?.bytes ?? 0);
} finally {
db.close();
}
} catch {
return 0;
}
}
/**
* Compute real-bytes stats across one session, one project (worktree
* filter), or every session on disk (lifetime).
@@ -1035,6 +1095,13 @@ export function getRealBytesStats(opts: {
sessionId?: string;
sessionsDir?: string;
worktreeHash?: string;
/**
* v1.0.133 Slice 3: when set alongside `sessionId`, the function joins
* the FTS5 content DB at this path and folds chunk bytes into
* `bytesAvoided` + `totalSavedTokens` + `contentBytes`. Render-time
* only — no DB writes.
*/
contentDbPath?: string;
loadDatabase?: () => unknown;
}): RealBytesStats {
const empty: RealBytesStats = {
@@ -1042,6 +1109,7 @@ export function getRealBytesStats(opts: {
bytesAvoided: 0,
bytesReturned: 0,
snapshotBytes: 0,
contentBytes: 0,
totalSavedTokens: 0,
};
@@ -1128,11 +1196,27 @@ export function getRealBytesStats(opts: {
} catch { /* missing tables / corrupt — skip */ }
}
// v1.0.133 Slice 3: fold content DB chunk bytes for this session into
// bytesAvoided. Skipped silently when caller didn't pass contentDbPath
// (lifetime / project tiers, or pre-Slice-3 callers). Treated as
// "avoided" because indexed chunks are bytes that would have been
// re-inflated into context on every search if the model had to
// re-read raw files.
let contentBytes = 0;
if (opts.sessionId && opts.contentDbPath) {
contentBytes = getContentBytesForSession(
opts.sessionId,
opts.contentDbPath,
{ loadDatabase: opts.loadDatabase },
);
bytesAvoided += contentBytes;
}
const totalSavedTokens = Math.floor(
(eventDataBytes + bytesAvoided + snapshotBytes) / 4,
);
return { eventDataBytes, bytesAvoided, bytesReturned, snapshotBytes, totalSavedTokens };
return { eventDataBytes, bytesAvoided, bytesReturned, snapshotBytes, contentBytes, totalSavedTokens };
}
// ─────────────────────────────────────────────────────────
@@ -1381,6 +1465,7 @@ export function getMultiAdapterRealBytesStats(opts?: {
bytesAvoided: 0,
bytesReturned: 0,
snapshotBytes: 0,
contentBytes: 0,
totalSavedTokens: 0,
};
const perAdapter: MultiAdapterRealBytesStats["perAdapter"] = [];
+48
View File
@@ -3583,6 +3583,54 @@ describe("buildFetchCode — embedded SSRF guard contract", () => {
expect(generated).toMatch(/delete process\.env\.all_proxy/);
});
test("embedded SSRF classifier is callable as `classifyIp` even when bundler renames the export (#bug-v1.0.133)", () => {
// REGRESSION: esbuild renames top-level `classifyIp` to a short name
// (e.g. `_h`) in server.bundle.mjs. The previous implementation embedded
// `classifyIp.toString()` directly, which yielded `function _h(t){...}`
// — but the subprocess template invokes `classifyIp(...)` literally and
// the function's own internal recursion uses the bundler-mangled name.
// Result was 100% failure of ctx_fetch_and_index in the published build:
// ReferenceError: classifyIp is not defined
// at patchedPromisesLookup (.../script.js:71:19)
// The fix must: (1) expose the canonical `classifyIp` identifier in the
// embedded scope, AND (2) preserve recursion under whatever name the
// bundler chose. Validate by evaluating the embedded source in an
// isolated scope and confirming `classifyIp` resolves and works for
// both direct calls AND the recursive IPv4-mapped-IPv6 path.
expect(generated).toMatch(/var\s+classifyIp\s*=/);
// Extract the self-contained classifier declaration block (var classifyIp
// = function classifyIp(rawIp){...};) and evaluate it in a fresh function
// scope where neither `classifyIp` nor any bundler alias exists in the
// outer closure. The canonical name MUST resolve and behave.
const classifyIpDeclMatch = generated.match(
/var\s+\w+\s*=\s*function\s+\w+\s*\(\s*rawIp\s*\)[\s\S]+?\n\};(?:\s*var\s+classifyIp\s*=\s*\w+;)?/,
);
expect(
classifyIpDeclMatch,
"embedded classifyIp declaration block must be extractable from buildFetchCode output",
).not.toBeNull();
const classifierBlock = classifyIpDeclMatch![0];
// eslint-disable-next-line @typescript-eslint/no-implied-eval
const probe = new Function(`
${classifierBlock}
return {
imds: classifyIp("169.254.169.254"),
loopback: classifyIp("127.0.0.1"),
publicIp: classifyIp("8.8.8.8"),
mapped: classifyIp("::ffff:169.254.169.254"),
};
`);
const result = probe() as Record<string, string>;
expect(result.imds).toBe("block");
expect(result.loopback).toBe("private");
expect(result.publicIp).toBe("public");
// IPv4-mapped IPv6 forces the internal recursive call path; if recursion
// is broken (bundler-mangled self-reference unresolved), this throws.
expect(result.mapped).toBe("block");
});
test("patches dns/promises lookup (separate function reference from dns.lookup)", () => {
// Patching dns.lookup does NOT affect dnsPromises.lookup. Today undici
// uses callback-form dns.lookup so default fetch is covered, but the
+105 -1
View File
@@ -26,7 +26,8 @@ import { join } from "node:path";
import { randomUUID } from "node:crypto";
import { afterAll, describe, expect, test } from "vitest";
import { SessionDB } from "../../src/session/db.js";
import { getRealBytesStats } from "../../src/session/analytics.js";
import { getContentBytesForSession, getRealBytesStats } from "../../src/session/analytics.js";
import { ContentStore } from "../../src/store.js";
const cleanups: Array<() => void> = [];
@@ -167,4 +168,107 @@ describe("getRealBytesStats (Phase 8 renderer source-of-truth)", () => {
expect(r.bytesReturned).toBe(0);
expect(r.totalSavedTokens).toBe(0);
});
// ── v1.0.133: stats bar reads content DB chunks (Slice 3 — render-time only) ──
//
// v1.0.132 wired chunks.session_id (Slice 1) so new chunks carry the FK.
// The render path still ignored the content DB, leaving the per-conversation
// bar invisible (≈200 B of event metadata). Slice 3 closes the loop with a
// read-only join: when ctx_stats fires, sum LENGTH(title)+LENGTH(content)
// FROM chunks WHERE session_id = ? and fold it into the bar formula.
//
// Architect-safe choice: legacy chunks (empty session_id) are NOT backfilled.
// Old sessions stay low; new sessions populate honestly.
test("8.7 getContentBytesForSession sums LENGTH(title)+LENGTH(content) for FK-attributed chunks", () => {
const sid = `chunk-${randomUUID()}`;
const contentDbPath = join(mkSessionsDir(), `content-${randomUUID()}.db`);
const store = new ContentStore(contentDbPath);
try {
// Two attributed chunks for the target session.
store.indexPlainText(
"alpha line one\nalpha line two",
"src/alpha.ts",
20,
{ sessionId: sid, eventId: "evt-1" },
);
store.indexPlainText(
"beta payload that should be summed",
"src/beta.ts",
20,
{ sessionId: sid, eventId: "evt-2" },
);
// One chunk attributed to a DIFFERENT session — must be excluded.
store.indexPlainText(
"noise from a sibling session",
"src/noise.ts",
20,
{ sessionId: "other-session", eventId: "evt-x" },
);
// One legacy chunk with empty session_id — must be excluded (no backfill).
store.indexPlainText(
"legacy chunk no FK",
"src/legacy.ts",
20,
);
} finally {
store.close();
}
const bytes = getContentBytesForSession(sid, contentDbPath);
// Two chunks for `sid`: titles "src/alpha.ts" + "src/beta.ts" plus
// bodies. Exact arithmetic depends on the markdown chunker (titles may
// be re-derived from headings), so assert a sane lower bound that
// still proves both attributed chunks were summed, plus an upper
// bound that would fail if noise or legacy rows leaked in (they'd
// push >200B easily).
expect(bytes).toBeGreaterThan(60);
expect(bytes).toBeLessThan(200);
});
test("8.8 getContentBytesForSession returns 0 for missing DB or unknown session", () => {
expect(getContentBytesForSession("any-sid", join(tmpdir(), `missing-${randomUUID()}.db`))).toBe(0);
const contentDbPath = join(mkSessionsDir(), `content-${randomUUID()}.db`);
const store = new ContentStore(contentDbPath);
try {
store.indexPlainText("payload", "src/x.ts", 20, { sessionId: "real-sid", eventId: "evt" });
} finally {
store.close();
}
expect(getContentBytesForSession("no-such-session", contentDbPath)).toBe(0);
});
test("8.9 getRealBytesStats with contentDbPath folds chunk bytes into bytesAvoided + totalSavedTokens", () => {
const dir = mkSessionsDir();
const sid = `int-${randomUUID()}`;
const dbPath = dbPathFor(dir, "cafebabecafebabe");
seed(dbPath, sid, [
{ type: "sandbox-execute", category: "sandbox", data: "ctx_execute", bytesReturned: 1_000 },
]);
const contentDbPath = join(dir, `content-${randomUUID()}.db`);
const store = new ContentStore(contentDbPath);
try {
// Big enough payload that the chunk byte sum dwarfs event-data noise
// and proves the value flowed through, not just got rounded in.
store.indexPlainText(
"X".repeat(10_000),
"fixture.txt",
20,
{ sessionId: sid, eventId: "evt-int" },
);
} finally {
store.close();
}
const baseline = getRealBytesStats({ sessionId: sid, sessionsDir: dir });
const withChunks = getRealBytesStats({ sessionId: sid, sessionsDir: dir, contentDbPath });
expect(withChunks.bytesAvoided).toBeGreaterThan(baseline.bytesAvoided + 9_000);
expect(withChunks.totalSavedTokens).toBeGreaterThan(baseline.totalSavedTokens + 2_000);
// bytesReturned untouched — content DB doesn't represent re-served bytes.
expect(withChunks.bytesReturned).toBe(baseline.bytesReturned);
});
});