fix(coding-agent): compact without provider usage

closes #8328
This commit is contained in:
Vegard Stikbakke
2026-08-19 13:44:22 +02:00
parent 3de00332f7
commit 4495469a5e
3 changed files with 95 additions and 11 deletions
+1
View File
@@ -21,6 +21,7 @@
- Fixed hung pi.dev model catalog requests consuming the entire refresh deadline without retrying ([#8198](https://github.com/earendil-works/pi/issues/8198)).
- Fixed inherited Xiaomi model catalogs listing shut-down MiMo V2 models in `/model` and `--list-models` ([#8187](https://github.com/earendil-works/pi/issues/8187)).
- Fixed branch summary entries recording the navigation destination in `fromId` instead of the pre-navigation source leaf.
- Fixed threshold auto-compaction being skipped when providers omit streaming usage data ([#8328](https://github.com/earendil-works/pi/issues/8328)).
## [0.84.2] - 2026-08-14
+14 -11
View File
@@ -2099,17 +2099,20 @@ export class AgentSession {
if (assistantMessage.stopReason === "error" || directContextTokens === 0) {
const messages = this.agent.state.messages;
const estimate = estimateContextTokens(messages);
if (estimate.lastUsageIndex === null) return false; // No usage data at all
// Verify the usage source is post-compaction. Kept pre-compaction messages
// have stale usage reflecting the old (larger) context and would falsely
// trigger compaction right after one just finished.
const usageMsg = messages[estimate.lastUsageIndex];
if (
compactionEntry &&
usageMsg.role === "assistant" &&
(usageMsg as AssistantMessage).timestamp <= new Date(compactionEntry.timestamp).getTime()
) {
return false;
// Without provider usage, estimate.tokens is the pure message-size estimate.
// Only usage-backed estimates need the stale pre-compaction check.
if (estimate.lastUsageIndex !== null) {
// Verify the usage source is post-compaction. Kept pre-compaction messages
// have stale usage reflecting the old (larger) context and would falsely
// trigger compaction right after one just finished.
const usageMsg = messages[estimate.lastUsageIndex];
if (
compactionEntry &&
usageMsg.role === "assistant" &&
(usageMsg as AssistantMessage).timestamp <= new Date(compactionEntry.timestamp).getTime()
) {
return false;
}
}
contextTokens = estimate.tokens;
} else {
@@ -0,0 +1,80 @@
import type { AssistantMessage } from "@earendil-works/pi-ai";
import { afterEach, describe, expect, it, vi } from "vitest";
import { createHarness, type Harness } from "../harness.ts";
type SessionWithCompactionInternals = {
_checkCompaction: (assistantMessage: AssistantMessage) => Promise<boolean>;
_runAutoCompaction: (reason: "overflow" | "threshold", willRetry: boolean) => Promise<boolean>;
};
function createZeroUsageAssistant(harness: Harness): AssistantMessage {
const model = harness.getModel();
return {
role: "assistant",
content: [{ type: "text", text: "response" }],
api: model.api,
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: Date.now(),
};
}
describe("issue #8328 zero-usage auto-compaction", () => {
const harnesses: Harness[] = [];
afterEach(() => {
vi.restoreAllMocks();
while (harnesses.length > 0) {
harnesses.pop()?.cleanup();
}
});
async function createCompactionHarness(): Promise<Harness> {
const harness = await createHarness({
models: [{ id: "faux-1", contextWindow: 100, maxTokens: 20 }],
settings: { compaction: { enabled: true, reserveTokens: 10 } },
});
harnesses.push(harness);
return harness;
}
it("uses the message estimate when no assistant has reported usage", async () => {
const harness = await createCompactionHarness();
const assistant = createZeroUsageAssistant(harness);
harness.session.agent.state.messages = [
{ role: "user", content: [{ type: "text", text: "x".repeat(400) }], timestamp: Date.now() - 1 },
assistant,
];
const sessionInternals = harness.session as unknown as SessionWithCompactionInternals;
const runAutoCompactionSpy = vi.spyOn(sessionInternals, "_runAutoCompaction").mockResolvedValue(false);
await sessionInternals._checkCompaction(assistant);
expect(runAutoCompactionSpy).toHaveBeenCalledOnce();
expect(runAutoCompactionSpy).toHaveBeenCalledWith("threshold", false);
});
it("does not compact when the zero-usage message estimate is below the threshold", async () => {
const harness = await createCompactionHarness();
const assistant = createZeroUsageAssistant(harness);
harness.session.agent.state.messages = [
{ role: "user", content: [{ type: "text", text: "short" }], timestamp: Date.now() - 1 },
assistant,
];
const sessionInternals = harness.session as unknown as SessionWithCompactionInternals;
const runAutoCompactionSpy = vi.spyOn(sessionInternals, "_runAutoCompaction").mockResolvedValue(false);
await sessionInternals._checkCompaction(assistant);
expect(runAutoCompactionSpy).not.toHaveBeenCalled();
});
});