diff --git a/action.yml b/action.yml index 24693cca..d695d831 100644 --- a/action.yml +++ b/action.yml @@ -94,9 +94,11 @@ inputs: max_tokens_budget: description: >- Total token cap passed to `ocr review --max-tokens-budget`. Base-10 - integer; empty or 0 means unlimited. Once the cap is exceeded, dispatch - stops, skipped files are reported as failed(budget), partial results - are still published, and the review exits 0. + integer; empty or 0 means unlimited. Checked before every LLM round: + a group already over the cap gets one final round to submit findings, + no further groups are dispatched, over-budget and skipped files are + reported as failed(budget), partial results are still published, and + the review exits 0. required: false default: '' stream_progress: diff --git a/cmd/opencodereview/shared_flags.go b/cmd/opencodereview/shared_flags.go index 907f5503..8040943d 100644 --- a/cmd/opencodereview/shared_flags.go +++ b/cmd/opencodereview/shared_flags.go @@ -55,7 +55,7 @@ func addConcurrencyFlags(cmd *cobra.Command, concurrency, timeout, maxTools, max cmd.Flags().IntVar(maxTools, "max-tools", 0, "max tool call rounds per subtask (0 = template default; min 50)") cmd.Flags().IntVar(maxGitProcs, "max-git-procs", 16, "max concurrent git subprocesses") cmd.Flags().IntVar(maxTokens, "max-tokens", 0, "per-group prompt token ceiling (0 = configured or template default)") - cmd.Flags().IntVar(maxTokensBudget, "max-tokens-budget", 0, "cap total token usage (input+output) for this review; dispatch stops once exceeded and skipped files are reported as failed(budget). Partial results are published and review exits 0; it exits non-zero only if every selected item failed (0 = unlimited)") + cmd.Flags().IntVar(maxTokensBudget, "max-tokens-budget", 0, "cap total token usage (input+output) for this review; checked before every LLM round, so a group already over budget gets one final round to submit findings and is reported as failed(budget), and no further groups are dispatched. Partial results are published and review exits 0; it exits non-zero only if every selected item failed (0 = unlimited)") } func addModelFlag(cmd *cobra.Command, target *string) { @@ -233,7 +233,7 @@ func registerScanFlags(cmd *cobra.Command, opts *scanOptions) { cmd.Flags().IntVar(&opts.maxTools, "max-tools", 0, "max tool call rounds per subtask; only takes effect when greater than template default") cmd.Flags().IntVar(&opts.maxGitProcs, "max-git-procs", 16, "max concurrent git subprocesses") cmd.Flags().IntVar(&opts.maxTokens, "max-tokens", 0, "per-file prompt token ceiling (0 = configured or template default)") - cmd.Flags().IntVar(&opts.maxTokensBudget, "max-tokens-budget", 0, "cap total token usage; dispatch stops once exceeded (0 = unlimited)") + cmd.Flags().IntVar(&opts.maxTokensBudget, "max-tokens-budget", 0, "cap total token usage; checked before every LLM round and at dispatch (0 = unlimited)") cmd.Flags().StringVarP(&opts.background, "background", "b", "", "optional requirement/business context for the scan") cmd.Flags().BoolVarP(&opts.preview, "preview", "p", false, "preview which files will be scanned without running the LLM") cmd.Flags().BoolVar(&opts.noPlan, "no-plan", false, "skip the per-file PLAN_TASK pre-pass") diff --git a/examples/github_actions/README.md b/examples/github_actions/README.md index cde52cd3..ec3f96fc 100644 --- a/examples/github_actions/README.md +++ b/examples/github_actions/README.md @@ -214,7 +214,7 @@ The task and request timeouts are independent: | Input | Default | Description | |-------|---------|-------------| | `effort` | `''` | Review effort preset passed to `ocr review --effort`: `low`, `medium`, or `high` (case-insensitive). Higher effort runs more review rounds. Empty keeps the CLI default (the configured value, or medium). | -| `max_tokens_budget` | `''` | Total token cap (input+output) passed to `ocr review --max-tokens-budget`. Empty or `'0'` means unlimited. Once the cap is exceeded, dispatch stops, skipped files are reported as failed(budget), partial results are still published, and the review exits 0. | +| `max_tokens_budget` | `''` | Total token cap (input+output) passed to `ocr review --max-tokens-budget`. Empty or `'0'` means unlimited. Checked before every LLM round: a group already over the cap gets one final round to submit findings, no further groups are dispatched, over-budget and skipped files are reported as failed(budget), partial results are still published, and the review exits 0. | ```yaml - uses: alibaba/open-code-review@main diff --git a/internal/agent/agent.go b/internal/agent/agent.go index 3cd16cf9..66d7a1c2 100644 --- a/internal/agent/agent.go +++ b/internal/agent/agent.go @@ -247,7 +247,8 @@ func New(args Args) *Agent { AllDiffs: a.allDiffs, // Non-nil only here: the same Runner serves scan, whose requests must // stay out of the retry report. See newRequestMeta. - NewRequestMeta: a.newRequestMeta, + NewRequestMeta: a.newRequestMeta, + MaxTokensBudget: args.MaxTokensBudget, }) return a } @@ -1203,19 +1204,20 @@ func classifyItemError(err error) (session.FailureClass, string) { } // classifyMainLoopStop maps a non-error, non-completed main-loop stop to an item -// failure class and a safe reason. Only the configured max-tool-request budget is -// a declared budget stop, so only it may use the budget classification; every -// other stop keeps the unknown class, because the FailureClass taxonomy has no -// category that fits an empty-round or compression exit. Stating that as "not -// max-rounds" rather than case-by-case is deliberate: a stop added to the enum -// later must default to the honest catch-all class, never inherit "budget". +// failure class and a safe reason. Only the configured limits — the +// max-tool-request rounds and the aggregate token budget — are declared budget +// stops, so only they may use the budget classification; every other stop keeps +// the unknown class, because the FailureClass taxonomy has no category that fits +// an empty-round or compression exit. Stating that as "not a declared limit" +// rather than case-by-case is deliberate: a stop added to the enum later must +// default to the honest catch-all class, never inherit "budget". // // The reason text comes from stop.Reason(), shared with the scan path so the // same stop cannot read differently in the two commands' output. In --format // json runs the progress lines that would say why an item stopped are discarded, // so that string is the only stop diagnostic that leaves a CI runner. func classifyMainLoopStop(stop llmloop.MainLoopStop) (session.FailureClass, string) { - if stop == llmloop.StopMaxRounds { + if stop == llmloop.StopMaxRounds || stop == llmloop.StopTokenBudget { return session.FailureBudget, stop.Reason() } return session.FailureUnknown, stop.Reason() @@ -1446,8 +1448,14 @@ func (a *Agent) executeGroupSubtask(ctx context.Context, g FileGroup) (bool, *su return false, nil, ctx.Err() } - if round > 1 && a.args.MaxTokensBudget > 0 && a.budgetExceeded.Load() { + if round > 1 && a.args.MaxTokensBudget > 0 && (a.budgetExceeded.Load() || a.runner.TotalTokensUsed() > a.args.MaxTokensBudget) { fmt.Fprintf(stdout.Writer(), "[ocr] Aggregate budget exceeded, skipping round %d for group %q\n", round, groupKey) + // A group can finish a round over budget with no other gate noticing, + // so record it here too or the run would report the budget as intact. + if a.budgetExceeded.CompareAndSwap(false, true) { + a.recordWarning("token_budget_reached", g.Diffs[0].NewPath, + fmt.Sprintf("skipped round %d of group %q: used %d tokens exceeds budget %d", round, groupKey, a.runner.TotalTokensUsed(), a.args.MaxTokensBudget)) + } break } @@ -1508,6 +1516,15 @@ func (a *Agent) executeGroupSubtask(ctx context.Context, g FileGroup) (bool, *su confirmed = append(confirmed, newlyConfirmed...) if !mainCompleted { + if mainStop == llmloop.StopTokenBudget { + // The runner stopped this conversation on the aggregate budget. Surface + // it the same way the dispatch gate does, so BudgetExceeded() and the + // warning list agree with the item's failed(budget) classification. + if a.budgetExceeded.CompareAndSwap(false, true) { + a.recordWarning("token_budget_reached", g.Diffs[0].NewPath, + fmt.Sprintf("stopped group %q mid-review: used %d tokens exceeds budget %d", groupKey, a.runner.TotalTokensUsed(), a.args.MaxTokensBudget)) + } + } class, reason := classifyMainLoopStop(mainStop) lastStop = &subtaskStop{ class: class, diff --git a/internal/agent/budget_test.go b/internal/agent/budget_test.go index 3c16ccb1..4fde6051 100644 --- a/internal/agent/budget_test.go +++ b/internal/agent/budget_test.go @@ -274,3 +274,200 @@ func TestDispatchSubtasks_UnlimitedBudget(t *testing.T) { t.Error("unlimited budget must not set BudgetExceeded") } } + +// fakeFirstDoneThenNeverClient completes the first conversation it sees with +// task_done at negligible cost, then never calls task_done again: every later +// response is an empty assistant turn that still reports perCallTokens of +// usage. It drives the in-conversation budget stop against a run that also has +// one healthy, completed group, so the result is a partial run rather than an +// all-failed one. +type fakeFirstDoneThenNeverClient struct { + perCallTokens int64 + calls int64 // atomic +} + +func (f *fakeFirstDoneThenNeverClient) CompletionsWithCtx(_ context.Context, _ llm.ChatRequest) (*llm.ChatResponse, error) { + n := atomic.AddInt64(&f.calls, 1) + if n == 1 { + return &llm.ChatResponse{ + Choices: []llm.Choice{{ + Message: llm.ResponseMessage{Role: "assistant", ToolCalls: []llm.ToolCall{{ + ID: "1", Type: "function", Function: llm.FunctionCall{Name: "task_done", Arguments: "{}"}, + }}}, + FinishReason: "tool_calls", + }}, + Model: "fake", + Usage: &llm.UsageInfo{PromptTokens: 10, TotalTokens: 10}, + }, nil + } + content := "" + return &llm.ChatResponse{ + Choices: []llm.Choice{{Message: llm.ResponseMessage{Role: "assistant", Content: &content}}}, + Model: "fake", + Usage: &llm.UsageInfo{PromptTokens: f.perCallTokens, TotalTokens: f.perCallTokens}, + }, nil +} + +// TestDispatchSubtasks_TokenBudgetStopsInFlightGroup pins the gap the dispatch +// look-ahead cannot close: a group admitted while usage is still low can outgrow +// the budget on its own, and only an in-conversation check can stop it. The +// stopped group must read as failed(budget) while the completed one stays +// completed, BudgetExceeded() must be true, the warning must be recorded once, +// and the run must be a controlled partial stop rather than an error. +func TestDispatchSubtasks_TokenBudgetStopsInFlightGroup(t *testing.T) { + setTestHome(t, t.TempDir()) + diffs := makeBudgetDiffs(2) + // The gate admits both groups (estimate <= budget). The second group's + // second round pushes usage past the budget, so its third round is refused. + budget := estimateDiffFileTokens(diffs[1]) * 4 + perCall := budget/2 + 1 + fake := &fakeFirstDoneThenNeverClient{perCallTokens: perCall} + a := New(Args{ + LLMClient: fake, + Model: "fake", + CommentCollector: tool.NewCommentCollector(), + Tools: tool.NewRegistry(), + MaxConcurrency: 1, // serialize so the call sequence is deterministic + MaxTokensBudget: budget, + Template: budgetAgentTestTemplate(), + MainToolDefs: []llm.ToolDef{ + {Type: "function", Function: llm.FunctionDef{Name: "task_done", Description: "done"}}, + }, + }) + t.Cleanup(func() { _ = a.Session().Finalize() }) + a.diffs = diffs + a.currentDate = "2025-06-26 10:00" + a.args.Tools.Freeze() + + comments, err := a.dispatchSubtasks(context.Background()) + if err != nil { + t.Fatalf("dispatchSubtasks must not return an error for a controlled stop: %v", err) + } + if comments == nil { + t.Error("expected a non-nil (empty) comments slice, got nil") + } + + // One call for the completed group, then two admitted rounds and one grace + // round for the stopped group. Without the in-loop check the template's + // five rounds would all run. + if calls := atomic.LoadInt64(&fake.calls); calls != 4 { + t.Fatalf("LLM calls = %d, want 4 (1 done + 2 rounds + grace)", calls) + } + if !a.BudgetExceeded() { + t.Error("expected BudgetExceeded()==true after an in-flight budget stop") + } + var warnings int + for _, w := range a.Warnings() { + if w.Type == "token_budget_reached" { + warnings++ + } + } + if warnings != 1 { + t.Errorf("token_budget_reached warnings = %d, want 1", warnings) + } + + if err := a.finalizeManifest(); err != nil { + t.Fatalf("finalize manifest: %v", err) + } + manifest := a.RunManifest() + if manifest == nil || manifest.TerminalState != session.StatePartial { + t.Fatalf("manifest terminal = %v, want partial", manifest) + } + if manifest.RunFailure != nil { + t.Fatalf("run_failure = %+v, want nil for a controlled budget stop", manifest.RunFailure) + } + if got := len(manifest.Coverage.Completed); got != 1 { + t.Fatalf("completed coverage = %d, want 1", got) + } + if got := len(manifest.Coverage.Failed); got != 1 { + t.Fatalf("failed coverage = %d, want 1", got) + } + if got := manifest.Coverage.Failed[0].Classification; got != session.FailureBudget { + t.Fatalf("stopped group classification = %q, want budget", got) + } +} + +// fakeCommentAndDoneClient answers the first request with a code_comment and a +// task_done in the same turn, reporting perCallTokens of usage. Its round +// therefore completes with a new finding, which is what lets the round loop +// reach round 2 at all, while leaving the run over budget. +type fakeCommentAndDoneClient struct { + perCallTokens int64 + path string + calls int64 // atomic +} + +func (f *fakeCommentAndDoneClient) CompletionsWithCtx(_ context.Context, _ llm.ChatRequest) (*llm.ChatResponse, error) { + atomic.AddInt64(&f.calls, 1) + return &llm.ChatResponse{ + Choices: []llm.Choice{{ + Message: llm.ResponseMessage{Role: "assistant", ToolCalls: []llm.ToolCall{ + {ID: "1", Type: "function", Function: llm.FunctionCall{ + Name: "code_comment", + Arguments: `{"comments":[{"path":"` + f.path + `","content":"missing a nil check here"}]}`, + }}, + {ID: "2", Type: "function", Function: llm.FunctionCall{Name: "task_done", Arguments: "{}"}}, + }}, + FinishReason: "tool_calls", + }}, + Model: "fake", + Usage: &llm.UsageInfo{PromptTokens: f.perCallTokens, TotalTokens: f.perCallTokens}, + }, nil +} + +// TestDispatchSubtasks_TokenBudgetSkipsNextRoundAfterCompletedRound covers the +// round gate rather than the in-conversation check: a group whose round 1 ends +// in task_done already over budget must not start round 2, must not get a grace +// round (it has nothing unreported), and must stay completed, while the run +// still reports that the budget was exceeded. +func TestDispatchSubtasks_TokenBudgetSkipsNextRoundAfterCompletedRound(t *testing.T) { + setTestHome(t, t.TempDir()) + diffs := makeBudgetDiffs(1) + budget := estimateDiffFileTokens(diffs[0]) * 4 + fake := &fakeCommentAndDoneClient{perCallTokens: budget + 1, path: diffs[0].NewPath} + collector := tool.NewCommentCollector() + reg := tool.NewRegistry() + reg.Register(&tool.CodeCommentProvider{Collector: collector}) + tpl := budgetAgentTestTemplate() + tpl.MaxReviewRounds = 2 + a := New(Args{ + LLMClient: fake, + Model: "fake", + CommentCollector: collector, + Tools: reg, + MaxConcurrency: 1, + MaxTokensBudget: budget, + SkipFilter: true, // the filter is an LLM call of its own + Template: tpl, + MainToolDefs: []llm.ToolDef{ + {Type: "function", Function: llm.FunctionDef{Name: "task_done", Description: "done"}}, + {Type: "function", Function: llm.FunctionDef{Name: "code_comment", Description: "comment"}}, + }, + }) + t.Cleanup(func() { _ = a.Session().Finalize() }) + a.diffs = diffs + a.currentDate = "2025-06-26 10:00" + a.args.Tools.Freeze() + + comments, err := a.dispatchSubtasks(context.Background()) + if err != nil { + t.Fatalf("dispatchSubtasks: %v", err) + } + if calls := atomic.LoadInt64(&fake.calls); calls != 1 { + t.Fatalf("LLM calls = %d, want 1: round 2 and any grace round must be skipped", calls) + } + if len(comments) != 1 { + t.Errorf("comments = %d, want the round 1 finding", len(comments)) + } + if !a.BudgetExceeded() { + t.Error("expected BudgetExceeded()==true when the round gate skips a round") + } + + if err := a.finalizeManifest(); err != nil { + t.Fatalf("finalize manifest: %v", err) + } + manifest := a.RunManifest() + if manifest == nil || len(manifest.Coverage.Completed) != 1 || len(manifest.Coverage.Failed) != 0 { + t.Fatalf("coverage = %+v, want the group completed", manifest) + } +} diff --git a/internal/llmloop/loop.go b/internal/llmloop/loop.go index b96b871f..88d056b7 100644 --- a/internal/llmloop/loop.go +++ b/internal/llmloop/loop.go @@ -63,6 +63,15 @@ type Deps struct { // requestNo must be the RequestNo of the session.TaskRecord already created // for this request, so the report joins against the session JSONL. NewRequestMeta func(filePath string, taskType session.TaskType, requestNo int) llm.RequestMeta + + // MaxTokensBudget, when > 0, is the run's aggregate token budget + // (input+output across every LLM call on this Runner). RunMainTask checks + // it before each round and stops with StopTokenBudget once the running + // total is over it, so a single long conversation cannot outrun a budget + // that the caller's dispatch gate only evaluates between subtasks. 0 = + // unlimited, which is also what scan and every existing caller get by + // default. + MaxTokensBudget int64 } // requestCtx returns ctx carrying the identity of one logical LLM request, or @@ -287,6 +296,10 @@ const ( // StopCompression — context compression exceeded its threshold, so the loop // could not continue. Token/context driven but not a declared budget. StopCompression + // StopTokenBudget — the run's aggregate token budget (Deps.MaxTokensBudget) + // was already exceeded when the next round was about to start. This is a + // declared budget limit, like StopMaxRounds. + StopTokenBudget ) // String names the stop for diagnostics — telemetry attributes, log lines and @@ -302,6 +315,8 @@ func (s MainLoopStop) String() string { return "empty_rounds" case StopCompression: return "compression" + case StopTokenBudget: + return "token_budget" default: return fmt.Sprintf("MainLoopStop(%d)", int(s)) } @@ -331,6 +346,8 @@ func (s MainLoopStop) Reason() string { return "stopped after repeated rounds without a usable tool result" case StopCompression: return "stopped because context compression exceeded its threshold" + case StopTokenBudget: + return "reached the aggregate token budget before finishing" default: return fmt.Sprintf("main task stopped for an unrecognized reason (stop=%d)", int(s)) } @@ -373,8 +390,8 @@ func (r *Runner) RunMainTask(ctx context.Context, messages []llm.Message, taskKe defer r.cancelPendingCompression(st) // stop defaults to StopMaxRounds: if the for-loop exits because toolReqCount - // reached zero, the run stopped on the round budget. The empty-round and - // compression breaks overwrite it at their trigger points. + // reached zero, the run stopped on the round budget. The empty-round, + // compression and token-budget breaks overwrite it at their trigger points. stop := StopMaxRounds for toolReqCount > 0 { select { @@ -383,6 +400,15 @@ func (r *Runner) RunMainTask(ctx context.Context, messages []llm.Message, taskKe default: } + // The aggregate budget is checked here, before the request that would + // spend past it, rather than only at subtask dispatch: a conversation + // grows with every round and re-sends its whole history, so the one + // long group is exactly the spender a between-subtasks gate never sees. + if r.tokenBudgetExceeded() { + stop = StopTokenBudget + break + } + toolReqCount-- fs := r.deps.Session.GetOrCreateFileSession(taskKey) @@ -494,13 +520,25 @@ func (r *Runner) RunMainTask(ctx context.Context, messages []llm.Message, taskKe } } - if stop == StopMaxRounds { + switch stop { + case StopMaxRounds: fmt.Fprintf(stdout.Writer(), "[ocr] Max tool requests reached for %s.\n", taskKey) r.runGraceRound(ctx, messages, taskKey, sessionID) + case StopTokenBudget: + fmt.Fprintf(stdout.Writer(), "[ocr] Token budget exceeded (used %d > budget %d) for %s.\n", + r.TotalTokensUsed(), r.deps.MaxTokensBudget, taskKey) + r.runGraceRound(ctx, messages, taskKey, sessionID) } return false, stop, nil } +// tokenBudgetExceeded reports whether the run's aggregate token usage is past +// Deps.MaxTokensBudget. A zero budget never trips. +func (r *Runner) tokenBudgetExceeded() bool { + budget := r.deps.MaxTokensBudget + return budget > 0 && r.TotalTokensUsed() > budget +} + // runGraceRound performs one final LLM call after the tool-request budget is // exhausted, giving the model a chance to submit any findings it identified // but did not yet report via code_comment. @@ -511,7 +549,7 @@ func (r *Runner) runGraceRound(ctx context.Context, messages []llm.Message, task } messages = append(messages, llm.NewTextMessage("user", - "Your tool-call budget is exhausted. This is your FINAL round. You may ONLY:\n"+ + "Your review budget is exhausted. This is your FINAL round. You may ONLY:\n"+ "- Call code_comment to submit any findings you have identified but not yet reported.\n"+ "- Call task_done if you have nothing more to report.\n"+ "No other tools are available. Do not attempt further analysis.")) diff --git a/internal/llmloop/loop_test.go b/internal/llmloop/loop_test.go index 7999b2ab..a75410f7 100644 --- a/internal/llmloop/loop_test.go +++ b/internal/llmloop/loop_test.go @@ -853,12 +853,12 @@ func TestMainLoopStopStringAndReason(t *testing.T) { // stop its own String() and Reason() case instead of letting it fall through to // a message that says nothing. func TestMainLoopStopUnknownValue(t *testing.T) { - unknown := StopCompression + 1 + unknown := StopTokenBudget + 1 - if got, want := unknown.String(), "MainLoopStop(4)"; got != want { + if got, want := unknown.String(), "MainLoopStop(5)"; got != want { t.Errorf("String() = %q, want %q; a new constant needs its own case in String() and Reason()", got, want) } - if got, want := unknown.Reason(), "main task stopped for an unrecognized reason (stop=4)"; got != want { + if got, want := unknown.Reason(), "main task stopped for an unrecognized reason (stop=5)"; got != want { t.Errorf("Reason() = %q, want %q; a new constant needs its own case in String() and Reason()", got, want) } if unknown.Reason() == StopNone.Reason() { @@ -972,3 +972,77 @@ func TestExecuteToolCall_CodeCommentWellFormedArgsDoesNotWarn(t *testing.T) { t.Errorf("warnings = %+v, want none", w) } } + +// TestRunMainTask_TokenBudgetStopsBeforeNextRound pins the in-conversation +// budget check: once the Runner's aggregate usage is over Deps.MaxTokensBudget +// the next round is not sent, the stop is classified StopTokenBudget, and the +// model still gets its one grace round to submit findings. +func TestRunMainTask_TokenBudgetStopsBeforeNextRound(t *testing.T) { + // Every round reads a file and reports 600 tokens; budget 1000 admits two + // rounds (0 and 600 are both within budget) and refuses the third (1200). + client := &fakeClient{responses: []*llm.ChatResponse{ + withUsage(fileReadToolCallResponse("call_1", `{"path":"main.go"}`), 600), + withUsage(fileReadToolCallResponse("call_2", `{"path":"main.go"}`), 600), + withUsage(fileReadToolCallResponse("call_3", `{"path":"main.go"}`), 600), + withUsage(fileReadToolCallResponse("call_4", `{"path":"main.go"}`), 600), + }} + deps := newTestDeps(client) + deps.MaxTokensBudget = 1000 + deps.MainToolDefs = []llm.ToolDef{ + {Type: "function", Function: llm.FunctionDef{Name: "file_read", Description: "read"}}, + {Type: "function", Function: llm.FunctionDef{Name: "task_done", Description: "done"}}, + } + runner := NewRunner(deps) + + msgs := []llm.Message{llm.NewTextMessage("user", "review")} + completed, stop, err := runner.RunMainTask(context.Background(), msgs, "main.go") + if err != nil { + t.Fatalf("RunMainTask: %v", err) + } + if completed { + t.Fatal("RunMainTask completed without task_done") + } + if stop != StopTokenBudget { + t.Fatalf("expected StopTokenBudget, got %v", stop) + } + // Two review rounds plus exactly one grace round; the budget must not be + // spent on a third review round and the grace round must not be skipped. + if got := len(client.requests); got != 3 { + t.Fatalf("expected 3 LLM requests (2 rounds + grace), got %d", got) + } + last := client.requests[2] + for _, def := range last.Tools { + if def.Function.Name == "file_read" { + t.Fatalf("grace round must not offer file_read, got tools %+v", last.Tools) + } + } + if runner.TotalTokensUsed() <= deps.MaxTokensBudget { + t.Fatalf("usage %d should exceed the budget %d after the stop", runner.TotalTokensUsed(), deps.MaxTokensBudget) + } +} + +// TestRunMainTask_ZeroTokenBudgetNeverStops guards the default: callers that +// never set a budget keep the pre-existing round-only behaviour. +func TestRunMainTask_ZeroTokenBudgetNeverStops(t *testing.T) { + client := &fakeClient{responses: []*llm.ChatResponse{ + withUsage(fileReadToolCallResponse("call_1", `{"path":"main.go"}`), 5000), + withUsage(fileReadToolCallResponse("call_2", `{"path":"main.go"}`), 5000), + taskDoneResponse(), + }} + deps := newTestDeps(client) + runner := NewRunner(deps) + + msgs := []llm.Message{llm.NewTextMessage("user", "review")} + completed, stop, err := runner.RunMainTask(context.Background(), msgs, "main.go") + if err != nil { + t.Fatalf("RunMainTask: %v", err) + } + if !completed || stop != StopNone { + t.Fatalf("expected completion with StopNone, got completed=%v stop=%v", completed, stop) + } +} + +func withUsage(resp *llm.ChatResponse, prompt int64) *llm.ChatResponse { + resp.Usage = &llm.UsageInfo{PromptTokens: prompt, CompletionTokens: 0, TotalTokens: prompt} + return resp +} diff --git a/internal/scan/agent.go b/internal/scan/agent.go index adb8a28d..be4a764c 100644 --- a/internal/scan/agent.go +++ b/internal/scan/agent.go @@ -146,7 +146,8 @@ func NewAgent(args Args) *Agent { // DiffLookup returns a synthetic Diff so the code_comment tool's // line-number resolver (resolveFromFileContent) can match against // the full file content of the scanned file. - DiffLookup: a.lookupDiff, + DiffLookup: a.lookupDiff, + MaxTokensBudget: args.MaxTokensBudget, // NewRequestMeta is deliberately left nil. The retry report describes // ocr review; scan shares this Runner, and a nil factory is what keeps // scan's requests out of the report. See llmloop.Deps.NewRequestMeta. diff --git a/pages/src/content/docs/en/cli-reference.md b/pages/src/content/docs/en/cli-reference.md index 02149bb1..a329ed4b 100644 --- a/pages/src/content/docs/en/cli-reference.md +++ b/pages/src/content/docs/en/cli-reference.md @@ -126,7 +126,7 @@ staged + unstaged + untracked changes in the current directory's repo. | `--rule ` | — | — | Path to a custom JSON review rule file. Overrides the project-level and global `rule.json`. | | `--max-tools ` | — | template default | Max tool-call rounds per subtask. `0` uses the template default (`100`); values 1–49 are clamped up to `50`. The flag only ever *raises* the cap — a value below the template default is ignored. | | `--max-tokens ` | — | config or template default | Prompt (input) token ceiling per subtask; the template default is `200000`. Overrides the saved `max_tokens` setting for this run. Does not change the output cap — see `MAX_COMPLETION_TOKENS`. | -| `--max-tokens-budget ` | — | `0` (unlimited) | Cap total input + output token usage for the review. Dispatch stops once the budget is exceeded and partial results are still published. | +| `--max-tokens-budget ` | — | `0` (unlimited) | Cap total input + output token usage for the review. Checked before every LLM round: a subtask already over budget gets one final round to submit findings and is reported as `failed(budget)`, no further subtasks are dispatched, and partial results are still published. | | `--provider ` | — | — | Select a configured provider for this run. Names under both `providers` and `custom_providers` are accepted. | | `--model ` | — | — | Override the resolved LLM model for this run (e.g., `claude-opus-4-6`). | | `--max-git-procs ` | — | `16` | Maximum number of concurrent git subprocesses. | diff --git a/pages/src/content/docs/en/integrations/ci.md b/pages/src/content/docs/en/integrations/ci.md index b2967b21..4021338d 100644 --- a/pages/src/content/docs/en/integrations/ci.md +++ b/pages/src/content/docs/en/integrations/ci.md @@ -117,7 +117,7 @@ step: | Input | Default | Description | |---|---|---| | `effort` | `''` | Review effort preset passed to `ocr review --effort`: `low`, `medium`, or `high` (case-insensitive). Empty keeps the CLI default (the configured value, or medium). Requires OCR v1.10.0 or newer; the action fails early with a clear error on older versions. | -| `max_tokens_budget` | `''` | Total token cap (input + output) passed to `ocr review --max-tokens-budget`. Empty or `'0'` means unlimited. Once the cap is exceeded, dispatch stops, skipped files are reported as `failed(budget)`, partial results are still published, and the review exits 0. | +| `max_tokens_budget` | `''` | Total token cap (input + output) passed to `ocr review --max-tokens-budget`. Empty or `'0'` means unlimited. Checked before every LLM round: a subtask already over the cap gets one final round to submit findings, no further subtasks are dispatched, over-budget and skipped files are reported as `failed(budget)`, partial results are still published, and the review exits 0. | | `llm_reasoning_effort` | `''` | Reasoning depth for models with a steerable `reasoning_effort` request field (e.g. GLM-5.x, OpenAI reasoning models): `minimal`, `low`, `medium`, `high`, `max` (case-insensitive). Merged into the request body through `llm_extra_body`, so it works with every published CLI version; an explicit `reasoning_effort` key in `llm_extra_body` wins over this input. Empty (default) sends nothing. OpenAI-compatible protocols only — the Anthropic API rejects unknown body fields, so the action fails fast on that protocol; steer Anthropic thinking through an explicit `llm_extra_body` key instead. | | `stream_progress` | `'false'` | `'true'` streams live `[ocr]` progress lines to the workflow log (human audience on stderr) instead of staying silent until the run finishes. Display-only: stderr is still captured to a file for artifacts and comment posting. | diff --git a/pages/src/content/docs/ja/cli-reference.md b/pages/src/content/docs/ja/cli-reference.md index 87473a85..306415b6 100644 --- a/pages/src/content/docs/ja/cli-reference.md +++ b/pages/src/content/docs/ja/cli-reference.md @@ -119,7 +119,7 @@ ocr r [flags] (alias) | `--rule ` | — | — | カスタム JSON レビュールールファイルのパス。プロジェクトレベルおよびグローバルの `rule.json` を上書きします。 | | `--max-tools ` | — | テンプレートのデフォルト | サブタスクごとの最大ツール呼び出し回数。`0` はテンプレートのデフォルト(`100`)を使用します。1〜49 は `50` に引き上げられます。解決後の値はテンプレートのデフォルトを**上回る場合にのみ**適用されます(引き上げのみ可能で、引き下げはできません)。 | | `--max-tokens ` | — | 設定またはテンプレートのデフォルト | サブタスクごとの**プロンプト**トークン上限(review のデフォルトは `200000`)。この実行で保存済みの `max_tokens` 設定を上書きします。出力の上限には影響しません。そちらは `MAX_COMPLETION_TOKENS`(`16384`)が個別に制御します。 | -| `--max-tokens-budget ` | — | `0`(無制限) | レビュー全体の入力 + 出力トークン使用量を制限します。予算を超えると処理の割り当てを停止し、部分的な結果は引き続き公開されます。 | +| `--max-tokens-budget ` | — | `0`(無制限) | レビュー全体の入力 + 出力トークン使用量を制限します。LLM の各ラウンドの前に確認されます: すでに予算を超えたサブタスクには発見を提出するための最終ラウンドが 1 回与えられ、`failed(budget)` として報告されます。以降のサブタスクは割り当てられず、部分的な結果は引き続き公開されます。 | | `--effort ` | — | 設定または `medium` | レビューの労力プリセット: `low` = main ループ 1 ラウンド、`medium` = 2 ラウンド(デフォルト)、`high` = 3 ラウンド。ラウンドが多いほど recall は上がりますが、時間とトークンも増えます。`ocr config set effort ` で永続化できます。 | | `--provider ` | — | — | 今回の実行で設定済み provider を選択します。`providers` と `custom_providers` の両方の名前を使用できます。 | | `--model ` | — | — | 今回の実行で解決済みの LLM model を上書きします(例: `claude-opus-4-6`)。 | diff --git a/pages/src/content/docs/ja/integrations/ci.md b/pages/src/content/docs/ja/integrations/ci.md index 3c016215..c5b0bfd4 100644 --- a/pages/src/content/docs/ja/integrations/ci.md +++ b/pages/src/content/docs/ja/integrations/ci.md @@ -79,7 +79,7 @@ curl -o .github/workflows/ocr-review.yml \ | 入力 | デフォルト | 説明 | |---|---|---| | `effort` | `''` | `ocr review --effort` に渡すレビュー強度プリセット:`low`、`medium`、`high`(大文字小文字を区別しません)。空の場合は CLI のデフォルト(設定済みの値、なければ medium)を使います。OCR v1.10.0 以降が必要で、それより古いバージョンではアクションが明確なエラーで早期に失敗します。 | -| `max_tokens_budget` | `''` | `ocr review --max-tokens-budget` に渡すトークン総量(入力 + 出力)の上限。空または `'0'` は無制限です。上限を超えるとディスパッチが停止し、スキップされたファイルは `failed(budget)` として報告され、部分的な結果は引き続き公開され、レビューは 0 で終了します。 | +| `max_tokens_budget` | `''` | `ocr review --max-tokens-budget` に渡すトークン総量(入力 + 出力)の上限。空または `'0'` は無制限です。LLM の各ラウンドの前に確認されます: すでに上限を超えたサブタスクには発見を提出するための最終ラウンドが 1 回与えられ、以降のサブタスクはディスパッチされず、予算超過およびスキップされたファイルは `failed(budget)` として報告され、部分的な結果は引き続き公開され、レビューは 0 で終了します。 | | `llm_reasoning_effort` | `''` | `reasoning_effort` リクエストフィールドを調整できるモデル(GLM-5.x、OpenAI reasoning モデルなど)の推論深度:`minimal`、`low`、`medium`、`high`、`max`(大文字小文字を区別しません)。`llm_extra_body` 経由でリクエストボディにマージされるため、公開済みのすべての CLI バージョンで動作します。`llm_extra_body` 内の明示的な `reasoning_effort` キーがこの入力より優先されます。空(デフォルト)の場合は送信しません。OpenAI 互換プロトコル専用です——Anthropic API は未知のボディフィールドを拒否するため、そのプロトコルではアクションが即座に失敗します。Anthropic の thinking 制御には `llm_extra_body` の明示的なキーを使ってください。 | | `stream_progress` | `'false'` | `'true'` にすると、レビューが終了するまで沈黙する代わりに、`[ocr]` の進捗行をワークフローログへライブで流します(stderr の human audience)。表示のみの切り替えで、stderr は引き続きファイルにキャプチャされ、アーティファクトとコメント投稿に使われます。 | diff --git a/pages/src/content/docs/ko/cli-reference.md b/pages/src/content/docs/ko/cli-reference.md index e5fa6586..bbd520f8 100644 --- a/pages/src/content/docs/ko/cli-reference.md +++ b/pages/src/content/docs/ko/cli-reference.md @@ -125,7 +125,7 @@ ocr r [flags] (alias) | `--rule ` | — | — | 커스텀 JSON 리뷰 규칙 파일 경로. 프로젝트 수준과 전역 `rule.json`을 덮어씁니다. | | `--max-tools ` | — | 템플릿 기본값 | 서브태스크당 최대 도구 호출 라운드 수. `0`이면 템플릿 기본값(`100`)을 쓰고, 1~49는 `50`으로 올려 맞춥니다. 이 플래그는 상한을 *올리기만* 합니다. 템플릿 기본값보다 낮은 값은 무시됩니다. | | `--max-tokens ` | — | 설정 또는 템플릿 기본값 | 서브태스크당 프롬프트(입력) 토큰 상한이며 템플릿 기본값은 `200000`입니다. 이 실행에 한해 저장된 `max_tokens` 설정을 덮어씁니다. 출력 상한은 바뀌지 않습니다. `MAX_COMPLETION_TOKENS`를 참고하세요. | -| `--max-tokens-budget ` | — | `0`(무제한) | 리뷰 전체의 입력+출력 토큰 사용량을 제한합니다. 예산을 넘기면 작업 전달을 멈추지만 그때까지의 결과는 그대로 내보냅니다. | +| `--max-tokens-budget ` | — | `0`(무제한) | 리뷰 전체의 입력+출력 토큰 사용량을 제한합니다. LLM 라운드마다 먼저 확인하며, 이미 예산을 넘긴 하위 작업은 발견 사항을 제출할 마지막 라운드를 한 번 받고 `failed(budget)`로 보고됩니다. 이후 하위 작업은 전달되지 않지만 그때까지의 결과는 그대로 내보냅니다. | | `--provider ` | — | — | 이 실행에 쓸 프로바이더를 고릅니다. `providers`와 `custom_providers` 양쪽의 이름을 모두 받습니다. | | `--model ` | — | — | 이 실행에 한해 해석된 LLM 모델을 덮어씁니다(예: `claude-opus-4-6`). | | `--max-git-procs ` | — | `16` | 동시에 띄울 git 서브프로세스의 최대 개수. | diff --git a/pages/src/content/docs/ko/integrations/ci.md b/pages/src/content/docs/ko/integrations/ci.md index 685de613..4c24949c 100644 --- a/pages/src/content/docs/ko/integrations/ci.md +++ b/pages/src/content/docs/ko/integrations/ci.md @@ -107,7 +107,7 @@ curl -o .github/workflows/ocr-review.yml \ | 입력 | 기본값 | 설명 | |---|---|---| | `effort` | `''` | `ocr review --effort`로 전달되는 리뷰 강도 프리셋: `low`, `medium`, `high`(대소문자 무시). 비워 두면 CLI 기본값(설정된 값, 없으면 medium)을 따릅니다. OCR v1.10.0 이상이 필요하며, 더 낮은 버전에서는 액션이 명확한 오류와 함께 일찍 실패합니다. | -| `max_tokens_budget` | `''` | `ocr review --max-tokens-budget`으로 전달되는 총 토큰(입력 + 출력) 상한. 비어 있거나 `'0'`이면 무제한입니다. 상한을 넘으면 디스패치가 멈추고, 건너뛴 파일은 `failed(budget)`로 보고되며, 부분 결과는 그대로 게시되고, 리뷰는 0으로 종료합니다. | +| `max_tokens_budget` | `''` | `ocr review --max-tokens-budget`으로 전달되는 총 토큰(입력 + 출력) 상한. 비어 있거나 `'0'`이면 무제한입니다. LLM 라운드마다 먼저 확인하며, 이미 상한을 넘긴 하위 작업은 발견 사항을 제출할 마지막 라운드를 한 번 받고, 이후 하위 작업은 디스패치되지 않으며, 예산을 넘기거나 건너뛴 파일은 `failed(budget)`로 보고되고, 부분 결과는 그대로 게시되며, 리뷰는 0으로 종료합니다. | | `llm_reasoning_effort` | `''` | `reasoning_effort` 요청 필드를 조절할 수 있는 모델(예: GLM-5.x, OpenAI reasoning 모델)의 추론 깊이: `minimal`, `low`, `medium`, `high`, `max`(대소문자 무시). `llm_extra_body`를 통해 요청 본문에 병합되므로 이미 배포된 모든 CLI 버전에서 동작합니다. `llm_extra_body` 안의 명시적 `reasoning_effort` 키가 이 입력보다 우선합니다. 비어 있으면(기본값) 아무것도 보내지 않습니다. OpenAI 호환 프로토콜 전용입니다 — Anthropic API는 알 수 없는 본문 필드를 거부하므로 해당 프로토콜에서는 액션이 즉시 실패합니다. Anthropic의 thinking 제어는 `llm_extra_body`의 명시적 키를 사용하세요. | | `stream_progress` | `'false'` | `'true'`로 설정하면 실행이 끝날 때까지 조용히 기다리는 대신 `[ocr]` 진행 라인을 워크플로 로그에 실시간으로 흘려보냅니다(stderr의 human audience). 표시 전용 토글이며 stderr는 여전히 파일에 캡처되어 아티팩트와 코멘트 게시에 사용됩니다. | diff --git a/pages/src/content/docs/ru/cli-reference.md b/pages/src/content/docs/ru/cli-reference.md index bc139cdb..2e1aaae0 100644 --- a/pages/src/content/docs/ru/cli-reference.md +++ b/pages/src/content/docs/ru/cli-reference.md @@ -125,7 +125,7 @@ ocr r [flags] (alias) | `--rule ` | — | — | Путь к пользовательскому JSON-файлу правил ревью. Переопределяет проектный и глобальный `rule.json`. | | `--max-tools ` | — | значение шаблона | Максимальное число раундов вызова инструментов для каждой подзадачи. `0` использует значение шаблона (`100`); значения 1–49 повышаются до `50`; итоговое значение применяется только если оно **больше** значения шаблона (предел можно лишь повысить, но не понизить). | | `--max-tokens ` | — | значение конфигурации или шаблона | Предел токенов **промпта** для каждой подзадачи (по умолчанию `200000` для review). Переопределяет сохранённое значение `max_tokens` для этого запуска. На предел вывода не влияет — он задаётся отдельно параметром `MAX_COMPLETION_TOKENS` (`16384`). | -| `--max-tokens-budget ` | — | `0` (без ограничения) | Ограничивает общее число входных + выходных токенов ревью. После превышения бюджета новые задачи не запускаются, а частичные результаты всё равно публикуются. | +| `--max-tokens-budget ` | — | `0` (без ограничения) | Ограничивает общее число входных + выходных токенов ревью. Проверяется перед каждым раундом LLM: подзадача, уже превысившая бюджет, получает один финальный раунд, чтобы отправить найденное, и отмечается как `failed(budget)`; новые подзадачи не запускаются, а частичные результаты всё равно публикуются. | | `--effort ` | — | значение конфигурации или `medium` | Предустановка усилий ревью: `low` = 1 раунд основного цикла, `medium` = 2 раунда (по умолчанию), `high` = 3 раунда. Больше раундов — выше полнота находок, но больше времени и токенов. Сохранить можно командой `ocr config set effort `. | | `--provider ` | — | — | Выбирает настроенного провайдера для этого запуска. Поддерживаются имена из `providers` и `custom_providers`. | | `--model ` | — | — | Переопределяет выбранную LLM-модель для этого ревью (например, `claude-opus-4-6`). | diff --git a/pages/src/content/docs/ru/integrations/ci.md b/pages/src/content/docs/ru/integrations/ci.md index afcb73ac..18b9030a 100644 --- a/pages/src/content/docs/ru/integrations/ci.md +++ b/pages/src/content/docs/ru/integrations/ci.md @@ -115,7 +115,7 @@ action: | Параметр | По умолчанию | Описание | |---|---|---| | `effort` | `''` | Пресет интенсивности ревью, передаваемый в `ocr review --effort`: `low`, `medium` или `high` (без учёта регистра). Пустое значение сохраняет значение CLI по умолчанию (настроенное значение или medium). Требуется OCR v1.10.0 или новее — на более старых версиях action завершается раньше с понятной ошибкой. | -| `max_tokens_budget` | `''` | Общий лимит токенов (ввод + вывод), передаваемый в `ocr review --max-tokens-budget`. Пустое значение или `'0'` означает «без ограничений». После превышения лимита диспетчеризация останавливается, пропущенные файлы отмечаются как `failed(budget)`, частичные результаты по-прежнему публикуются, а ревью завершается с кодом 0. | +| `max_tokens_budget` | `''` | Общий лимит токенов (ввод + вывод), передаваемый в `ocr review --max-tokens-budget`. Пустое значение или `'0'` означает «без ограничений». Проверяется перед каждым раундом LLM: подзадача, уже превысившая лимит, получает один финальный раунд, чтобы отправить найденное, новые подзадачи не запускаются, файлы сверх бюджета и пропущенные файлы отмечаются как `failed(budget)`, частичные результаты по-прежнему публикуются, а ревью завершается с кодом 0. | | `llm_reasoning_effort` | `''` | Глубина рассуждений для моделей с управляемым полем запроса `reasoning_effort` (например, GLM-5.x, reasoning-модели OpenAI): `minimal`, `low`, `medium`, `high`, `max` (без учёта регистра). Значение вливается в тело запроса через `llm_extra_body`, поэтому работает со всеми опубликованными версиями CLI; явный ключ `reasoning_effort` в `llm_extra_body` имеет приоритет над этим параметром. Пустое значение (по умолчанию) ничего не отправляет. Только для OpenAI-совместимых протоколов — Anthropic API отклоняет неизвестные поля тела, поэтому на протоколе Anthropic action завершается сразу; управляйте режимом мышления Anthropic через явный ключ в `llm_extra_body`. | | `stream_progress` | `'false'` | Значение `'true'` транслирует живые строки прогресса `[ocr]` в лог workflow (human audience в stderr) вместо молчания до конца запуска. Влияет только на отображение: stderr по-прежнему записывается в файл для артефактов и публикации комментариев. | diff --git a/pages/src/content/docs/zh/cli-reference.md b/pages/src/content/docs/zh/cli-reference.md index 778be88a..169ab771 100644 --- a/pages/src/content/docs/zh/cli-reference.md +++ b/pages/src/content/docs/zh/cli-reference.md @@ -120,7 +120,7 @@ unstaged + untracked 变更。 | `--rule ` | — | — | 自定义 JSON 评审规则文件路径。覆盖项目级与全局 `rule.json`。 | | `--max-tools ` | — | 模板默认 | 每个子任务的最大工具调用轮数。`0` 用模板默认(`100`);1–49 会被上调到 `50`;解析后的值只在**大于**模板默认值时才生效(即只能上调,不能下调)。 | | `--max-tokens ` | — | 配置或模板默认 | 每个子任务的**提示词** token 上限(review 默认 `200000`)。覆盖本次运行已保存的 `max_tokens` 设置。不影响输出上限——那由 `MAX_COMPLETION_TOKENS`(`16384`)单独控制。 | -| `--max-tokens-budget ` | — | `0`(无限制) | 限制本次评审的输入 + 输出 token 总量。超出预算后停止分发,并仍会发布部分结果。 | +| `--max-tokens-budget ` | — | `0`(无限制) | 限制本次评审的输入 + 输出 token 总量。每次 LLM 轮次前都会检查:已超出预算的子任务会获得最后一轮来提交发现,并记为 `failed(budget)`;不再分发新的子任务,部分结果仍会发布。 | | `--effort ` | — | 配置或 `medium` | 评审投入档位:`low` = 1 轮 main 循环,`medium` = 2 轮(默认),`high` = 3 轮。轮数越多召回越高、耗时与 token 也越多。可用 `ocr config set effort ` 持久化。 | | `--provider ` | — | — | 为本次运行选择已配置的 provider。支持 `providers` 和 `custom_providers` 中的名称。 | | `--model ` | — | — | 为本次运行覆盖已解析出的 LLM model(如 `claude-opus-4-6`)。 | diff --git a/pages/src/content/docs/zh/integrations/ci.md b/pages/src/content/docs/zh/integrations/ci.md index f10a085e..b174d651 100644 --- a/pages/src/content/docs/zh/integrations/ci.md +++ b/pages/src/content/docs/zh/integrations/ci.md @@ -97,7 +97,7 @@ composite action | Input | 默认值 | 说明 | |---|---|---| | `effort` | `''` | 传给 `ocr review --effort` 的评审强度预设:`low`、`medium` 或 `high`(不区分大小写)。留空则沿用 CLI 默认值(已配置的值,否则为 medium)。需要 OCR v1.10.0 或更新版本;在更旧版本上 action 会提前以明确报错失败。 | -| `max_tokens_budget` | `''` | 传给 `ocr review --max-tokens-budget` 的 token 总量上限(输入 + 输出)。留空或 `'0'` 表示不限。超过上限后停止派发,被跳过的文件记为 `failed(budget)`,已产生的部分结果仍会发布,评审以 0 退出。 | +| `max_tokens_budget` | `''` | 传给 `ocr review --max-tokens-budget` 的 token 总量上限(输入 + 输出)。留空或 `'0'` 表示不限。每次 LLM 轮次前都会检查:已超出上限的子任务会获得最后一轮来提交发现,不再派发新的子任务,超出预算和被跳过的文件记为 `failed(budget)`,已产生的部分结果仍会发布,评审以 0 退出。 | | `llm_reasoning_effort` | `''` | 面向支持 `reasoning_effort` 请求字段的模型(如 GLM-5.x、OpenAI reasoning 模型)的推理深度:`minimal`、`low`、`medium`、`high`、`max`(不区分大小写)。经 `llm_extra_body` 合并进请求体,因此所有已发布的 CLI 版本均可使用;`llm_extra_body` 中显式的 `reasoning_effort` 键优先于此 input。留空(默认)则不发送。仅适用于 OpenAI 兼容协议——Anthropic API 会拒绝未知请求体字段,action 在该协议下会快速失败;Anthropic 的 thinking 控制请改用 `llm_extra_body` 中的显式键。 | | `stream_progress` | `'false'` | 设为 `'true'` 时,把 `[ocr]` 实时进度行流入工作流日志(stderr 上的 human audience),而不是在评审结束前保持静默。仅影响展示:stderr 仍会写入文件,供产物上传与评论张贴使用。 | diff --git a/plugins/open-code-review/skills/open-code-review/SKILL.md b/plugins/open-code-review/skills/open-code-review/SKILL.md index c4161efe..66a09c01 100644 --- a/plugins/open-code-review/skills/open-code-review/SKILL.md +++ b/plugins/open-code-review/skills/open-code-review/SKILL.md @@ -194,7 +194,7 @@ Beyond the common flags above, `ocr review` exposes a few groups of controls. Ru **Budget** - `--max-tokens ` — per-group prompt ceiling; defaults to the configured value or the template default (`200000`). -- `--max-tokens-budget ` — cap total input + output tokens for the run. Once exceeded, dispatch stops, partial results are still published, and skipped files are reported as `failed(budget)`. +- `--max-tokens-budget ` — cap total input + output tokens for the run. Checked before every LLM round: a group already over budget gets one final round to submit findings, no further groups are dispatched, partial results are still published, and skipped files are reported as `failed(budget)`. - `--no-filter` — keep all review comments and skip the LLM post-filtering call. ## Gotchas diff --git a/skills/open-code-review/SKILL.md b/skills/open-code-review/SKILL.md index 7425bd3c..01317b74 100644 --- a/skills/open-code-review/SKILL.md +++ b/skills/open-code-review/SKILL.md @@ -189,7 +189,7 @@ Beyond the common flags above, `ocr review` exposes a few groups of controls. Ru **Budget** - `--max-tokens ` — per-group prompt ceiling; defaults to the configured value or the template default (`200000`). -- `--max-tokens-budget ` — cap total input + output tokens for the run. Once exceeded, dispatch stops, partial results are still published, and skipped files are reported as `failed(budget)`. +- `--max-tokens-budget ` — cap total input + output tokens for the run. Checked before every LLM round: a group already over budget gets one final round to submit findings, no further groups are dispatched, partial results are still published, and skipped files are reported as `failed(budget)`. - `--no-filter` — keep all review comments and skip the LLM post-filtering call. ## Gotchas