mirror of
https://github.com/ayghri/i-have-adhd.git
synced 2026-10-02 04:14:46 +08:00
Validate cost measurements and report input overhead
This commit is contained in:
+20
-1
@@ -46,7 +46,26 @@ Aggregate token usage, reported cost, and response length from the completed res
|
||||
python3 scripts/run_evals.py measure evals/results/responses.jsonl
|
||||
```
|
||||
|
||||
The command refuses to compare conditions produced by different runners or conditions with unequal `(case_id, trial)` coverage. This is the same comparability rule the release gate applies to judged scores.
|
||||
The summary reports input/output token totals, reported generation cost, stored
|
||||
response length, and candidate-minus-baseline deltas. Positive deltas mean more
|
||||
usage or cost; negative deltas mean less. Read these alongside quality scores.
|
||||
|
||||
Comparisons require the same runner and identical `(case_id, trial)` coverage,
|
||||
without duplicates. If rows include `model`, every row must name the same model.
|
||||
Older runner output does not record model metadata: `model: null` means model
|
||||
comparability is unverified. Check the original model/CLI settings yourself;
|
||||
a matching runner alias alone does not establish the same model or configuration.
|
||||
|
||||
Missing costs or token counts produce `null` totals and deltas, not zero. Invalid
|
||||
negative, boolean, non-finite, or fractional token counts are rejected. A zero
|
||||
baseline has no meaningful percentage change, so that percentage is `null`.
|
||||
Claude input totals include cache creation/read tokens; Codex cached input is
|
||||
already part of its input count and is not added again.
|
||||
|
||||
This measures recorded generation rows, not total provider billing: judge costs
|
||||
and unrecorded failed/retried calls are excluded. Scenario captures currently omit
|
||||
token usage, and their response length includes transcript JSON and user prompts.
|
||||
The command is read-only and makes no model calls.
|
||||
|
||||
## Judge and score
|
||||
|
||||
|
||||
+51
-28
@@ -4,6 +4,7 @@
|
||||
import argparse
|
||||
from contextlib import contextmanager
|
||||
import json
|
||||
import math
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -188,7 +189,24 @@ def summarize_scores(scores: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
|
||||
def summarize_usage(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
for row in rows:
|
||||
for index, row in enumerate(rows, start=1):
|
||||
for field in ("case_id", "runner", "condition"):
|
||||
if not isinstance(row.get(field), str) or not row[field]:
|
||||
raise ValueError(f"Response {index}: {field} must be a non-empty string")
|
||||
if row["condition"] not in CONDITIONS:
|
||||
raise ValueError(f"Response {index}: unsupported condition")
|
||||
if type(row.get("trial")) is not int or row["trial"] < 1:
|
||||
raise ValueError(f"Response {index}: trial must be a positive integer")
|
||||
if not isinstance(row.get("response"), str):
|
||||
raise ValueError(f"Response {index}: response must be a string")
|
||||
cost = row.get("cost_usd")
|
||||
if cost is not None and (
|
||||
type(cost) not in (int, float) or cost < 0 or not math.isfinite(cost)
|
||||
):
|
||||
raise ValueError(f"Response {index}: cost_usd must be finite and non-negative")
|
||||
model = row.get("model")
|
||||
if model is not None and (not isinstance(model, str) or not model.strip()):
|
||||
raise ValueError(f"Response {index}: model must be a non-empty string")
|
||||
grouped[row["condition"]].append(row)
|
||||
if "baseline" not in grouped or "candidate" not in grouped:
|
||||
raise ValueError("Responses must include baseline and candidate conditions")
|
||||
@@ -197,6 +215,9 @@ def summarize_usage(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
if len(runners) > 1:
|
||||
names = ", ".join(sorted(str(runner) for runner in runners))
|
||||
raise ValueError(f"Responses must use the same runner; found: {names}")
|
||||
models = {row.get("model") for row in rows}
|
||||
if len(models) > 1:
|
||||
raise ValueError("Responses must report the same model on every row, or omit it on all rows")
|
||||
_check_pairing(grouped)
|
||||
|
||||
conditions: dict[str, dict[str, Any]] = {}
|
||||
@@ -208,14 +229,14 @@ def summarize_usage(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
output_total = _reported_token_total(output_counts)
|
||||
response_chars = sum(len(row["response"]) for row in condition_rows)
|
||||
reported_costs = [row.get("cost_usd") for row in condition_rows]
|
||||
unreported_costs = sum(
|
||||
not isinstance(cost, (int, float)) for cost in reported_costs
|
||||
)
|
||||
unreported_costs = sum(cost is None for cost in reported_costs)
|
||||
cost_total = (
|
||||
None
|
||||
if unreported_costs
|
||||
else sum(float(cost) for cost in reported_costs)
|
||||
)
|
||||
if cost_total is not None and not math.isfinite(cost_total):
|
||||
raise ValueError(f"{condition}: cost total is not finite")
|
||||
summary = {
|
||||
"rows": len(condition_rows),
|
||||
"input_tokens": input_total,
|
||||
@@ -239,25 +260,18 @@ def summarize_usage(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
for condition, summary in sorted(conditions.items()):
|
||||
if condition == "baseline":
|
||||
continue
|
||||
output_delta = _difference(summary["output_tokens"], baseline["output_tokens"])
|
||||
cost_delta = _difference(summary["cost_usd"], baseline["cost_usd"])
|
||||
response_chars_delta = _difference(
|
||||
summary["mean_response_chars"], baseline["mean_response_chars"]
|
||||
)
|
||||
deltas[condition] = {
|
||||
"output_tokens": output_delta,
|
||||
"output_tokens_pct": _percent_change(
|
||||
output_delta, baseline["output_tokens"]
|
||||
),
|
||||
"cost_usd": cost_delta,
|
||||
"cost_usd_pct": _percent_change(cost_delta, baseline["cost_usd"]),
|
||||
"mean_response_chars": response_chars_delta,
|
||||
"mean_response_chars_pct": _percent_change(
|
||||
response_chars_delta, baseline["mean_response_chars"]
|
||||
),
|
||||
}
|
||||
deltas[condition] = {}
|
||||
for metric in ("input_tokens", "output_tokens", "cost_usd", "mean_response_chars"):
|
||||
delta = _difference(summary[metric], baseline[metric])
|
||||
deltas[condition][metric] = delta
|
||||
deltas[condition][f"{metric}_pct"] = _percent_change(delta, baseline[metric])
|
||||
|
||||
return {"runner": next(iter(runners)), "conditions": conditions, "delta": deltas}
|
||||
return {
|
||||
"runner": next(iter(runners)),
|
||||
"model": next(iter(models)),
|
||||
"conditions": conditions,
|
||||
"delta": deltas,
|
||||
}
|
||||
|
||||
|
||||
def _reported_token_total(values: list[int | None]) -> int | None:
|
||||
@@ -345,12 +359,21 @@ _INPUT_TOKEN_KEYS = (
|
||||
)
|
||||
|
||||
|
||||
def _usage_tokens(usage: dict[str, Any]) -> tuple[int | None, int | None]:
|
||||
def _usage_tokens(usage: Optional[dict[str, Any]]) -> tuple[int | None, int | None]:
|
||||
"""Return (input, output) token counts, or None where they were not reported."""
|
||||
input_values = [usage[key] for key in _INPUT_TOKEN_KEYS if key in usage]
|
||||
input_tokens = sum(input_values) if input_values else None
|
||||
output_tokens = usage["output_tokens"] if "output_tokens" in usage else None
|
||||
return input_tokens, output_tokens
|
||||
if usage is None:
|
||||
return None, None
|
||||
if not isinstance(usage, dict):
|
||||
raise ValueError("usage must be an object or null")
|
||||
for key in (*_INPUT_TOKEN_KEYS, "cached_input_tokens", "output_tokens"):
|
||||
value = usage.get(key)
|
||||
if value is not None and (type(value) is not int or value < 0):
|
||||
raise ValueError(f"{key} must be a non-negative integer or null")
|
||||
# Claude cache counts are additional input; Codex cached_input_tokens is a subset.
|
||||
input_values = [usage.get("input_tokens")] + [
|
||||
usage.get(key, 0) for key in _INPUT_TOKEN_KEYS[1:]
|
||||
]
|
||||
return _reported_token_total(input_values), usage.get("output_tokens")
|
||||
|
||||
|
||||
def run_evaluations(args: argparse.Namespace) -> int:
|
||||
@@ -518,7 +541,7 @@ def main(argv: Optional[list[str]] = None) -> int:
|
||||
print(json.dumps(summarize_scores(read_jsonl(args.scores)), indent=2))
|
||||
return 0
|
||||
if args.command == "measure":
|
||||
print(json.dumps(summarize_usage(read_jsonl(args.responses)), indent=2))
|
||||
print(json.dumps(summarize_usage(read_jsonl(args.responses)), indent=2, allow_nan=False))
|
||||
return 0
|
||||
parser.error("unknown command")
|
||||
return 2
|
||||
|
||||
@@ -269,6 +269,93 @@ class EvaluationHarnessTest(unittest.TestCase):
|
||||
)
|
||||
self.assertEqual((None, None), run_evals._usage_tokens({}))
|
||||
|
||||
def test_usage_rejects_invalid_measurements(self):
|
||||
for cost in (-1, True, False, float("nan"), float("inf"), "0.1"):
|
||||
with self.subTest(cost=cost):
|
||||
rows = [self._usage_row("a", condition, "ok", cost, {})
|
||||
for condition in ("baseline", "candidate")]
|
||||
with self.assertRaisesRegex(ValueError, "cost_usd"):
|
||||
run_evals.summarize_usage(rows)
|
||||
for key in ("input_tokens", "output_tokens", "cache_creation_input_tokens",
|
||||
"cache_read_input_tokens", "cached_input_tokens"):
|
||||
for value in (-1, True, 1.5, float("nan"), "10"):
|
||||
with self.subTest(key=key, value=value):
|
||||
with self.assertRaisesRegex(ValueError, key):
|
||||
run_evals._usage_tokens({key: value})
|
||||
with self.assertRaisesRegex(ValueError, "usage"):
|
||||
run_evals._usage_tokens([])
|
||||
|
||||
def test_usage_rejects_overflowing_cost_total(self):
|
||||
rows = [self._usage_row(case, condition, "ok", 1e308, {})
|
||||
for case in ("a", "b") for condition in ("baseline", "candidate")]
|
||||
with self.assertRaisesRegex(ValueError, "cost total is not finite"):
|
||||
run_evals.summarize_usage(rows)
|
||||
|
||||
def test_usage_distinguishes_missing_counts_from_zero(self):
|
||||
self.assertEqual((None, None), run_evals._usage_tokens(None))
|
||||
self.assertEqual((None, 2), run_evals._usage_tokens(
|
||||
{"cache_read_input_tokens": 10, "output_tokens": 2}))
|
||||
self.assertEqual((None, 2), run_evals._usage_tokens(
|
||||
{"input_tokens": 10, "cache_read_input_tokens": None, "output_tokens": 2}))
|
||||
rows = [self._usage_row("a", condition, "", 0,
|
||||
{"input_tokens": 0, "output_tokens": 0})
|
||||
for condition in ("baseline", "candidate")]
|
||||
result = run_evals.summarize_usage(rows)
|
||||
self.assertEqual(0, result["conditions"]["candidate"]["input_tokens"])
|
||||
self.assertEqual(0, result["delta"]["candidate"]["cost_usd"])
|
||||
self.assertIsNone(result["delta"]["candidate"]["cost_usd_pct"])
|
||||
rows[1]["usage"] = None
|
||||
self.assertIsNone(run_evals.summarize_usage(rows)["delta"]["candidate"]["input_tokens"])
|
||||
|
||||
def test_usage_rejects_duplicate_or_invalid_rows(self):
|
||||
rows = [self._usage_row("a", condition, "ok", 0, {})
|
||||
for condition in ("baseline", "candidate")]
|
||||
with self.assertRaisesRegex(ValueError, "duplicate"):
|
||||
run_evals.summarize_usage(rows + [rows[0]])
|
||||
for field, value in (("condition", "unknown"), ("case_id", None),
|
||||
("trial", True), ("trial", 0), ("runner", ""),
|
||||
("response", None)):
|
||||
with self.subTest(field=field):
|
||||
invalid = [dict(rows[0]), dict(rows[1])]
|
||||
invalid[1][field] = value
|
||||
with self.assertRaises(ValueError):
|
||||
run_evals.summarize_usage(invalid)
|
||||
|
||||
def test_usage_checks_model_metadata_when_available(self):
|
||||
rows = [self._usage_row("a", condition, "ok", 0, {})
|
||||
for condition in ("baseline", "candidate")]
|
||||
self.assertIsNone(run_evals.summarize_usage(rows)["model"])
|
||||
rows[0]["model"] = "model-a"
|
||||
with self.assertRaisesRegex(ValueError, "same model"):
|
||||
run_evals.summarize_usage(rows)
|
||||
rows[1]["model"] = "model-b"
|
||||
with self.assertRaisesRegex(ValueError, "same model"):
|
||||
run_evals.summarize_usage(rows)
|
||||
rows[1]["model"] = "model-a"
|
||||
self.assertEqual("model-a", run_evals.summarize_usage(rows)["model"])
|
||||
|
||||
def test_measure_cli_reports_increased_input_and_cost(self):
|
||||
rows = [
|
||||
self._usage_row("a", "baseline", "long answer", 0.1,
|
||||
{"input_tokens": 100, "output_tokens": 20}),
|
||||
self._usage_row("a", "candidate", "short", 0.15,
|
||||
{"input_tokens": 150, "output_tokens": 10}),
|
||||
]
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
path = Path(temporary) / "responses.jsonl"
|
||||
contents = "".join(json.dumps(row) + "\n" for row in rows)
|
||||
path.write_text(contents, encoding="utf-8")
|
||||
result = subprocess.run(
|
||||
[sys.executable, str(ROOT / "scripts/run_evals.py"), "measure", str(path)],
|
||||
check=True, capture_output=True, text=True, timeout=10,
|
||||
)
|
||||
self.assertEqual(contents, path.read_text(encoding="utf-8"))
|
||||
delta = json.loads(result.stdout)["delta"]["candidate"]
|
||||
self.assertEqual(50, delta["input_tokens"])
|
||||
self.assertEqual(50, delta["input_tokens_pct"])
|
||||
self.assertEqual(-50, delta["output_tokens_pct"])
|
||||
self.assertAlmostEqual(50, delta["cost_usd_pct"])
|
||||
|
||||
@staticmethod
|
||||
def _score_row(case_id, condition, value, trial=1):
|
||||
return {
|
||||
|
||||
Reference in New Issue
Block a user