mirror of
https://github.com/superdesigndev/treg.git
synced 2026-10-02 03:24:35 +08:00
MCP catalog_search answered from the lexical ranker with the v1 judge's say over it. The judge recovered a quarter of the searches the lexical gate admitted nothing for and barely moved the rest: its page was the lexical candidates filtered and reordered, with no vendor the words did not reach, and an empty page told the agent to try other words, so half of all searches were followed by another search. The engine behind /catalog/find (recall by job and by meaning, one judge request, every vendor of a fitting job, a verdict that can say the catalog lacks it) now answers agents in a new experiment mode. - search_experiment gains mode v2: arm_for deals v2 as the majority arm, the same two holdouts keep the pure lexical and the pure v1 judged page, so v2 is read against what it replaces. Arms are dealt by team and email where the search resolved them (an OAuth token rotates hourly), else by token. - catalog_find: decide takes the verdict for its rule 8 (a person gets none, an agent keyword: its input always means something), an abstain keeps the judge's reason, expand returns its groups (expand_groups) and answer_v2 splits into judge_and_decide and lay_out, so a holdout that only records lays out nothing. store gains routed_discovery_on and routed_parent, the one reader of each where four were. - application/catalog_search lays the answer out for an agent (agent_page): the best limit // 2 jobs, rows dealt round-robin and laid out job by job, members the judge rated on their own leading their job, a routed parent leading a strong or closest job's group, a listed hub tool joining its job with no lexical gate, and per job the vendors the page left out. - The verdicts an agent sees: strong, closest, name, and none only for a catalog gap (an empty page that says so, with catalog_request and no near misses). Not-a-task, an abstaining judge, a failure or a caller past the new per-caller cap (search_judge_max_per_caller_hour, the only bound on what an unmetered search can spend; it guards every dealt caller before any judge) serve the lexical page under keyword. - The response adds verdict, reason, jobs, and per row job and more_providers; score is null on a judged answer (no probability reaches an agent); the hint says only what to do next. Both MCP surfaces answer alike. - SearchLog: every served row records its job in every mode, so the report credits a call to any vendor of a job the page showed (a v2 hint sends the agent there); v2 rows carry find's readings and the verdict; the lexical holdout still has v2 judged and recorded, the counterfactual on a false none. The report gains conversion by job per arm and verdict, calls after a none, and the re-query rate per verdict. - scripts/search_agent_bench.py scores the v2 answer offline against searches agents made and what they called next (job-hit, hit@limit, false-none by arm), beside the pages the log served, on find_bench's harness (the judge and the query vectors cached on disk, the card vectors warmed once). - find_index: stored vectors decode as arrays, the index builds off the loop. The HTTP route and the CLI still answer from the lexical ranker; they follow once the route's hub read holds no session through a judge call and the CLI sends its token. Fragments: search-experiment.md, find.md, catalog.md, mcp-oauth.md; llms.txt and skill.md (and the generated SKILL.md copies) say what the verdict means.
67 lines
3.4 KiB
Python
67 lines
3.4 KiB
Python
"""The agent search bench (scripts/search_agent_bench.py): scores the job-first answer against what
|
|
agents called next, beside the pages the log served. Pinned on synthetic cases and a fake judge."""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
from scripts import search_agent_bench as sb
|
|
from tests.test_catalog_find import _fake_v2
|
|
from treg.config import get_settings
|
|
from treg.domain.catalog import store
|
|
from treg.infra import judge as judge_infra
|
|
|
|
JOB = "people.email.find"
|
|
|
|
|
|
def _cases(tmp_path: Path, cat) -> Path:
|
|
members = [e["id"] for e in cat.endpoints if e.get("capability") == JOB and store.browsable(e)]
|
|
other = next(e["id"] for e in cat.endpoints if e.get("capability") == "people.search" and store.browsable(e))
|
|
rows = [
|
|
# the caller called a vendor of the job v2 finds: a job-hit for v2 whichever row it shows first
|
|
{"id": "a", "q": "work email of a person at a company", "arm": "interleave", "baseline_empty": True,
|
|
"baseline_ids": [], "judged": [[members[-1], 0.8]], "called": members[-1], "called_on_shown": True},
|
|
# the caller called something else: a miss for every page
|
|
{"id": "b", "q": "work email finder", "arm": "baseline", "baseline_empty": False,
|
|
"baseline_ids": [other], "judged": [], "called": other, "called_on_shown": True},
|
|
# no call: counts for the verdict distribution only
|
|
{"id": "c", "q": "book a table for two", "arm": "interleave", "baseline_empty": True,
|
|
"baseline_ids": [], "judged": [], "called": None, "called_on_shown": False},
|
|
]
|
|
p = tmp_path / "agent.jsonl"
|
|
p.write_text("\n".join(json.dumps(r) for r in rows) + "\n")
|
|
return p
|
|
|
|
|
|
def test_the_bench_scores_job_hits_against_the_logged_pages(tmp_path, monkeypatch):
|
|
cat = store.load()
|
|
monkeypatch.setattr(get_settings(), "typesafe_api_key", "test-key", raising=False)
|
|
|
|
async def judge(query, cands, **kw):
|
|
if query.startswith("book"):
|
|
return await _fake_v2({}, plat=("none", 0.9))(query, cands, **kw)
|
|
return await _fake_v2({JOB: 0.9})(query, cands, **kw)
|
|
monkeypatch.setattr(judge_infra, "judge", judge)
|
|
|
|
cases = sb.load_cases(_cases(tmp_path, cat))
|
|
run = sb.bench(cases, limit=4, cache=None)
|
|
s = run["summary"]
|
|
assert s["n"] == 3 and s["labeled"] == 2
|
|
assert s["verdicts"] == {"none:gap": 1, "strong": 2}
|
|
assert s["job_hit_all"] == {"v2": [1, 2], "baseline": [1, 2], "judged": [1, 2]}
|
|
assert s["hit_all"]["baseline"] == [1, 2] and s["hit_all"]["judged"] == [1, 2]
|
|
assert s["job_hit_empty"]["v2"] == [1, 1] and s["job_hit_nonempty"]["v2"] == [0, 1]
|
|
assert s["false_none"] == [0, 2] and s["false_none_by_arm"] == {"baseline": [0, 1], "interleave": [0, 1]}
|
|
a = run["cases"]["a"]
|
|
assert a["job_hit"] is True and a["jobs"] == [JOB] and len(a["rows"]) == 4
|
|
assert run["cases"]["c"]["verdict"] == "none" and "job_hit" not in run["cases"]["c"]
|
|
|
|
|
|
def test_a_gap_where_the_caller_called_is_a_false_none(tmp_path, monkeypatch):
|
|
cat = store.load()
|
|
monkeypatch.setattr(get_settings(), "typesafe_api_key", "test-key", raising=False)
|
|
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, plat=("none", 0.9)))
|
|
run = sb.bench(sb.load_cases(_cases(tmp_path, cat)), limit=8, cache=None)
|
|
assert run["summary"]["false_none"] == [2, 2] and run["summary"]["false_none_by_arm"]["baseline"] == [1, 1]
|
|
assert run["summary"]["job_hit_all"]["v2"] == [0, 2]
|