From df292b1669a32612a76875d04445b3dc4239f402 Mon Sep 17 00:00:00 2001 From: SToneX Date: Wed, 30 Sep 2026 23:14:30 +0800 Subject: [PATCH] fix(find): an empty recall still asks whether it is a name or a gap With no units the judge sent nothing and returned no extra answers, so decide fell through to not_task: an out-of-catalog word like "weather" hid the request link and filed its miss as not_task. The judge now asks the extra questions alone when there are no candidates, and v2 passes its name Noul and platform Choice, so an empty recall can be a gap. v1 still sends no request for an empty recall. The bench records each case's embedding error, so a query that ran without the semantic channel is visible in the run file. --- docs/context/architecture/find.md | 4 +++- scripts/find_bench.py | 2 ++ src/treg/application/catalog_find.py | 6 ++++-- src/treg/infra/judge.py | 7 ++++--- tests/test_catalog_find.py | 14 ++++++++++++++ tests/test_search_experiment.py | 16 ++++++++++++++++ 6 files changed, 43 insertions(+), 6 deletions(-) diff --git a/docs/context/architecture/find.md b/docs/context/architecture/find.md index bba13af4..9ae2b426 100644 --- a/docs/context/architecture/find.md +++ b/docs/context/architecture/find.md @@ -104,7 +104,9 @@ representatives of those jobs in fused order (`find_delta`), then uncatalogued e platform, vendor count, a few names) and is asked whether tools that do the job accomplish the task, under `JOB_CRITERIA`; an endpoint keeps the v1 question and view. Two extra questions ride along: the v1 name Noul, and off a shelf a Choice over the platforms plus `none` (`platform_question`). No -second model request, ever; the Choice only classifies and records. +second model request, ever; the Choice only classifies and records. A recall with no units still +sends the two extra questions (a request of extras only), so an empty recall is told apart as a +catalog gap or not a task instead of defaulting to `not_task`. **The name table** (`find_recall.name_of`) is string lookup, because a name is a lookup, not a judgement: a platform (exactly its name or slug, listed with the others the name matches: exact diff --git a/scripts/find_bench.py b/scripts/find_bench.py index 3d5bda7c..386f2dc3 100644 --- a/scripts/find_bench.py +++ b/scripts/find_bench.py @@ -504,6 +504,8 @@ def bench(cases: list[Case], *, engine: str, tier: str, cache: Path | None = Non row = {"q": c.q, "stratum": c.stratum, "candidates": a.candidates} if a.reach is not None: row["reach"] = a.reach + if a.embed_error: + row["embed_error"] = a.embed_error if tier == "recall": row["hit"] = any(view.is_gold(i, c.gold) for i in a.candidates) if c.gold else None else: diff --git a/src/treg/application/catalog_find.py b/src/treg/application/catalog_find.py index 032a3e1e..ca483b7c 100644 --- a/src/treg/application/catalog_find.py +++ b/src/treg/application/catalog_find.py @@ -199,7 +199,7 @@ async def judge(query: str, cands: list[tuple[dict, float]], cat: catalog_store. for ep, _ in cands] j = await judge_infra.judge(query, views, api_key=s.typesafe_api_key, model=s.typesafe_model, url=s.typesafe_url, timeout_s=float(s.find_timeout_s), - criteria=FIT_CRITERIA, extra={"name": NAME_QUESTION}) + criteria=FIT_CRITERIA, extra={"name": NAME_QUESTION} if views else None) if j.probs is None: page, _, _ = catalog_store.rank_band(query, cat, 25, platform) return Judged(KEYWORD, [(ep, None) for ep, _ in page[:25]], j) @@ -375,7 +375,9 @@ async def recall_with_meaning(query: str, cat: catalog_store.Catalog, provider_d async def judge_v2(query: str, cands: list[find_recall.Candidate], cat: catalog_store.Catalog, provider_display, platform: str | None = None) -> judge_infra.Judgement: - """One request: a fit per unit, "is it only a name?", and (off a shelf) which platform.""" + """One request: a fit per unit, "is it only a name?", and (off a shelf) which platform. With no + units the two extra questions are still asked, so an empty recall can still be told apart as a + catalog gap or not a task.""" s = get_settings() extra = {"name": NAME_QUESTION} if platform is None: diff --git a/src/treg/infra/judge.py b/src/treg/infra/judge.py index 38080c03..104c2913 100644 --- a/src/treg/infra/judge.py +++ b/src/treg/infra/judge.py @@ -129,9 +129,10 @@ async def judge(query: str, candidates: list[dict], *, api_key: str, model: str, `criteria` (Noul `true`/`false` descriptions) is attached to every endpoint question. A candidate in `job_view`'s shape (it carries `job`) is asked the job question instead, with `job_criteria`. `extra` maps ids to further questions about the same state (the query alone, say), Noul or - Choice; they ride in the same request and come back as `Judgement.extra`. None of these changes - what a caller passing none sends.""" - if not candidates: + Choice; they ride in the same request and come back as `Judgement.extra`, and are asked even + with no candidates (a request of extras only). None of these changes what a caller passing none + sends.""" + if not candidates and not extra: return Judgement(probs=[], ms=0, extra={}) ids = [c["id"] for c in candidates] kinds = ["job" if "job" in c else "endpoint" for c in candidates] diff --git a/tests/test_catalog_find.py b/tests/test_catalog_find.py index c21e0982..a2417f18 100644 --- a/tests/test_catalog_find.py +++ b/tests/test_catalog_find.py @@ -344,6 +344,20 @@ def test_v2_a_shelf_none_is_scope_never_a_gap(): assert (found.verdict, found.reason) == (F.NONE, F.SCOPE) +async def test_v2_with_no_units_still_tells_a_gap_from_not_a_task(clients, monkeypatch): + seen = [] + _on(monkeypatch, find_engine="v2") + monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, seen, plat=("none", 0.9))) + q = "zzqx qqxz" # no word on any card, no vectors + _, (first, judged) = await _find(clients, q) + assert first["units"] == [] and (judged["verdict"], judged["reason"]) == ("none", "gap") + (_, views, kw), = seen + assert views == [] and set(kw["extra"]) == {"name", "plat"} + monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, plat=("people", 0.9))) + _, (_, judged) = await _find(clients, q) + assert (judged["verdict"], judged["reason"]) == ("none", "not_task") + + async def test_v2_on_a_shelf_reads_that_shelf_and_asks_no_platform(clients, monkeypatch): seen = [] _on(monkeypatch, find_engine="v2") diff --git a/tests/test_search_experiment.py b/tests/test_search_experiment.py index a008ea5d..0f156709 100644 --- a/tests/test_search_experiment.py +++ b/tests/test_search_experiment.py @@ -156,6 +156,22 @@ async def test_judge_asks_a_job_its_own_question_and_answers_a_choice(): assert len(seen) == 2 +async def test_with_no_candidates_only_the_extra_questions_are_asked(): + judge_infra.clear_cache() + seen = [] + + def handler(request): + seen.append(json.loads(request.content)) + return httpx.Response(200, json={"answers": {"x_name": {"noul": 0.1}}}) + + kw = dict(api_key="k", model="jev-latest", url="https://judge.test/v1", timeout_s=1.0, + transport=_transport(handler)) + assert (await judge_infra.judge("weather", [], **kw)).probs == [] and not seen # nothing to ask + v = await judge_infra.judge("weather", [], extra={"name": {"type": "noul", "instructions": "a name"}}, **kw) + assert v.probs == [] and v.extra == {"name": 0.1} + assert set(seen[0]["questions"]) == {"x_name"} and seen[0]["state"]["candidates"] == [] + + async def test_judge_abstains_on_timeout_http_error_and_bad_body(): judge_infra.clear_cache()