fix(find): an empty recall still asks whether it is a name or a gap

With no units the judge sent nothing and returned no extra answers, so
decide fell through to not_task: an out-of-catalog word like "weather"
hid the request link and filed its miss as not_task. The judge now asks
the extra questions alone when there are no candidates, and v2 passes
its name Noul and platform Choice, so an empty recall can be a gap. v1
still sends no request for an empty recall.

The bench records each case's embedding error, so a query that ran
without the semantic channel is visible in the run file.
This commit is contained in:
SToneX
2026-09-30 23:14:30 +08:00
parent 2929951ff3
commit df292b1669
6 changed files with 43 additions and 6 deletions
+3 -1
View File
@@ -104,7 +104,9 @@ representatives of those jobs in fused order (`find_delta`), then uncatalogued e
platform, vendor count, a few names) and is asked whether tools that do the job accomplish the task,
under `JOB_CRITERIA`; an endpoint keeps the v1 question and view. Two extra questions ride along:
the v1 name Noul, and off a shelf a Choice over the platforms plus `none` (`platform_question`). No
second model request, ever; the Choice only classifies and records.
second model request, ever; the Choice only classifies and records. A recall with no units still
sends the two extra questions (a request of extras only), so an empty recall is told apart as a
catalog gap or not a task instead of defaulting to `not_task`.
**The name table** (`find_recall.name_of`) is string lookup, because a name is a lookup, not a
judgement: a platform (exactly its name or slug, listed with the others the name matches: exact
+2
View File
@@ -504,6 +504,8 @@ def bench(cases: list[Case], *, engine: str, tier: str, cache: Path | None = Non
row = {"q": c.q, "stratum": c.stratum, "candidates": a.candidates}
if a.reach is not None:
row["reach"] = a.reach
if a.embed_error:
row["embed_error"] = a.embed_error
if tier == "recall":
row["hit"] = any(view.is_gold(i, c.gold) for i in a.candidates) if c.gold else None
else:
+4 -2
View File
@@ -199,7 +199,7 @@ async def judge(query: str, cands: list[tuple[dict, float]], cat: catalog_store.
for ep, _ in cands]
j = await judge_infra.judge(query, views, api_key=s.typesafe_api_key, model=s.typesafe_model,
url=s.typesafe_url, timeout_s=float(s.find_timeout_s),
criteria=FIT_CRITERIA, extra={"name": NAME_QUESTION})
criteria=FIT_CRITERIA, extra={"name": NAME_QUESTION} if views else None)
if j.probs is None:
page, _, _ = catalog_store.rank_band(query, cat, 25, platform)
return Judged(KEYWORD, [(ep, None) for ep, _ in page[:25]], j)
@@ -375,7 +375,9 @@ async def recall_with_meaning(query: str, cat: catalog_store.Catalog, provider_d
async def judge_v2(query: str, cands: list[find_recall.Candidate], cat: catalog_store.Catalog,
provider_display, platform: str | None = None) -> judge_infra.Judgement:
"""One request: a fit per unit, "is it only a name?", and (off a shelf) which platform."""
"""One request: a fit per unit, "is it only a name?", and (off a shelf) which platform. With no
units the two extra questions are still asked, so an empty recall can still be told apart as a
catalog gap or not a task."""
s = get_settings()
extra = {"name": NAME_QUESTION}
if platform is None:
+4 -3
View File
@@ -129,9 +129,10 @@ async def judge(query: str, candidates: list[dict], *, api_key: str, model: str,
`criteria` (Noul `true`/`false` descriptions) is attached to every endpoint question. A candidate
in `job_view`'s shape (it carries `job`) is asked the job question instead, with `job_criteria`.
`extra` maps ids to further questions about the same state (the query alone, say), Noul or
Choice; they ride in the same request and come back as `Judgement.extra`. None of these changes
what a caller passing none sends."""
if not candidates:
Choice; they ride in the same request and come back as `Judgement.extra`, and are asked even
with no candidates (a request of extras only). None of these changes what a caller passing none
sends."""
if not candidates and not extra:
return Judgement(probs=[], ms=0, extra={})
ids = [c["id"] for c in candidates]
kinds = ["job" if "job" in c else "endpoint" for c in candidates]
+14
View File
@@ -344,6 +344,20 @@ def test_v2_a_shelf_none_is_scope_never_a_gap():
assert (found.verdict, found.reason) == (F.NONE, F.SCOPE)
async def test_v2_with_no_units_still_tells_a_gap_from_not_a_task(clients, monkeypatch):
seen = []
_on(monkeypatch, find_engine="v2")
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, seen, plat=("none", 0.9)))
q = "zzqx qqxz" # no word on any card, no vectors
_, (first, judged) = await _find(clients, q)
assert first["units"] == [] and (judged["verdict"], judged["reason"]) == ("none", "gap")
(_, views, kw), = seen
assert views == [] and set(kw["extra"]) == {"name", "plat"}
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, plat=("people", 0.9)))
_, (_, judged) = await _find(clients, q)
assert (judged["verdict"], judged["reason"]) == ("none", "not_task")
async def test_v2_on_a_shelf_reads_that_shelf_and_asks_no_platform(clients, monkeypatch):
seen = []
_on(monkeypatch, find_engine="v2")
+16
View File
@@ -156,6 +156,22 @@ async def test_judge_asks_a_job_its_own_question_and_answers_a_choice():
assert len(seen) == 2
async def test_with_no_candidates_only_the_extra_questions_are_asked():
judge_infra.clear_cache()
seen = []
def handler(request):
seen.append(json.loads(request.content))
return httpx.Response(200, json={"answers": {"x_name": {"noul": 0.1}}})
kw = dict(api_key="k", model="jev-latest", url="https://judge.test/v1", timeout_s=1.0,
transport=_transport(handler))
assert (await judge_infra.judge("weather", [], **kw)).probs == [] and not seen # nothing to ask
v = await judge_infra.judge("weather", [], extra={"name": {"type": "noul", "instructions": "a name"}}, **kw)
assert v.probs == [] and v.extra == {"name": 0.1}
assert set(seen[0]["questions"]) == {"x_name"} and seen[0]["state"]["candidates"] == []
async def test_judge_abstains_on_timeout_http_error_and_bad_body():
judge_infra.clear_cache()