mirror of
https://github.com/superdesigndev/treg.git
synced 2026-10-02 03:24:35 +08:00
fix(find): an empty recall still asks whether it is a name or a gap
With no units the judge sent nothing and returned no extra answers, so decide fell through to not_task: an out-of-catalog word like "weather" hid the request link and filed its miss as not_task. The judge now asks the extra questions alone when there are no candidates, and v2 passes its name Noul and platform Choice, so an empty recall can be a gap. v1 still sends no request for an empty recall. The bench records each case's embedding error, so a query that ran without the semantic channel is visible in the run file.
This commit is contained in:
@@ -104,7 +104,9 @@ representatives of those jobs in fused order (`find_delta`), then uncatalogued e
|
||||
platform, vendor count, a few names) and is asked whether tools that do the job accomplish the task,
|
||||
under `JOB_CRITERIA`; an endpoint keeps the v1 question and view. Two extra questions ride along:
|
||||
the v1 name Noul, and off a shelf a Choice over the platforms plus `none` (`platform_question`). No
|
||||
second model request, ever; the Choice only classifies and records.
|
||||
second model request, ever; the Choice only classifies and records. A recall with no units still
|
||||
sends the two extra questions (a request of extras only), so an empty recall is told apart as a
|
||||
catalog gap or not a task instead of defaulting to `not_task`.
|
||||
|
||||
**The name table** (`find_recall.name_of`) is string lookup, because a name is a lookup, not a
|
||||
judgement: a platform (exactly its name or slug, listed with the others the name matches: exact
|
||||
|
||||
@@ -504,6 +504,8 @@ def bench(cases: list[Case], *, engine: str, tier: str, cache: Path | None = Non
|
||||
row = {"q": c.q, "stratum": c.stratum, "candidates": a.candidates}
|
||||
if a.reach is not None:
|
||||
row["reach"] = a.reach
|
||||
if a.embed_error:
|
||||
row["embed_error"] = a.embed_error
|
||||
if tier == "recall":
|
||||
row["hit"] = any(view.is_gold(i, c.gold) for i in a.candidates) if c.gold else None
|
||||
else:
|
||||
|
||||
@@ -199,7 +199,7 @@ async def judge(query: str, cands: list[tuple[dict, float]], cat: catalog_store.
|
||||
for ep, _ in cands]
|
||||
j = await judge_infra.judge(query, views, api_key=s.typesafe_api_key, model=s.typesafe_model,
|
||||
url=s.typesafe_url, timeout_s=float(s.find_timeout_s),
|
||||
criteria=FIT_CRITERIA, extra={"name": NAME_QUESTION})
|
||||
criteria=FIT_CRITERIA, extra={"name": NAME_QUESTION} if views else None)
|
||||
if j.probs is None:
|
||||
page, _, _ = catalog_store.rank_band(query, cat, 25, platform)
|
||||
return Judged(KEYWORD, [(ep, None) for ep, _ in page[:25]], j)
|
||||
@@ -375,7 +375,9 @@ async def recall_with_meaning(query: str, cat: catalog_store.Catalog, provider_d
|
||||
|
||||
async def judge_v2(query: str, cands: list[find_recall.Candidate], cat: catalog_store.Catalog,
|
||||
provider_display, platform: str | None = None) -> judge_infra.Judgement:
|
||||
"""One request: a fit per unit, "is it only a name?", and (off a shelf) which platform."""
|
||||
"""One request: a fit per unit, "is it only a name?", and (off a shelf) which platform. With no
|
||||
units the two extra questions are still asked, so an empty recall can still be told apart as a
|
||||
catalog gap or not a task."""
|
||||
s = get_settings()
|
||||
extra = {"name": NAME_QUESTION}
|
||||
if platform is None:
|
||||
|
||||
@@ -129,9 +129,10 @@ async def judge(query: str, candidates: list[dict], *, api_key: str, model: str,
|
||||
`criteria` (Noul `true`/`false` descriptions) is attached to every endpoint question. A candidate
|
||||
in `job_view`'s shape (it carries `job`) is asked the job question instead, with `job_criteria`.
|
||||
`extra` maps ids to further questions about the same state (the query alone, say), Noul or
|
||||
Choice; they ride in the same request and come back as `Judgement.extra`. None of these changes
|
||||
what a caller passing none sends."""
|
||||
if not candidates:
|
||||
Choice; they ride in the same request and come back as `Judgement.extra`, and are asked even
|
||||
with no candidates (a request of extras only). None of these changes what a caller passing none
|
||||
sends."""
|
||||
if not candidates and not extra:
|
||||
return Judgement(probs=[], ms=0, extra={})
|
||||
ids = [c["id"] for c in candidates]
|
||||
kinds = ["job" if "job" in c else "endpoint" for c in candidates]
|
||||
|
||||
@@ -344,6 +344,20 @@ def test_v2_a_shelf_none_is_scope_never_a_gap():
|
||||
assert (found.verdict, found.reason) == (F.NONE, F.SCOPE)
|
||||
|
||||
|
||||
async def test_v2_with_no_units_still_tells_a_gap_from_not_a_task(clients, monkeypatch):
|
||||
seen = []
|
||||
_on(monkeypatch, find_engine="v2")
|
||||
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, seen, plat=("none", 0.9)))
|
||||
q = "zzqx qqxz" # no word on any card, no vectors
|
||||
_, (first, judged) = await _find(clients, q)
|
||||
assert first["units"] == [] and (judged["verdict"], judged["reason"]) == ("none", "gap")
|
||||
(_, views, kw), = seen
|
||||
assert views == [] and set(kw["extra"]) == {"name", "plat"}
|
||||
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, plat=("people", 0.9)))
|
||||
_, (_, judged) = await _find(clients, q)
|
||||
assert (judged["verdict"], judged["reason"]) == ("none", "not_task")
|
||||
|
||||
|
||||
async def test_v2_on_a_shelf_reads_that_shelf_and_asks_no_platform(clients, monkeypatch):
|
||||
seen = []
|
||||
_on(monkeypatch, find_engine="v2")
|
||||
|
||||
@@ -156,6 +156,22 @@ async def test_judge_asks_a_job_its_own_question_and_answers_a_choice():
|
||||
assert len(seen) == 2
|
||||
|
||||
|
||||
async def test_with_no_candidates_only_the_extra_questions_are_asked():
|
||||
judge_infra.clear_cache()
|
||||
seen = []
|
||||
|
||||
def handler(request):
|
||||
seen.append(json.loads(request.content))
|
||||
return httpx.Response(200, json={"answers": {"x_name": {"noul": 0.1}}})
|
||||
|
||||
kw = dict(api_key="k", model="jev-latest", url="https://judge.test/v1", timeout_s=1.0,
|
||||
transport=_transport(handler))
|
||||
assert (await judge_infra.judge("weather", [], **kw)).probs == [] and not seen # nothing to ask
|
||||
v = await judge_infra.judge("weather", [], extra={"name": {"type": "noul", "instructions": "a name"}}, **kw)
|
||||
assert v.probs == [] and v.extra == {"name": 0.1}
|
||||
assert set(seen[0]["questions"]) == {"x_name"} and seen[0]["state"]["candidates"] == []
|
||||
|
||||
|
||||
async def test_judge_abstains_on_timeout_http_error_and_bad_body():
|
||||
judge_infra.clear_cache()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user