fix(find): shadow mode files one search miss, from the served engine

In shadow mode v1 and v2 both wrote a SearchMiss for the same empty
query, and searchmiss had no column to tell them apart, so every
double miss counted twice in the demand report. Only the served engine
now files a miss (v2's empty answers stay visible in its SearchLog
row), and searchmiss gains `engine` in migration 0054, which no
deployment has applied yet.
This commit is contained in:
SToneX
2026-09-30 23:15:13 +08:00
parent df292b1669
commit abbea14e49
7 changed files with 39 additions and 14 deletions
+2 -1
View File
@@ -344,7 +344,8 @@ uses this metadata, never the encrypted token's shape.
- **`SearchMiss`** - a catalog search that returned **nothing**: `query` (capped to 300 chars),
`source` (`api` for the HTTP route that serves web + CLI + raw API; `mcp` for the team MCP; or
`claude-connector` for V2; `web-find` for `/catalog/find`), `created_at`, and on a find `reason`
(0054: `gap` | `not_task` | `judge_off` | `scope`, see [find](find.md)). The demand
(0054: `gap` | `not_task` | `judge_off` | `scope`, see [find](find.md)) and `engine` (the find
engine served; a shadow answer files none). The demand
signal one step before a `ToolRequest`: most agents that miss never file, so the query text is all
they leave. Written fire-and-forget through `audit.record_search_miss` (dropped rows cost
analytics, never a search) from both search paths - `GET /catalog/search` and the in-process MCP
+3 -1
View File
@@ -187,7 +187,9 @@ units reach, so the pages light the right platforms and vendors); `judged` gains
`baseline_ids` is every endpoint the units reach, `judged` the kept units. `embed_ms` and
`embed_error` record the query's vector (`off` without a key, `not_ready` while the card vectors
build, else the client's reason). The `judged` event carries the same as `embed: {ms, error}`. `SearchMiss` gains `reason` (`gap`,
`not_task`, `judge_off` for an empty keyword fallback, `scope` for a shelf's `none`). Migration 0054. Fire-and-forget through
`not_task`, `judge_off` for an empty keyword fallback, `scope` for a shelf's `none`) and `engine`.
Only the served engine files a miss: in `shadow` v1 does, and v2's empty answers show in its
SearchLog row only, so a find never counts twice in the misses. Migration 0054. Fire-and-forget through
`audit`, like every row there.
## The pages
@@ -7,7 +7,8 @@ Create Date: 2026-09-30
`searchlog` gains the engine that answered a web find (v1 | v2; shadow mode writes one row for
each) and v2's own readings: the judge's platform choice and its confidence, the name probability,
recall and embedding times, the embedding error, and every unit the judge read as [kind, id, p].
`searchmiss` gains why a find came back empty (gap | not_task | judge_off | scope). All nullable, no
`searchmiss` gains why a find came back empty (gap | not_task | judge_off | scope) and which engine
served it (shadow mode files a miss for the served engine only). All nullable, no
defaults: metadata-only on Postgres, and rows written before read as unknown.
"""
from collections.abc import Sequence
@@ -36,10 +37,12 @@ def upgrade() -> None:
for name, type_ in _SEARCHLOG:
op.add_column("searchlog", sa.Column(name, type_, nullable=True))
op.add_column("searchmiss", sa.Column("reason", sa.String(), nullable=True))
op.add_column("searchmiss", sa.Column("engine", sa.String(), nullable=True))
def downgrade() -> None:
with op.batch_alter_table("searchmiss") as batch:
batch.drop_column("engine")
batch.drop_column("reason")
with op.batch_alter_table("searchlog") as batch:
for name, _ in reversed(_SEARCHLOG):
+10 -7
View File
@@ -274,7 +274,7 @@ def _log(query: str, *, source: str, baseline_total: int, cands: list[tuple[dict
baseline_total=int(baseline_total), differs=False,
judge_ms=j.ms, judge_tokens_in=j.tokens_in, judge_tokens_out=j.tokens_out, judge_error=j.error)
if judged.verdict == NONE or (judged.verdict == KEYWORD and not judged.rows):
audit.record_search_miss(query=query, source=source,
audit.record_search_miss(query=query, source=source, engine="v1",
reason=JUDGE_OFF if judged.verdict == KEYWORD else None)
@@ -532,7 +532,7 @@ def _v2_row(r: dict, cat: catalog_store.Catalog, provider_display) -> dict:
async def _stream_v2(query: str, provider_display, platform: str | None,
evidence: Evidence | None) -> AsyncIterator[dict]:
evidence: Evidence | None, served: bool = True) -> AsyncIterator[dict]:
cat = catalog_store.load()
r = await recall_with_meaning(query, cat, provider_display, platform)
reached = _candidate_endpoints(r.cands, cat)
@@ -544,13 +544,13 @@ async def _stream_v2(query: str, provider_display, platform: str | None,
"rows": [_v2_row(row, cat, provider_display) for row in found.rows],
"reason": found.reason, "platform": found.platform, "engine": "v2",
"embed": {"ms": r.embed.ms, "error": r.embed.error}}
_log_v2(query, cat, found, r, reached)
_log_v2(query, cat, found, r, reached, served)
async def _shadow_v2(query: str, provider_display, platform: str | None, evidence: Evidence | None) -> None:
"""v2 beside a served v1 answer, for the log only: the v2 stream, its events unread. Never raises."""
try:
async for _ in _stream_v2(query, provider_display, platform, evidence):
async for _ in _stream_v2(query, provider_display, platform, evidence, served=False):
pass
except asyncio.CancelledError:
raise
@@ -558,7 +558,10 @@ async def _shadow_v2(query: str, provider_display, platform: str | None, evidenc
log.warning("find shadow failed", exc_info=True)
def _log_v2(query: str, cat: catalog_store.Catalog, found: Found, r: Recalled, reached: list[dict]) -> None:
def _log_v2(query: str, cat: catalog_store.Catalog, found: Found, r: Recalled, reached: list[dict],
served: bool = True) -> None:
"""The v2 SearchLog row, and - when v2's answer is the one served - its SearchMiss: a shadow
answer never files a miss beside the served engine's, so each find files at most one."""
j = found.judgement
_, baseline_total = catalog_store.search(query, cat, 0)
probs = j.probs or [None] * len(found.cands)
@@ -576,6 +579,6 @@ def _log_v2(query: str, cat: catalog_store.Catalog, found: Found, r: Recalled, r
name_p=None if found.name_p is None else round(found.name_p, 3), recall_ms=round(r.recall_ms),
embed_ms=r.embed.ms, embed_error=r.embed.error,
units=[[c.unit.kind, c.unit.id, None if p is None else round(p, 3)] for c, p in zip(found.cands, probs)])
if found.verdict == NONE or (found.verdict == KEYWORD and not found.rows):
audit.record_search_miss(query=query, source="web-find",
if served and (found.verdict == NONE or (found.verdict == KEYWORD and not found.rows)):
audit.record_search_miss(query=query, source="web-find", engine="v2",
reason=found.reason or (JUDGE_OFF if found.verdict == KEYWORD else None))
+6 -4
View File
@@ -106,11 +106,13 @@ def _known_fields(model, telemetry: dict | None) -> dict:
return known
def record_search_miss(*, query: str, source: str, reason: str | None = None) -> None:
def record_search_miss(*, query: str, source: str, reason: str | None = None,
engine: str | None = None) -> None:
"""A catalog search that matched nothing — logged so the misses can steer ingest (see
models.SearchMiss). `reason` says why a find answer was empty. Same contract as every write
here: fire-and-forget, and a dropped row under load costs a data point, never a search response."""
_enqueue(SearchMiss, dict(query=query[:300], source=source, reason=reason))
models.SearchMiss). On a find, `reason` says why the answer was empty and `engine` which find
answered. Same contract as every write here: fire-and-forget, and a dropped row under load
costs a data point, never a search response."""
_enqueue(SearchMiss, dict(query=query[:300], source=source, reason=reason, engine=engine))
def record_search(*, query: str, source: str, org_id: int | None, user_email: str | None,
+2
View File
@@ -1507,6 +1507,8 @@ class SearchMiss(SQLModel, table=True):
# web-find only: why the answer was empty - gap (the catalog lacks it) | not_task | judge_off |
# scope (a shelf's find read that shelf only)
reason: str | None = Field(default=None)
# web-find only: the engine whose answer was served and empty (v1 | v2); a shadow files no miss
engine: str | None = Field(default=None)
class SearchLog(SQLModel, table=True):
+12
View File
@@ -414,3 +414,15 @@ async def test_v2_with_the_semantic_channel_reports_it_on_the_event_and_the_log(
assert [(r.embed_ms, r.embed_error) for r in rows] == [(None, "not_ready"), (3, None)]
finally:
find_index.reset()
async def test_shadow_files_one_miss_from_the_engine_it_serves(clients, monkeypatch):
_on(monkeypatch, find_engine="shadow")
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, plat=("none", 0.9))) # both engines find nothing
await _find(clients, JOB)
await audit.drain()
async with session_maker() as s:
misses = (await s.execute(select(SearchMiss))).scalars().all()
logs = (await s.execute(select(SearchLog))).scalars().all()
assert [(m.engine, m.source) for m in misses] == [("v1", "web-find")]
assert sorted(r.engine for r in logs) == ["v1", "v2"]