mirror of
https://github.com/superdesigndev/treg.git
synced 2026-10-02 03:24:35 +08:00
fix(find): shadow mode files one search miss, from the served engine
In shadow mode v1 and v2 both wrote a SearchMiss for the same empty query, and searchmiss had no column to tell them apart, so every double miss counted twice in the demand report. Only the served engine now files a miss (v2's empty answers stay visible in its SearchLog row), and searchmiss gains `engine` in migration 0054, which no deployment has applied yet.
This commit is contained in:
@@ -344,7 +344,8 @@ uses this metadata, never the encrypted token's shape.
|
||||
- **`SearchMiss`** - a catalog search that returned **nothing**: `query` (capped to 300 chars),
|
||||
`source` (`api` for the HTTP route that serves web + CLI + raw API; `mcp` for the team MCP; or
|
||||
`claude-connector` for V2; `web-find` for `/catalog/find`), `created_at`, and on a find `reason`
|
||||
(0054: `gap` | `not_task` | `judge_off` | `scope`, see [find](find.md)). The demand
|
||||
(0054: `gap` | `not_task` | `judge_off` | `scope`, see [find](find.md)) and `engine` (the find
|
||||
engine served; a shadow answer files none). The demand
|
||||
signal one step before a `ToolRequest`: most agents that miss never file, so the query text is all
|
||||
they leave. Written fire-and-forget through `audit.record_search_miss` (dropped rows cost
|
||||
analytics, never a search) from both search paths - `GET /catalog/search` and the in-process MCP
|
||||
|
||||
@@ -187,7 +187,9 @@ units reach, so the pages light the right platforms and vendors); `judged` gains
|
||||
`baseline_ids` is every endpoint the units reach, `judged` the kept units. `embed_ms` and
|
||||
`embed_error` record the query's vector (`off` without a key, `not_ready` while the card vectors
|
||||
build, else the client's reason). The `judged` event carries the same as `embed: {ms, error}`. `SearchMiss` gains `reason` (`gap`,
|
||||
`not_task`, `judge_off` for an empty keyword fallback, `scope` for a shelf's `none`). Migration 0054. Fire-and-forget through
|
||||
`not_task`, `judge_off` for an empty keyword fallback, `scope` for a shelf's `none`) and `engine`.
|
||||
Only the served engine files a miss: in `shadow` v1 does, and v2's empty answers show in its
|
||||
SearchLog row only, so a find never counts twice in the misses. Migration 0054. Fire-and-forget through
|
||||
`audit`, like every row there.
|
||||
|
||||
## The pages
|
||||
|
||||
@@ -7,7 +7,8 @@ Create Date: 2026-09-30
|
||||
`searchlog` gains the engine that answered a web find (v1 | v2; shadow mode writes one row for
|
||||
each) and v2's own readings: the judge's platform choice and its confidence, the name probability,
|
||||
recall and embedding times, the embedding error, and every unit the judge read as [kind, id, p].
|
||||
`searchmiss` gains why a find came back empty (gap | not_task | judge_off | scope). All nullable, no
|
||||
`searchmiss` gains why a find came back empty (gap | not_task | judge_off | scope) and which engine
|
||||
served it (shadow mode files a miss for the served engine only). All nullable, no
|
||||
defaults: metadata-only on Postgres, and rows written before read as unknown.
|
||||
"""
|
||||
from collections.abc import Sequence
|
||||
@@ -36,10 +37,12 @@ def upgrade() -> None:
|
||||
for name, type_ in _SEARCHLOG:
|
||||
op.add_column("searchlog", sa.Column(name, type_, nullable=True))
|
||||
op.add_column("searchmiss", sa.Column("reason", sa.String(), nullable=True))
|
||||
op.add_column("searchmiss", sa.Column("engine", sa.String(), nullable=True))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
with op.batch_alter_table("searchmiss") as batch:
|
||||
batch.drop_column("engine")
|
||||
batch.drop_column("reason")
|
||||
with op.batch_alter_table("searchlog") as batch:
|
||||
for name, _ in reversed(_SEARCHLOG):
|
||||
|
||||
@@ -274,7 +274,7 @@ def _log(query: str, *, source: str, baseline_total: int, cands: list[tuple[dict
|
||||
baseline_total=int(baseline_total), differs=False,
|
||||
judge_ms=j.ms, judge_tokens_in=j.tokens_in, judge_tokens_out=j.tokens_out, judge_error=j.error)
|
||||
if judged.verdict == NONE or (judged.verdict == KEYWORD and not judged.rows):
|
||||
audit.record_search_miss(query=query, source=source,
|
||||
audit.record_search_miss(query=query, source=source, engine="v1",
|
||||
reason=JUDGE_OFF if judged.verdict == KEYWORD else None)
|
||||
|
||||
|
||||
@@ -532,7 +532,7 @@ def _v2_row(r: dict, cat: catalog_store.Catalog, provider_display) -> dict:
|
||||
|
||||
|
||||
async def _stream_v2(query: str, provider_display, platform: str | None,
|
||||
evidence: Evidence | None) -> AsyncIterator[dict]:
|
||||
evidence: Evidence | None, served: bool = True) -> AsyncIterator[dict]:
|
||||
cat = catalog_store.load()
|
||||
r = await recall_with_meaning(query, cat, provider_display, platform)
|
||||
reached = _candidate_endpoints(r.cands, cat)
|
||||
@@ -544,13 +544,13 @@ async def _stream_v2(query: str, provider_display, platform: str | None,
|
||||
"rows": [_v2_row(row, cat, provider_display) for row in found.rows],
|
||||
"reason": found.reason, "platform": found.platform, "engine": "v2",
|
||||
"embed": {"ms": r.embed.ms, "error": r.embed.error}}
|
||||
_log_v2(query, cat, found, r, reached)
|
||||
_log_v2(query, cat, found, r, reached, served)
|
||||
|
||||
|
||||
async def _shadow_v2(query: str, provider_display, platform: str | None, evidence: Evidence | None) -> None:
|
||||
"""v2 beside a served v1 answer, for the log only: the v2 stream, its events unread. Never raises."""
|
||||
try:
|
||||
async for _ in _stream_v2(query, provider_display, platform, evidence):
|
||||
async for _ in _stream_v2(query, provider_display, platform, evidence, served=False):
|
||||
pass
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
@@ -558,7 +558,10 @@ async def _shadow_v2(query: str, provider_display, platform: str | None, evidenc
|
||||
log.warning("find shadow failed", exc_info=True)
|
||||
|
||||
|
||||
def _log_v2(query: str, cat: catalog_store.Catalog, found: Found, r: Recalled, reached: list[dict]) -> None:
|
||||
def _log_v2(query: str, cat: catalog_store.Catalog, found: Found, r: Recalled, reached: list[dict],
|
||||
served: bool = True) -> None:
|
||||
"""The v2 SearchLog row, and - when v2's answer is the one served - its SearchMiss: a shadow
|
||||
answer never files a miss beside the served engine's, so each find files at most one."""
|
||||
j = found.judgement
|
||||
_, baseline_total = catalog_store.search(query, cat, 0)
|
||||
probs = j.probs or [None] * len(found.cands)
|
||||
@@ -576,6 +579,6 @@ def _log_v2(query: str, cat: catalog_store.Catalog, found: Found, r: Recalled, r
|
||||
name_p=None if found.name_p is None else round(found.name_p, 3), recall_ms=round(r.recall_ms),
|
||||
embed_ms=r.embed.ms, embed_error=r.embed.error,
|
||||
units=[[c.unit.kind, c.unit.id, None if p is None else round(p, 3)] for c, p in zip(found.cands, probs)])
|
||||
if found.verdict == NONE or (found.verdict == KEYWORD and not found.rows):
|
||||
audit.record_search_miss(query=query, source="web-find",
|
||||
if served and (found.verdict == NONE or (found.verdict == KEYWORD and not found.rows)):
|
||||
audit.record_search_miss(query=query, source="web-find", engine="v2",
|
||||
reason=found.reason or (JUDGE_OFF if found.verdict == KEYWORD else None))
|
||||
|
||||
+6
-4
@@ -106,11 +106,13 @@ def _known_fields(model, telemetry: dict | None) -> dict:
|
||||
return known
|
||||
|
||||
|
||||
def record_search_miss(*, query: str, source: str, reason: str | None = None) -> None:
|
||||
def record_search_miss(*, query: str, source: str, reason: str | None = None,
|
||||
engine: str | None = None) -> None:
|
||||
"""A catalog search that matched nothing — logged so the misses can steer ingest (see
|
||||
models.SearchMiss). `reason` says why a find answer was empty. Same contract as every write
|
||||
here: fire-and-forget, and a dropped row under load costs a data point, never a search response."""
|
||||
_enqueue(SearchMiss, dict(query=query[:300], source=source, reason=reason))
|
||||
models.SearchMiss). On a find, `reason` says why the answer was empty and `engine` which find
|
||||
answered. Same contract as every write here: fire-and-forget, and a dropped row under load
|
||||
costs a data point, never a search response."""
|
||||
_enqueue(SearchMiss, dict(query=query[:300], source=source, reason=reason, engine=engine))
|
||||
|
||||
|
||||
def record_search(*, query: str, source: str, org_id: int | None, user_email: str | None,
|
||||
|
||||
@@ -1507,6 +1507,8 @@ class SearchMiss(SQLModel, table=True):
|
||||
# web-find only: why the answer was empty - gap (the catalog lacks it) | not_task | judge_off |
|
||||
# scope (a shelf's find read that shelf only)
|
||||
reason: str | None = Field(default=None)
|
||||
# web-find only: the engine whose answer was served and empty (v1 | v2); a shadow files no miss
|
||||
engine: str | None = Field(default=None)
|
||||
|
||||
|
||||
class SearchLog(SQLModel, table=True):
|
||||
|
||||
@@ -414,3 +414,15 @@ async def test_v2_with_the_semantic_channel_reports_it_on_the_event_and_the_log(
|
||||
assert [(r.embed_ms, r.embed_error) for r in rows] == [(None, "not_ready"), (3, None)]
|
||||
finally:
|
||||
find_index.reset()
|
||||
|
||||
|
||||
async def test_shadow_files_one_miss_from_the_engine_it_serves(clients, monkeypatch):
|
||||
_on(monkeypatch, find_engine="shadow")
|
||||
monkeypatch.setattr(judge_infra, "judge", _fake_v2({}, plat=("none", 0.9))) # both engines find nothing
|
||||
await _find(clients, JOB)
|
||||
await audit.drain()
|
||||
async with session_maker() as s:
|
||||
misses = (await s.execute(select(SearchMiss))).scalars().all()
|
||||
logs = (await s.execute(select(SearchLog))).scalars().all()
|
||||
assert [(m.engine, m.source) for m in misses] == [("v1", "web-find")]
|
||||
assert sorted(r.engine for r in logs) == ["v1", "v2"]
|
||||
|
||||
Reference in New Issue
Block a user