feat(phase-19/50): add hypothesis-generator deep capstone

This commit is contained in:
Rohit Ghumare
2026-05-26 19:31:37 +01:00
parent c1374e1d5a
commit c1ea434a77
4 changed files with 657 additions and 0 deletions
@@ -0,0 +1,311 @@
"""Hypothesis generator: temperature ramped sampling, novelty filter, ranked queue.
Conceptual references:
- ./docs/en.md (this lesson)
- Phase 19 Track A lessons 20-29 (agent harness primitives)
Stdlib only. Run: python3 code/main.py
"""
from __future__ import annotations
import hashlib
import json
import math
import re
from dataclasses import dataclass, field
from typing import Callable
HASH_DIM = 128
TAG_RE = re.compile(
r"<hypothesis>\s*"
r"<text>(?P<text>.*?)</text>\s*"
r"<variables>(?P<variables>.*?)</variables>\s*"
r"<metric>(?P<metric>.*?)</metric>\s*"
r"(?:<baseline>(?P<baseline>.*?)</baseline>\s*)?"
r"</hypothesis>",
re.DOTALL,
)
@dataclass
class Hypothesis:
id: int
text: str
variables: list[str]
metric: str
baseline_ref: str | None
draft_pass: int
temperature: float
novelty_score: float = 0.0
rank_score: float = 0.0
def to_dict(self) -> dict:
return {
"id": self.id,
"text": self.text,
"variables": list(self.variables),
"metric": self.metric,
"baseline_ref": self.baseline_ref,
"draft_pass": self.draft_pass,
"temperature": round(self.temperature, 3),
"novelty_score": round(self.novelty_score, 4),
"rank_score": round(self.rank_score, 4),
}
class ParserError(ValueError):
"""Raised when a sampler response does not match the hypothesis tag schema."""
def _tokenise(text: str) -> list[str]:
return re.findall(r"[a-z0-9]+", text.lower())
def hashed_embed(text: str, dim: int = HASH_DIM) -> list[float]:
"""Hashed bag of tokens embedding, L2 normalised. Deterministic stdlib only."""
vec = [0.0] * dim
for tok in _tokenise(text):
h = hashlib.md5(tok.encode("utf-8")).digest()
idx = int.from_bytes(h[:4], "big") % dim
sign = 1.0 if (h[4] & 1) == 0 else -1.0
vec[idx] += sign
norm = math.sqrt(sum(v * v for v in vec))
if norm == 0.0:
return vec
return [v / norm for v in vec]
def cosine_distance(a: list[float], b: list[float]) -> float:
dot = sum(x * y for x, y in zip(a, b))
dot = max(-1.0, min(1.0, dot))
return 1.0 - dot
def parse_response(raw: str) -> dict:
match = TAG_RE.search(raw)
if match is None:
raise ParserError("no hypothesis block found")
text = match.group("text").strip()
if not text:
raise ParserError("empty text")
metric = match.group("metric").strip()
if not metric:
raise ParserError("empty metric")
raw_vars = match.group("variables").strip()
variables = [v.strip() for v in raw_vars.split(",") if v.strip()]
if not variables:
raise ParserError("empty variables")
baseline = match.group("baseline")
baseline_ref = baseline.strip() if baseline and baseline.strip() else None
return {
"text": text,
"variables": variables,
"metric": metric,
"baseline_ref": baseline_ref,
}
def temperature_bucket(temperature: float) -> int:
"""Map a continuous temperature to a discrete bucket index."""
if temperature < 0.35:
return 0
if temperature < 0.65:
return 1
if temperature < 0.95:
return 2
return 3
class MockLLM:
"""Scripted sampler keyed on (prompt_signature, temperature_bucket).
The seed is folded into the response so identical prompts and buckets with
different seeds yield distinct drafts. Unknown keys return an unparseable
fallback so the parser-failure path is reachable from tests.
"""
def __init__(self, scripts: dict[tuple[str, int], list[str]]) -> None:
self._scripts = dict(scripts)
@staticmethod
def prompt_signature(prompt: str) -> str:
return hashlib.sha1(prompt.encode("utf-8")).hexdigest()[:10]
def sample(self, prompt: str, temperature: float, seed: int) -> str:
key = (self.prompt_signature(prompt), temperature_bucket(temperature))
bank = self._scripts.get(key)
if not bank:
return "<noise>untagged drift</noise>"
return bank[seed % len(bank)]
@dataclass
class GeneratorConfig:
n_passes: int = 6
t_min: float = 0.2
t_max: float = 1.2
novelty_threshold: float = 0.25
target_variable_count: int = 3
w_novelty: float = 0.4
w_specificity: float = 0.3
w_testability: float = 0.3
base_seed: int = 0
def schedule(self) -> list[float]:
if self.n_passes <= 0:
return []
if self.n_passes == 1:
return [self.t_min]
step = (self.t_max - self.t_min) / (self.n_passes - 1)
return [self.t_min + i * step for i in range(self.n_passes)]
@dataclass
class GenerationLog:
pass_index: int
temperature: float
seed: int
accepted_id: int | None
reject_reason: str | None
raw_excerpt: str
def to_dict(self) -> dict:
return {
"pass": self.pass_index,
"temperature": round(self.temperature, 3),
"seed": self.seed,
"accepted_id": self.accepted_id,
"reject_reason": self.reject_reason,
"raw_excerpt": self.raw_excerpt[:80],
}
class HypothesisGenerator:
"""Drives the mock LLM over a temperature schedule and ranks the survivors."""
def __init__(
self,
llm: MockLLM,
config: GeneratorConfig | None = None,
embedder: Callable[[str], list[float]] = hashed_embed,
) -> None:
self._llm = llm
self._cfg = config or GeneratorConfig()
self._embed = embedder
def _specificity_score(self, h: Hypothesis) -> float:
target = max(1, self._cfg.target_variable_count)
return min(1.0, len(h.variables) / target)
def _testability_score(self, h: Hypothesis) -> float:
if h.metric and h.baseline_ref:
return 1.0
if h.metric:
return 0.5
return 0.0
def _score(self, h: Hypothesis) -> float:
return (
self._cfg.w_novelty * h.novelty_score
+ self._cfg.w_specificity * self._specificity_score(h)
+ self._cfg.w_testability * self._testability_score(h)
)
def _novelty(self, candidate: list[float], survivors: list[list[float]]) -> float:
if not survivors:
return 1.0
return min(cosine_distance(candidate, s) for s in survivors)
def run(self, seed_prompt: str) -> tuple[list[Hypothesis], list[GenerationLog]]:
survivors: list[Hypothesis] = []
survivor_vecs: list[list[float]] = []
logs: list[GenerationLog] = []
next_id = 1
for pass_index, temperature in enumerate(self._cfg.schedule()):
seed = self._cfg.base_seed + pass_index
raw = self._llm.sample(seed_prompt, temperature, seed)
try:
parsed = parse_response(raw)
except ParserError as exc:
logs.append(GenerationLog(pass_index, temperature, seed, None, f"parse:{exc}", raw))
continue
vec = self._embed(parsed["text"])
novelty = self._novelty(vec, survivor_vecs)
if novelty < self._cfg.novelty_threshold:
logs.append(GenerationLog(pass_index, temperature, seed, None, "duplicate", raw))
continue
hypothesis = Hypothesis(
id=next_id,
text=parsed["text"],
variables=parsed["variables"],
metric=parsed["metric"],
baseline_ref=parsed["baseline_ref"],
draft_pass=pass_index,
temperature=temperature,
novelty_score=novelty,
)
hypothesis.rank_score = self._score(hypothesis)
survivors.append(hypothesis)
survivor_vecs.append(vec)
logs.append(GenerationLog(pass_index, temperature, seed, next_id, None, raw))
next_id += 1
survivors.sort(key=lambda h: (-h.rank_score, h.id))
return survivors, logs
def build_demo_scripts() -> dict[tuple[str, int], list[str]]:
"""Scripted responses for the demo seed prompt across temperature buckets."""
seed_prompt = "Investigate attention sparsity in small transformers"
sig = MockLLM.prompt_signature(seed_prompt)
return {
(sig, 0): [
"<hypothesis>"
"<text>Lowering attention head count from 8 to 4 raises validation loss by less than 2 percent on a 12M parameter model.</text>"
"<variables>head_count, validation_loss</variables>"
"<metric>validation_loss</metric>"
"<baseline>head_count_8</baseline>"
"</hypothesis>",
],
(sig, 1): [
"<hypothesis>"
"<text>Top-k sparse attention with k equal to 16 matches dense attention on perplexity at 12M parameters.</text>"
"<variables>k, perplexity, parameter_count</variables>"
"<metric>perplexity</metric>"
"<baseline>dense_attention</baseline>"
"</hypothesis>",
],
(sig, 2): [
"<hypothesis>"
"<text>Routing attention through a learned gate reduces flops by 30 percent without harming downstream accuracy.</text>"
"<variables>gate_temperature, flops, accuracy</variables>"
"<metric>downstream_accuracy</metric>"
"<baseline>dense_attention</baseline>"
"</hypothesis>",
],
(sig, 3): [
"<hypothesis>"
"<text>Block sparse attention with block size 32 lowers wall clock training time by 18 percent on consumer GPUs.</text>"
"<variables>block_size, training_seconds, hardware</variables>"
"<metric>training_seconds</metric>"
"<baseline>dense_attention</baseline>"
"</hypothesis>",
],
}
def _demo() -> None:
llm = MockLLM(build_demo_scripts())
config = GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1)
generator = HypothesisGenerator(llm, config)
queue, logs = generator.run("Investigate attention sparsity in small transformers")
print(json.dumps({
"queue_size": len(queue),
"queue": [h.to_dict() for h in queue],
"logs": [log.to_dict() for log in logs],
}, indent=2))
if __name__ == "__main__":
_demo()
@@ -0,0 +1,156 @@
"""Tests for HypothesisGenerator: linear queue, dedup, parser, schedule, rank order."""
from __future__ import annotations
import os
import sys
import unittest
HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, os.path.dirname(HERE))
from main import ( # noqa: E402
GeneratorConfig,
HypothesisGenerator,
MockLLM,
ParserError,
build_demo_scripts,
cosine_distance,
hashed_embed,
parse_response,
temperature_bucket,
)
SEED_PROMPT = "Investigate attention sparsity in small transformers"
class TestParser(unittest.TestCase):
def test_parses_full_block(self) -> None:
raw = (
"<hypothesis><text>x</text><variables>a, b</variables>"
"<metric>m</metric><baseline>r</baseline></hypothesis>"
)
parsed = parse_response(raw)
self.assertEqual(parsed["text"], "x")
self.assertEqual(parsed["variables"], ["a", "b"])
self.assertEqual(parsed["metric"], "m")
self.assertEqual(parsed["baseline_ref"], "r")
def test_baseline_optional(self) -> None:
raw = "<hypothesis><text>x</text><variables>a</variables><metric>m</metric></hypothesis>"
self.assertIsNone(parse_response(raw)["baseline_ref"])
def test_rejects_unparseable(self) -> None:
with self.assertRaises(ParserError):
parse_response("plain text no tags")
def test_rejects_empty_variables(self) -> None:
raw = "<hypothesis><text>x</text><variables> </variables><metric>m</metric></hypothesis>"
with self.assertRaises(ParserError):
parse_response(raw)
class TestEmbedding(unittest.TestCase):
def test_unit_norm(self) -> None:
vec = hashed_embed("attention sparsity small transformer")
n = sum(v * v for v in vec) ** 0.5
self.assertAlmostEqual(n, 1.0, places=5)
def test_distance_self_zero(self) -> None:
v = hashed_embed("identical text identical text")
self.assertAlmostEqual(cosine_distance(v, v), 0.0, places=5)
def test_distance_disjoint_high(self) -> None:
a = hashed_embed("attention sparsity transformer")
b = hashed_embed("dataloader checkpoint scheduler")
self.assertGreater(cosine_distance(a, b), 0.5)
class TestTemperatureRamp(unittest.TestCase):
def test_schedule_endpoints(self) -> None:
cfg = GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1)
schedule = cfg.schedule()
self.assertEqual(len(schedule), 4)
self.assertAlmostEqual(schedule[0], 0.2)
self.assertAlmostEqual(schedule[-1], 1.1)
def test_schedule_one_pass(self) -> None:
cfg = GeneratorConfig(n_passes=1, t_min=0.5, t_max=1.2)
self.assertEqual(cfg.schedule(), [0.5])
def test_schedule_zero_passes(self) -> None:
self.assertEqual(GeneratorConfig(n_passes=0).schedule(), [])
def test_bucket_boundaries(self) -> None:
self.assertEqual(temperature_bucket(0.2), 0)
self.assertEqual(temperature_bucket(0.5), 1)
self.assertEqual(temperature_bucket(0.8), 2)
self.assertEqual(temperature_bucket(1.1), 3)
class TestGenerator(unittest.TestCase):
def test_demo_path_produces_queue(self) -> None:
gen = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
queue, logs = gen.run(SEED_PROMPT)
self.assertEqual(len(queue), 4)
self.assertEqual(len(logs), 4)
for log in logs:
self.assertIsNone(log.reject_reason)
ids = [h.id for h in queue]
self.assertEqual(sorted(ids), [1, 2, 3, 4])
def test_queue_sorted_by_rank_desc(self) -> None:
gen = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
queue, _ = gen.run(SEED_PROMPT)
scores = [h.rank_score for h in queue]
self.assertEqual(scores, sorted(scores, reverse=True))
def test_duplicate_rejected(self) -> None:
sig = MockLLM.prompt_signature(SEED_PROMPT)
repeated = (
"<hypothesis><text>head count eight to four loss two percent</text>"
"<variables>head_count, loss</variables><metric>loss</metric>"
"<baseline>head_count_8</baseline></hypothesis>"
)
scripts = {(sig, 0): [repeated], (sig, 1): [repeated], (sig, 2): [repeated], (sig, 3): [repeated]}
gen = HypothesisGenerator(MockLLM(scripts), GeneratorConfig(n_passes=4))
queue, logs = gen.run(SEED_PROMPT)
self.assertEqual(len(queue), 1)
reject_reasons = [log.reject_reason for log in logs if log.reject_reason]
self.assertEqual(reject_reasons, ["duplicate", "duplicate", "duplicate"])
def test_parser_failure_logged(self) -> None:
sig = MockLLM.prompt_signature(SEED_PROMPT)
scripts = {(sig, 0): ["plain text"], (sig, 1): build_demo_scripts()[(sig, 1)]}
gen = HypothesisGenerator(MockLLM(scripts), GeneratorConfig(n_passes=2, t_min=0.2, t_max=0.6))
queue, logs = gen.run(SEED_PROMPT)
self.assertEqual(len(queue), 1)
self.assertTrue(logs[0].reject_reason.startswith("parse:"))
def test_unknown_prompt_falls_back_and_drops(self) -> None:
gen = HypothesisGenerator(MockLLM({}), GeneratorConfig(n_passes=3))
queue, logs = gen.run("never seen prompt")
self.assertEqual(queue, [])
self.assertTrue(all(log.reject_reason and log.reject_reason.startswith("parse:") for log in logs))
def test_specificity_weight_changes_rank(self) -> None:
cfg_a = GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1, w_specificity=1.0, w_novelty=0.0, w_testability=0.0)
gen = HypothesisGenerator(MockLLM(build_demo_scripts()), cfg_a)
queue, _ = gen.run(SEED_PROMPT)
for h in queue:
self.assertGreaterEqual(h.rank_score, 0.0)
self.assertLessEqual(h.rank_score, 1.0)
class TestDeterminism(unittest.TestCase):
def test_two_runs_identical(self) -> None:
gen_a = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
gen_b = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
queue_a, _ = gen_a.run(SEED_PROMPT)
queue_b, _ = gen_b.run(SEED_PROMPT)
self.assertEqual([h.to_dict() for h in queue_a], [h.to_dict() for h in queue_b])
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,112 @@
# Hypothesis Generator
> A research agent that asks the same question twice is wasting tokens. The trick is forcing each draft to land somewhere new.
**Type:** Build
**Languages:** Python
**Prerequisites:** Phase 19 Track A lessons 20-29
**Time:** ~90 minutes
## Learning Objectives
- Drive a sampler from a seed prompt and turn its outputs into typed hypothesis records.
- Ramp the sampler temperature on each pass so the next draft drifts further from the last.
- Filter near duplicates with a small embedding model and a cosine distance threshold.
- Rank the survivors with a scoring function that blends novelty, specificity, and testability.
- Hold every step deterministic so the same seed always produces the same queue.
## Why generate, then filter
A planner that asks one model one time gets one hypothesis. That is fine for a worked example. For a research loop it is the wrong shape. The loop wants a ranked queue with depth, so when the first hypothesis fails the runner has the next one ready without paying for another full sampling pass.
Two ideas combine to produce that queue. The first is temperature ramping: each pass through the sampler raises the temperature a notch, so later drafts are encouraged to wander. The second is novelty filtering: after each draft, the generator measures the embedding distance from every prior survivor and rejects anything inside the cluster.
The lesson ships a mock language model that returns scripted token sequences for fixed prompts. The mock is enough to exercise the full path: seed prompt in, temperature ramp applied, candidates parsed, novelty filter run, ranked queue out.
## The Hypothesis shape
```text
Hypothesis
id : int (monotonic within a run)
text : str (the claim)
variables : list[str] (what changes between conditions)
metric : str (what the runner will measure)
baseline_ref : str | None (which paper or run the comparison cites)
draft_pass : int (which sampler pass produced this)
temperature : float (the sampler setting at draft time)
novelty_score : float (distance from prior survivors, 0..1)
rank_score : float (weighted sum used for ordering)
```
`variables` and `metric` are not free text. The parser pulls them from a tagged response. The runner in lesson fifty-two reads these fields directly when it builds the experiment config.
`baseline_ref` is optional but recommended. The evaluator in lesson fifty-three needs a baseline to compare against. If the hypothesis omits one, the evaluator falls back to the previous run on the same metric.
## Architecture
```mermaid
flowchart TD
A[seed prompt] --> B[temperature ramp]
B --> C[mock language model draft]
C --> D[parse tagged response]
D --> E{novelty filter}
E -- duplicate --> F[discard]
E -- novel --> G[append to survivors]
G --> H{pass budget hit}
H -- no --> B
H -- yes --> I[rank survivors]
I --> J[hypothesis queue]
```
The loop is straight forward. The interesting part is each box has a hard contract.
## Temperature ramp
Start at `t_min`, end at `t_max`, step `(t_max - t_min) / passes`. Each pass calls the sampler at the current temperature. The mock model honors temperature by switching between a small set of scripted responses keyed on `(prompt, temp_bucket)`. The buckets are open intervals so a small change in temperature picks a different bucket and produces a different draft. In production the sampler would be a real model with `temperature=t` passed through.
The default schedule is six passes from `0.2` to `1.2`. Six is enough to fill the queue without paying for samples that the novelty filter will reject anyway. Below `0.2` the model parrots the seed back. Above `1.2` the responses tend to drift off topic and fail the parser.
## Novelty filter
After each draft is parsed, the generator embeds the text and compares against every accepted hypothesis. The embedding is a small hashed bag of word tokens, normalised to unit length. Cosine distance between two unit vectors is `1 - dot(a, b)`. A draft passes if its minimum distance to any prior survivor is above `novelty_threshold`. Default is `0.25`.
The hashed embedding is not fancy. It is deterministic, has zero dependencies, and is enough to catch the obvious case: two drafts that share most of their nouns. A production deployment would swap in a small sentence model. The interface stays the same.
## Rank score
```text
rank_score = w_novelty * novelty_score
+ w_specificity * specificity_score
+ w_testability * testability_score
```
Three sub scores. `novelty_score` is the minimum embedding distance from prior survivors. `specificity_score` is the count of concrete variables in the hypothesis divided by a target count. `testability_score` is one if the hypothesis specifies both a metric and a baseline, half if it only has a metric, zero otherwise.
Default weights are `0.4`, `0.3`, `0.3`. The weights live in the generator config so a downstream lesson can shift them without forking the code.
## Mock language model
```python
class MockLLM:
def sample(self, prompt: str, temperature: float, seed: int) -> str:
...
```
The sampler is deterministic given a `(prompt, temperature, seed)` triple. The mock keeps a scripted response table keyed on `(prompt_signature, temperature_bucket)`. If the table has no entry for a key, the sampler returns a fallback that fails the parser. The fallback path is exercised by one of the tests.
The seed is mixed into the response so the same `(prompt, temperature)` pair with different seeds produces different drafts. In tests we pin the seed to keep results reproducible. In a real deployment the seed would come from a system clock or a counter.
## Output queue
The output is a list of `Hypothesis` records sorted by `rank_score` descending. The runner in lesson fifty-two pops the head, runs the experiment, and the evaluator in lesson fifty-three writes a verdict back. If the verdict says the hypothesis was wrong, the runner pops the next one.
The queue is finite. When it is empty the orchestrator can either widen the seed prompt and run the generator again or stop and report the budget exhausted.
## How to read the code
`code/main.py` defines `Hypothesis`, `MockLLM`, `HypothesisGenerator`, and a deterministic demo. The generator exposes a single `run(seed_prompt, n_passes)` method that returns a sorted queue. The embedding is a hashed bag of tokens. The novelty filter is a single function. The rank score is a single function. Nothing depends on `numpy`; the embedding math is pure stdlib so the lesson stays portable.
`code/tests/test_generator.py` covers the linear path, the duplicate rejection path, the parser failure path, the temperature ramp boundaries, and the rank ordering.
## Where this slots in
Lesson fifty produces the queue. Lesson fifty-one takes the head of the queue and runs a literature search to confirm or refute it. Lesson fifty-two takes the same head and runs an actual experiment. Lesson fifty-three reads both outputs and writes a verdict. The four lessons compose into a research loop with no human in it; a human can step in at any boundary.
@@ -0,0 +1,78 @@
{
"lesson": "50-hypothesis-generator",
"title": "Hypothesis Generator",
"questions": [
{
"stage": "pre",
"question": "Why does the generator produce a ranked queue instead of a single hypothesis?",
"options": [
"Because the parser cannot read a single block",
"Because the runner needs depth so it can pop the next hypothesis when the first one fails",
"Because the mock model only emits lists",
"Because the embedding requires more than one input"
],
"correct": 1,
"explanation": "The point of generating a queue is to amortise sampling cost across the loop. When the first hypothesis fails the runner pops the next without a fresh sampling pass."
},
{
"stage": "pre",
"question": "What does the temperature ramp accomplish on each pass?",
"options": [
"It raises the parser tolerance",
"It widens the sampling distribution so later drafts can land further from the seed",
"It increases the embedding dimension",
"It triples the seed value"
],
"correct": 1,
"explanation": "Higher temperature widens the sampling distribution. The ramp encourages each pass to drift further so the novelty filter has something to do."
},
{
"stage": "check",
"question": "When does the novelty filter reject a draft?",
"options": [
"When its rank score is below the threshold",
"When its minimum cosine distance to any prior survivor falls below the novelty threshold",
"When its parser passes but its tag count is wrong",
"When its draft pass is greater than the queue length"
],
"correct": 1,
"explanation": "Novelty is the minimum distance to prior survivors. If that distance is below the threshold the draft is a near duplicate and is dropped."
},
{
"stage": "check",
"question": "Which three components combine in the rank score?",
"options": [
"Latency, throughput, cost",
"Novelty, specificity, testability",
"Temperature, seed, pass index",
"Variables, metric, baseline length"
],
"correct": 1,
"explanation": "The rank score is a weighted sum of novelty, specificity, and testability. Each sub score lives between zero and one."
},
{
"stage": "check",
"question": "Why is the mock language model keyed on a temperature bucket rather than the raw float?",
"options": [
"Because floats cannot be hashed",
"Because buckets make the schedule discrete so a small temperature change can pick a different scripted draft",
"Because the parser needs an integer",
"Because the embedding requires it"
],
"correct": 1,
"explanation": "Buckets discretise the continuous schedule. Two adjacent temperatures can map to different buckets and pull different drafts from the scripted bank, which is how the mock simulates varied sampling."
},
{
"stage": "check",
"question": "What happens if every draft from the mock model fails the parser?",
"options": [
"The generator raises a hard error",
"The queue is empty and each pass logs a parse rejection so the failure mode is auditable",
"The runner retries with a fresh prompt",
"The novelty threshold is lowered automatically"
],
"correct": 1,
"explanation": "Parser failures are recorded as logs with a parse reject reason. The queue can come back empty without crashing the loop, and the logs explain why."
}
]
}