mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 01:54:39 +08:00
feat(phase-19/50): add hypothesis-generator deep capstone
This commit is contained in:
@@ -0,0 +1,311 @@
|
||||
"""Hypothesis generator: temperature ramped sampling, novelty filter, ranked queue.
|
||||
|
||||
Conceptual references:
|
||||
- ./docs/en.md (this lesson)
|
||||
- Phase 19 Track A lessons 20-29 (agent harness primitives)
|
||||
|
||||
Stdlib only. Run: python3 code/main.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Callable
|
||||
|
||||
|
||||
HASH_DIM = 128
|
||||
TAG_RE = re.compile(
|
||||
r"<hypothesis>\s*"
|
||||
r"<text>(?P<text>.*?)</text>\s*"
|
||||
r"<variables>(?P<variables>.*?)</variables>\s*"
|
||||
r"<metric>(?P<metric>.*?)</metric>\s*"
|
||||
r"(?:<baseline>(?P<baseline>.*?)</baseline>\s*)?"
|
||||
r"</hypothesis>",
|
||||
re.DOTALL,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Hypothesis:
|
||||
id: int
|
||||
text: str
|
||||
variables: list[str]
|
||||
metric: str
|
||||
baseline_ref: str | None
|
||||
draft_pass: int
|
||||
temperature: float
|
||||
novelty_score: float = 0.0
|
||||
rank_score: float = 0.0
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"id": self.id,
|
||||
"text": self.text,
|
||||
"variables": list(self.variables),
|
||||
"metric": self.metric,
|
||||
"baseline_ref": self.baseline_ref,
|
||||
"draft_pass": self.draft_pass,
|
||||
"temperature": round(self.temperature, 3),
|
||||
"novelty_score": round(self.novelty_score, 4),
|
||||
"rank_score": round(self.rank_score, 4),
|
||||
}
|
||||
|
||||
|
||||
class ParserError(ValueError):
|
||||
"""Raised when a sampler response does not match the hypothesis tag schema."""
|
||||
|
||||
|
||||
def _tokenise(text: str) -> list[str]:
|
||||
return re.findall(r"[a-z0-9]+", text.lower())
|
||||
|
||||
|
||||
def hashed_embed(text: str, dim: int = HASH_DIM) -> list[float]:
|
||||
"""Hashed bag of tokens embedding, L2 normalised. Deterministic stdlib only."""
|
||||
vec = [0.0] * dim
|
||||
for tok in _tokenise(text):
|
||||
h = hashlib.md5(tok.encode("utf-8")).digest()
|
||||
idx = int.from_bytes(h[:4], "big") % dim
|
||||
sign = 1.0 if (h[4] & 1) == 0 else -1.0
|
||||
vec[idx] += sign
|
||||
norm = math.sqrt(sum(v * v for v in vec))
|
||||
if norm == 0.0:
|
||||
return vec
|
||||
return [v / norm for v in vec]
|
||||
|
||||
|
||||
def cosine_distance(a: list[float], b: list[float]) -> float:
|
||||
dot = sum(x * y for x, y in zip(a, b))
|
||||
dot = max(-1.0, min(1.0, dot))
|
||||
return 1.0 - dot
|
||||
|
||||
|
||||
def parse_response(raw: str) -> dict:
|
||||
match = TAG_RE.search(raw)
|
||||
if match is None:
|
||||
raise ParserError("no hypothesis block found")
|
||||
text = match.group("text").strip()
|
||||
if not text:
|
||||
raise ParserError("empty text")
|
||||
metric = match.group("metric").strip()
|
||||
if not metric:
|
||||
raise ParserError("empty metric")
|
||||
raw_vars = match.group("variables").strip()
|
||||
variables = [v.strip() for v in raw_vars.split(",") if v.strip()]
|
||||
if not variables:
|
||||
raise ParserError("empty variables")
|
||||
baseline = match.group("baseline")
|
||||
baseline_ref = baseline.strip() if baseline and baseline.strip() else None
|
||||
return {
|
||||
"text": text,
|
||||
"variables": variables,
|
||||
"metric": metric,
|
||||
"baseline_ref": baseline_ref,
|
||||
}
|
||||
|
||||
|
||||
def temperature_bucket(temperature: float) -> int:
|
||||
"""Map a continuous temperature to a discrete bucket index."""
|
||||
if temperature < 0.35:
|
||||
return 0
|
||||
if temperature < 0.65:
|
||||
return 1
|
||||
if temperature < 0.95:
|
||||
return 2
|
||||
return 3
|
||||
|
||||
|
||||
class MockLLM:
|
||||
"""Scripted sampler keyed on (prompt_signature, temperature_bucket).
|
||||
|
||||
The seed is folded into the response so identical prompts and buckets with
|
||||
different seeds yield distinct drafts. Unknown keys return an unparseable
|
||||
fallback so the parser-failure path is reachable from tests.
|
||||
"""
|
||||
|
||||
def __init__(self, scripts: dict[tuple[str, int], list[str]]) -> None:
|
||||
self._scripts = dict(scripts)
|
||||
|
||||
@staticmethod
|
||||
def prompt_signature(prompt: str) -> str:
|
||||
return hashlib.sha1(prompt.encode("utf-8")).hexdigest()[:10]
|
||||
|
||||
def sample(self, prompt: str, temperature: float, seed: int) -> str:
|
||||
key = (self.prompt_signature(prompt), temperature_bucket(temperature))
|
||||
bank = self._scripts.get(key)
|
||||
if not bank:
|
||||
return "<noise>untagged drift</noise>"
|
||||
return bank[seed % len(bank)]
|
||||
|
||||
|
||||
@dataclass
|
||||
class GeneratorConfig:
|
||||
n_passes: int = 6
|
||||
t_min: float = 0.2
|
||||
t_max: float = 1.2
|
||||
novelty_threshold: float = 0.25
|
||||
target_variable_count: int = 3
|
||||
w_novelty: float = 0.4
|
||||
w_specificity: float = 0.3
|
||||
w_testability: float = 0.3
|
||||
base_seed: int = 0
|
||||
|
||||
def schedule(self) -> list[float]:
|
||||
if self.n_passes <= 0:
|
||||
return []
|
||||
if self.n_passes == 1:
|
||||
return [self.t_min]
|
||||
step = (self.t_max - self.t_min) / (self.n_passes - 1)
|
||||
return [self.t_min + i * step for i in range(self.n_passes)]
|
||||
|
||||
|
||||
@dataclass
|
||||
class GenerationLog:
|
||||
pass_index: int
|
||||
temperature: float
|
||||
seed: int
|
||||
accepted_id: int | None
|
||||
reject_reason: str | None
|
||||
raw_excerpt: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"pass": self.pass_index,
|
||||
"temperature": round(self.temperature, 3),
|
||||
"seed": self.seed,
|
||||
"accepted_id": self.accepted_id,
|
||||
"reject_reason": self.reject_reason,
|
||||
"raw_excerpt": self.raw_excerpt[:80],
|
||||
}
|
||||
|
||||
|
||||
class HypothesisGenerator:
|
||||
"""Drives the mock LLM over a temperature schedule and ranks the survivors."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
llm: MockLLM,
|
||||
config: GeneratorConfig | None = None,
|
||||
embedder: Callable[[str], list[float]] = hashed_embed,
|
||||
) -> None:
|
||||
self._llm = llm
|
||||
self._cfg = config or GeneratorConfig()
|
||||
self._embed = embedder
|
||||
|
||||
def _specificity_score(self, h: Hypothesis) -> float:
|
||||
target = max(1, self._cfg.target_variable_count)
|
||||
return min(1.0, len(h.variables) / target)
|
||||
|
||||
def _testability_score(self, h: Hypothesis) -> float:
|
||||
if h.metric and h.baseline_ref:
|
||||
return 1.0
|
||||
if h.metric:
|
||||
return 0.5
|
||||
return 0.0
|
||||
|
||||
def _score(self, h: Hypothesis) -> float:
|
||||
return (
|
||||
self._cfg.w_novelty * h.novelty_score
|
||||
+ self._cfg.w_specificity * self._specificity_score(h)
|
||||
+ self._cfg.w_testability * self._testability_score(h)
|
||||
)
|
||||
|
||||
def _novelty(self, candidate: list[float], survivors: list[list[float]]) -> float:
|
||||
if not survivors:
|
||||
return 1.0
|
||||
return min(cosine_distance(candidate, s) for s in survivors)
|
||||
|
||||
def run(self, seed_prompt: str) -> tuple[list[Hypothesis], list[GenerationLog]]:
|
||||
survivors: list[Hypothesis] = []
|
||||
survivor_vecs: list[list[float]] = []
|
||||
logs: list[GenerationLog] = []
|
||||
next_id = 1
|
||||
for pass_index, temperature in enumerate(self._cfg.schedule()):
|
||||
seed = self._cfg.base_seed + pass_index
|
||||
raw = self._llm.sample(seed_prompt, temperature, seed)
|
||||
try:
|
||||
parsed = parse_response(raw)
|
||||
except ParserError as exc:
|
||||
logs.append(GenerationLog(pass_index, temperature, seed, None, f"parse:{exc}", raw))
|
||||
continue
|
||||
vec = self._embed(parsed["text"])
|
||||
novelty = self._novelty(vec, survivor_vecs)
|
||||
if novelty < self._cfg.novelty_threshold:
|
||||
logs.append(GenerationLog(pass_index, temperature, seed, None, "duplicate", raw))
|
||||
continue
|
||||
hypothesis = Hypothesis(
|
||||
id=next_id,
|
||||
text=parsed["text"],
|
||||
variables=parsed["variables"],
|
||||
metric=parsed["metric"],
|
||||
baseline_ref=parsed["baseline_ref"],
|
||||
draft_pass=pass_index,
|
||||
temperature=temperature,
|
||||
novelty_score=novelty,
|
||||
)
|
||||
hypothesis.rank_score = self._score(hypothesis)
|
||||
survivors.append(hypothesis)
|
||||
survivor_vecs.append(vec)
|
||||
logs.append(GenerationLog(pass_index, temperature, seed, next_id, None, raw))
|
||||
next_id += 1
|
||||
survivors.sort(key=lambda h: (-h.rank_score, h.id))
|
||||
return survivors, logs
|
||||
|
||||
|
||||
def build_demo_scripts() -> dict[tuple[str, int], list[str]]:
|
||||
"""Scripted responses for the demo seed prompt across temperature buckets."""
|
||||
seed_prompt = "Investigate attention sparsity in small transformers"
|
||||
sig = MockLLM.prompt_signature(seed_prompt)
|
||||
return {
|
||||
(sig, 0): [
|
||||
"<hypothesis>"
|
||||
"<text>Lowering attention head count from 8 to 4 raises validation loss by less than 2 percent on a 12M parameter model.</text>"
|
||||
"<variables>head_count, validation_loss</variables>"
|
||||
"<metric>validation_loss</metric>"
|
||||
"<baseline>head_count_8</baseline>"
|
||||
"</hypothesis>",
|
||||
],
|
||||
(sig, 1): [
|
||||
"<hypothesis>"
|
||||
"<text>Top-k sparse attention with k equal to 16 matches dense attention on perplexity at 12M parameters.</text>"
|
||||
"<variables>k, perplexity, parameter_count</variables>"
|
||||
"<metric>perplexity</metric>"
|
||||
"<baseline>dense_attention</baseline>"
|
||||
"</hypothesis>",
|
||||
],
|
||||
(sig, 2): [
|
||||
"<hypothesis>"
|
||||
"<text>Routing attention through a learned gate reduces flops by 30 percent without harming downstream accuracy.</text>"
|
||||
"<variables>gate_temperature, flops, accuracy</variables>"
|
||||
"<metric>downstream_accuracy</metric>"
|
||||
"<baseline>dense_attention</baseline>"
|
||||
"</hypothesis>",
|
||||
],
|
||||
(sig, 3): [
|
||||
"<hypothesis>"
|
||||
"<text>Block sparse attention with block size 32 lowers wall clock training time by 18 percent on consumer GPUs.</text>"
|
||||
"<variables>block_size, training_seconds, hardware</variables>"
|
||||
"<metric>training_seconds</metric>"
|
||||
"<baseline>dense_attention</baseline>"
|
||||
"</hypothesis>",
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _demo() -> None:
|
||||
llm = MockLLM(build_demo_scripts())
|
||||
config = GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1)
|
||||
generator = HypothesisGenerator(llm, config)
|
||||
queue, logs = generator.run("Investigate attention sparsity in small transformers")
|
||||
print(json.dumps({
|
||||
"queue_size": len(queue),
|
||||
"queue": [h.to_dict() for h in queue],
|
||||
"logs": [log.to_dict() for log in logs],
|
||||
}, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_demo()
|
||||
@@ -0,0 +1,156 @@
|
||||
"""Tests for HypothesisGenerator: linear queue, dedup, parser, schedule, rank order."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
sys.path.insert(0, os.path.dirname(HERE))
|
||||
|
||||
from main import ( # noqa: E402
|
||||
GeneratorConfig,
|
||||
HypothesisGenerator,
|
||||
MockLLM,
|
||||
ParserError,
|
||||
build_demo_scripts,
|
||||
cosine_distance,
|
||||
hashed_embed,
|
||||
parse_response,
|
||||
temperature_bucket,
|
||||
)
|
||||
|
||||
|
||||
SEED_PROMPT = "Investigate attention sparsity in small transformers"
|
||||
|
||||
|
||||
class TestParser(unittest.TestCase):
|
||||
def test_parses_full_block(self) -> None:
|
||||
raw = (
|
||||
"<hypothesis><text>x</text><variables>a, b</variables>"
|
||||
"<metric>m</metric><baseline>r</baseline></hypothesis>"
|
||||
)
|
||||
parsed = parse_response(raw)
|
||||
self.assertEqual(parsed["text"], "x")
|
||||
self.assertEqual(parsed["variables"], ["a", "b"])
|
||||
self.assertEqual(parsed["metric"], "m")
|
||||
self.assertEqual(parsed["baseline_ref"], "r")
|
||||
|
||||
def test_baseline_optional(self) -> None:
|
||||
raw = "<hypothesis><text>x</text><variables>a</variables><metric>m</metric></hypothesis>"
|
||||
self.assertIsNone(parse_response(raw)["baseline_ref"])
|
||||
|
||||
def test_rejects_unparseable(self) -> None:
|
||||
with self.assertRaises(ParserError):
|
||||
parse_response("plain text no tags")
|
||||
|
||||
def test_rejects_empty_variables(self) -> None:
|
||||
raw = "<hypothesis><text>x</text><variables> </variables><metric>m</metric></hypothesis>"
|
||||
with self.assertRaises(ParserError):
|
||||
parse_response(raw)
|
||||
|
||||
|
||||
class TestEmbedding(unittest.TestCase):
|
||||
def test_unit_norm(self) -> None:
|
||||
vec = hashed_embed("attention sparsity small transformer")
|
||||
n = sum(v * v for v in vec) ** 0.5
|
||||
self.assertAlmostEqual(n, 1.0, places=5)
|
||||
|
||||
def test_distance_self_zero(self) -> None:
|
||||
v = hashed_embed("identical text identical text")
|
||||
self.assertAlmostEqual(cosine_distance(v, v), 0.0, places=5)
|
||||
|
||||
def test_distance_disjoint_high(self) -> None:
|
||||
a = hashed_embed("attention sparsity transformer")
|
||||
b = hashed_embed("dataloader checkpoint scheduler")
|
||||
self.assertGreater(cosine_distance(a, b), 0.5)
|
||||
|
||||
|
||||
class TestTemperatureRamp(unittest.TestCase):
|
||||
def test_schedule_endpoints(self) -> None:
|
||||
cfg = GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1)
|
||||
schedule = cfg.schedule()
|
||||
self.assertEqual(len(schedule), 4)
|
||||
self.assertAlmostEqual(schedule[0], 0.2)
|
||||
self.assertAlmostEqual(schedule[-1], 1.1)
|
||||
|
||||
def test_schedule_one_pass(self) -> None:
|
||||
cfg = GeneratorConfig(n_passes=1, t_min=0.5, t_max=1.2)
|
||||
self.assertEqual(cfg.schedule(), [0.5])
|
||||
|
||||
def test_schedule_zero_passes(self) -> None:
|
||||
self.assertEqual(GeneratorConfig(n_passes=0).schedule(), [])
|
||||
|
||||
def test_bucket_boundaries(self) -> None:
|
||||
self.assertEqual(temperature_bucket(0.2), 0)
|
||||
self.assertEqual(temperature_bucket(0.5), 1)
|
||||
self.assertEqual(temperature_bucket(0.8), 2)
|
||||
self.assertEqual(temperature_bucket(1.1), 3)
|
||||
|
||||
|
||||
class TestGenerator(unittest.TestCase):
|
||||
def test_demo_path_produces_queue(self) -> None:
|
||||
gen = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
|
||||
queue, logs = gen.run(SEED_PROMPT)
|
||||
self.assertEqual(len(queue), 4)
|
||||
self.assertEqual(len(logs), 4)
|
||||
for log in logs:
|
||||
self.assertIsNone(log.reject_reason)
|
||||
ids = [h.id for h in queue]
|
||||
self.assertEqual(sorted(ids), [1, 2, 3, 4])
|
||||
|
||||
def test_queue_sorted_by_rank_desc(self) -> None:
|
||||
gen = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
|
||||
queue, _ = gen.run(SEED_PROMPT)
|
||||
scores = [h.rank_score for h in queue]
|
||||
self.assertEqual(scores, sorted(scores, reverse=True))
|
||||
|
||||
def test_duplicate_rejected(self) -> None:
|
||||
sig = MockLLM.prompt_signature(SEED_PROMPT)
|
||||
repeated = (
|
||||
"<hypothesis><text>head count eight to four loss two percent</text>"
|
||||
"<variables>head_count, loss</variables><metric>loss</metric>"
|
||||
"<baseline>head_count_8</baseline></hypothesis>"
|
||||
)
|
||||
scripts = {(sig, 0): [repeated], (sig, 1): [repeated], (sig, 2): [repeated], (sig, 3): [repeated]}
|
||||
gen = HypothesisGenerator(MockLLM(scripts), GeneratorConfig(n_passes=4))
|
||||
queue, logs = gen.run(SEED_PROMPT)
|
||||
self.assertEqual(len(queue), 1)
|
||||
reject_reasons = [log.reject_reason for log in logs if log.reject_reason]
|
||||
self.assertEqual(reject_reasons, ["duplicate", "duplicate", "duplicate"])
|
||||
|
||||
def test_parser_failure_logged(self) -> None:
|
||||
sig = MockLLM.prompt_signature(SEED_PROMPT)
|
||||
scripts = {(sig, 0): ["plain text"], (sig, 1): build_demo_scripts()[(sig, 1)]}
|
||||
gen = HypothesisGenerator(MockLLM(scripts), GeneratorConfig(n_passes=2, t_min=0.2, t_max=0.6))
|
||||
queue, logs = gen.run(SEED_PROMPT)
|
||||
self.assertEqual(len(queue), 1)
|
||||
self.assertTrue(logs[0].reject_reason.startswith("parse:"))
|
||||
|
||||
def test_unknown_prompt_falls_back_and_drops(self) -> None:
|
||||
gen = HypothesisGenerator(MockLLM({}), GeneratorConfig(n_passes=3))
|
||||
queue, logs = gen.run("never seen prompt")
|
||||
self.assertEqual(queue, [])
|
||||
self.assertTrue(all(log.reject_reason and log.reject_reason.startswith("parse:") for log in logs))
|
||||
|
||||
def test_specificity_weight_changes_rank(self) -> None:
|
||||
cfg_a = GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1, w_specificity=1.0, w_novelty=0.0, w_testability=0.0)
|
||||
gen = HypothesisGenerator(MockLLM(build_demo_scripts()), cfg_a)
|
||||
queue, _ = gen.run(SEED_PROMPT)
|
||||
for h in queue:
|
||||
self.assertGreaterEqual(h.rank_score, 0.0)
|
||||
self.assertLessEqual(h.rank_score, 1.0)
|
||||
|
||||
|
||||
class TestDeterminism(unittest.TestCase):
|
||||
def test_two_runs_identical(self) -> None:
|
||||
gen_a = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
|
||||
gen_b = HypothesisGenerator(MockLLM(build_demo_scripts()), GeneratorConfig(n_passes=4, t_min=0.2, t_max=1.1))
|
||||
queue_a, _ = gen_a.run(SEED_PROMPT)
|
||||
queue_b, _ = gen_b.run(SEED_PROMPT)
|
||||
self.assertEqual([h.to_dict() for h in queue_a], [h.to_dict() for h in queue_b])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,112 @@
|
||||
# Hypothesis Generator
|
||||
|
||||
> A research agent that asks the same question twice is wasting tokens. The trick is forcing each draft to land somewhere new.
|
||||
|
||||
**Type:** Build
|
||||
**Languages:** Python
|
||||
**Prerequisites:** Phase 19 Track A lessons 20-29
|
||||
**Time:** ~90 minutes
|
||||
|
||||
## Learning Objectives
|
||||
- Drive a sampler from a seed prompt and turn its outputs into typed hypothesis records.
|
||||
- Ramp the sampler temperature on each pass so the next draft drifts further from the last.
|
||||
- Filter near duplicates with a small embedding model and a cosine distance threshold.
|
||||
- Rank the survivors with a scoring function that blends novelty, specificity, and testability.
|
||||
- Hold every step deterministic so the same seed always produces the same queue.
|
||||
|
||||
## Why generate, then filter
|
||||
|
||||
A planner that asks one model one time gets one hypothesis. That is fine for a worked example. For a research loop it is the wrong shape. The loop wants a ranked queue with depth, so when the first hypothesis fails the runner has the next one ready without paying for another full sampling pass.
|
||||
|
||||
Two ideas combine to produce that queue. The first is temperature ramping: each pass through the sampler raises the temperature a notch, so later drafts are encouraged to wander. The second is novelty filtering: after each draft, the generator measures the embedding distance from every prior survivor and rejects anything inside the cluster.
|
||||
|
||||
The lesson ships a mock language model that returns scripted token sequences for fixed prompts. The mock is enough to exercise the full path: seed prompt in, temperature ramp applied, candidates parsed, novelty filter run, ranked queue out.
|
||||
|
||||
## The Hypothesis shape
|
||||
|
||||
```text
|
||||
Hypothesis
|
||||
id : int (monotonic within a run)
|
||||
text : str (the claim)
|
||||
variables : list[str] (what changes between conditions)
|
||||
metric : str (what the runner will measure)
|
||||
baseline_ref : str | None (which paper or run the comparison cites)
|
||||
draft_pass : int (which sampler pass produced this)
|
||||
temperature : float (the sampler setting at draft time)
|
||||
novelty_score : float (distance from prior survivors, 0..1)
|
||||
rank_score : float (weighted sum used for ordering)
|
||||
```
|
||||
|
||||
`variables` and `metric` are not free text. The parser pulls them from a tagged response. The runner in lesson fifty-two reads these fields directly when it builds the experiment config.
|
||||
|
||||
`baseline_ref` is optional but recommended. The evaluator in lesson fifty-three needs a baseline to compare against. If the hypothesis omits one, the evaluator falls back to the previous run on the same metric.
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A[seed prompt] --> B[temperature ramp]
|
||||
B --> C[mock language model draft]
|
||||
C --> D[parse tagged response]
|
||||
D --> E{novelty filter}
|
||||
E -- duplicate --> F[discard]
|
||||
E -- novel --> G[append to survivors]
|
||||
G --> H{pass budget hit}
|
||||
H -- no --> B
|
||||
H -- yes --> I[rank survivors]
|
||||
I --> J[hypothesis queue]
|
||||
```
|
||||
|
||||
The loop is straight forward. The interesting part is each box has a hard contract.
|
||||
|
||||
## Temperature ramp
|
||||
|
||||
Start at `t_min`, end at `t_max`, step `(t_max - t_min) / passes`. Each pass calls the sampler at the current temperature. The mock model honors temperature by switching between a small set of scripted responses keyed on `(prompt, temp_bucket)`. The buckets are open intervals so a small change in temperature picks a different bucket and produces a different draft. In production the sampler would be a real model with `temperature=t` passed through.
|
||||
|
||||
The default schedule is six passes from `0.2` to `1.2`. Six is enough to fill the queue without paying for samples that the novelty filter will reject anyway. Below `0.2` the model parrots the seed back. Above `1.2` the responses tend to drift off topic and fail the parser.
|
||||
|
||||
## Novelty filter
|
||||
|
||||
After each draft is parsed, the generator embeds the text and compares against every accepted hypothesis. The embedding is a small hashed bag of word tokens, normalised to unit length. Cosine distance between two unit vectors is `1 - dot(a, b)`. A draft passes if its minimum distance to any prior survivor is above `novelty_threshold`. Default is `0.25`.
|
||||
|
||||
The hashed embedding is not fancy. It is deterministic, has zero dependencies, and is enough to catch the obvious case: two drafts that share most of their nouns. A production deployment would swap in a small sentence model. The interface stays the same.
|
||||
|
||||
## Rank score
|
||||
|
||||
```text
|
||||
rank_score = w_novelty * novelty_score
|
||||
+ w_specificity * specificity_score
|
||||
+ w_testability * testability_score
|
||||
```
|
||||
|
||||
Three sub scores. `novelty_score` is the minimum embedding distance from prior survivors. `specificity_score` is the count of concrete variables in the hypothesis divided by a target count. `testability_score` is one if the hypothesis specifies both a metric and a baseline, half if it only has a metric, zero otherwise.
|
||||
|
||||
Default weights are `0.4`, `0.3`, `0.3`. The weights live in the generator config so a downstream lesson can shift them without forking the code.
|
||||
|
||||
## Mock language model
|
||||
|
||||
```python
|
||||
class MockLLM:
|
||||
def sample(self, prompt: str, temperature: float, seed: int) -> str:
|
||||
...
|
||||
```
|
||||
|
||||
The sampler is deterministic given a `(prompt, temperature, seed)` triple. The mock keeps a scripted response table keyed on `(prompt_signature, temperature_bucket)`. If the table has no entry for a key, the sampler returns a fallback that fails the parser. The fallback path is exercised by one of the tests.
|
||||
|
||||
The seed is mixed into the response so the same `(prompt, temperature)` pair with different seeds produces different drafts. In tests we pin the seed to keep results reproducible. In a real deployment the seed would come from a system clock or a counter.
|
||||
|
||||
## Output queue
|
||||
|
||||
The output is a list of `Hypothesis` records sorted by `rank_score` descending. The runner in lesson fifty-two pops the head, runs the experiment, and the evaluator in lesson fifty-three writes a verdict back. If the verdict says the hypothesis was wrong, the runner pops the next one.
|
||||
|
||||
The queue is finite. When it is empty the orchestrator can either widen the seed prompt and run the generator again or stop and report the budget exhausted.
|
||||
|
||||
## How to read the code
|
||||
|
||||
`code/main.py` defines `Hypothesis`, `MockLLM`, `HypothesisGenerator`, and a deterministic demo. The generator exposes a single `run(seed_prompt, n_passes)` method that returns a sorted queue. The embedding is a hashed bag of tokens. The novelty filter is a single function. The rank score is a single function. Nothing depends on `numpy`; the embedding math is pure stdlib so the lesson stays portable.
|
||||
|
||||
`code/tests/test_generator.py` covers the linear path, the duplicate rejection path, the parser failure path, the temperature ramp boundaries, and the rank ordering.
|
||||
|
||||
## Where this slots in
|
||||
|
||||
Lesson fifty produces the queue. Lesson fifty-one takes the head of the queue and runs a literature search to confirm or refute it. Lesson fifty-two takes the same head and runs an actual experiment. Lesson fifty-three reads both outputs and writes a verdict. The four lessons compose into a research loop with no human in it; a human can step in at any boundary.
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"lesson": "50-hypothesis-generator",
|
||||
"title": "Hypothesis Generator",
|
||||
"questions": [
|
||||
{
|
||||
"stage": "pre",
|
||||
"question": "Why does the generator produce a ranked queue instead of a single hypothesis?",
|
||||
"options": [
|
||||
"Because the parser cannot read a single block",
|
||||
"Because the runner needs depth so it can pop the next hypothesis when the first one fails",
|
||||
"Because the mock model only emits lists",
|
||||
"Because the embedding requires more than one input"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "The point of generating a queue is to amortise sampling cost across the loop. When the first hypothesis fails the runner pops the next without a fresh sampling pass."
|
||||
},
|
||||
{
|
||||
"stage": "pre",
|
||||
"question": "What does the temperature ramp accomplish on each pass?",
|
||||
"options": [
|
||||
"It raises the parser tolerance",
|
||||
"It widens the sampling distribution so later drafts can land further from the seed",
|
||||
"It increases the embedding dimension",
|
||||
"It triples the seed value"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "Higher temperature widens the sampling distribution. The ramp encourages each pass to drift further so the novelty filter has something to do."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "When does the novelty filter reject a draft?",
|
||||
"options": [
|
||||
"When its rank score is below the threshold",
|
||||
"When its minimum cosine distance to any prior survivor falls below the novelty threshold",
|
||||
"When its parser passes but its tag count is wrong",
|
||||
"When its draft pass is greater than the queue length"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "Novelty is the minimum distance to prior survivors. If that distance is below the threshold the draft is a near duplicate and is dropped."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "Which three components combine in the rank score?",
|
||||
"options": [
|
||||
"Latency, throughput, cost",
|
||||
"Novelty, specificity, testability",
|
||||
"Temperature, seed, pass index",
|
||||
"Variables, metric, baseline length"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "The rank score is a weighted sum of novelty, specificity, and testability. Each sub score lives between zero and one."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "Why is the mock language model keyed on a temperature bucket rather than the raw float?",
|
||||
"options": [
|
||||
"Because floats cannot be hashed",
|
||||
"Because buckets make the schedule discrete so a small temperature change can pick a different scripted draft",
|
||||
"Because the parser needs an integer",
|
||||
"Because the embedding requires it"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "Buckets discretise the continuous schedule. Two adjacent temperatures can map to different buckets and pull different drafts from the scripted bank, which is how the mock simulates varied sampling."
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "What happens if every draft from the mock model fails the parser?",
|
||||
"options": [
|
||||
"The generator raises a hard error",
|
||||
"The queue is empty and each pass logs a parse rejection so the failure mode is auditable",
|
||||
"The runner retries with a fresh prompt",
|
||||
"The novelty threshold is lowered automatically"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": "Parser failures are recorded as logs with a parse reject reason. The queue can come back empty without crashing the loop, and the logs explain why."
|
||||
}
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user