feat(phase-19/57): add end-to-end-research-demo deep capstone

This commit is contained in:
Rohit Ghumare
2026-05-26 19:30:42 +01:00
parent e63bd43368
commit 15fc990cec
4 changed files with 595 additions and 0 deletions
@@ -0,0 +1,281 @@
"""End-to-end auto-research demo: seed -> scheduler -> critic loop -> paper writer.
Conceptual references:
- ./docs/en.md (this lesson)
- Phase 19 lesson 54 (paper writer)
- Phase 19 lesson 55 (critic loop)
- Phase 19 lesson 56 (iteration scheduler)
- Phase 19 lessons 50-53 (earlier auto-research stages; the seed/runner stub here stands in for them)
Stdlib + numpy only. Run: python3 code/main.py
"""
from __future__ import annotations
import asyncio
import json
import os
import sys
import tempfile
from dataclasses import dataclass, field
from typing import Awaitable, Callable
HERE = os.path.dirname(os.path.abspath(__file__))
LESSON_ROOT = os.path.dirname(os.path.dirname(HERE))
def _add(path: str) -> None:
if path not in sys.path:
sys.path.insert(0, path)
_add(os.path.join(LESSON_ROOT, "54-paper-writer", "code"))
_add(os.path.join(LESSON_ROOT, "55-critic-loop", "code"))
_add(os.path.join(LESSON_ROOT, "56-iteration-scheduler", "code"))
import importlib
import importlib.util
def _load_module(name: str, file_path: str):
spec = importlib.util.spec_from_file_location(name, file_path)
if spec is None or spec.loader is None:
raise ImportError(f"cannot load {name} from {file_path}")
mod = importlib.util.module_from_spec(spec)
sys.modules[name] = mod
spec.loader.exec_module(mod)
return mod
paper_writer_mod = _load_module(
"_t_d2_paper_writer",
os.path.join(LESSON_ROOT, "54-paper-writer", "code", "main.py"),
)
critic_loop_mod = _load_module(
"_t_d2_critic_loop",
os.path.join(LESSON_ROOT, "55-critic-loop", "code", "main.py"),
)
scheduler_mod = _load_module(
"_t_d2_scheduler",
os.path.join(LESSON_ROOT, "56-iteration-scheduler", "code", "main.py"),
)
Paper = paper_writer_mod.Paper
Section = paper_writer_mod.Section
Figure = paper_writer_mod.Figure
BibEntry = paper_writer_mod.BibEntry
PaperWriter = paper_writer_mod.PaperWriter
PaperValidationError = paper_writer_mod.PaperValidationError
MockProseGenerator = paper_writer_mod.MockProseGenerator
MiniPaper = critic_loop_mod.MiniPaper
MiniSection = critic_loop_mod.MiniSection
CriticLoop = critic_loop_mod.CriticLoop
deterministic_critic = critic_loop_mod.deterministic_critic
deterministic_reviser = critic_loop_mod.deterministic_reviser
Hypothesis = scheduler_mod.Hypothesis
Result = scheduler_mod.Result
IterationScheduler = scheduler_mod.IterationScheduler
SchedulerReport = scheduler_mod.SchedulerReport
make_deterministic_runner = scheduler_mod.make_deterministic_runner
class NoTriggerError(Exception):
"""No branch crossed the paper threshold; demo cannot pick a best result."""
class BestResultError(Exception):
"""Picker received an empty trigger list."""
@dataclass
class DemoReport:
scheduler_report: dict
best_branch: str
best_reward: float
critic_result: dict
paper_manifest: dict
stop_reason: str
def to_dict(self) -> dict:
return {
"scheduler_report": self.scheduler_report,
"best_branch": self.best_branch,
"best_reward": self.best_reward,
"critic_result": self.critic_result,
"paper_manifest": self.paper_manifest,
"stop_reason": self.stop_reason,
}
def make_seed_hypotheses() -> list[Hypothesis]:
"""Three seed hypotheses, one per research branch. Stand-in for lessons 50-53."""
return [
Hypothesis(id="h-alpha-1", branch="alpha", payload={"q": "method-x"}),
Hypothesis(id="h-beta-1", branch="beta", payload={"q": "method-y"}),
Hypothesis(id="h-gamma-1", branch="gamma", payload={"q": "method-z"}),
]
def pick_best_branch(scheduler_report: SchedulerReport) -> tuple[str, float]:
"""Pick the branch with the highest mean reward among triggered branches.
Ties break alphabetically by branch id (deterministic).
"""
if not scheduler_report.paper_triggers:
raise NoTriggerError("no branch crossed the paper threshold")
by_branch = {b.branch: b for b in scheduler_report.branches}
triggered = [by_branch[name] for name in scheduler_report.paper_triggers if name in by_branch]
if not triggered:
raise BestResultError("trigger list empty after lookup")
triggered.sort(key=lambda b: (-b.mean, b.branch))
best = triggered[0]
return best.branch, best.mean
def _originality_for_reward(reward: float) -> str:
if reward >= 0.8:
return "high"
if reward >= 0.6:
return "medium"
return "low"
def build_mini_paper(branch: str, reward: float) -> MiniPaper:
return MiniPaper(
title=f"Auto-Research Findings on Branch {branch}",
abstract=f"We summarise the best yielding branch {branch} from the auto-research loop.",
sections=[
MiniSection(id="intro", title="Introduction", body="initial observations"),
MiniSection(id="results", title="Results", body=""),
],
originality_tag=_originality_for_reward(reward),
)
def mini_to_full_paper(mini: MiniPaper, branch: str) -> Paper:
"""Promote a converged MiniPaper into a full Paper for the writer.
Adds one figure and one bibliography entry per used citation key. Every
section's cite list is preserved; the bibliography is built from the union.
"""
cites: list[str] = []
for sec in mini.sections:
for c in sec.cites:
if c not in cites:
cites.append(c)
bib = [
BibEntry(
key=key, entry_type="article",
fields={"title": f"Source {key}", "author": "Synthesised", "year": "2026"},
)
for key in cites
]
if not bib:
bib = [BibEntry(
key=f"{branch}-baseline", entry_type="article",
fields={"title": "Baseline", "author": "Synthesised", "year": "2026"},
)]
for sec in mini.sections:
if sec.id == "intro":
sec.cites.append(f"{branch}-baseline")
break
fig = Figure(
id=f"{branch}-results",
path=f"figs/{branch}.pdf",
caption=f"Reward trajectory on branch {branch}",
)
sections = []
for s in mini.sections:
figure_refs = [fig.id] if s.id == "results" else []
sections.append(Section(
id=s.id, title=s.title, body=s.body,
cites=list(s.cites), figure_refs=figure_refs,
))
return Paper(
title=mini.title,
authors=["Auto-Research Demo"],
abstract=mini.abstract,
sections=sections,
figures=[fig],
bibliography=bib,
)
async def _run_demo_async(out_dir: str, seed: int = 11) -> DemoReport:
seed_list = make_seed_hypotheses()
if not seed_list:
raise BestResultError("seed list is empty")
runner = make_deterministic_runner(
base_rewards={"alpha": 0.82, "beta": 0.55, "gamma": 0.15},
noise=0.04,
delay_ms=2.0,
seed=seed,
)
sched = IterationScheduler(
runner=runner, slots=3, max_experiments=6,
paper_threshold=0.7, prune_floor=0.2, prune_after_runs=3,
expander=scheduler_mod.deterministic_expander,
)
sched_report = await sched.run(seed_list)
branch, reward = pick_best_branch(sched_report)
mini = build_mini_paper(branch, reward)
loop = CriticLoop(
critic=deterministic_critic,
reviser=deterministic_reviser,
max_rounds=6,
target_score=8.0,
)
critic_result = loop.run(mini)
full_paper = mini_to_full_paper(critic_result.paper, branch)
prose = MockProseGenerator(outlines={
"intro": f"motivation for branch {branch}",
"results": f"summary of reward trajectory on branch {branch}",
"method": "method description",
"related-work": "related work survey",
})
writer = PaperWriter(prose=prose)
manifest = writer.write(full_paper, out_dir)
return DemoReport(
scheduler_report=sched_report.to_dict(),
best_branch=branch,
best_reward=reward,
critic_result=critic_result.to_dict(),
paper_manifest=manifest,
stop_reason=sched_report.stop_reason,
)
def run_demo(out_dir: str | None = None, seed: int = 11) -> DemoReport:
if out_dir is None:
out_dir = tempfile.mkdtemp(prefix="auto-research-demo-")
return asyncio.run(_run_demo_async(out_dir, seed=seed))
def demo() -> dict:
return run_demo().to_dict()
if __name__ == "__main__":
rep = demo()
print(json.dumps({
"stop_reason": rep["stop_reason"],
"best_branch": rep["best_branch"],
"best_reward": rep["best_reward"],
"critic_status": rep["critic_result"]["status"],
"critic_rounds": rep["critic_result"]["rounds_used"],
"paper_sections": [s["id"] for s in rep["paper_manifest"]["sections"]],
"paper_figures": [f["id"] for f in rep["paper_manifest"]["figures"]],
"experiments_run": rep["scheduler_report"]["experiments_run"],
}, indent=2))
@@ -0,0 +1,130 @@
"""Tests for the end-to-end auto-research demo: composition, determinism, failure modes."""
from __future__ import annotations
import os
import sys
import tempfile
import unittest
HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, os.path.dirname(HERE))
from main import ( # noqa: E402
BestResultError,
DemoReport,
NoTriggerError,
PaperValidationError,
SchedulerReport,
build_mini_paper,
make_seed_hypotheses,
mini_to_full_paper,
pick_best_branch,
run_demo,
)
import main as e2e # noqa: E402
class TestComposition(unittest.TestCase):
def test_demo_runs_to_completion(self) -> None:
with tempfile.TemporaryDirectory() as td:
rep = run_demo(out_dir=td)
self.assertIsInstance(rep, DemoReport)
self.assertTrue(rep.best_branch)
self.assertGreater(rep.best_reward, 0.0)
self.assertEqual(rep.critic_result["status"], "converged")
self.assertGreaterEqual(rep.scheduler_report["experiments_run"], 3)
self.assertGreaterEqual(len(rep.paper_manifest["sections"]), 2)
self.assertGreaterEqual(len(rep.paper_manifest["figures"]), 1)
def test_paper_files_emitted(self) -> None:
with tempfile.TemporaryDirectory() as td:
rep = run_demo(out_dir=td)
tex = rep.paper_manifest["tex_path"]
bib = rep.paper_manifest["bib_path"]
self.assertTrue(os.path.exists(tex))
self.assertTrue(os.path.exists(bib))
class TestDeterminism(unittest.TestCase):
def test_two_runs_with_same_seed_produce_same_branch(self) -> None:
with tempfile.TemporaryDirectory() as td1, tempfile.TemporaryDirectory() as td2:
r1 = run_demo(out_dir=td1, seed=11)
r2 = run_demo(out_dir=td2, seed=11)
self.assertEqual(r1.best_branch, r2.best_branch)
self.assertAlmostEqual(r1.best_reward, r2.best_reward, places=6)
self.assertEqual(
[s["id"] for s in r1.paper_manifest["sections"]],
[s["id"] for s in r2.paper_manifest["sections"]],
)
class TestPicker(unittest.TestCase):
def test_no_trigger_raises(self) -> None:
rep = SchedulerReport(
stop_reason="queue_empty", experiments_run=3, wall_seconds=0.01,
branches=[], paper_triggers=[], trace=[],
)
with self.assertRaises(NoTriggerError):
pick_best_branch(rep)
def test_orphan_trigger_raises_best_result_error(self) -> None:
rep = SchedulerReport(
stop_reason="queue_empty", experiments_run=3, wall_seconds=0.01,
branches=[], paper_triggers=["ghost"], trace=[],
)
with self.assertRaises(BestResultError):
pick_best_branch(rep)
def test_ties_break_alphabetically(self) -> None:
from main import scheduler_mod
BranchStats = scheduler_mod.BranchStats
rep = SchedulerReport(
stop_reason="queue_empty", experiments_run=2, wall_seconds=0.01,
branches=[
BranchStats(branch="beta", runs=2, reward_sum=1.6),
BranchStats(branch="alpha", runs=2, reward_sum=1.6),
],
paper_triggers=["beta", "alpha"], trace=[],
)
branch, _ = pick_best_branch(rep)
self.assertEqual(branch, "alpha")
class TestPaperWriterContract(unittest.TestCase):
def test_validation_error_propagates(self) -> None:
mini = build_mini_paper("alpha", 0.9)
full = mini_to_full_paper(mini, "alpha")
full.title = ""
from main import MockProseGenerator, PaperWriter
writer = PaperWriter(prose=MockProseGenerator(outlines={}))
with tempfile.TemporaryDirectory() as td:
with self.assertRaises(PaperValidationError):
writer.write(full, td)
class TestSeedAndScheduler(unittest.TestCase):
def test_seed_has_three_branches(self) -> None:
seed = make_seed_hypotheses()
self.assertEqual(len(seed), 3)
self.assertEqual(len({h.branch for h in seed}), 3)
def test_demo_stop_reason_is_known(self) -> None:
with tempfile.TemporaryDirectory() as td:
rep = run_demo(out_dir=td)
self.assertIn(
rep.stop_reason,
{"queue_empty", "max_experiments", "deadline"},
)
class TestPicker_BestBranchIsAlpha(unittest.TestCase):
def test_alpha_wins_under_default_seed(self) -> None:
with tempfile.TemporaryDirectory() as td:
rep = run_demo(out_dir=td, seed=11)
self.assertEqual(rep.best_branch, "alpha")
self.assertGreater(rep.best_reward, 0.7)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,106 @@
# End-to-End Research Demo
> A demo is the place where every contract you wrote earlier has to compose. If any one of them leaks, the demo is the lesson that catches it.
**Type:** Build
**Languages:** Python
**Prerequisites:** Phase 19 lessons 50-53
**Time:** ~90 minutes
## Learning Objectives
- Wire the auto-research loop end to end: hypothesis seed, experiment runner, scheduler, critic loop, paper writer.
- Compose the primitives from the four earlier Track D lessons through plain Python imports, not a framework.
- Run the loop to a self-terminating end and emit a single demo report that lists every stage's output.
- Keep the demo deterministic so the test suite can assert the final shape.
- Surface a clear failure mode when any stage's contract breaks, so the next stage does not run with a broken input.
## What composes here
```mermaid
flowchart LR
Seed[Seed hypotheses] --> Sched[Iteration scheduler]
Sched --> Exp[Experiment runner]
Exp --> Bus[Result bus]
Bus --> Sched
Bus --> Trig[Paper trigger]
Trig --> Pick[Best result picker]
Pick --> Critic[Critic loop]
Critic --> Writer[Paper writer]
Writer --> Report[Demo report]
```
Five stages. The seed is a list of three hypotheses. The scheduler runs six experiments across them with three parallel slots. The bus reports one or more paper triggers. The picker selects the single best result. The critic loop iterates on a draft built from that result. The paper writer emits the final LaTeX, BibTeX, and manifest.
## Why import, not copy
Each earlier lesson ships a `main.py` with public dataclasses and functions. The demo imports them by adjusting `sys.path` to the parent directory of each lesson. This is not framework wiring; it is the same import the test files in the earlier lessons already use.
```mermaid
flowchart TB
Demo[57: end-to-end demo] --> A[54: PaperWriter]
Demo --> B[55: CriticLoop]
Demo --> C[56: IterationScheduler]
Demo --> Inline[Inline stub: seed and runner]
```
The inline stub stands in for lessons fifty through fifty-three: a small generator of seed hypotheses and a synchronous reward function. The user can swap the inline stub for the real primitives from those lessons by adjusting two imports.
## Determinism guarantees
The demo is deterministic by construction. The experiment runner is seeded numpy. The critic loop's reviser walks fixed dimensions in fixed order. The paper writer's prose generator is the mocked one from lesson fifty-four. The scheduler's UCB picker breaks ties on iteration order, not random choice.
Given the same seed, the demo emits the same report. The test asserts this property by running the demo twice and comparing the manifest.
## The demo report shape
```mermaid
flowchart TB
Rep[DemoReport] --> Sch[scheduler_report]
Rep --> Pick[best_branch and best_reward]
Rep --> Cri[critic_result]
Rep --> Pap[paper_manifest]
Rep --> Term[stop_reason]
```
Each field comes verbatim from the upstream stage. The demo does not transform any output; it composes them. That is the test the demo is.
## Failure mode handling
Each stage either succeeds or raises a typed error.
```text
Scheduler ........ returns SchedulerReport with stop_reason
in {queue_empty, max_experiments, deadline}
Best-result pick . raises NoTriggerError if no paper trigger fired
Critic loop ...... returns LoopResult with status converged or stopped
Paper writer ..... raises PaperValidationError on contract break
```
A failure in any stage short-circuits the demo with a typed exception. The test asserts that a deliberately broken seed (empty hypothesis list) raises the right error and never touches the writer.
## The best-result picker
The scheduler emits paper triggers per branch. The picker selects the branch with the highest mean reward across all triggers. Ties break alphabetically by branch id so the demo is deterministic. The picker is a small pure function; the test pins it on a fixed scheduler report.
## Wiring the critic loop
The critic loop in lesson fifty-five operates on a `MiniPaper`. The demo builds a `MiniPaper` from the picked branch by populating the abstract with the branch id, seeding two sections (Introduction and Results), and setting `originality_tag` from the branch's mean reward (high if `>= 0.8`, medium if `>= 0.6`, low otherwise).
The reviser then iterates the draft to convergence. The output goes into the paper writer.
## Wiring the paper writer
The paper writer in lesson fifty-four operates on the full `Paper` shape with figures and bibliography. The demo upgrades the converged `MiniPaper` by attaching one figure per high-yield branch and a small synthetic bibliography that points at made-up keys the critic suggested. Every cite the demo adds is also added to the bibliography list, so validation passes.
## How to read the code
`code/main.py` defines `BestResultError`, `NoTriggerError`, `DemoReport`, `pick_best_branch`, `build_paper_from_result`, and `run_demo`. The imports at the top adjust `sys.path` once and pull `PaperWriter`, `CriticLoop`, and `IterationScheduler` from their lessons.
`code/tests/test_e2e.py` covers: demo runs end to end and emits a report with all five fields populated, determinism across two runs, NoTriggerError when no branch crosses the threshold, PaperValidationError when the writer's contract breaks, paper manifest contains the picked branch's figure, and the scheduler stop reason is one of the expected values.
## Going further
Three extensions worth wiring once the demo is green. First, persistent state: each stage's result writes to a small JSON store so a restart can resume without re-running the cheap stages. Second, a dashboard: the trace events from the scheduler and critic loop render as a single timeline. Third, real model calls: swap the mocked prose generator and the deterministic critic for model-driven ones; the wiring does not change.
The demo's job is to prove that composition is the architecture. Five lessons, four imports, one report. The next time you add a stage, the wiring grows by exactly one line.
@@ -0,0 +1,78 @@
{
"lesson": "57-end-to-end-research-demo",
"title": "End-to-End Research Demo",
"questions": [
{
"stage": "pre",
"question": "What is the demo's job in this lesson?",
"options": [
"To replace the earlier lessons with a single new framework",
"To compose the primitives from the earlier lessons through plain Python imports",
"To benchmark the scheduler against a baseline",
"To call a real language model on each branch"
],
"correct": 1,
"explanation": "The demo is a composition test. Each stage exists in its own lesson. The demo imports them and wires the report. No new framework, no new abstractions."
},
{
"stage": "pre",
"question": "Why does the lesson load earlier lessons via importlib instead of a normal package import?",
"options": [
"Because importlib is faster",
"Because each earlier lesson is a standalone code/ folder, not an installed package; importlib loads them by file path",
"Because asyncio requires importlib",
"Because the lessons share a module name and would collide"
],
"correct": 1,
"explanation": "The lessons are not installed packages. Each ships a main.py inside its own code/ folder. importlib.util.spec_from_file_location loads them by path without polluting the global module namespace."
},
{
"stage": "check",
"question": "How does the best-result picker decide between branches when more than one triggered a paper?",
"options": [
"It picks the branch with the most experiments run",
"It picks the branch with the highest mean reward, breaking ties alphabetically by branch id",
"It picks at random for fairness",
"It picks the first triggered branch"
],
"correct": 1,
"explanation": "Highest mean wins. Ties break on branch id so the demo is deterministic. Random would defeat the determinism test."
},
{
"stage": "check",
"question": "What error does the picker raise when the scheduler reports zero paper triggers?",
"options": [
"ValueError",
"NoTriggerError, which short-circuits the demo before the writer runs",
"PaperValidationError",
"It returns None"
],
"correct": 1,
"explanation": "NoTriggerError is the typed failure mode. Returning None would let the writer run on a malformed input and break later with a less actionable error."
},
{
"stage": "post",
"question": "Why does the demo run twice with the same seed in the determinism test?",
"options": [
"To exercise asyncio twice",
"To assert that the picked branch, reward, and paper section ids are identical across runs",
"To catch a memory leak",
"To prove that the scheduler is fair"
],
"correct": 1,
"explanation": "Determinism is part of the contract. Two runs with the same seed must produce the same artifacts. The test pins that."
},
{
"stage": "post",
"question": "What does the demo report contain after a successful run?",
"options": [
"Only the final LaTeX file",
"scheduler_report, best_branch, best_reward, critic_result, paper_manifest, and stop_reason",
"Just the experiment count and a status string",
"A model prompt and completion log"
],
"correct": 1,
"explanation": "Every stage's output is preserved. The report is a composition of upstream outputs, not a transformed summary. Downstream tooling reads the manifest; debuggers read the trace inside scheduler_report."
}
]
}