mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 01:54:39 +08:00
feat(phase-14/40): multi-session handoff packet generator
This commit is contained in:
@@ -0,0 +1,142 @@
|
||||
"""Generate a handoff packet from workbench artifacts.
|
||||
|
||||
Reads state, verdict, review, and feedback (here stubbed in-memory),
|
||||
writes handoff.md for humans and handoff.json for the next agent.
|
||||
|
||||
Run: python3 code/main.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).parent
|
||||
TAIL_K = 5
|
||||
|
||||
|
||||
@dataclass
|
||||
class WorkbenchSnapshot:
|
||||
task_id: str
|
||||
state: dict[str, object]
|
||||
verdict: dict[str, object]
|
||||
review: dict[str, object]
|
||||
feedback: list[dict[str, object]]
|
||||
diff_summary: dict[str, list[str]]
|
||||
|
||||
|
||||
@dataclass
|
||||
class HandoffPayload:
|
||||
task_id: str
|
||||
summary: str
|
||||
changed_files: list[str]
|
||||
commands_run: list[str]
|
||||
failed_attempts: list[str]
|
||||
open_risks: list[dict[str, str]]
|
||||
next_action: str
|
||||
verdict_pointer: dict[str, str]
|
||||
feedback_tail: list[dict[str, object]] = field(default_factory=list)
|
||||
|
||||
|
||||
def trim_feedback(records: list[dict[str, object]]) -> list[dict[str, object]]:
|
||||
tail = records[-TAIL_K:]
|
||||
nonzero = [r for r in records if r.get("exit_code") not in (0, None)]
|
||||
out: list[dict[str, object]] = []
|
||||
seen: set[int] = set()
|
||||
for r in tail + nonzero:
|
||||
key = id(r)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
out.append(r)
|
||||
return out
|
||||
|
||||
|
||||
def derive_risks(snapshot: WorkbenchSnapshot) -> list[dict[str, str]]:
|
||||
risks: list[dict[str, str]] = []
|
||||
for f in snapshot.verdict.get("findings", []) or []:
|
||||
if isinstance(f, dict) and f.get("severity") in ("warn", "block"):
|
||||
risks.append({"severity": str(f.get("severity")), "detail": str(f.get("detail"))})
|
||||
for blocker in snapshot.state.get("blockers") or []:
|
||||
risks.append({"severity": "warn", "detail": f"open blocker: {blocker}"})
|
||||
if int(snapshot.review.get("total", 10)) < 7:
|
||||
risks.append({"severity": "warn", "detail": f"review total {snapshot.review.get('total')} below 7"})
|
||||
return risks
|
||||
|
||||
|
||||
def generate_handoff(snapshot: WorkbenchSnapshot) -> tuple[str, HandoffPayload]:
|
||||
next_action = str(snapshot.state.get("next_action") or "no next_action recorded; needs human")
|
||||
payload = HandoffPayload(
|
||||
task_id=snapshot.task_id,
|
||||
summary=f"task {snapshot.task_id}: review={snapshot.review.get('verdict')}, gate={snapshot.verdict.get('passed')}",
|
||||
changed_files=snapshot.diff_summary.get("touched", []),
|
||||
commands_run=[str(r.get("command")) for r in snapshot.feedback],
|
||||
failed_attempts=[
|
||||
f"{r.get('command')} -> exit {r.get('exit_code')}"
|
||||
for r in snapshot.feedback
|
||||
if r.get("exit_code") not in (0, None)
|
||||
],
|
||||
open_risks=derive_risks(snapshot),
|
||||
next_action=next_action,
|
||||
verdict_pointer={
|
||||
"verdict": f"outputs/verification/{snapshot.task_id}.json",
|
||||
"review": f"outputs/review/{snapshot.task_id}.json",
|
||||
},
|
||||
feedback_tail=trim_feedback(snapshot.feedback),
|
||||
)
|
||||
|
||||
md_lines = [
|
||||
f"# Handoff: {payload.task_id}",
|
||||
"",
|
||||
f"**Summary.** {payload.summary}",
|
||||
"",
|
||||
"## Changed files",
|
||||
*(f"- `{f}`" for f in payload.changed_files),
|
||||
"",
|
||||
"## Commands run",
|
||||
*(f"- `{c}`" for c in payload.commands_run),
|
||||
"",
|
||||
"## Failed attempts",
|
||||
*(f"- {f}" for f in payload.failed_attempts) or ["- none"],
|
||||
"",
|
||||
"## Open risks",
|
||||
*(f"- [{r['severity']}] {r['detail']}" for r in payload.open_risks) or ["- none"],
|
||||
"",
|
||||
f"## Next action",
|
||||
f"{payload.next_action}",
|
||||
"",
|
||||
"## Receipts",
|
||||
f"- verdict: `{payload.verdict_pointer['verdict']}`",
|
||||
f"- review: `{payload.verdict_pointer['review']}`",
|
||||
]
|
||||
return "\n".join(md_lines) + "\n", payload
|
||||
|
||||
|
||||
def main() -> None:
|
||||
snapshot = WorkbenchSnapshot(
|
||||
task_id="T-001",
|
||||
state={
|
||||
"active_task_id": None,
|
||||
"blockers": ["awaiting decision on rate-limit window"],
|
||||
"next_action": "open PR with current diff and request review",
|
||||
},
|
||||
verdict={"passed": True, "findings": [{"severity": "warn", "detail": "off-scope: README.md"}]},
|
||||
review={"verdict": "pass", "total": 8},
|
||||
feedback=[
|
||||
{"command": "pytest", "exit_code": 0},
|
||||
{"command": "ruff check .", "exit_code": 0},
|
||||
{"command": "pytest test_signup.py", "exit_code": 1},
|
||||
{"command": "pytest test_signup.py", "exit_code": 0},
|
||||
],
|
||||
diff_summary={"touched": ["app/signup.py", "tests/test_signup.py", "README.md"]},
|
||||
)
|
||||
|
||||
md, payload = generate_handoff(snapshot)
|
||||
(HERE / "handoff.md").write_text(md)
|
||||
(HERE / "handoff.json").write_text(json.dumps(asdict(payload), indent=2) + "\n")
|
||||
print(md)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,116 @@
|
||||
# Multi-Session Handoff
|
||||
|
||||
> The session is going to end. The work is not. The handoff packet is the artifact that turns "the agent worked for an hour" into "the next session is productive in the first minute." Build it on purpose, not as an afterthought.
|
||||
|
||||
**Type:** Build
|
||||
**Languages:** Python (stdlib)
|
||||
**Prerequisites:** Phase 14 · 34 (Repo Memory), Phase 14 · 38 (Verification), Phase 14 · 39 (Reviewer)
|
||||
**Time:** ~50 minutes
|
||||
|
||||
## Learning Objectives
|
||||
|
||||
- Identify the seven fields every handoff packet needs.
|
||||
- Generate a handoff from the workbench artifacts without hand-writing prose.
|
||||
- Trim large feedback logs into a handoff-sized summary.
|
||||
- Make the next session's first action deterministic.
|
||||
|
||||
## The Problem
|
||||
|
||||
The session ends. The agent says "great, we made progress." The next session opens. The next agent asks "where did we leave off?" The first agent's answer is gone. The next agent rediscovers, re-runs the same commands, re-asks the human the same questions, and burns thirty minutes recovering the last thirty seconds of the previous session.
|
||||
|
||||
The cost of a bad handoff is paid every session for the life of the task. The fix is a packet generated automatically at session end: what changed, why, what was tried, what failed, what is left, what to do first next time.
|
||||
|
||||
## The Concept
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
State[agent_state.json] --> Generator[generate_handoff.py]
|
||||
Verdict[verification_report.json] --> Generator
|
||||
Review[review_report.json] --> Generator
|
||||
Feedback[feedback_record.jsonl] --> Generator
|
||||
Generator --> Handoff[handoff.md + handoff.json]
|
||||
Handoff --> Next[Next Session]
|
||||
```
|
||||
|
||||
### Seven fields every handoff carries
|
||||
|
||||
| Field | Question it answers |
|
||||
|-------|---------------------|
|
||||
| `summary` | One paragraph of what was done |
|
||||
| `changed_files` | The diff at a glance |
|
||||
| `commands_run` | What was actually executed |
|
||||
| `failed_attempts` | What was tried and why it did not work |
|
||||
| `open_risks` | What could bite next session, with severity |
|
||||
| `next_action` | The first concrete step next session takes |
|
||||
| `verdict_pointer` | Path to the verification + review reports |
|
||||
|
||||
The `next_action` field is the load-bearing one. A handoff with everything except `next_action` is a status report, not a handoff.
|
||||
|
||||
### Handoffs are generated, not written
|
||||
|
||||
A hand-written handoff is a handoff that gets skipped on a hard day. The generator reads the workbench artifacts and emits the packet. The agent's job is to leave the workbench in a state the generator can summarize, not to write the summary.
|
||||
|
||||
### Two forms: human-readable and machine-readable
|
||||
|
||||
`handoff.md` is what the human reads. `handoff.json` is what the next agent loads. Both come from the same source artifacts. If they diverge, the JSON wins.
|
||||
|
||||
### Feedback log trimming
|
||||
|
||||
The full `feedback_record.jsonl` may be hundreds of entries. The handoff carries only the last K plus every entry with a non-zero exit. The next session loads the full log if it needs to, but the packet stays small.
|
||||
|
||||
## Build It
|
||||
|
||||
`code/main.py` implements:
|
||||
|
||||
- A loader that gathers state, verdict, review, and feedback into a single `WorkbenchSnapshot`.
|
||||
- A `generate_handoff(snapshot) -> (markdown, payload)` function.
|
||||
- A filter that picks the last K feedback entries plus all non-zero exits.
|
||||
- A demo run that writes `handoff.md` and `handoff.json` next to the script.
|
||||
|
||||
Run it:
|
||||
|
||||
```
|
||||
python3 code/main.py
|
||||
```
|
||||
|
||||
Output: a printed handoff body, plus both files on disk.
|
||||
|
||||
## Use It
|
||||
|
||||
Production patterns:
|
||||
|
||||
- **Session-end hook.** The runtime fires the generator when the user closes the chat. The packet goes into `outputs/handoff/<session_id>/`.
|
||||
- **PR template.** The generator's markdown is also a PR body. Reviewers read it without opening five other files.
|
||||
- **Cross-agent handoff.** Build with one product (Claude Code), continue with another (Codex). The packet is the lingua franca.
|
||||
|
||||
The packet is small, regular, and cheap to produce. The cost saving compounds with every session.
|
||||
|
||||
## Ship It
|
||||
|
||||
`outputs/skill-handoff-generator.md` produces a generator tuned to a project's artifact paths, an end-of-session hook that runs it, and a `handoff.json` schema the next agent reads on startup.
|
||||
|
||||
## Exercises
|
||||
|
||||
1. Add an `assumptions_to_validate` field that surfaces every assumption the builder logged but the reviewer did not score above 1.
|
||||
2. Trim the feedback summary differently for failing runs versus passing ones. Defend the asymmetry.
|
||||
3. Include a "questions for the human" list. What is the threshold for a question to make it into the packet versus into a chat message?
|
||||
4. Make the generator idempotent: running it twice produces the same packet. What needs to be stable for that to hold?
|
||||
5. Add a "next session prereqs" section listing exactly the artifacts the next session must load before acting.
|
||||
|
||||
## Key Terms
|
||||
|
||||
| Term | What people say | What it actually means |
|
||||
|------|----------------|------------------------|
|
||||
| Handoff packet | "Session summary" | Generated artifact carrying the seven fields, both markdown and JSON |
|
||||
| Next action | "What to do first" | The one concrete step that starts the next session |
|
||||
| Feedback trim | "Log summary" | Last K records plus every non-zero exit |
|
||||
| Status report | "What we did" | A document missing `next_action`; useful, but not a handoff |
|
||||
| Verdict pointer | "Receipt" | Path to the verification + review reports for traceability |
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Anthropic, Effective harnesses for long-running agents](https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents)
|
||||
- [OpenAI Agents SDK handoffs](https://platform.openai.com/docs/guides/agents-sdk/handoffs)
|
||||
- Phase 14 · 34 — the state file the generator reads
|
||||
- Phase 14 · 38 — the verification verdict the packet points at
|
||||
- Phase 14 · 39 — the reviewer report bundled into the packet
|
||||
+49
@@ -0,0 +1,49 @@
|
||||
---
|
||||
name: handoff-generator
|
||||
description: Generate end-of-session handoff packets from workbench artifacts, producing both human-readable Markdown and machine-readable JSON keyed to the seven canonical fields.
|
||||
version: 1.0.0
|
||||
phase: 14
|
||||
lesson: 40
|
||||
tags: [handoff, generator, session-end, packet, next-action]
|
||||
---
|
||||
|
||||
Given a workbench (state, verdict, review, feedback log, diff), produce a session-end handoff generator wired into the agent runtime.
|
||||
|
||||
Produce:
|
||||
|
||||
1. `tools/generate_handoff.py` exposing `generate_handoff(snapshot) -> (markdown, payload)`.
|
||||
2. `outputs/handoff/<session_id>/handoff.md` and `handoff.json`.
|
||||
3. `handoff.schema.json` covering the seven required fields and the feedback tail format.
|
||||
4. Session-end hook script that runs the generator and refuses to close the session if any field is missing.
|
||||
5. `docs/handoff.md` listing the seven fields, their sources, and the trimming policy.
|
||||
|
||||
Hard rejects:
|
||||
|
||||
- A handoff without a `next_action`. Status reports masquerading as handoffs poison the next session.
|
||||
- A generator that hand-writes the summary. The agent's job is to leave the workbench in a generatable state.
|
||||
- A markdown packet that diverges from the JSON. JSON is the source; markdown is a render of JSON.
|
||||
- A feedback tail longer than 30 entries. The full log is in version control; the packet must stay small.
|
||||
|
||||
Refusal rules:
|
||||
|
||||
- If the verification report is missing, refuse to generate the packet. A handoff without a verdict is a wish.
|
||||
- If the review report is missing and a human reviewer was expected, refuse and require the review pass first.
|
||||
- If the diff summary is empty but the session ran longer than 5 minutes, surface the anomaly before generating; suspect a wedged session rather than a real no-op.
|
||||
|
||||
Output structure:
|
||||
|
||||
```
|
||||
<repo>/
|
||||
├── outputs/handoff/<session_id>/
|
||||
│ ├── handoff.md
|
||||
│ └── handoff.json
|
||||
├── tools/generate_handoff.py
|
||||
├── handoff.schema.json
|
||||
└── docs/handoff.md
|
||||
```
|
||||
|
||||
End with "what to read next" pointing to:
|
||||
|
||||
- Lesson 41 for end-to-end exercise on a real-style sample app.
|
||||
- Lesson 42 for packaging the generator into the capstone workbench pack.
|
||||
- Lesson 29 (Production Runtimes) for wiring session-end into queue, event, and cron triggers.
|
||||
Reference in New Issue
Block a user