Files
Rohit Ghumare 3ecf630b0f feat(projects): add 48 builds and a 52-project roadmap (#497)
* feat(projects): workflow-hooks stage 1

* feat(projects): workflow-hooks stage 2

* feat(projects): workflow-hooks stage 3

* feat(projects): workflow-hooks stage 4

* fix(projects): preserve sponsor navigation in project footer

* feat(projects): agent-budget-planner stage 1

* feat(projects): agent-budget-planner stage 2

* feat(projects): agent-budget-planner stage 3

* feat(projects): agent-budget-planner stage 4

* feat(projects): agent-trace-debugger stage 1

* feat(projects): agent-trace-debugger stage 2

* feat(projects): agent-trace-debugger stage 3

* feat(projects): agent-trace-debugger stage 4

* feat(projects): browser-agent stage 1

* feat(projects): browser-agent stage 2

* feat(projects): browser-agent stage 3

* feat(projects): browser-agent stage 4

* feat(projects): calendar-focus-planner stage 1

* feat(projects): calendar-focus-planner stage 2

* feat(projects): calendar-focus-planner stage 3

* feat(projects): calendar-focus-planner stage 4

* feat(projects): changelog-writer-from-git stage 1

* feat(projects): changelog-writer-from-git stage 2

* feat(projects): changelog-writer-from-git stage 3

* feat(projects): changelog-writer-from-git stage 4

* feat(projects): cloud-agent-with-aws-strands stage 1

* feat(projects): cloud-agent-with-aws-strands stage 2

* feat(projects): cloud-agent-with-aws-strands stage 3

* feat(projects): cloud-agent-with-aws-strands stage 4

* feat(projects): csv-sql-question-workbench stage 1

* feat(projects): csv-sql-question-workbench stage 2

* feat(projects): csv-sql-question-workbench stage 3

* feat(projects): csv-sql-question-workbench stage 4

* feat(projects): dataset-split-auditor stage 1

* feat(projects): dataset-split-auditor stage 2

* feat(projects): dataset-split-auditor stage 3

* feat(projects): dataset-split-auditor stage 4

* feat(projects): desktop-control stage 1

* feat(projects): desktop-control stage 2

* feat(projects): desktop-control stage 3

* feat(projects): desktop-control stage 4

* feat(projects): distributed-eval-farm stage 1

* feat(projects): distributed-eval-farm stage 2

* feat(projects): distributed-eval-farm stage 3

* feat(projects): distributed-eval-farm stage 4

* feat(projects): doc-qa-with-citations stage 1

* feat(projects): doc-qa-with-citations stage 2

* feat(projects): doc-qa-with-citations stage 3

* feat(projects): doc-qa-with-citations stage 4

* feat(projects): document-extraction-desk stage 1

* feat(projects): document-extraction-desk stage 2

* feat(projects): document-extraction-desk stage 3

* feat(projects): document-extraction-desk stage 4

* feat(projects): durable-agent-jobs stage 1

* feat(projects): durable-agent-jobs stage 2

* feat(projects): durable-agent-jobs stage 3

* feat(projects): durable-agent-jobs stage 4

* feat(projects): feedback-theme-board stage 1

* feat(projects): feedback-theme-board stage 2

* feat(projects): feedback-theme-board stage 3

* feat(projects): feedback-theme-board stage 4

* feat(projects): harness-bench stage 1

* feat(projects): harness-bench stage 2

* feat(projects): harness-bench stage 3

* feat(projects): harness-bench stage 4

* feat(projects): inbox-triage-desk stage 1

* feat(projects): inbox-triage-desk stage 2

* feat(projects): inbox-triage-desk stage 3

* feat(projects): inbox-triage-desk stage 4

* feat(projects): json-schema-output-guard stage 1

* feat(projects): json-schema-output-guard stage 2

* feat(projects): json-schema-output-guard stage 3

* feat(projects): json-schema-output-guard stage 4

* feat(projects): llm-gateway-with-fallbacks stage 1

* feat(projects): llm-gateway-with-fallbacks stage 2

* feat(projects): llm-gateway-with-fallbacks stage 3

* feat(projects): llm-gateway-with-fallbacks stage 4

* feat(projects): local-model-eval-harness stage 1

* feat(projects): local-model-eval-harness stage 2

* feat(projects): local-model-eval-harness stage 3

* feat(projects): local-model-eval-harness stage 4

* feat(projects): mcp-at-scale stage 1

* feat(projects): mcp-at-scale stage 2

* feat(projects): mcp-at-scale stage 3

* feat(projects): mcp-at-scale stage 4

* feat(projects): mcp-at-scale stage 5

* feat(projects): meeting-notes-to-actions stage 1

* feat(projects): meeting-notes-to-actions stage 2

* feat(projects): meeting-notes-to-actions stage 3

* feat(projects): meeting-notes-to-actions stage 4

* feat(projects): memory-server stage 1

* feat(projects): memory-server stage 2

* feat(projects): memory-server stage 3

* feat(projects): memory-server stage 4

* feat(projects): multi-agent-code-review-panel stage 1

* feat(projects): multi-agent-code-review-panel stage 2

* feat(projects): multi-agent-code-review-panel stage 3

* feat(projects): multi-agent-code-review-panel stage 4

* feat(projects): postmortem-writer stage 1

* feat(projects): postmortem-writer stage 2

* feat(projects): postmortem-writer stage 3

* feat(projects): postmortem-writer stage 4

* feat(projects): pr-review-reporter stage 1

* feat(projects): pr-review-reporter stage 2

* feat(projects): pr-review-reporter stage 3

* feat(projects): pr-review-reporter stage 4

* feat(projects): prompt-regression-tester stage 1

* feat(projects): prompt-regression-tester stage 2

* feat(projects): prompt-regression-tester stage 3

* feat(projects): prompt-regression-tester stage 4

* feat(projects): rag-freshness-pipeline stage 1

* feat(projects): rag-freshness-pipeline stage 2

* feat(projects): rag-freshness-pipeline stage 3

* feat(projects): rag-freshness-pipeline stage 4

* feat(projects): report-judge stage 1

* feat(projects): report-judge stage 2

* feat(projects): report-judge stage 3

* feat(projects): report-judge stage 4

* feat(projects): research-report-agent stage 1

* feat(projects): research-report-agent stage 2

* feat(projects): research-report-agent stage 3

* feat(projects): research-report-agent stage 4

* feat(projects): research-report-agent stage 5

* feat(projects): research-report-agent stage 6

* feat(projects): research-report-agent stage 7

* feat(projects): retrieval-evaluation-lab stage 1

* feat(projects): retrieval-evaluation-lab stage 2

* feat(projects): retrieval-evaluation-lab stage 3

* feat(projects): retrieval-evaluation-lab stage 4

* feat(projects): rust-agent-shell stage 1

* feat(projects): rust-agent-shell stage 2

* feat(projects): rust-agent-shell stage 3

* feat(projects): rust-agent-shell stage 4

* feat(projects): sandbox-ladder stage 1

* feat(projects): sandbox-ladder stage 2

* feat(projects): sandbox-ladder stage 3

* feat(projects): sandbox-ladder stage 4

* feat(projects): self-improving-skill-loop stage 1

* feat(projects): self-improving-skill-loop stage 2

* feat(projects): self-improving-skill-loop stage 3

* feat(projects): self-improving-skill-loop stage 4

* feat(projects): semantic-notes-search stage 1

* feat(projects): semantic-notes-search stage 2

* feat(projects): semantic-notes-search stage 3

* feat(projects): semantic-notes-search stage 4

* feat(projects): skill-installer stage 1

* feat(projects): skill-installer stage 2

* feat(projects): skill-installer stage 3

* feat(projects): skill-installer stage 4

* feat(projects): skill-router stage 1

* feat(projects): skill-router stage 2

* feat(projects): skill-router stage 3

* feat(projects): skill-router stage 4

* feat(projects): skill-scanner stage 1

* feat(projects): skill-scanner stage 2

* feat(projects): skill-scanner stage 3

* feat(projects): skill-scanner stage 4

* feat(projects): skill-validator stage 1

* feat(projects): skill-validator stage 2

* feat(projects): skill-validator stage 3

* feat(projects): skill-validator stage 4

* feat(projects): source-grounded-study-coach stage 1

* feat(projects): source-grounded-study-coach stage 2

* feat(projects): source-grounded-study-coach stage 3

* feat(projects): source-grounded-study-coach stage 4

* feat(projects): support-agent-with-google-adk stage 1

* feat(projects): support-agent-with-google-adk stage 2

* feat(projects): support-agent-with-google-adk stage 3

* feat(projects): support-agent-with-google-adk stage 4

* feat(projects): tiny-coding-agent stage 1

* feat(projects): tiny-coding-agent stage 2

* feat(projects): tiny-coding-agent stage 3

* feat(projects): tiny-coding-agent stage 4

* feat(projects): token-counter-and-cost-meter stage 1

* feat(projects): token-counter-and-cost-meter stage 2

* feat(projects): token-counter-and-cost-meter stage 3

* feat(projects): token-counter-and-cost-meter stage 4

* feat(projects): tool-call-firewall stage 1

* feat(projects): tool-call-firewall stage 2

* feat(projects): tool-call-firewall stage 3

* feat(projects): tool-call-firewall stage 4

* feat(projects): typed-workflow-agent-with-mastra stage 1

* feat(projects): typed-workflow-agent-with-mastra stage 2

* feat(projects): typed-workflow-agent-with-mastra stage 3

* feat(projects): typed-workflow-agent-with-mastra stage 4

* feat(projects): visual-evidence-library stage 1

* feat(projects): visual-evidence-library stage 2

* feat(projects): visual-evidence-library stage 3

* feat(projects): visual-evidence-library stage 4

* feat(projects): voice-note-transcriber-pipeline stage 1

* feat(projects): voice-note-transcriber-pipeline stage 2

* feat(projects): voice-note-transcriber-pipeline stage 3

* feat(projects): voice-note-transcriber-pipeline stage 4

* feat(projects): web-change-brief stage 1

* feat(projects): web-change-brief stage 2

* feat(projects): web-change-brief stage 3

* feat(projects): web-change-brief stage 4

* feat(projects): workflow-hooks stage 1

* feat(projects): workflow-hooks stage 2

* feat(projects): workflow-hooks stage 3

* feat(projects): workflow-hooks stage 4

* feat(projects): publish the practical application catalog

* feat(projects): add 52 planned builds to the roadmap

* fix(projects): make lesson animations explain computed changes

Stage mechanisms should expose the decisions learners are implementing. Preserve record identity as state changes and keep project context below the selected lesson.

* fix(projects): agent-budget-planner stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-budget-planner stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-budget-planner stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-budget-planner stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-trace-debugger stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): browser-agent stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): browser-agent stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): browser-agent stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): calendar-focus-planner stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): cloud-agent-with-aws-strands stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): cloud-agent-with-aws-strands stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): cloud-agent-with-aws-strands stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): desktop-control stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): distributed-eval-farm stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): distributed-eval-farm stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): distributed-eval-farm stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): document-extraction-desk stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): feedback-theme-board stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): inbox-triage-desk stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): llm-gateway-with-fallbacks stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): llm-gateway-with-fallbacks stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): mcp-at-scale review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): meeting-notes-to-actions stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): meeting-notes-to-actions stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): memory-server stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): pr-review-reporter stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): pr-review-reporter stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): pr-review-reporter stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): research-report-agent stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): retrieval-evaluation-lab review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): rust-agent-shell stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): rust-agent-shell stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): semantic-notes-search stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): skill-installer review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): skill-router stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): skill-router stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): tiny-coding-agent stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): voice-note-transcriber-pipeline stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): web-change-brief stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): workflow-hooks stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): site review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* docs(projects): record verified review corrections
2026-09-29 21:38:54 +05:30

1463 lines
81 KiB
JSON

{
"planned": [
{
"id": "bookmark-path-organizer",
"title": "Bookmark Path Organizer",
"level": 1,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Make a useful reading path from saved links.",
"summary": "Help a reader recover a useful collection from an exported bookmark pile. Normalize URLs conservatively, merge exact duplicates, score user-defined topics from saved titles and notes, and export an editable reading path as HTML and JSON.",
"output": "A standalone reading-path page and bookmarks.json with retained original URLs",
"prerequisiteProjects": [],
"milestones": [
"Import bookmarks with their original hierarchy",
"Normalize links and explain duplicate groups",
"Rank topics and assemble a reading path",
"Export an editable collection and progress file"
],
"references": [],
"distinctFrom": "Semantic Notes Search retrieves passages; this organizes link collections and produces an ordered reading artifact.",
"firstDemo": "Turn twelve authored links about community mapping into a short beginner reading path with duplicate explanations.",
"originalFixture": "Authored bookmark export using example.invalid URLs and original descriptive notes"
},
{
"id": "csv-repair-workbench",
"title": "CSV Repair Workbench",
"level": 1,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Fix messy tables without losing the original cells.",
"summary": "Help a spreadsheet owner repair inconsistent labels, dates and missing values through an explicit sequence of transformations. Preview every changed cell, keep ambiguous cases for review, and export a cleaned CSV with a reusable transformation recipe and HTML comparison.",
"output": "cleaned.csv, recipe.json and a cell-level HTML before/after report",
"prerequisiteProjects": [],
"milestones": [
"Profile columns and retain original cells",
"Propose explicit normalization rules",
"Preview changes and review ambiguous values",
"Export a reusable repair recipe and cleaned table"
],
"references": [],
"distinctFrom": "CSV Question Workbench queries a table; this project creates a reviewed, replayable data-cleaning workflow.",
"firstDemo": "Normalize three spellings of a community garden location while preserving two ambiguous dates for review.",
"originalFixture": "Authored garden volunteer attendance rows with deliberate label and date inconsistencies"
},
{
"id": "download-folder-sort-desk",
"title": "Download Folder Sort Desk",
"level": 1,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Turn a crowded folder into a reviewable filing plan.",
"summary": "Help someone organize local downloads by combining file metadata, content hashes and an explainable category score. Generate collision-free destination proposals and duplicate groups, then export an HTML review desk and a JSON move plan that another file manager can consume.",
"output": "A searchable HTML filing plan, duplicate groups and moves.json",
"prerequisiteProjects": [],
"milestones": [
"Inventory files and compute content identities",
"Score categories from visible file features",
"Plan destinations and surface naming conflicts",
"Export a reviewed filing plan for a file manager"
],
"references": [],
"distinctFrom": "Inbox Triage Desk works with message threads and replies; this handles local file identity, duplicate content and destination planning.",
"firstDemo": "Organize an authored folder of workshop handouts, images and duplicate attachments into an inspectable plan.",
"originalFixture": "New small text, image and tabular files with deliberately duplicated content and conflicting names"
},
{
"id": "glossary-hovercard-publisher",
"title": "Glossary Hovercard Publisher",
"level": 1,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Explain unfamiliar terms inside the page people are reading.",
"summary": "Help an educator add contextual definitions to existing lesson HTML using a reviewed glossary and longest-phrase matching. Preserve code and links, support keyboard-accessible definitions, and export a standalone annotated lesson with a reusable glossary file.",
"output": "An annotated HTML lesson, glossary.json and a term-coverage report",
"prerequisiteProjects": [],
"milestones": [
"Validate terms, aliases and original definitions",
"Match phrases inside eligible text nodes",
"Render accessible definitions in context",
"Export the lesson and a reusable glossary bundle"
],
"references": [],
"distinctFrom": "Source-Grounded Study Coach schedules practice; this improves comprehension inside an existing reading surface.",
"firstDemo": "Add five original mapping definitions to a short lesson while leaving its code example untouched.",
"originalFixture": "An original introductory map-making lesson and a small authored glossary"
},
{
"id": "photo-burst-contact-sheet",
"title": "Photo Burst Contact Sheet",
"level": 1,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Compare similar photos before choosing what to keep.",
"summary": "Help a creator review repeated shots by computing simple pixel fingerprints and an inspectable sharpness measure. Group candidate duplicates, preserve the photographer's final choice, and export a contact sheet plus a selection manifest for an editor.",
"output": "HTML and PNG contact sheets, similarity explanations and selections.json",
"prerequisiteProjects": [],
"milestones": [
"Decode images and preserve source dimensions",
"Compute pixel fingerprints and sharpness measures",
"Group similar shots for manual selection",
"Export a contact sheet and editor selection manifest"
],
"references": [],
"distinctFrom": "Visual Evidence Library searches supplied text regions; this compares actual image pixels for photo selection.",
"firstDemo": "Group original geometric scene images with blur, crop and exposure variations while leaving a different scene separate.",
"originalFixture": "Original generated geometric scene images with explicitly authored transformations"
},
{
"id": "vocabulary-autocomplete-pad",
"title": "Vocabulary Autocomplete Pad",
"level": 1,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Build suggestions that explain which words they learned from.",
"summary": "Help a writer reuse their own terminology with a small local autocomplete model built from token counts, prefixes and short word histories. Show the evidence behind each suggestion and export an interactive writing pad plus a portable vocabulary model for another editor.",
"output": "A local autocomplete writing page and vocabulary-model.json",
"prerequisiteProjects": [],
"milestones": [
"Tokenize an authored corpus and count word histories",
"Rank prefix and next-word candidates",
"Accept suggestions in a local writing pad",
"Export the learned model and editor-facing suggestion function"
],
"references": [],
"distinctFrom": "Semantic Notes Search returns documents; this learns conditional next-word suggestions for a text input.",
"firstDemo": "Teach the pad a fictional garden club's vocabulary and inspect why a two-word continuation appears.",
"originalFixture": "Original short garden club announcements with repeated terminology"
},
{
"id": "api-tutorial-runner",
"title": "API Tutorial Runner",
"level": 2,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"TypeScript"
],
"tagline": "Check that an API tutorial still works from its first request to its final result.",
"summary": "Help a documentation author verify a sequence of HTTP examples. Parse explicit request blocks, substitute only declared variables from prior responses, and run against a local fixture or an opted-in test endpoint. Produce a source-linked report that exposes stale examples and lets an assistant propose a reviewed correction.",
"output": "Replayable request manifest, JUnit results and annotated tutorial HTML",
"prerequisiteProjects": [
"json-schema-output-guard"
],
"milestones": [
"Extract explicit requests from an authored tutorial",
"Resolve typed variables across steps",
"Run bounded requests against a test service",
"Export source-linked failures and correction proposals"
],
"references": [],
"distinctFrom": "MCP Tool Discovery imports endpoint descriptions; this executes an ordered tutorial with response variables and source-linked assertions.",
"firstDemo": "Replay a fictional lending API tutorial and identify the first request broken by an outdated field name."
},
{
"id": "catalog-record-linker",
"title": "Catalog Record Linker",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Reconcile two catalogs without silently merging different things.",
"summary": "Help a librarian or organizer reconcile inconsistent item records from two CSV exports. Generate candidate pairs using normalized fields and token similarity, collect explicit match decisions, and export a source-preserving crosswalk and reviewed merged catalog.",
"output": "An HTML pair-review queue, crosswalk.csv and merged-catalog.json",
"prerequisiteProjects": [
"csv-repair-workbench"
],
"milestones": [
"Import two catalogs with stable source identifiers",
"Generate candidate pairs and explain similarity",
"Review matches and resolve conflicting fields",
"Export the crosswalk and merged catalog"
],
"references": [],
"distinctFrom": "CSV Repair Workbench transforms individual cells; this resolves entity identity across separate collections.",
"firstDemo": "Link differently named items in two original lending-library catalogs and preserve a near-match as a separate item.",
"originalFixture": "Two authored lending-library exports with aliases, conflicting attributes and deliberate near-matches"
},
{
"id": "chart-storyboard-builder",
"title": "Chart Storyboard Builder",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Turn a table into a visual explanation people can inspect.",
"summary": "Help a community organizer explain a small dataset by selecting explicit aggregations and chart forms from column roles. Tie each annotation to the plotted values, expose missing data and axis choices, and export an SVG chart sequence with HTML and machine-readable chart specifications.",
"output": "SVG charts, a standalone HTML storyboard and chart-specs.json",
"prerequisiteProjects": [
"csv-sql-question-workbench"
],
"milestones": [
"Profile columns and declare their roles and units",
"Compute aggregates and choose a chart grammar",
"Attach annotations to the actual plotted values",
"Export an accessible visual storyboard and data receipt"
],
"references": [],
"distinctFrom": "CSV Question Workbench answers with a query and table; this teaches visual encoding, annotation grounding and portable chart publication.",
"firstDemo": "Explain an authored repair-event attendance dataset with two charts and reveal how a missing week changes the story.",
"originalFixture": "Original repair-event attendance data with deliberate missing periods and explicit units"
},
{
"id": "configuration-drift-explainer",
"title": "Configuration Drift Explainer",
"level": 2,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Go"
],
"tagline": "Explain why two environments resolve the same setting differently.",
"summary": "Help a developer compare declared defaults, environment overrides and observed JSON configuration. Preserve precedence and redaction rules, expose shadowed values, and attach every explanation to its source file or variable name. Export a drift report that an assistant can consume without receiving secret values.",
"output": "Redacted configuration diff and precedence evidence JSON",
"prerequisiteProjects": [
"json-schema-output-guard"
],
"milestones": [
"Load named configuration layers",
"Resolve precedence with provenance",
"Compare expected and observed effective values",
"Export a redacted explanation and CI result"
],
"references": [],
"distinctFrom": "JSON Schema Output Guard checks data shape; this explains effective values through layered precedence and environment drift.",
"firstDemo": "Compare two authored environments and reveal a stale override that shadows the shared timeout setting."
},
{
"id": "dependency-upgrade-impact-map",
"title": "Dependency Upgrade Impact Map",
"level": 2,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Python",
"TypeScript"
],
"tagline": "Show which call sites need attention when a dependency contract changes.",
"summary": "Help a maintainer review an upgrade using supplied before/after API signatures and a bounded source parser. Map changed symbols to actual call sites, distinguish confirmed incompatibilities from unresolved uses, and emit a targeted test checklist. An optional model can explain evidence but cannot invent removed APIs.",
"output": "Call-site impact graph, SARIF findings and targeted test checklist",
"prerequisiteProjects": [
"pr-review-reporter"
],
"milestones": [
"Read supplied old and new API contracts",
"Index supported import and call-site forms",
"Link changed signatures to affected code",
"Export findings with unresolved cases visible"
],
"references": [],
"distinctFrom": "PR Review Reporter checks changed lines; this connects old and new dependency contracts to call sites throughout a repository.",
"firstDemo": "Change an authored library signature and identify two affected callers plus one unresolved dynamic call."
},
{
"id": "diagram-reading-companion",
"title": "Diagram Reading Companion",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Give a diagram an inspectable reading order and explanation.",
"summary": "Help an educator turn an authored SVG diagram and an explicit node-edge manifest into a keyboard-navigable explanation. Check that referenced nodes exist, expose branches and cycles, let the author review the reading order, and export an accessible HTML walkthrough plus graph JSON.",
"output": "A navigable HTML diagram walkthrough and a reviewed graph manifest",
"prerequisiteProjects": [
"glossary-hovercard-publisher"
],
"milestones": [
"Load an SVG and validate its declared nodes and edges",
"Compute branches, cycles and candidate reading order",
"Review explanations beside highlighted diagram parts",
"Export the navigable walkthrough and graph data"
],
"references": [
"https://svgwg.org/svg2-draft/struct.html"
],
"distinctFrom": "Visual Evidence Library indexes text rectangles; this works with explicitly declared diagram relationships and reading order.",
"firstDemo": "Navigate an original seed-exchange process diagram and reveal a branch that a linear paragraph would omit.",
"originalFixture": "An original SVG process diagram with separately authored node and edge metadata"
},
{
"id": "field-notes-map-builder",
"title": "Field Notes Map Builder",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Turn location-tagged observations into a map another tool can use.",
"summary": "Help a community mapping group explore observations supplied as coordinates, notes and optional photos. Validate locations, group nearby records using an explicit distance rule, preserve original observations during review, and export GeoJSON with a standalone map and observation table.",
"output": "A GeoJSON FeatureCollection, a standalone HTML map and reviewed observation groups",
"prerequisiteProjects": [
"csv-repair-workbench"
],
"milestones": [
"Import coordinates, timestamps and observation evidence",
"Compute distances and explain nearby-record groups",
"Review the map alongside the original observations",
"Export GeoJSON and a portable map report"
],
"references": [
"https://www.rfc-editor.org/rfc/rfc7946"
],
"distinctFrom": "Visual Evidence Library locates text inside images; this organizes supplied geographic observations and produces a mapping interchange artifact.",
"firstDemo": "Map authored bench and shade observations, then inspect why two nearby records are grouped without merging their evidence.",
"originalFixture": "Original fictional field observations and simple authored photographs or illustrations"
},
{
"id": "gesture-shortcut-pad",
"title": "Gesture Shortcut Pad",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Teach a small app your own drawn shortcuts.",
"summary": "Help a presenter or educator control a local practice application with a few self-recorded gestures. Normalize pointer strokes, compare them with an explicit sequence-distance algorithm, expose uncertain matches, and export a portable gesture profile plus a small browser integration demo.",
"output": "An interactive gesture pad, gesture-profile.json and label events for another browser app",
"prerequisiteProjects": [],
"milestones": [
"Capture and normalize original pointer strokes",
"Compare strokes with an inspectable distance function",
"Review uncertain matches and add better examples",
"Export the gesture profile and demonstrate app integration"
],
"references": [],
"distinctFrom": "Desktop Control executes coordinate actions; this learns labels from pointer gestures and integrates them into a local application.",
"firstDemo": "Draw three original gesture classes to navigate a fictional slide deck and inspect an ambiguous stroke.",
"originalFixture": "New recorded pointer strokes and a tiny authored browser slideshow"
},
{
"id": "infrastructure-plan-explainer",
"title": "Infrastructure Plan Explainer",
"level": 2,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Python"
],
"tagline": "Explain replacements, unknown values and dependencies before an infrastructure change.",
"summary": "Read a saved Terraform plan JSON for a developer reviewing a proposed change. Preserve unknown and sensitive-value markers, separate replacement from in-place updates, and trace declared dependencies. Produce a review brief whose explanations link to exact plan addresses; never apply the plan.",
"output": "Markdown change brief and JSON resource-impact graph",
"prerequisiteProjects": [
"json-schema-output-guard"
],
"milestones": [
"Load a versioned plan without exposing sensitive values",
"Classify create, update, replace and delete actions",
"Trace declared dependencies and unknown outcomes",
"Export address-linked review explanations"
],
"references": [
"https://developer.hashicorp.com/terraform/internals/json-format"
],
"distinctFrom": "Cloud Agent reads runtime state; this explains proposed infrastructure changes and values that cannot yet be known.",
"firstDemo": "Inspect a plan that replaces a service dependency while one computed endpoint remains unknown."
},
{
"id": "log-volume-reduction-lab",
"title": "Log Volume Reduction Lab",
"level": 2,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Rust",
"Python"
],
"tagline": "Shrink a noisy log stream while keeping evidence of rare failures.",
"summary": "Help an engineer build compact log context for an assistant. Learn bounded template grouping, counts and representative examples, then compare uniform and rarity-aware sampling on authored failure fixtures. Export compressed JSONL and a coverage report showing which important events were lost.",
"output": "Compact log context JSONL and rare-event coverage comparison",
"prerequisiteProjects": [
"token-counter-and-cost-meter"
],
"milestones": [
"Parse and bound log records",
"Group repeated templates with source locators",
"Compare sampling policies on rare events",
"Export compact context with retained-coverage evidence"
],
"references": [],
"distinctFrom": "Agent Trace Debugger analyzes structured spans; this compresses repetitive log text and measures loss of rare diagnostic events.",
"firstDemo": "Shrink an authored 200-line retry burst while retaining a single disk-full event that uniform sampling misses."
},
{
"id": "plain-language-rewrite-desk",
"title": "Plain Language Rewrite Desk",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Make an explanation easier to read while checking what changed.",
"summary": "Help a technical author revise dense prose using visible sentence and vocabulary measures, protected terms and an optional real rewriting provider. Compare original and proposed paragraphs, surface changed numbers and missing definitions, and export only the author's accepted revisions with a Markdown change record.",
"output": "Accepted Markdown, a paragraph-level HTML comparison and revision-decisions.json",
"prerequisiteProjects": [
"glossary-hovercard-publisher"
],
"milestones": [
"Segment prose and record protected terms and facts",
"Produce simpler candidate paragraphs",
"Compare omissions, numbers and readability measures",
"Export accepted revisions and an editorial change record"
],
"references": [],
"distinctFrom": "Report Judge audits citations across reports; this is an authoring workflow for paragraph revisions and technical vocabulary preservation.",
"firstDemo": "Rewrite an original map-projection explanation while catching a changed numeric example before acceptance.",
"originalFixture": "Original dense and simplified instructional paragraphs with deliberate factual changes in rejected candidates"
},
{
"id": "service-ownership-navigator",
"title": "Service Ownership Navigator",
"level": 2,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"TypeScript"
],
"tagline": "Find a service's owner and runbook with evidence for every answer.",
"summary": "Combine an authored service catalog, repository ownership rules and runbook links into a local directory. Resolve ownership at the requested path, surface conflicts and missing runbooks, and answer handoff questions with file-level evidence. Export a static site and JSON lookup interface without sending notifications.",
"output": "Static service directory and source-backed ownership lookup JSON",
"prerequisiteProjects": [
"semantic-notes-search",
"doc-qa-with-citations"
],
"milestones": [
"Load service metadata and a documented ownership-rule subset",
"Resolve paths and conflicting ownership evidence",
"Check runbook references",
"Export a searchable directory and cited handoff packet"
],
"references": [],
"distinctFrom": "Document QA searches prose; this resolves explicit ownership rules and service relationships with conflict handling.",
"firstDemo": "Find the owner and runbook of a fictional image service while surfacing two conflicting path rules."
},
{
"id": "subtitle-localization-desk",
"title": "Subtitle Localization Desk",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Translate captions while keeping the timing and terminology visible.",
"summary": "Help a course creator localize supplied WebVTT captions through a glossary, an optional real translation provider and an explicit review step. Preserve cue identities and timestamps, flag text that exceeds the chosen reading budget, and export reviewed captions with a side-by-side HTML desk.",
"output": "Reviewed WebVTT files, translation-decisions.json and a bilingual HTML review page",
"prerequisiteProjects": [],
"milestones": [
"Parse subtitle cues and preserve their timing",
"Apply a glossary to translation proposals",
"Review cue length, terminology and meaning",
"Export localized captions and their review decisions"
],
"references": [
"https://www.w3.org/TR/webvtt1/"
],
"distinctFrom": "Voice Note Transcriber Pipeline obtains and reviews source speech; this starts from captions and manages language localization and timing constraints.",
"firstDemo": "Review an authored six-cue lesson translation and fix one overlong caption without shifting the audio timeline.",
"originalFixture": "Original short lesson captions with independently authored translations and a reviewed term glossary"
},
{
"id": "ui-string-localization-workbench",
"title": "UI String Localization Workbench",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Review translated interface text without breaking its placeholders.",
"summary": "Help a small product team localize JSON message catalogs while retaining stable keys, variable placeholders and contextual notes. Compare recorded or real provider proposals in a UI preview, require review of ambiguous strings, and export an approved locale catalog and unresolved-string queue.",
"output": "Reviewed locale JSON, an interactive message preview and unresolved-strings.json",
"prerequisiteProjects": [
"glossary-hovercard-publisher"
],
"milestones": [
"Read message keys, placeholders and context",
"Generate bounded translation proposals",
"Review strings in a rendered interface preview",
"Export the approved locale catalog and remaining questions"
],
"references": [
"https://html.spec.whatwg.org/multipage/dom.html#the-lang-and-xml:lang-attributes"
],
"distinctFrom": "Subtitle Localization Desk preserves temporal cues; this preserves message keys, interpolation contracts and interface context.",
"firstDemo": "Translate a fictional borrowing app's strings and catch a proposal that drops its item-count placeholder.",
"originalFixture": "Original borrowing-app messages and small HTML previews with reviewed translations"
},
{
"id": "workshop-materials-planner",
"title": "Workshop Materials Planner",
"level": 2,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Scale a workshop plan and show exactly which supplies are missing.",
"summary": "Help a workshop host reconcile per-participant materials, reusable equipment and an inventory snapshot. Use explicit unit conversions and declared sharing rules to compute shortages, keep unconvertible quantities visible, and export a packing list with a participant-count comparison.",
"output": "packing-list.csv, shortages.json and an HTML what-if planner",
"prerequisiteProjects": [
"csv-repair-workbench"
],
"milestones": [
"Model participants, materials and inventory units",
"Scale consumables and shared equipment separately",
"Compute shortages and review uncertain conversions",
"Export packing lists and participant-count scenarios"
],
"references": [],
"distinctFrom": "Calendar Focus Planner allocates time; this reconciles material quantities and equipment-sharing constraints.",
"firstDemo": "Change an original paper-craft workshop from eight participants to twelve and inspect the exact extra supplies required.",
"originalFixture": "Original paper-craft bills of materials, reusable-tool counts and inventory snapshots"
},
{
"id": "ci-failure-reproducer",
"title": "CI Failure Reproducer",
"level": 3,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Python",
"TypeScript"
],
"tagline": "Turn a failed build log into a small, reviewable reproduction recipe.",
"summary": "Help a contributor reproduce a failed job from an exported workflow log, commit identity and declared environment. Identify the first failing command, distinguish infrastructure failures from test failures, and produce an allowlisted local replay recipe. Retain the original log lines beside every generated instruction.",
"output": "Reproduction manifest, shell-free replay CLI and HTML failure brief",
"prerequisiteProjects": [
"tiny-coding-agent",
"pr-review-reporter"
],
"milestones": [
"Normalize job steps and source line ranges",
"Separate the initiating failure from downstream noise",
"Build an explicit replay recipe",
"Run an approved local command and compare evidence"
],
"references": [
"https://docs.github.com/en/actions/how-tos/monitor-workflows/use-workflow-run-logs"
],
"distinctFrom": "Tiny Coding Agent patches a local workspace; this recovers the environment and minimal replay steps from a failed CI job.",
"firstDemo": "Reduce an authored failed job to a three-step local reproduction while retaining the original failure lines."
},
{
"id": "community-learning-path-finder",
"title": "Community Learning Path Finder",
"level": 3,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Build a learning path that explains its prerequisites and tradeoffs.",
"summary": "Help a learner choose a feasible sequence from an explicit catalog of workshops and resources. Combine prerequisite graphs, stated goals and a time budget, explain why each resource was selected, and export an editable HTML learning map with a machine-readable path for another course platform.",
"output": "An HTML learning map, path.json and unmet-prerequisites.json",
"prerequisiteProjects": [
"bookmark-path-organizer",
"semantic-notes-search"
],
"milestones": [
"Model resources, goals and prerequisite relationships",
"Find feasible paths and expose missing prerequisites",
"Rank alternatives under an explicit time budget",
"Export an editable learning map and integration contract"
],
"references": [],
"distinctFrom": "Source-Grounded Study Coach schedules cards within a reading; this plans prerequisite-aware progression across complete learning resources.",
"firstDemo": "Plan a short route through an authored neighborhood-mapping curriculum and explain the extra prerequisite required for one advanced workshop.",
"originalFixture": "An original miniature curriculum with explicit durations, prerequisite edges and learner goals"
},
{
"id": "database-migration-rehearsal",
"title": "Database Migration Rehearsal",
"level": 3,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Python"
],
"tagline": "Rehearse a proposed migration on a copy and show exactly what changes.",
"summary": "Let a developer or coding agent propose SQLite migrations against a disposable database copy. Record schema and row-count invariants before and after, test declared rollback steps, and expose irreversible changes. Export a machine-readable rehearsal receipt and review page without touching the source database.",
"output": "Schema diff, invariant results and reversible-migration receipt",
"prerequisiteProjects": [
"csv-sql-question-workbench",
"json-schema-output-guard"
],
"milestones": [
"Snapshot a local database and define invariants",
"Apply migrations only to the disposable copy",
"Exercise rollback and detect data loss",
"Export evidence for an explicit release decision"
],
"references": [],
"distinctFrom": "CSV Question Workbench performs read-only queries; this rehearses schema writes and checks rollback on a disposable database.",
"firstDemo": "Rehearse an authored column migration, catch a lost row and demonstrate the failed rollback check."
},
{
"id": "decision-tradeoff-explorer",
"title": "Decision Tradeoff Explorer",
"level": 3,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"TypeScript"
],
"tagline": "Show when a different priority changes the preferred option.",
"summary": "Help a community team compare practical project options using a user-supplied criteria table and supporting notes. Apply explicit constraints, weighted scores and sensitivity analysis, keep missing values visible, and export an interactive HTML decision board and a reusable decision record.",
"output": "An interactive tradeoff board, criteria.csv and decision-record.json",
"prerequisiteProjects": [
"csv-sql-question-workbench",
"chart-storyboard-builder"
],
"milestones": [
"Declare options, criteria, constraints and source notes",
"Compute comparable scores while preserving unknowns",
"Explore weight sensitivity and dominated options",
"Export the chosen scenario and its assumptions"
],
"references": [],
"distinctFrom": "Research Report Agent gathers and cites evidence; this begins with an explicit evidence table and computes how stated preferences affect a decision.",
"firstDemo": "Compare three original community exhibition layouts and reveal which priority change reverses their ranking.",
"originalFixture": "Original exhibition-layout options with declared space, accessibility, setup-time and visitor-flow criteria"
},
{
"id": "kubernetes-event-storyboard",
"title": "Kubernetes Event Storyboard",
"level": 3,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Go",
"TypeScript"
],
"tagline": "Turn workload events into a timeline with the exact objects behind each clue.",
"summary": "Help an on-call engineer inspect exported Kubernetes events and workload snapshots. Group repeated events, distinguish observation time from occurrence time, and link each suggested investigation to object identity and rollout revision. Export an HTML timeline and a redacted JSON context pack for an assistant.",
"output": "HTML workload timeline and versioned JSON context pack",
"prerequisiteProjects": [
"agent-trace-debugger",
"postmortem-writer"
],
"milestones": [
"Parse event and workload snapshots",
"Group repetitions without losing timestamps",
"Connect evidence to workload revisions",
"Export a redacted assistant context pack"
],
"references": [
"https://kubernetes.io/docs/reference/kubernetes-api/core/event-v1/"
],
"distinctFrom": "Incident Postmortem Writer reviews an incident ledger; this reconstructs workload context from raw Kubernetes objects and event repetitions.",
"firstDemo": "Replay an authored rollout with a failed image pull, repeated warnings and a later successful pod."
},
{
"id": "label-adjudication-desk",
"title": "Label Adjudication Desk",
"level": 3,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python",
"TypeScript"
],
"tagline": "Turn annotator disagreements into a traceable evaluation dataset.",
"summary": "Help a small team create useful evaluation labels without silently replacing disagreement with majority vote. Import independently labeled records, blind annotator identities during review, group disagreements and capture an explicit final decision with its evidence. Export a local review page and versioned JSONL that preserves both original labels and adjudication history.",
"output": "Local adjudication UI, agreement report and provenance-preserving labeled JSONL",
"prerequisiteProjects": [
"dataset-split-auditor",
"feedback-theme-board"
],
"milestones": [
"Import independent labels and preserve record provenance",
"Measure agreement and build a blinded review queue",
"Capture evidence-backed adjudication decisions",
"Export versioned labels with unresolved cases visible"
],
"references": [],
"distinctFrom": "Dataset Split Auditor checks dataset separation and Feedback Theme Board groups comments; this project produces accountable human reference labels for evaluations.",
"firstDemo": "Review three disagreements in an authored annotation set and export both accepted decisions and one unresolved case.",
"implementationBoundary": "Retain unresolved judgments and raw agreement counts. Inter-annotator agreement measures consistency, not the objective truth of a label."
},
{
"id": "lost-item-match-desk",
"title": "Lost Item Match Desk",
"level": 3,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Compare likely item matches without pretending a score proves ownership.",
"summary": "Help a community venue match lost-item descriptions with photographed found objects. Combine explicit text features, simple image-color descriptors and optional reviewed vision proposals, then export an HTML candidate-pair desk and a decision file for the venue's existing records.",
"output": "An HTML object-pair review desk, feature explanations and match-decisions.json",
"prerequisiteProjects": [
"catalog-record-linker",
"photo-burst-contact-sheet"
],
"milestones": [
"Import object descriptions, photos and intake identifiers",
"Compute text and image features independently",
"Rank candidate pairs and review conflicting evidence",
"Export match decisions with their source records"
],
"references": [],
"distinctFrom": "Catalog Record Linker reconciles structured records; this teaches multimodal candidate matching with separate text and image evidence.",
"firstDemo": "Compare original photos of two similar bottles and one umbrella against authored descriptions and leave an ambiguous bottle unresolved.",
"originalFixture": "Original object-only photographs and fictional lost/found intake records"
},
{
"id": "presentation-rehearsal-coach",
"title": "Presentation Rehearsal Coach",
"level": 3,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "See which slide ideas your rehearsal actually covered.",
"summary": "Help a speaker improve a rehearsal by aligning supplied slide text and timed transcript segments. Compute explicit pacing and topic-coverage signals, show the source phrases behind suggested practice targets, and export a slide-by-slide HTML report with a reusable rehearsal comparison file.",
"output": "A slide-linked HTML rehearsal report, coverage.json and a comparison between two rehearsals",
"prerequisiteProjects": [
"voice-note-transcriber-pipeline",
"semantic-notes-search"
],
"milestones": [
"Import slide text and timed rehearsal segments",
"Align spoken segments to candidate slides",
"Measure pacing and review apparent coverage gaps",
"Export a rehearsal report and compare a second attempt"
],
"references": [],
"distinctFrom": "Meeting Notes to Actions extracts commitments; this aligns a speaker's delivery to planned visual material and supports deliberate rehearsal.",
"firstDemo": "Compare two original three-slide rehearsals and identify a skipped definition with its supporting transcript ranges.",
"originalFixture": "An original three-slide talk with two independently authored timed rehearsal transcripts"
},
{
"id": "retry-storm-lab",
"title": "Retry Storm Lab",
"level": 3,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Go",
"TypeScript"
],
"tagline": "Watch many reasonable retries combine into an overloaded service.",
"summary": "Help SDK authors choose retry behavior when many clients fail together. Replay concurrent clients against a deterministic service model, compare bounded backoff and jitter with a shared retry budget, then run the same policy against a local HTTP fixture. Export amplification and recovery timelines with a reusable client policy.",
"output": "Retry policy library, deterministic workload replay and amplification timeline",
"prerequisiteProjects": [
"llm-gateway-with-fallbacks",
"agent-budget-planner"
],
"milestones": [
"Model concurrent failures with a deterministic clock",
"Compare bounded backoff and jitter",
"Enforce shared retry budgets and server delays",
"Replay the policy against a local HTTP service"
],
"references": [],
"distinctFrom": "LLM Gateway With Fallbacks follows one request through provider failures; this project measures the aggregate load created by many clients retrying together.",
"firstDemo": "Start twenty fixture clients at the same failure instant and compare synchronized retries with seeded jitter and a shared retry cap.",
"implementationBoundary": "Seeded simulation explains mechanisms; local measurements do not establish the capacity or recovery behavior of an external service."
},
{
"id": "semantic-cache-correctness-lab",
"title": "Semantic Cache Correctness Lab",
"level": 3,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Go",
"Python"
],
"tagline": "Measure when a similar question can reuse an answer and when it must miss.",
"summary": "Build an opt-in answer cache that binds entries to tenant, model, source revision and expiry. Compare exact and semantic candidate matching using authored near-miss questions, including changed numbers and negation. Export a local HTTP adapter and a false-hit report; similarity alone must not authorize cross-user reuse.",
"output": "Local cache adapter and false-hit, latency and invalidation report",
"prerequisiteProjects": [
"semantic-notes-search",
"rag-freshness-pipeline"
],
"milestones": [
"Define identity, freshness and privacy boundaries",
"Implement exact-match lookup and invalidation",
"Evaluate semantic candidates against adversarial pairs",
"Export a bounded HTTP adapter and reuse evidence"
],
"references": [
"https://www.rfc-editor.org/rfc/rfc9111.html"
],
"distinctFrom": "RAG Freshness Pipeline versions source content; this decides whether a prior answer can be reused for a new query under identity and freshness constraints.",
"firstDemo": "Show a cache hit for a paraphrase and a miss when the question changes a critical number or tenant."
},
{
"id": "streaming-answer-recovery",
"title": "Streaming Answer Recovery",
"level": 3,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"TypeScript",
"Go"
],
"tagline": "Reconnect a streamed answer without duplicating text or hiding an incomplete result.",
"summary": "Build a browser answer view and local event stream for developers integrating assistant output. Handle split UTF-8 bytes, event IDs, reconnection, cancellation and partial citations. Preserve the difference between an interrupted draft and a completed answer, with a downloadable event trace to reproduce rendering failures.",
"output": "Reusable browser component, SSE fixture server and replayable event trace",
"prerequisiteProjects": [
"agent-trace-debugger",
"json-schema-output-guard"
],
"milestones": [
"Decode chunked bytes into complete events",
"Render text and citations by stable event identity",
"Handle reconnect and cancellation without duplicate output",
"Export and replay an interrupted-answer trace"
],
"references": [
"https://html.spec.whatwg.org/multipage/server-sent-events.html"
],
"distinctFrom": "Streaming Agent Shell handles a terminal tool loop; this teaches browser rendering, reconnection and partial-answer state over an event stream.",
"firstDemo": "Disconnect an authored answer midway through a Unicode character, reconnect and replay without duplicate text."
},
{
"id": "tool-contract-migration-checker",
"title": "Tool Contract Migration Checker",
"level": 3,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"TypeScript",
"Python"
],
"tagline": "Find the saved tool calls that a proposed schema change would break.",
"summary": "Help tool authors evolve a callable interface while existing agents still hold old argument shapes. Compare a documented subset of before-and-after schemas, generate counterexamples for required fields and narrowed values, and replay saved calls through both validators. Export a compatibility report and reviewable migration fixtures for CI.",
"output": "Schema compatibility CLI, breaking-call fixtures and CI report",
"prerequisiteProjects": [
"json-schema-output-guard"
],
"milestones": [
"Load versioned tool contracts and saved calls",
"Classify supported schema changes",
"Generate and replay breaking-call counterexamples",
"Export migration fixtures and a compatibility gate"
],
"references": [],
"distinctFrom": "JSON Schema Output Guard validates one payload; this project reasons about compatibility between two tool contracts and supplies concrete calls that change validity. It does not map package upgrades to source call sites.",
"firstDemo": "Add a required argument to a fictional catalog tool and generate a saved call that passes the old contract but fails the new one.",
"implementationBoundary": "Publish the supported schema subset and mark unhandled keywords unresolved instead of declaring full JSON Schema compatibility."
},
{
"id": "video-highlight-storyboard",
"title": "Video Highlight Storyboard",
"level": 3,
"source": "core",
"status": "planned",
"track": "applications",
"languages": [
"Python"
],
"tagline": "Review a clip sequence before spending time editing video.",
"summary": "Help a tutorial creator assemble short excerpts from a supplied video, frame samples and timed transcript. Combine visible shot changes with transcript topic scores, preserve exact source time ranges, and export a playable storyboard and edit-decision JSON for a video editor.",
"output": "A timestamped HTML storyboard, selected thumbnails and edit-decisions.json",
"prerequisiteProjects": [
"voice-note-transcriber-pipeline",
"photo-burst-contact-sheet"
],
"milestones": [
"Align video metadata, frame samples and transcript time",
"Generate shot and topic boundary candidates",
"Review clip selections under a duration budget",
"Export the storyboard and editor-facing time ranges"
],
"references": [],
"distinctFrom": "Voice Note Transcriber Pipeline creates captions; this combines visual and transcript timing to select an editable video sequence.",
"firstDemo": "Select a concise introduction and demonstration from an original paper-folding tutorial without cutting a sentence midway.",
"originalFixture": "An original short recorded craft tutorial, its authored transcript and explicit frame timestamps"
},
{
"id": "a2a-task-handoff-workbench",
"title": "A2A Task Handoff Workbench",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python",
"TypeScript"
],
"tagline": "Trace a task as two independently running agents exchange work and artifacts.",
"summary": "Help teams connect agents that expose different capabilities and task lifecycles. Read Agent Cards, choose a mutually supported binding, and run a handoff that includes a request for more input, cancellation and a final artifact. Export a local client-server pair and a replayable receipt showing every state transition and artifact owner.",
"output": "Two local agent endpoints, handoff client and task/artifact replay bundle",
"prerequisiteProjects": [
"typed-workflow-agent-with-mastra",
"agent-trace-debugger"
],
"milestones": [
"Read Agent Cards and select supported interfaces",
"Create a task and preserve message identity",
"Handle input requests and cancellation",
"Exchange artifacts and replay the complete handoff"
],
"references": [
"https://a2a-protocol.org/latest/specification/"
],
"distinctFrom": "Multi-Agent Code Review Panel coordinates local reviewers for one purpose; this project tests task exchange between independent protocol participants.",
"firstDemo": "Have two local agents assemble an authored workshop brief, pause for one missing input and exchange the final artifact.",
"implementationBoundary": "Select and pin the A2A protocol revision, binding and SDK versions at implementation time. Verify their wire compatibility and capability declarations with real local endpoints before claiming support; optional extensions remain explicit."
},
{
"id": "abstention-calibration-workbench",
"title": "Abstention Calibration Workbench",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python"
],
"tagline": "Choose when an assistant should defer using a measured error-versus-coverage tradeoff.",
"summary": "Help teams set a defensible accept-or-defer rule for a bounded task. Tune a score threshold on labeled development records under explicit error costs, freeze that policy, and evaluate held-out risk and coverage across slices. Export a decision-policy file and audit that shows how many requests need human review.",
"output": "Frozen accept/defer policy JSON, risk-coverage curves and held-out slice audit",
"prerequisiteProjects": [
"dataset-split-auditor",
"local-model-eval-harness"
],
"milestones": [
"Validate separated development and holdout records",
"Tune thresholds under declared error costs",
"Freeze and evaluate the policy on holdout slices",
"Export a decision adapter and review-volume estimate"
],
"references": [],
"distinctFrom": "Local Model Evaluation Harness reports model performance and calibration; this project turns scores into an operational defer policy with a separate threshold-selection phase.",
"firstDemo": "Tune a defer threshold on an authored development split, freeze it and reveal its different coverage on the held-out split.",
"implementationBoundary": "Input scores are not automatically probabilities. Report empirical results and uncertainty without promising a deployment-wide error bound from a small dataset."
},
{
"id": "artifact-provenance-gate",
"title": "Artifact Provenance Gate",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Go",
"Python"
],
"tagline": "Explain whether an artifact matches its signed build record and your trust policy.",
"summary": "Help a maintainer inspect a downloaded agent plugin or release artifact before integrating it. Verify the subject digest and attestation signature, evaluate declared builder and source identities against an explicit policy, and preserve the reason for every rejection. Export a CI gate with authored signed fixtures and a separately verified external-verifier adapter.",
"output": "Artifact verification CLI, trust-policy file and provenance decision JSON",
"prerequisiteProjects": [
"skill-installer",
"skill-validator"
],
"milestones": [
"Bind artifact bytes to the attestation subject",
"Verify signatures against an explicit trust root",
"Evaluate builder and source-material policies",
"Export a CI gate with reproducible rejection fixtures"
],
"references": [
"https://slsa.dev/spec/v1.2/provenance"
],
"distinctFrom": "Cross-Agent Skill Installer writes a local skill package; this project checks artifact origin and build evidence before accepting a package or binary from an external build process.",
"firstDemo": "Verify an authored signed artifact, then change one byte and separately reject a signature from an untrusted builder.",
"implementationBoundary": "Pin the supported attestation formats and verifier versions. Provenance verification establishes particular origin claims, not content safety or automatic SLSA certification."
},
{
"id": "context-compaction-verifier",
"title": "Context Compaction Verifier",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python",
"Rust"
],
"tagline": "Check which task constraints survive when an agent compresses its working context.",
"summary": "Help agent developers avoid losing a user constraint during a long session. Track explicit facts and constraints with source locators, create an extractive compressed context, and test candidate summaries against a declared retention contract. Export compaction middleware that keeps the prior snapshot when required evidence disappears.",
"output": "Compaction middleware, constraint-retention receipt and restorable context snapshots",
"prerequisiteProjects": [
"memory-server",
"token-counter-and-cost-meter"
],
"milestones": [
"Build a source-linked fact and constraint ledger",
"Create a bounded extractive context snapshot",
"Check retention and authored contradiction cases",
"Commit or restore context with an audit receipt"
],
"references": [],
"distinctFrom": "Persistent Memory Server stores persistent records; this project validates one active-session compression step against explicit constraints before replacing working context.",
"firstDemo": "Compress an authored session whose required output format is easy to omit and restore the previous snapshot when that constraint disappears.",
"implementationBoundary": "The baseline verifies explicit structured constraints and authored lexical relations. Optional model summaries require separate evaluation and do not turn these checks into general semantic-equivalence proof."
},
{
"id": "hedged-request-lab",
"title": "Hedged Request Lab",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Go"
],
"tagline": "Measure the extra work required to escape a slow read request.",
"summary": "Help client developers evaluate delayed duplicate requests for safe reads. Send a second attempt only after a configurable delay, accept the first valid response, and record cancellation, late completion and duplicate service work separately. Export a bounded HTTP client and a latency-versus-work report from controlled local endpoints.",
"output": "Hedged-read client, controllable endpoint fixtures and latency/work comparison receipt",
"prerequisiteProjects": [
"llm-gateway-with-fallbacks",
"agent-trace-debugger"
],
"milestones": [
"Measure the latency distribution of safe reads",
"Schedule delayed duplicate attempts",
"Validate winners and account for late work",
"Export latency and duplicate-work tradeoffs"
],
"references": [],
"distinctFrom": "LLM Gateway With Fallbacks makes sequential attempts after failure; this project studies overlapping attempts triggered by slowness and their additional resource cost.",
"firstDemo": "Delay one local read endpoint and measure how a second attempt changes both latency and total service work.",
"implementationBoundary": "Restrict the default adapter to declared safe reads. Client cancellation does not prove that an upstream service stopped computation or billing."
},
{
"id": "local-inference-capacity-planner",
"title": "Local Inference Capacity Planner",
"level": 4,
"source": "core",
"status": "planned",
"track": "platform",
"languages": [
"Python",
"Go"
],
"tagline": "Measure how concurrency changes queueing, latency and successful local requests.",
"summary": "Help a team choose a concurrency limit for its own model endpoint. Replay an authored arrival schedule, record queue wait separately from service time, and compare observed percentiles and timeouts across bounded runs. Export a capacity curve and editable operating limit based on measured requests, not model-name lookup tables.",
"output": "Load-replay manifest, capacity curve and measured operating-limit receipt",
"prerequisiteProjects": [
"local-model-eval-harness",
"agent-budget-planner"
],
"milestones": [
"Define an arrival schedule and resource limits",
"Measure queue and service time independently",
"Compare concurrency settings on the same workload",
"Export a reproducible capacity decision"
],
"references": [],
"distinctFrom": "Local Model Evaluation Harness scores outputs; this measures queue behavior under controlled arrival schedules and concurrency.",
"firstDemo": "Replay the same authored arrivals at three concurrency limits and compare queue wait with service latency."
},
{
"id": "mcp-protocol-compatibility-doctor",
"title": "MCP Compatibility Doctor",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"TypeScript",
"Python"
],
"tagline": "Find the first protocol disagreement between an MCP client and server.",
"summary": "Help an integration author diagnose a client that connects but cannot use a server correctly. Run authored negotiation, request-correlation and capability probes through a bounded transport adapter. Export a compatibility matrix and the smallest wire transcript that explains a failure, with credentials removed.",
"output": "Protocol probe CLI, compatibility JSON and redacted reproduction transcript",
"prerequisiteProjects": [
"mcp-at-scale",
"agent-trace-debugger"
],
"milestones": [
"Record initialization and negotiated capabilities",
"Probe supported operations and request correlation",
"Reduce a failing exchange to its required messages",
"Export a client-server compatibility receipt"
],
"references": [
"https://modelcontextprotocol.io/specification/2025-11-25/basic/lifecycle"
],
"distinctFrom": "MCP Tool Discovery Workbench teaches serving and discovering a tool inventory; this project diagnoses disagreements between independently implemented clients and servers.",
"firstDemo": "Connect two authored local participants whose declared capabilities disagree and export the first failing exchange.",
"implementationBoundary": "Pin the selected MCP specification revision and transport during implementation; validate probes against that revision and report unsupported capabilities explicitly. Planned probes are not a universal conformance certification."
},
{
"id": "metamorphic-evaluation-workbench",
"title": "Metamorphic Evaluation Workbench",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python",
"TypeScript"
],
"tagline": "Test relationships between answers when a single golden answer is not enough.",
"summary": "Help evaluators catch inconsistent behavior across related inputs. Author transformations such as unit conversion, record reordering and equivalent formatting, define the output relationship each should preserve, and shrink violating examples. Export a property-based evaluation runner and small failure bundles that can be consumed by CI.",
"output": "Transformation/property runner, minimized counterexample JSON and CI results",
"prerequisiteProjects": [
"prompt-regression-tester",
"json-schema-output-guard"
],
"milestones": [
"Define input transformations and expected relations",
"Run paired cases with recorded response provenance",
"Check relations and shrink violations",
"Export reproducible property failures for CI"
],
"references": [],
"distinctFrom": "Prompt Regression Tester compares known cases across prompt revisions; this project generates related inputs and checks domain invariants between their outputs.",
"firstDemo": "Reorder an authored table, change equivalent unit notation and isolate the transformation that violates the answer contract.",
"implementationBoundary": "Relations are authored assumptions about a particular task. Passing them does not establish general model quality, and stochastic adapters must retain repeated-run evidence."
},
{
"id": "oauth-resource-boundary-lab",
"title": "OAuth Resource Boundary Lab",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Go",
"TypeScript"
],
"tagline": "Show why a valid token can still be wrong for the requested resource.",
"summary": "Help developers integrate delegated authorization without accepting a token intended for another service. Build local authorization and resource fixtures, follow metadata discovery, bind authorization to the requested resource, and check audience and scope at the server. Export reusable validation middleware and a redacted explanation of rejected requests.",
"output": "Local authorization fixture, resource-server middleware and authorization decision transcript",
"prerequisiteProjects": [
"mcp-at-scale",
"tool-call-firewall"
],
"milestones": [
"Model the client, issuer and resource identities",
"Discover metadata and bind a proof-key authorization flow",
"Reject wrong audiences and insufficient scopes",
"Integrate middleware and export redacted decisions"
],
"references": [
"https://modelcontextprotocol.io/specification/2025-11-25/basic/authorization",
"https://www.rfc-editor.org/info/rfc8707/"
],
"distinctFrom": "Tool Call Firewall checks a proposed tool action; this project establishes which resource and permissions a delegated credential actually authorizes.",
"firstDemo": "Use a locally issued token against two fictional resources and explain why the second resource rejects it.",
"implementationBoundary": "Pin the MCP authorization revision and underlying OAuth requirements before coding. Use a local issuer by default and verify any external identity-provider adapter separately; do not describe OAuth 2.1 draft material as a finalized RFC."
},
{
"id": "outbound-request-policy-proxy",
"title": "Outbound Request Policy Proxy",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Go"
],
"tagline": "Explain every destination decision before an agent's HTTP request leaves the process.",
"summary": "Help developers constrain a fetch tool that follows URLs supplied by untrusted documents. Evaluate the scheme, host, resolved address and each redirect, then connect using the validated destination under explicit size and time limits. Export a local egress adapter and loopback cases that reveal redirect and address-policy mistakes.",
"output": "Policy-aware HTTP adapter, destination decision receipts and adversarial local fixtures",
"prerequisiteProjects": [
"llm-gateway-with-fallbacks",
"tool-call-firewall"
],
"milestones": [
"Parse destinations and define address policies",
"Bind resolution decisions to actual connections",
"Revalidate redirects and bound responses",
"Export an adapter and replay denied requests"
],
"references": [],
"distinctFrom": "LLM Gateway With Fallbacks selects configured providers; this project constrains network destinations for arbitrary URL-fetch operations, including redirect transitions.",
"firstDemo": "Follow an authored local redirect chain whose final destination violates the declared address policy and inspect the blocked hop.",
"implementationBoundary": "Document the supported DNS, proxy and network assumptions. A process-level adapter is not an operating-system sandbox or a guarantee about traffic that bypasses it."
},
{
"id": "release-canary-decision-lab",
"title": "Release Canary Decision Lab",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python",
"Go"
],
"tagline": "Make a candidate release earn more traffic through visible evidence and stopping rules.",
"summary": "Help a team decide whether an agent behavior change is ready for wider exposure. Assign stable request cohorts, compare candidate and baseline error and latency windows, and distinguish insufficient evidence from a failed gate. Export replayable decision receipts and a local shadow adapter that exercises the policy without deploying anything.",
"output": "Canary policy runner, local shadow adapter and stop/continue/insufficient-evidence receipts",
"prerequisiteProjects": [
"harness-bench",
"agent-trace-debugger"
],
"milestones": [
"Assign stable cohorts and define release signals",
"Accumulate comparable baseline and candidate windows",
"Apply minimum-sample and stopping rules",
"Replay decisions through a local shadow adapter"
],
"references": [],
"distinctFrom": "Harness Bench compares policies on a fixed dataset; this project studies sequential release decisions, cohort assignment and evidence accumulated over time.",
"firstDemo": "Replay authored baseline and candidate requests through windows that first lack evidence and then trigger an explicit stop decision.",
"implementationBoundary": "Export recommendations only. State the chosen statistical assumptions and do not interpret a small passing window as universal release safety."
},
{
"id": "secret-injection-sidecar",
"title": "Secret Injection Sidecar",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Rust",
"Python"
],
"tagline": "Let an agent request an authenticated operation using an opaque credential handle.",
"summary": "Help developers keep service credentials out of model-visible tool arguments and routine traces. Build a local broker that resolves opaque handles, binds each handle to a destination and operation, and inserts environment-supplied credentials at the network boundary. Test echoed credential markers and export decision receipts with bounded response filtering.",
"output": "Local credential broker, handle-based client adapter and redacted request receipts",
"prerequisiteProjects": [
"tool-call-firewall",
"rust-agent-shell"
],
"milestones": [
"Define opaque handles and destination-bound grants",
"Inject credentials inside the broker boundary",
"Exercise response and trace redaction fixtures",
"Integrate a client using handles alone"
],
"references": [],
"distinctFrom": "Tool Call Firewall authorizes actions, while this project removes raw credentials from the agent-facing interface and tests the broker's observable boundaries.",
"firstDemo": "Call a local fixture using an opaque handle and demonstrate that its exact echoed credential marker is absent from the returned trace.",
"implementationBoundary": "Use environment secrets and loopback test services. Exact-marker filtering demonstrates a bounded property and cannot guarantee that an arbitrary upstream service never leaks a transformed secret."
},
{
"id": "signed-webhook-inbox",
"title": "Signed Webhook Inbox",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Go"
],
"tagline": "Accept a delivery once and retain the evidence needed to replay it safely.",
"summary": "Help developers feed external events into an agent workflow without losing retries or accepting altered payloads. Verify an authored signature scheme over the raw body, enforce timestamp bounds, and store stable delivery identities before dispatch. Export an HTTP receiver and a replay CLI that distinguishes a transport retry from a new business event.",
"output": "Local webhook receiver, delivery ledger and explicit replay CLI",
"prerequisiteProjects": [
"workflow-hooks",
"durable-agent-jobs"
],
"milestones": [
"Preserve raw request bytes and delivery identity",
"Verify signatures and bounded timestamps",
"Deduplicate retries with a durable inbox",
"Replay accepted deliveries with processing receipts"
],
"references": [],
"distinctFrom": "Durable Agent Jobs handles work after submission; this project establishes delivery authenticity, identity and replay behavior at the inbound HTTP boundary.",
"firstDemo": "Deliver the same authored event twice, alter one payload byte and inspect which deliveries enter the durable inbox.",
"implementationBoundary": "The default signature format is an explicitly authored teaching contract. Any provider adapter must implement and test that provider's actual signature rules."
},
{
"id": "tenant-fairness-scheduler",
"title": "Tenant Fairness Scheduler",
"level": 4,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Rust",
"Python"
],
"tagline": "Keep one busy tenant from consuming the entire agent execution queue.",
"summary": "Help platform developers share a finite worker pool across tenants with different job sizes. Implement weighted deficit scheduling, bounded admission and aging, then compare per-tenant wait distributions under authored bursts. Export a queue adapter and workload receipts that make starvation and unused capacity visible.",
"output": "Fair queue adapter, workload replay CLI and per-tenant wait/throughput report",
"prerequisiteProjects": [
"durable-agent-jobs",
"agent-budget-planner"
],
"milestones": [
"Represent tenant queues and declared job costs",
"Implement weighted deficit scheduling",
"Bound admission and detect starvation",
"Compare fairness on replayed burst workloads"
],
"references": [],
"distinctFrom": "Agent Budget Planner accounts for caller-supplied usage and deadlines; this project decides which tenant receives the next shared execution slot and measures fairness.",
"firstDemo": "Replay a burst from one fictional tenant while a second submits small jobs, then compare their waits under two queue policies.",
"implementationBoundary": "Define fairness against declared or measured service units and expose errors in cost estimates. The local adapter does not claim distributed consensus."
},
{
"id": "agent-resource-deadlock-lab",
"title": "Agent Resource Deadlock Lab",
"level": 5,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Rust",
"TypeScript"
],
"tagline": "Replay the exact resource-acquisition order that leaves cooperating agents stuck.",
"summary": "Help developers coordinate agents that need several exclusive resources, such as a browser session and a workspace. Build wait-for graphs, explore bounded schedules, and compare ordered acquisition with timeout-and-release policies. Export deadlock traces and a resource-coordinator adapter with explicit acquisition receipts.",
"output": "Bounded schedule explorer, deadlock replay traces and resource-coordinator adapter",
"prerequisiteProjects": [
"durable-agent-jobs",
"multi-agent-code-review-panel"
],
"milestones": [
"Model resource ownership and wait-for edges",
"Explore bounded interleavings and detect cycles",
"Compare ordering and timeout-release policies",
"Export a coordinator and replayable deadlock evidence"
],
"references": [],
"distinctFrom": "Durable Agent Jobs coordinates queue leases; this project studies circular wait when tasks hold multiple scarce resources while requesting others.",
"firstDemo": "Replay two local agents acquiring a browser and workspace in opposite order, then demonstrate a schedule that completes.",
"implementationBoundary": "A bounded explorer proves only the schedules it actually examines. Define resource semantics explicitly and distinguish deadlock from slow work, starvation and crashed ownership."
},
{
"id": "judge-bias-calibration-lab",
"title": "Judge Bias Calibration Lab",
"level": 5,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Python",
"TypeScript"
],
"tagline": "Find when a judge prefers answer position or length over the evidence in its rubric.",
"summary": "Help evaluators decide whether an automated pairwise judge is useful for their task. Blind candidate identities, swap answer positions, run authored length counterfactuals and compare decisions with held-out human anchors. Export disagreement and bias breakdowns plus a versioned rubric and judge configuration receipt.",
"output": "Pairwise judging runner, bias scorecard and versioned rubric/configuration bundle",
"prerequisiteProjects": [
"report-judge",
"dataset-split-auditor",
"local-model-eval-harness"
],
"milestones": [
"Create blinded pairs and separated human anchors",
"Run position swaps and length counterfactuals",
"Measure disagreement and slice-specific bias",
"Export a rubric with held-out validation evidence"
],
"references": [],
"distinctFrom": "Report Judge checks report evidence and citation support; this project evaluates the reliability and biases of the evaluator itself.",
"firstDemo": "Swap the positions of two authored answers and expose a recorded judge that changes preference without new evidence.",
"implementationBoundary": "Recorded judge responses form the offline baseline. Any live model judge must retain its model/configuration receipt, and calibration findings apply only to the tested task distribution."
},
{
"id": "trajectory-counterexample-minimizer",
"title": "Trajectory Counterexample Minimizer",
"level": 5,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Rust",
"Python"
],
"tagline": "Reduce a long agent run to the smallest event sequence that still breaks a rule.",
"summary": "Help agent developers diagnose failures that depend on action order. Express bounded temporal rules such as approval preceding an effect, check event traces, and remove irrelevant events while preserving declared causal dependencies. Export a minimal replay bundle with the violated rule and exact event identities.",
"output": "Temporal trace checker, dependency-preserving reducer and minimal replay bundle",
"prerequisiteProjects": [
"agent-trace-debugger",
"tool-call-firewall"
],
"milestones": [
"Normalize events and declared causal dependencies",
"Check bounded temporal rules",
"Minimize failures without breaking prerequisites",
"Export the smallest reproducible violation"
],
"references": [],
"distinctFrom": "Agent Trace Debugger explains execution timing; this project checks event-order properties and automatically reduces a failing trajectory.",
"firstDemo": "Reduce a twenty-event authored run to the approval and effect events that reproduce a stale-approval violation.",
"implementationBoundary": "Causal preservation depends on recorded dependencies and the supported rule language. A minimal trace is relative to the chosen reducer and replay oracle, not a proof of global minimality."
},
{
"id": "transactional-workspace-patch-broker",
"title": "Transactional Workspace Patch Broker",
"level": 5,
"source": "core",
"status": "planned",
"track": "systems",
"languages": [
"Rust",
"TypeScript"
],
"tagline": "Publish a coordinated set of agent edits only when its input revision still matches.",
"summary": "Help concurrent coding agents avoid overwriting each other's files halfway through a multi-file change. Bind a patch bundle to content hashes, apply it to a private versioned workspace, run declared checks, and publish by switching one workspace pointer. Export a transaction API with conflict evidence and an explicit rollback journal.",
"output": "Workspace transaction API, reviewable patch bundle and conflict/rollback journal",
"prerequisiteProjects": [
"tiny-coding-agent",
"sandbox-ladder"
],
"milestones": [
"Bind a multi-file patch to its input revision",
"Stage edits in a private versioned workspace",
"Reject conflicts and run declared validation",
"Publish a workspace pointer and record rollback"
],
"references": [],
"distinctFrom": "Tiny Coding Agent teaches a coding loop; this project makes publication and conflicts explicit when several agents produce overlapping multi-file changes.",
"firstDemo": "Let two local workers propose overlapping edits and show one complete publication alongside the other worker's revision conflict.",
"implementationBoundary": "Atomicity applies to the versioned workspace pointer and readers that honor it. Do not claim arbitrary multi-file writes into an existing checkout are atomic."
}
]
}