Files
Rohit Ghumare 3ecf630b0f feat(projects): add 48 builds and a 52-project roadmap (#497)
* feat(projects): workflow-hooks stage 1

* feat(projects): workflow-hooks stage 2

* feat(projects): workflow-hooks stage 3

* feat(projects): workflow-hooks stage 4

* fix(projects): preserve sponsor navigation in project footer

* feat(projects): agent-budget-planner stage 1

* feat(projects): agent-budget-planner stage 2

* feat(projects): agent-budget-planner stage 3

* feat(projects): agent-budget-planner stage 4

* feat(projects): agent-trace-debugger stage 1

* feat(projects): agent-trace-debugger stage 2

* feat(projects): agent-trace-debugger stage 3

* feat(projects): agent-trace-debugger stage 4

* feat(projects): browser-agent stage 1

* feat(projects): browser-agent stage 2

* feat(projects): browser-agent stage 3

* feat(projects): browser-agent stage 4

* feat(projects): calendar-focus-planner stage 1

* feat(projects): calendar-focus-planner stage 2

* feat(projects): calendar-focus-planner stage 3

* feat(projects): calendar-focus-planner stage 4

* feat(projects): changelog-writer-from-git stage 1

* feat(projects): changelog-writer-from-git stage 2

* feat(projects): changelog-writer-from-git stage 3

* feat(projects): changelog-writer-from-git stage 4

* feat(projects): cloud-agent-with-aws-strands stage 1

* feat(projects): cloud-agent-with-aws-strands stage 2

* feat(projects): cloud-agent-with-aws-strands stage 3

* feat(projects): cloud-agent-with-aws-strands stage 4

* feat(projects): csv-sql-question-workbench stage 1

* feat(projects): csv-sql-question-workbench stage 2

* feat(projects): csv-sql-question-workbench stage 3

* feat(projects): csv-sql-question-workbench stage 4

* feat(projects): dataset-split-auditor stage 1

* feat(projects): dataset-split-auditor stage 2

* feat(projects): dataset-split-auditor stage 3

* feat(projects): dataset-split-auditor stage 4

* feat(projects): desktop-control stage 1

* feat(projects): desktop-control stage 2

* feat(projects): desktop-control stage 3

* feat(projects): desktop-control stage 4

* feat(projects): distributed-eval-farm stage 1

* feat(projects): distributed-eval-farm stage 2

* feat(projects): distributed-eval-farm stage 3

* feat(projects): distributed-eval-farm stage 4

* feat(projects): doc-qa-with-citations stage 1

* feat(projects): doc-qa-with-citations stage 2

* feat(projects): doc-qa-with-citations stage 3

* feat(projects): doc-qa-with-citations stage 4

* feat(projects): document-extraction-desk stage 1

* feat(projects): document-extraction-desk stage 2

* feat(projects): document-extraction-desk stage 3

* feat(projects): document-extraction-desk stage 4

* feat(projects): durable-agent-jobs stage 1

* feat(projects): durable-agent-jobs stage 2

* feat(projects): durable-agent-jobs stage 3

* feat(projects): durable-agent-jobs stage 4

* feat(projects): feedback-theme-board stage 1

* feat(projects): feedback-theme-board stage 2

* feat(projects): feedback-theme-board stage 3

* feat(projects): feedback-theme-board stage 4

* feat(projects): harness-bench stage 1

* feat(projects): harness-bench stage 2

* feat(projects): harness-bench stage 3

* feat(projects): harness-bench stage 4

* feat(projects): inbox-triage-desk stage 1

* feat(projects): inbox-triage-desk stage 2

* feat(projects): inbox-triage-desk stage 3

* feat(projects): inbox-triage-desk stage 4

* feat(projects): json-schema-output-guard stage 1

* feat(projects): json-schema-output-guard stage 2

* feat(projects): json-schema-output-guard stage 3

* feat(projects): json-schema-output-guard stage 4

* feat(projects): llm-gateway-with-fallbacks stage 1

* feat(projects): llm-gateway-with-fallbacks stage 2

* feat(projects): llm-gateway-with-fallbacks stage 3

* feat(projects): llm-gateway-with-fallbacks stage 4

* feat(projects): local-model-eval-harness stage 1

* feat(projects): local-model-eval-harness stage 2

* feat(projects): local-model-eval-harness stage 3

* feat(projects): local-model-eval-harness stage 4

* feat(projects): mcp-at-scale stage 1

* feat(projects): mcp-at-scale stage 2

* feat(projects): mcp-at-scale stage 3

* feat(projects): mcp-at-scale stage 4

* feat(projects): mcp-at-scale stage 5

* feat(projects): meeting-notes-to-actions stage 1

* feat(projects): meeting-notes-to-actions stage 2

* feat(projects): meeting-notes-to-actions stage 3

* feat(projects): meeting-notes-to-actions stage 4

* feat(projects): memory-server stage 1

* feat(projects): memory-server stage 2

* feat(projects): memory-server stage 3

* feat(projects): memory-server stage 4

* feat(projects): multi-agent-code-review-panel stage 1

* feat(projects): multi-agent-code-review-panel stage 2

* feat(projects): multi-agent-code-review-panel stage 3

* feat(projects): multi-agent-code-review-panel stage 4

* feat(projects): postmortem-writer stage 1

* feat(projects): postmortem-writer stage 2

* feat(projects): postmortem-writer stage 3

* feat(projects): postmortem-writer stage 4

* feat(projects): pr-review-reporter stage 1

* feat(projects): pr-review-reporter stage 2

* feat(projects): pr-review-reporter stage 3

* feat(projects): pr-review-reporter stage 4

* feat(projects): prompt-regression-tester stage 1

* feat(projects): prompt-regression-tester stage 2

* feat(projects): prompt-regression-tester stage 3

* feat(projects): prompt-regression-tester stage 4

* feat(projects): rag-freshness-pipeline stage 1

* feat(projects): rag-freshness-pipeline stage 2

* feat(projects): rag-freshness-pipeline stage 3

* feat(projects): rag-freshness-pipeline stage 4

* feat(projects): report-judge stage 1

* feat(projects): report-judge stage 2

* feat(projects): report-judge stage 3

* feat(projects): report-judge stage 4

* feat(projects): research-report-agent stage 1

* feat(projects): research-report-agent stage 2

* feat(projects): research-report-agent stage 3

* feat(projects): research-report-agent stage 4

* feat(projects): research-report-agent stage 5

* feat(projects): research-report-agent stage 6

* feat(projects): research-report-agent stage 7

* feat(projects): retrieval-evaluation-lab stage 1

* feat(projects): retrieval-evaluation-lab stage 2

* feat(projects): retrieval-evaluation-lab stage 3

* feat(projects): retrieval-evaluation-lab stage 4

* feat(projects): rust-agent-shell stage 1

* feat(projects): rust-agent-shell stage 2

* feat(projects): rust-agent-shell stage 3

* feat(projects): rust-agent-shell stage 4

* feat(projects): sandbox-ladder stage 1

* feat(projects): sandbox-ladder stage 2

* feat(projects): sandbox-ladder stage 3

* feat(projects): sandbox-ladder stage 4

* feat(projects): self-improving-skill-loop stage 1

* feat(projects): self-improving-skill-loop stage 2

* feat(projects): self-improving-skill-loop stage 3

* feat(projects): self-improving-skill-loop stage 4

* feat(projects): semantic-notes-search stage 1

* feat(projects): semantic-notes-search stage 2

* feat(projects): semantic-notes-search stage 3

* feat(projects): semantic-notes-search stage 4

* feat(projects): skill-installer stage 1

* feat(projects): skill-installer stage 2

* feat(projects): skill-installer stage 3

* feat(projects): skill-installer stage 4

* feat(projects): skill-router stage 1

* feat(projects): skill-router stage 2

* feat(projects): skill-router stage 3

* feat(projects): skill-router stage 4

* feat(projects): skill-scanner stage 1

* feat(projects): skill-scanner stage 2

* feat(projects): skill-scanner stage 3

* feat(projects): skill-scanner stage 4

* feat(projects): skill-validator stage 1

* feat(projects): skill-validator stage 2

* feat(projects): skill-validator stage 3

* feat(projects): skill-validator stage 4

* feat(projects): source-grounded-study-coach stage 1

* feat(projects): source-grounded-study-coach stage 2

* feat(projects): source-grounded-study-coach stage 3

* feat(projects): source-grounded-study-coach stage 4

* feat(projects): support-agent-with-google-adk stage 1

* feat(projects): support-agent-with-google-adk stage 2

* feat(projects): support-agent-with-google-adk stage 3

* feat(projects): support-agent-with-google-adk stage 4

* feat(projects): tiny-coding-agent stage 1

* feat(projects): tiny-coding-agent stage 2

* feat(projects): tiny-coding-agent stage 3

* feat(projects): tiny-coding-agent stage 4

* feat(projects): token-counter-and-cost-meter stage 1

* feat(projects): token-counter-and-cost-meter stage 2

* feat(projects): token-counter-and-cost-meter stage 3

* feat(projects): token-counter-and-cost-meter stage 4

* feat(projects): tool-call-firewall stage 1

* feat(projects): tool-call-firewall stage 2

* feat(projects): tool-call-firewall stage 3

* feat(projects): tool-call-firewall stage 4

* feat(projects): typed-workflow-agent-with-mastra stage 1

* feat(projects): typed-workflow-agent-with-mastra stage 2

* feat(projects): typed-workflow-agent-with-mastra stage 3

* feat(projects): typed-workflow-agent-with-mastra stage 4

* feat(projects): visual-evidence-library stage 1

* feat(projects): visual-evidence-library stage 2

* feat(projects): visual-evidence-library stage 3

* feat(projects): visual-evidence-library stage 4

* feat(projects): voice-note-transcriber-pipeline stage 1

* feat(projects): voice-note-transcriber-pipeline stage 2

* feat(projects): voice-note-transcriber-pipeline stage 3

* feat(projects): voice-note-transcriber-pipeline stage 4

* feat(projects): web-change-brief stage 1

* feat(projects): web-change-brief stage 2

* feat(projects): web-change-brief stage 3

* feat(projects): web-change-brief stage 4

* feat(projects): workflow-hooks stage 1

* feat(projects): workflow-hooks stage 2

* feat(projects): workflow-hooks stage 3

* feat(projects): workflow-hooks stage 4

* feat(projects): publish the practical application catalog

* feat(projects): add 52 planned builds to the roadmap

* fix(projects): make lesson animations explain computed changes

Stage mechanisms should expose the decisions learners are implementing. Preserve record identity as state changes and keep project context below the selected lesson.

* fix(projects): agent-budget-planner stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-budget-planner stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-budget-planner stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-budget-planner stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): agent-trace-debugger stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): browser-agent stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): browser-agent stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): browser-agent stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): calendar-focus-planner stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): cloud-agent-with-aws-strands stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): cloud-agent-with-aws-strands stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): cloud-agent-with-aws-strands stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): desktop-control stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): distributed-eval-farm stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): distributed-eval-farm stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): distributed-eval-farm stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): document-extraction-desk stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): durable-agent-jobs stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): feedback-theme-board stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): inbox-triage-desk stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): llm-gateway-with-fallbacks stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): llm-gateway-with-fallbacks stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): mcp-at-scale review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): meeting-notes-to-actions stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): meeting-notes-to-actions stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): memory-server stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): pr-review-reporter stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): pr-review-reporter stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): pr-review-reporter stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): research-report-agent stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): retrieval-evaluation-lab review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): rust-agent-shell stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): rust-agent-shell stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): semantic-notes-search stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): skill-installer review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): skill-router stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): skill-router stage 04 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): tiny-coding-agent stage 02 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): voice-note-transcriber-pipeline stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): web-change-brief stage 03 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): workflow-hooks stage 01 contracts

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* fix(projects): site review corrections

Keep the stage contract, learner scaffold and verification evidence
consistent at the reviewed failure boundary.

* docs(projects): record verified review corrections
2026-09-29 21:38:54 +05:30

373 lines
20 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Grade cumulative project stages with standard-library test runners.
The authoring and completion-evidence contracts live in projects/AUTHORING.md.
Reference solutions are never eligible for learner completion certificates.
"""
import argparse
from datetime import datetime, timezone
import hashlib
import json
import os
from pathlib import Path
import re
import shutil
import signal
import subprocess
import sys
import tempfile
import time
REPO = Path(__file__).resolve().parent.parent
PROJECTS = REPO / "projects"
LANGUAGES = {"python", "typescript", "rust", "go"}
SLUG = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
def contained(root, candidate):
root, candidate = root.resolve(), candidate.resolve()
if candidate != root and root not in candidate.parents:
raise ValueError(f"path escapes {root}: {candidate}")
return candidate
def load_project(project_id):
if not SLUG.fullmatch(project_id):
raise ValueError(f"invalid project id: {project_id!r}")
root = contained(PROJECTS, PROJECTS / project_id)
manifest = contained(root, root / "project.json")
if not manifest.is_file():
raise ValueError(f"unknown project {project_id!r}")
meta = json.loads(manifest.read_text(encoding="utf-8"))
if meta.get("id") != project_id or not isinstance(meta.get("stages"), list) or not meta["stages"]:
raise ValueError(f"{manifest}: id must match directory and stages must be nonempty")
seen = set()
for stage in meta["stages"]:
sid = stage.get("id", "")
if not SLUG.fullmatch(sid) or sid in seen:
raise ValueError(f"{manifest}: invalid or duplicate stage id {sid!r}")
seen.add(sid)
stage_dir = contained(root, root / "stages" / sid)
contained(root, stage_dir / "tests")
runner_specs(stage, meta)
return root, meta
def stage_dirs(root, meta):
return [contained(root, root / "stages" / stage["id"]) for stage in meta["stages"]]
def runner_specs(stage, meta):
language = stage.get("language") or str((meta.get("languages") or ["Python"])[0]).lower()
specs = stage.get("runners")
if specs is None:
specs = [{"language": language, **({"argv": stage["runner"]} if "runner" in stage else {})}]
if not isinstance(specs, list) or not specs:
raise ValueError(f"{stage['id']}: runners must be a nonempty array")
result = []
for entry in specs:
if not isinstance(entry, dict):
raise ValueError(f"{stage['id']}: runner must be an object")
spec = dict(entry)
if "optional" in spec and not isinstance(spec["optional"], bool):
raise ValueError(f"{stage['id']}: optional must be boolean")
spec["language"] = spec.get("language", language).lower()
if spec["language"] not in LANGUAGES:
raise ValueError(f"{stage['id']}: unsupported language {spec['language']!r}")
timeout = spec.get("timeout", stage.get("timeout", 60))
if isinstance(timeout, bool) or not isinstance(timeout, (int, float)) or not 0 < timeout <= 600:
raise ValueError(f"{stage['id']}: timeout must be between 0 and 600 seconds")
spec["timeout"] = timeout
requirements = stage.get("requires", []) + spec.get("requires", [])
if not isinstance(requirements, list) or any(not isinstance(item, str) or not item for item in requirements):
raise ValueError(f"{stage['id']}: requires must list executable names")
spec["requires"] = list(dict.fromkeys(requirements))
if "argv" in spec and (not isinstance(spec["argv"], list) or not spec["argv"] or any(not isinstance(arg, str) or not arg for arg in spec["argv"])):
raise ValueError(f"{stage['id']}: runner argv must be a nonempty string array")
result.append(spec)
return result
def summarize_failure(output):
missing = list(dict.fromkeys(re.findall(r"NotImplementedError: (.+)|(?:not yet implemented: |panic: )?(Stage \d+: implement [A-Za-z_][\w./ -]*)", output)))
if missing:
return "not implemented yet: " + next(part for part in missing[0] if part).strip()
imports = re.findall(r"(?:ModuleNotFoundError|ImportError): (.+)", output)
if imports:
return "import failed: " + imports[-1].strip()
assertions = re.findall(r"^(?:AssertionError: .+|error.+|FAIL.+)", output, re.MULTILINE)
return (assertions[0][:180] if assertions else "tests failed; rerun with --verbose")
def execute(argv, cwd, env, timeout):
started = time.monotonic()
try:
process = subprocess.Popen(argv, cwd=cwd, env=env, stdout=subprocess.PIPE,
stderr=subprocess.STDOUT, text=True, start_new_session=os.name != "nt")
except (OSError, ValueError) as error:
return 1, str(error), False
try:
output, _ = process.communicate(timeout=timeout)
return process.returncode, output, False
except subprocess.TimeoutExpired:
if os.name != "nt":
os.killpg(process.pid, signal.SIGKILL)
else:
process.kill()
output, _ = process.communicate()
return 1, output + f"\ntimeout after {time.monotonic() - started:.1f}s", True
def test_counts(language, output):
if language == "python":
matches = re.findall(r"Ran (\d+) tests?", output)
skipped = sum(int(n) for n in re.findall(r"skipped=(\d+)", output))
return sum(map(int, matches)), skipped
if language == "typescript":
tests = re.findall(r"(?:#|ℹ) tests (\d+)", output)
skipped = re.findall(r"(?:#|ℹ) (?:skipped|todo) (\d+)", output)
return sum(map(int, tests)), sum(map(int, skipped))
if language == "rust":
summaries = re.findall(r"test result: \w+\. (\d+) passed; (\d+) failed; (\d+) ignored;", output)
return sum(int(a) + int(b) + int(c) for a, b, c in summaries), sum(int(c) for _, _, c in summaries)
total = skipped = 0
for line in output.splitlines():
try:
event = json.loads(line)
except json.JSONDecodeError:
continue
if event.get("Test") and event.get("Action") in ("pass", "fail", "skip"):
total += 1
skipped += event["Action"] == "skip"
return total, skipped
def runner_result(spec, root, stage_dir, workspace, verbose):
started = time.monotonic()
language = spec["language"]
result = {"language": language, "status": "fail", "tests": 0, "skippedTests": 0, "reason": "", "durationMs": 0}
runtime = {"python": sys.executable, "typescript": "node", "rust": "rustc", "go": "go"}[language]
substitutions = {"workspace": str(workspace), "project": str(root), "stage": str(stage_dir), "tests": str(stage_dir / "tests"), "python": sys.executable}
argv = [arg.format_map(substitutions) for arg in spec["argv"]] if "argv" in spec else None
packages = [name for name in spec["requires"] if name.startswith(("python:", "node:"))]
required = [runtime] + [name for name in spec["requires"] if name not in packages] + ([argv[0]] if argv else [])
missing = [name for name in dict.fromkeys(required) if shutil.which(name) is None]
if missing:
result.update(status="skip", reason="missing executable: " + ", ".join(missing))
return result
env = dict(os.environ, PROJECT_WORKSPACE=str(workspace), PROJECT_ROOT=str(root), PROJECT_STAGE=str(stage_dir), PYTHONDONTWRITEBYTECODE="1")
env.pop("NODE_TEST_CONTEXT", None)
env["PYTHONPATH"] = str(workspace) + (os.pathsep + env["PYTHONPATH"] if env.get("PYTHONPATH") else "")
for requirement in packages:
ecosystem, name = requirement.split(":", 1)
if not re.fullmatch(r"[A-Za-z0-9_@][A-Za-z0-9_./@-]*", name):
raise ValueError(f"invalid dependency requirement: {requirement}")
if ecosystem == "python":
probe = "import importlib.util, importlib.machinery, sys\nparts=sys.argv[1].split('.')\nspec=importlib.util.find_spec(parts[0])\nfor index in range(1,len(parts)):\n if spec is None or spec.submodule_search_locations is None: spec=None; break\n spec=importlib.machinery.PathFinder.find_spec('.'.join(parts[:index+1]),spec.submodule_search_locations)\nsys.exit(0 if spec is not None else 1)"
command = [sys.executable, "-c", probe, name]
distribution = {"google.adk": "google-adk", "langchain_text_splitters": "langchain-text-splitters", "strands": "strands-agents"}.get(name, name.replace("_", "-"))
hint = f"python3 -m pip install {distribution}"
else:
command = ["node", "--input-type=module", "-e", "try { import.meta.resolve(process.argv[1]); } catch { process.exit(1); }", name]
hint = f"npm install {name}"
code, _, _ = execute(command, workspace, env, min(10, spec["timeout"]))
if code:
result.update(status="skip", reason=f"missing dependency {requirement}; install with: {hint}")
return result
if language == "typescript":
code, version, timed_out = execute(["node", "--version"], workspace, env, min(10, spec["timeout"]))
match = re.fullmatch(r"v(\d+)\.(\d+)\.(\d+)", version.strip())
if code or not match or tuple(map(int, match.groups())) < (22, 18, 0):
result.update(status="skip", reason="Node.js 22.18 or later is required")
return result
tests = stage_dir / "tests"
outputs, commands = [], []
with tempfile.TemporaryDirectory(prefix="aifs-project-grade-") as temp:
temp_path = Path(temp)
cwd = workspace
if argv:
commands = [argv]
elif language == "python":
commands = [[sys.executable, "-m", "unittest", "discover", "-s", str(tests), "-p", "test_*.py", "-t", str(tests)]]
elif language == "typescript":
files = sorted([*tests.glob("*.test.ts"), *tests.glob("*.test.mjs")])
if files:
commands = [["node", "--experimental-strip-types", "--test-reporter=tap", "--test", *map(str, files)]]
elif language == "rust":
for index, file in enumerate(sorted(tests.glob("*.rs"))):
binary = str(temp_path / f"stage-test-{index}")
commands.extend([["rustc", "--edition", "2021", "--test", str(file), "-o", binary], [binary]])
elif language == "go":
shutil.copytree(workspace, temp_path / "workspace", ignore=shutil.ignore_patterns(".git", "node_modules", "target", "__pycache__"))
cwd = temp_path / "workspace"
for file in tests.rglob("*.go"):
target = contained(cwd, cwd / file.relative_to(tests))
target.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(contained(root, file), target)
if not (cwd / "go.mod").exists():
env["GO111MODULE"] = "off"
env["GOWORK"] = "off"
commands = [["go", "test", "-json", "./..."]]
code = 0
timed_out = False
for command in commands:
remaining = spec["timeout"] - (time.monotonic() - started)
if remaining <= 0:
code, timed_out = 1, True
break
code, output, timed_out = execute(command, cwd, env, remaining)
outputs.append(output)
if code:
break
output = "\n".join(outputs)
result["tests"], result["skippedTests"] = test_counts(language, output)
result["durationMs"] = round((time.monotonic() - started) * 1000)
if verbose and output:
print(output, file=sys.stderr)
if timed_out:
result["reason"] = f"timeout after {spec['timeout']} seconds"
elif not result["tests"] and code in (0, 5):
result["reason"] = "no tests ran; an empty suite is not a pass"
elif code:
result["reason"] = summarize_failure(output)
elif result["skippedTests"]:
result.update(status="skip", reason=f"{result['skippedTests']} tests skipped or pending")
else:
result["status"] = "pass"
return result
def run_stage(root, info, number, code_path, meta, verbose, optional=False):
started = time.monotonic()
stage_dir = contained(root, root / "stages" / info["id"])
result = {"id": info["id"], "number": number, "language": info.get("language", str(meta.get("languages", ["Python"])[0]).lower()),
"status": "fail", "tests": 0, "skippedTests": 0, "durationMs": 0, "reason": "", "runners": []}
if not code_path.is_dir():
result["reason"] = f"workspace does not exist: {code_path}"
return result
for spec in runner_specs(info, meta):
if spec.get("optional") and not optional:
continue
result["runners"].append(runner_result(spec, root, stage_dir, code_path, verbose))
statuses = [entry["status"] for entry in result["runners"]]
result["status"] = "fail" if not statuses or "fail" in statuses else "skip" if "skip" in statuses else "pass"
result["tests"] = sum(entry["tests"] for entry in result["runners"])
result["skippedTests"] = sum(entry["skippedTests"] for entry in result["runners"])
result["durationMs"] = round((time.monotonic() - started) * 1000)
result["reason"] = "; ".join(entry["reason"] for entry in result["runners"] if entry["reason"])
return result
def init_workspace(root, meta, target, force):
target = Path(target).resolve()
if target == root or root in target.parents or target == REPO or target == PROJECTS:
raise ValueError("initialize a learner workspace outside the project's source directory")
target.mkdir(parents=True, exist_ok=True)
copied, skipped = [], []
for stage in stage_dirs(root, meta):
starter = contained(root, stage / "starter")
for source in sorted(starter.rglob("*")):
if not source.is_file():
continue
contained(root, source)
destination = contained(target, target / source.relative_to(starter))
if destination.exists() and not force:
skipped.append(destination)
continue
destination.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(source, destination)
copied.append(destination)
print(f"workspace {target}")
for path in copied:
print(f" created {path.relative_to(target)}")
for path in skipped:
print(f" kept {path.relative_to(target)}")
print(f"next: python3 scripts/project_test.py {meta['id']} --stage 1 --path {target}")
def list_stages(meta):
print(f"{meta['title']} (level {meta['level']}, ~{meta['hours']}h)")
for number, stage in enumerate(meta["stages"], 1):
print(f" {number}. [{stage.get('difficulty', 'core'):<7}] {stage['title']} ~{stage['hours']}h")
def main(argv=None):
parser = argparse.ArgumentParser(description="Grade projects with offline, standard-library runners.")
parser.add_argument("project", nargs="?", help="project id")
parser.add_argument("--all", action="store_true", help="all stages, or all ready projects when no id is given")
parser.add_argument("--stage", type=int, help="grade stages 1..N")
parser.add_argument("--only", action="store_true", help="grade only --stage N")
parser.add_argument("--path", default=".", help="learner workspace; with all projects, a parent containing project-id workspaces")
parser.add_argument("--solution", action="store_true", help="grade reference solutions (no learner certificates)")
parser.add_argument("--strict", action="store_true", help="fail on missing runtimes or skipped tests")
parser.add_argument("--optional", action="store_true", help="include optional framework comparison runners")
parser.add_argument("--report", metavar="FILE", help="write JSON completion evidence")
parser.add_argument("--init", metavar="DIR", help="copy all stage starter files non-destructively")
parser.add_argument("--force", action="store_true", help="overwrite existing files with --init")
parser.add_argument("--list", action="store_true", help="list stages")
parser.add_argument("--verbose", action="store_true", help="show runner output")
args = parser.parse_args(argv)
if not args.project and not args.all:
parser.error("provide a project id or --all")
if args.only and args.stage is None:
parser.error("--only requires --stage")
if args.all and (args.stage is not None or args.only):
parser.error("--all cannot be combined with --stage or --only")
if args.force and not args.init:
parser.error("--force requires --init")
if args.init and not args.project:
parser.error("--init requires a project id")
try:
ids = [args.project] if args.project else sorted(p.name for p in PROJECTS.iterdir() if p.is_dir() and not p.name.startswith(("_", ".")) and (p / "project.json").is_file() and json.loads((p / "project.json").read_text()).get("status") == "ready")
if not ids:
raise ValueError("no ready projects found")
report = {"schemaVersion": 1, "generatedAt": datetime.now(timezone.utc).isoformat(), "projects": [], "certificateEligible": False}
failed = False
for project_id in ids:
root, meta = load_project(project_id)
if args.list:
list_stages(meta)
continue
if args.init:
init_workspace(root, meta, args.init, args.force)
continue
total = len(meta["stages"])
last = args.stage if args.stage is not None else total
if not 1 <= last <= total:
raise ValueError(f"{project_id}: --stage must be between 1 and {total}")
selected = [last] if args.only else list(range(1, last + 1))
workspace = contained(root, root / "solution") if args.solution else (Path(args.path) / (project_id if not args.project else "")).resolve()
reference = contained(root, root / "solution")
reference_mode = args.solution or workspace == reference or reference in workspace.parents
entry = {"id": project_id, "title": meta["title"], "mode": "solution" if reference_mode else "learner",
"optionalIncluded": args.optional,
"manifestHash": hashlib.sha256((root / "project.json").read_bytes()).hexdigest(),
"selectedStages": [meta["stages"][n - 1]["id"] for n in selected], "stages": [],
"allStagesPassed": False, "certificateEligible": False}
print(f"{meta['title']}: grading {entry['mode']}")
for number in selected:
info = meta["stages"][number - 1]
result = run_stage(root, info, number, workspace, meta, args.verbose, args.optional)
entry["stages"].append(result)
print(f" {result['status'].upper():4} stage {number}: {info['title']} ({result['tests']} tests)")
if result["reason"]:
print(f" {result['reason']}")
failed |= result["status"] == "fail" or (args.strict and result["status"] == "skip")
entry["allStagesPassed"] = len(selected) == total and all(stage["status"] == "pass" and stage["tests"] > 0 and stage["skippedTests"] == 0 for stage in entry["stages"])
entry["certificateEligible"] = entry["allStagesPassed"] and entry["mode"] == "learner"
report["projects"].append(entry)
report["certificateEligible"] = bool(report["projects"]) and all(entry["certificateEligible"] for entry in report["projects"])
if args.report:
output = Path(args.report).resolve()
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
print(f"report: {output}")
return int(failed)
except (ValueError, OSError, KeyError, TypeError) as error:
print(f"project grader: {error}", file=sys.stderr)
return 2
if __name__ == "__main__":
raise SystemExit(main())