Merge pull request #105871 from NousResearch/ethie/rip-publish-evidence

chore(ci): remove publish-e2e-evidence pipeline
This commit is contained in:
ethernet
2026-09-24 23:56:13 -04:00
committed by GitHub
6 changed files with 2 additions and 699 deletions
-14
View File
@@ -153,7 +153,6 @@ jobs:
python3 ../../scripts/ci/e2e_screenshot_status.py \
--results-dir test-results \
--manifest-output /tmp/e2e-screenshot-manifest.json \
--evidence-dir /tmp/e2e-evidence \
--artifact-url "$RESULTS_URL" \
--output /tmp/e2e-review-status.json
{
@@ -174,19 +173,6 @@ jobs:
overwrite: true
if-no-files-found: ignore
# The trusted workflow_run publisher consumes only this flat, bounded
# artifact. It turns selected images into GitHub attachment URLs; it
# never checks out or runs this PR's code.
- name: Upload inline E2E evidence
if: always() && github.ref_name != 'main'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: e2e-evidence-${{ github.sha }}
path: /tmp/e2e-evidence
retention-days: 14
overwrite: true
if-no-files-found: error
# ── Generate step summary with visual diff info ───────────────────
# Parse the JSON report + scan for diff images, then post a summary
# to the GitHub Actions step output so reviewers can see what changed
@@ -1,80 +0,0 @@
name: Publish E2E evidence
# This runs only from the default branch after CI completes. It intentionally
# checks out main, never the PR ref, and treats the downloaded artifact as
# untrusted input before uploading validated GitHub attachments.
on:
workflow_run:
workflows: [CI]
types: [completed]
permissions:
actions: read
contents: read
pull-requests: write
concurrency:
group: publish-e2e-evidence-${{ github.event.workflow_run.id }}
cancel-in-progress: false
jobs:
publish:
name: Publish inline E2E evidence
if: github.event.workflow_run.event == 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 10
environment: gh-image
steps:
- name: Check out trusted publisher
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ github.event.repository.default_branch }}
persist-credentials: false
# v1.2.0 resolves to 44f4b93ecbbe22de6c45fa2f62f519aee564ca8c.
- name: Install gh-image
env:
GH_TOKEN: ${{ github.token }}
run: gh extension install drogers0/gh-image --pin v1.2.0
- name: Download and attach evidence
env:
GH_TOKEN: ${{ github.token }}
GITHUB_TOKEN: ${{ github.token }}
GH_SESSION_TOKEN: ${{ secrets.GH_IMAGE_SESSION_TOKEN }}
SOURCE_REPO: ${{ github.repository }}
SOURCE_RUN_ID: ${{ github.event.workflow_run.id }}
HEAD_OWNER: ${{ github.event.workflow_run.head_repository.owner.login }}
HEAD_BRANCH: ${{ github.event.workflow_run.head_branch }}
HEAD_SHA: ${{ github.event.workflow_run.head_sha }}
run: |
set -euo pipefail
# The run's own ``pull_requests`` payload is always empty for a
# fork PR, so resolve the PR from its head reference instead.
# The head-SHA match skips runs that a newer push superseded.
PR_NUMBER=$(gh api -X GET "repos/$SOURCE_REPO/pulls" \
-f head="$HEAD_OWNER:$HEAD_BRANCH" -f state=open \
--jq '.[] | select(.head.sha == $ENV.HEAD_SHA) | .number' \
| head -n1)
if [ -z "$PR_NUMBER" ]; then
echo "No open pull request has head $HEAD_OWNER:$HEAD_BRANCH at $HEAD_SHA (CI run $SOURCE_RUN_ID)."
exit 0
fi
ARTIFACT_NAME=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID/artifacts" \
--jq '.artifacts[] | select(.expired == false and (.name | startswith("e2e-evidence-"))) | .name' \
| python3 -c 'import sys; print(next(iter(sys.stdin), "").strip())')
if [ -z "$ARTIFACT_NAME" ]; then
echo "No E2E evidence artifact was produced for CI run $SOURCE_RUN_ID."
exit 0
fi
EVIDENCE_DIR="$RUNNER_TEMP/e2e-evidence"
mkdir -p "$EVIDENCE_DIR"
gh run download "$SOURCE_RUN_ID" --repo "$SOURCE_REPO" --name "$ARTIFACT_NAME" --dir "$EVIDENCE_DIR"
python3 scripts/ci/publish_e2e_evidence.py \
--evidence-dir "$EVIDENCE_DIR" \
--source-repo "$SOURCE_REPO" \
--pr-number "$PR_NUMBER"
+1 -46
View File
@@ -4,14 +4,10 @@
from __future__ import annotations
import argparse
import hashlib
import json
import shutil
from pathlib import Path
SOURCE = "playwright e2e"
EVIDENCE_START = "<!-- hermes-e2e-evidence:start -->"
EVIDENCE_END = "<!-- hermes-e2e-evidence:end -->"
def _files(root: Path, pattern: str) -> list[Path]:
@@ -46,12 +42,6 @@ def _base_screenshot_names(path: Path | None) -> set[str] | None:
return {name for name in names if isinstance(name, str)}
def _stage_name(kind: str, path: Path, results_dir: Path) -> str:
relative = path.relative_to(results_dir).as_posix()
digest = hashlib.sha256(relative.encode("utf-8")).hexdigest()[:12]
return f"{kind}-{digest}-{path.name}"
def select_evidence(results_dir: Path, base_manifest: Path | None = None) -> dict:
"""Select only screenshots new to main, plus every generated visual diff."""
base_names = _base_screenshot_names(base_manifest)
@@ -71,40 +61,8 @@ def select_evidence(results_dir: Path, base_manifest: Path | None = None) -> dic
return {"screenshots": screenshots, "diffs": diffs}
def stage_evidence(results_dir: Path, evidence_dir: Path, selection: dict) -> dict:
"""Copy selected PNGs into a flat, path-safe evidence artifact."""
evidence_dir.mkdir(parents=True, exist_ok=True)
staged: dict[Path, str] = {}
def stage(kind: str, path: Path) -> str:
if path in staged:
return staged[path]
name = _stage_name(kind, path, results_dir)
shutil.copyfile(path, evidence_dir / name)
staged[path] = name
return name
manifest = {"version": 1, "screenshots": [], "diffs": []}
for screenshot in selection["screenshots"]:
manifest["screenshots"].append({
"name": screenshot.name,
"file": stage("screenshot", screenshot),
})
for diff in selection["diffs"]:
entry = {"name": diff["diff"].name.removesuffix("-diff.png"), "diff": stage("diff", diff["diff"])}
for kind in ("actual", "expected"):
if kind in diff:
entry[kind] = stage(kind, diff[kind])
manifest["diffs"].append(entry)
(evidence_dir / "e2e-evidence.json").write_text(
json.dumps(manifest, sort_keys=True) + "\n", encoding="utf-8"
)
return manifest
def build_status(selection: dict, artifact_url: str = "") -> list[dict]:
"""Return the review status. The trusted publisher replaces its marker."""
"""Return the review status for the unified CI review comment."""
screenshots = selection["screenshots"]
diffs = selection["diffs"]
if not screenshots and not diffs:
@@ -122,7 +80,6 @@ def build_status(selection: dict, artifact_url: str = "") -> list[dict]:
"kind": "info",
"title": "Desktop E2E visual evidence",
"summary": "; ".join(summary_parts) + ".",
"detail": "\n".join((EVIDENCE_START, "<sub>inline evidence is publishing...</sub>", EVIDENCE_END)),
}
if artifact_url:
result["link"] = artifact_url
@@ -135,7 +92,6 @@ def main() -> int:
parser.add_argument("--results-dir", type=Path, required=True)
parser.add_argument("--base-manifest", type=Path)
parser.add_argument("--manifest-output", type=Path, required=True)
parser.add_argument("--evidence-dir", type=Path, required=True)
parser.add_argument("--artifact-url", default="")
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
@@ -144,7 +100,6 @@ def main() -> int:
json.dumps(build_manifest(args.results_dir), sort_keys=True) + "\n", encoding="utf-8"
)
selection = select_evidence(args.results_dir, args.base_manifest)
stage_evidence(args.results_dir, args.evidence_dir, selection)
args.output.write_text(
json.dumps(build_status(selection, args.artifact_url)) + "\n", encoding="utf-8"
)
-337
View File
@@ -1,337 +0,0 @@
#!/usr/bin/env python3
"""Publish validated E2E evidence as GitHub attachments and update its PR comment.
This script only runs from the trusted ``workflow_run`` publisher. It never
checks out PR code: it accepts the small evidence artifact produced by the
untrusted E2E workflow, validates its manifest and PNG bytes, uploads the
approved files as GitHub attachments, and replaces the placeholder in the
source PR's CI review comment with those attachment URLs.
"""
from __future__ import annotations
import argparse
import html
import json
import os
import re
import subprocess
import sys
import time
import urllib.request
from dataclasses import dataclass
from pathlib import Path
from typing import Any
API_BASE = "https://api.github.com"
EVIDENCE_START = "<!-- hermes-e2e-evidence:start -->"
EVIDENCE_END = "<!-- hermes-e2e-evidence:end -->"
PNG_SIGNATURE = b"\x89PNG\r\n\x1a\n"
MAX_FILES = 20
MAX_FILE_BYTES = 5 * 1024 * 1024
MAX_TOTAL_BYTES = 20 * 1024 * 1024
MAX_DIMENSION = 8_000
COMMENT_LOOKUP_ATTEMPTS = 6
COMMENT_LOOKUP_DELAY_SECONDS = 2
_SAFE_FILE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*\.png$")
_ATTACHMENT_URL = re.compile(r"^!\[[^\]\r\n]*\]\((https://github\.com/user-attachments/assets/[0-9a-fA-F-]+)\)$")
@dataclass(frozen=True)
class EvidenceFile:
"""One validated PNG and the label used when rendering the PR comment."""
filename: str
label: str
def _api_request(
url: str,
token: str,
method: str = "GET",
payload: dict[str, Any] | None = None,
) -> dict[str, Any]:
"""Send one authenticated GitHub API request and return its JSON object."""
data = json.dumps(payload).encode("utf-8") if payload is not None else None
request = urllib.request.Request(
url,
data=data,
method=method,
headers={
"Authorization": f"Bearer {token}",
"Accept": "application/vnd.github+json",
"Content-Type": "application/json",
"X-GitHub-Api-Version": "2022-11-28",
"User-Agent": "hermes-e2e-evidence-publisher",
},
)
with urllib.request.urlopen(request) as response:
parsed = json.loads(response.read())
if not isinstance(parsed, dict):
raise ValueError(f"Expected an object from {url}")
return parsed
def _read_png(path: Path) -> bytes:
"""Read a bounded PNG, rejecting corrupt and unexpectedly large images."""
if not path.is_file() or path.is_symlink():
raise ValueError(f"Evidence file is not a regular file: {path.name}")
size = path.stat().st_size
if size == 0 or size > MAX_FILE_BYTES:
raise ValueError(f"Evidence file has invalid size: {path.name}")
data = path.read_bytes()
if not data.startswith(PNG_SIGNATURE) or len(data) < 24 or data[12:16] != b"IHDR":
raise ValueError(f"Evidence file is not a PNG: {path.name}")
width = int.from_bytes(data[16:20], "big")
height = int.from_bytes(data[20:24], "big")
if not 0 < width <= MAX_DIMENSION or not 0 < height <= MAX_DIMENSION:
raise ValueError(f"Evidence image has invalid dimensions: {path.name}")
return data
def _manifest_files(manifest: dict[str, Any]) -> list[EvidenceFile]:
"""Flatten a version-one manifest into ordered, reviewer-facing images."""
if manifest.get("version") != 1:
raise ValueError("Unsupported E2E evidence manifest version")
files: list[EvidenceFile] = []
screenshots = manifest.get("screenshots", [])
diffs = manifest.get("diffs", [])
if not isinstance(screenshots, list) or not isinstance(diffs, list):
raise ValueError("Evidence manifest lists are malformed")
for entry in screenshots:
if not isinstance(entry, dict) or not isinstance(entry.get("name"), str) or not isinstance(entry.get("file"), str):
raise ValueError("Evidence screenshot entry is malformed")
files.append(EvidenceFile(entry["file"], f"new screenshot: {entry['name']}"))
for entry in diffs:
if not isinstance(entry, dict) or not isinstance(entry.get("name"), str) or not isinstance(entry.get("diff"), str):
raise ValueError("Evidence visual-diff entry is malformed")
files.append(EvidenceFile(entry["diff"], f"visual diff: {entry['name']}"))
for kind in ("actual", "expected"):
value = entry.get(kind)
if value is not None:
if not isinstance(value, str):
raise ValueError("Evidence visual-diff companion is malformed")
files.append(EvidenceFile(value, f"visual {kind}: {entry['name']}"))
names = [item.filename for item in files]
if len(files) > MAX_FILES or len(set(names)) != len(names):
raise ValueError("Evidence manifest has too many or duplicate files")
if any(not _SAFE_FILE.fullmatch(name) for name in names):
raise ValueError("Evidence manifest contains an unsafe filename")
return files
def load_evidence(evidence_dir: Path) -> tuple[list[EvidenceFile], dict[str, bytes]]:
"""Load the manifest and return only the validated files it declares."""
manifest_path = evidence_dir / "e2e-evidence.json"
if not manifest_path.is_file() or manifest_path.is_symlink():
raise ValueError("E2E evidence manifest is missing")
try:
manifest = json.loads(manifest_path.read_text(encoding="utf-8-sig"))
except json.JSONDecodeError as exc:
raise ValueError("E2E evidence manifest is not JSON") from exc
if not isinstance(manifest, dict):
raise ValueError("E2E evidence manifest is not an object")
files = _manifest_files(manifest)
payloads: dict[str, bytes] = {}
total = 0
for item in files:
path = evidence_dir / item.filename
if path.parent != evidence_dir:
raise ValueError("Evidence file escaped its artifact directory")
payload = _read_png(path)
total += len(payload)
if total > MAX_TOTAL_BYTES:
raise ValueError("E2E evidence exceeds the total size limit")
payloads[item.filename] = payload
return files, payloads
def render_evidence(files: list[EvidenceFile], attachment_urls: dict[str, str]) -> str:
"""Render validated GitHub attachment URLs inside the review-comment marker."""
blocks = [EVIDENCE_START]
for item in files:
url = attachment_urls.get(item.filename)
if url is None:
raise ValueError(f"Missing attachment URL for {item.filename}")
blocks.extend((
"<details>",
f"<summary>{item.label}</summary>",
"",
f"![{item.label}]({url})",
"",
"</details>",
))
blocks.append(EVIDENCE_END)
return "\n".join(blocks)
def render_upload_failure(error: Exception) -> str:
"""Render an escaped upload error inside the review-comment marker."""
return "\n".join((
EVIDENCE_START,
"<sub>inline evidence upload failed.</sub>",
"",
f"<pre>{html.escape(str(error))}</pre>",
EVIDENCE_END,
))
def replace_evidence_marker(comment: str, evidence: str) -> str:
"""Replace exactly the pending-evidence region in a CI review comment."""
pattern = re.compile(f"{re.escape(EVIDENCE_START)}.*?{re.escape(EVIDENCE_END)}", re.DOTALL)
result, count = pattern.subn(evidence, comment, count=1)
if count != 1:
raise ValueError("CI review comment does not contain one evidence marker")
return result
def _find_review_comment(comments: object) -> dict[str, Any] | None:
"""Find a live CI review comment only after it contains this marker."""
if not isinstance(comments, list):
raise ValueError("GitHub comments response is malformed")
for item in comments:
if not isinstance(item, dict):
continue
body = str(item.get("body", ""))
if body.startswith("<!-- hermes-ci-review-bot -->") and EVIDENCE_START in body and EVIDENCE_END in body:
return item
return None
def _wait_for_review_comment(token: str, source_repo: str, pr_number: str) -> dict[str, Any] | None:
"""Wait briefly for GitHub's comment API to expose the completed marker."""
request = urllib.request.Request(
f"{API_BASE}/repos/{source_repo}/issues/{pr_number}/comments?per_page=100",
headers={
"Authorization": f"Bearer {token}",
"Accept": "application/vnd.github+json",
"X-GitHub-Api-Version": "2022-11-28",
"User-Agent": "hermes-e2e-evidence-publisher",
},
)
for attempt in range(COMMENT_LOOKUP_ATTEMPTS):
with urllib.request.urlopen(request) as response:
comment = _find_review_comment(json.loads(response.read()))
if comment is not None:
return comment
if attempt + 1 < COMMENT_LOOKUP_ATTEMPTS:
time.sleep(COMMENT_LOOKUP_DELAY_SECONDS)
return None
def upload_evidence(
files: list[EvidenceFile],
evidence_dir: Path,
source_repo: str,
session_token: str,
) -> dict[str, str]:
"""Upload validated files through gh-image and accept only attachment URLs."""
environment = os.environ.copy()
environment["GH_SESSION_TOKEN"] = session_token
attachment_urls: dict[str, str] = {}
for item in files:
try:
result = subprocess.run(
[
"gh",
"image",
"--repo",
source_repo,
str(evidence_dir / item.filename),
],
check=True,
capture_output=True,
text=True, encoding="utf-8", errors="replace",
env=environment,
)
except subprocess.CalledProcessError as exc:
output = "; ".join(
value.strip()
for value in (exc.stdout, exc.stderr)
if value and value.strip()
)
message = f"Failed to upload {item.filename} with gh image (exit code {exc.returncode})"
if output:
message = f"{message}: {output}"
print(message, file=sys.stderr)
raise RuntimeError(message) from exc
match = _ATTACHMENT_URL.fullmatch(result.stdout.strip())
if match is None:
raise ValueError(f"gh-image returned an invalid attachment reference for {item.filename}")
attachment_urls[item.filename] = match.group(1)
return attachment_urls
def publish(
token: str,
source_repo: str,
evidence_dir: Path,
pr_number: str,
session_token: str,
) -> bool:
"""Publish evidence and patch its source PR comment; false means nothing to show."""
files, _ = load_evidence(evidence_dir)
if not files:
print("No inline E2E evidence to publish.")
return False
comment = _wait_for_review_comment(token, source_repo, pr_number)
if comment is None:
# A fork PR gets no CI review comment (the live poller needs a
# write token there), so there is no marker to patch. The
# evidence stays available in the workflow artifact.
print(
f"PR #{pr_number} has no CI review comment with an E2E evidence "
"marker; the evidence stays in the workflow artifact."
)
return False
try:
attachment_urls = upload_evidence(
files, evidence_dir, source_repo, session_token
)
except Exception as exc:
body = replace_evidence_marker(
str(comment.get("body", "")), render_upload_failure(exc)
)
_api_request(
f"{API_BASE}/repos/{source_repo}/issues/comments/{comment['id']}",
token,
method="PATCH",
payload={"body": body},
)
raise
evidence = render_evidence(files, attachment_urls)
body = replace_evidence_marker(str(comment.get("body", "")), evidence)
_api_request(
f"{API_BASE}/repos/{source_repo}/issues/comments/{comment['id']}",
token,
method="PATCH",
payload={"body": body},
)
print(f"Published {len(files)} E2E evidence image attachment(s).")
return True
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--evidence-dir", type=Path, required=True)
parser.add_argument("--source-repo", required=True)
parser.add_argument("--pr-number", required=True)
args = parser.parse_args()
token = os.environ.get("GITHUB_TOKEN", "")
if not token:
parser.error("GITHUB_TOKEN is required")
session_token = os.environ.get("GH_SESSION_TOKEN", "")
if not session_token:
parser.error("GH_SESSION_TOKEN is required")
publish(token, args.source_repo, args.evidence_dir, args.pr_number, session_token)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+1 -5
View File
@@ -34,24 +34,20 @@ def test_status_selects_only_new_explicit_screenshots_and_all_diffs(tmp_path):
result = status[0]["results"][0]
assert result["kind"] == "info"
assert result["summary"] == "1 new screenshot vs main; 1 visual diff."
assert _mod.EVIDENCE_START in result["detail"]
assert "already-on-main.png" not in result["detail"]
assert "detail" not in result
assert result["link"] == "https://github.test/artifacts/1"
def test_cli_output_ends_with_newline_for_github_output_delimiter(tmp_path, monkeypatch):
output = tmp_path / "review-status.json"
manifest = tmp_path / "main-manifest.json"
evidence_dir = tmp_path / "evidence"
monkeypatch.setattr(sys, "argv", [
"e2e_screenshot_status.py",
"--results-dir", str(tmp_path),
"--manifest-output", str(manifest),
"--evidence-dir", str(evidence_dir),
"--output", str(output),
])
assert _mod.main() == 0
assert output.read_text(encoding="utf-8") == "[]\n"
assert manifest.read_text(encoding="utf-8") == '{"screenshot_names": [], "version": 1}\n'
assert evidence_dir.joinpath("e2e-evidence.json").is_file()
-217
View File
@@ -1,217 +0,0 @@
"""Tests for scripts/ci/publish_e2e_evidence.py."""
from __future__ import annotations
import importlib.util
import sys
from pathlib import Path
import pytest
_PATH = Path(__file__).resolve().parents[2] / "scripts" / "ci" / "publish_e2e_evidence.py"
_spec = importlib.util.spec_from_file_location("publish_e2e_evidence", _PATH)
if _spec is None or _spec.loader is None:
raise ImportError("Failed to load publish_e2e_evidence.py")
_mod = importlib.util.module_from_spec(_spec)
sys.modules["publish_e2e_evidence"] = _mod
_spec.loader.exec_module(_mod)
def _png(width: int = 4, height: int = 3) -> bytes:
return _mod.PNG_SIGNATURE + b"\x00\x00\x00\rIHDR" + width.to_bytes(4, "big") + height.to_bytes(4, "big")
def test_load_evidence_validates_manifest_and_pngs(tmp_path):
(tmp_path / "shot.png").write_bytes(_png())
(tmp_path / "diff.png").write_bytes(_png())
(tmp_path / "actual.png").write_bytes(_png())
(tmp_path / "expected.png").write_bytes(_png())
(tmp_path / "e2e-evidence.json").write_text(
"""{
"version": 1,
"screenshots": [{"name": "main-view.png", "file": "shot.png"}],
"diffs": [{"name": "main-view", "diff": "diff.png", "actual": "actual.png", "expected": "expected.png"}]
}""",
encoding="utf-8",
)
files, payloads = _mod.load_evidence(tmp_path)
assert [item.label for item in files] == [
"new screenshot: main-view.png",
"visual diff: main-view",
"visual actual: main-view",
"visual expected: main-view",
]
assert set(payloads) == {"shot.png", "diff.png", "actual.png", "expected.png"}
def test_load_evidence_rejects_path_escape_and_non_png(tmp_path):
(tmp_path / "e2e-evidence.json").write_text(
'{"version":1,"screenshots":[{"name":"bad","file":"../secret.png"}],"diffs":[]}',
encoding="utf-8",
)
with pytest.raises(ValueError, match="unsafe filename"):
_mod.load_evidence(tmp_path)
(tmp_path / "e2e-evidence.json").write_text(
'{"version":1,"screenshots":[{"name":"bad","file":"not-png.png"}],"diffs":[]}',
encoding="utf-8",
)
(tmp_path / "not-png.png").write_bytes(b"not a png")
with pytest.raises(ValueError, match="not a PNG"):
_mod.load_evidence(tmp_path)
def test_upload_evidence_accepts_only_attachment_urls(tmp_path, monkeypatch):
shot = tmp_path / "shot.png"
shot.write_bytes(_png())
calls = []
def fake_run(args, **kwargs):
calls.append((args, kwargs))
return _mod.subprocess.CompletedProcess(
args,
0,
stdout="![shot.png](https://github.com/user-attachments/assets/12345678-1234-1234-1234-123456789abc)\n",
)
monkeypatch.setattr(_mod.subprocess, "run", fake_run)
result = _mod.upload_evidence(
[_mod.EvidenceFile("shot.png", "new screenshot: shot.png")],
tmp_path,
"NousResearch/hermes-agent",
"bot-session-token",
)
assert result == {"shot.png": "https://github.com/user-attachments/assets/12345678-1234-1234-1234-123456789abc"}
assert calls[0][0] == ["gh", "image", "--repo", "NousResearch/hermes-agent", str(shot)]
assert calls[0][1]["env"]["GH_SESSION_TOKEN"] == "bot-session-token"
def test_upload_evidence_reports_gh_image_error(tmp_path, monkeypatch, capsys):
shot = tmp_path / "shot.png"
shot.write_bytes(_png())
def fake_run(args, **kwargs):
raise _mod.subprocess.CalledProcessError(
1,
args,
output="upload output",
stderr="upload error",
)
monkeypatch.setattr(_mod.subprocess, "run", fake_run)
with pytest.raises(RuntimeError, match="Failed to upload shot.png.*upload error"):
_mod.upload_evidence(
[_mod.EvidenceFile("shot.png", "new screenshot: shot.png")],
tmp_path,
"NousResearch/hermes-agent",
"bot-session-token",
)
captured = capsys.readouterr()
assert "Failed to upload shot.png" in captured.err
assert "upload output" in captured.err
assert "upload error" in captured.err
def test_publish_marks_evidence_upload_failure_in_pr_comment(tmp_path, monkeypatch):
comment = {
"id": 123,
"body": "before\n<!-- hermes-e2e-evidence:start -->\npending\n<!-- hermes-e2e-evidence:end -->\nafter",
}
updates = []
monkeypatch.setattr(
_mod,
"load_evidence",
lambda evidence_dir: (
[_mod.EvidenceFile("shot.png", "new screenshot: shot.png")],
{},
),
)
monkeypatch.setattr(_mod, "_wait_for_review_comment", lambda *args: comment)
monkeypatch.setattr(
_mod,
"upload_evidence",
lambda *args: (_ for _ in ()).throw(
RuntimeError("Failed to upload shot.png: bad <response>")
),
)
monkeypatch.setattr(
_mod,
"_api_request",
lambda url, token, method, payload: updates.append((
url,
token,
method,
payload,
)),
)
with pytest.raises(RuntimeError, match="Failed to upload shot.png"):
_mod.publish(
"github-token",
"NousResearch/hermes-agent",
tmp_path,
"69868",
"image-token",
)
assert updates == [
(
"https://api.github.com/repos/NousResearch/hermes-agent/issues/comments/123",
"github-token",
"PATCH",
{
"body": "before\n<!-- hermes-e2e-evidence:start -->\n<sub>inline evidence upload failed.</sub>\n\n<pre>Failed to upload shot.png: bad &lt;response&gt;</pre>\n<!-- hermes-e2e-evidence:end -->\nafter"
},
)
]
def test_publish_skips_when_no_review_comment_exists(tmp_path, monkeypatch, capsys):
monkeypatch.setattr(
_mod,
"load_evidence",
lambda evidence_dir: (
[_mod.EvidenceFile("shot.png", "new screenshot: shot.png")],
{},
),
)
monkeypatch.setattr(_mod, "_wait_for_review_comment", lambda *args: None)
monkeypatch.setattr(
_mod,
"upload_evidence",
lambda *args: (_ for _ in ()).throw(AssertionError("must not upload")),
)
assert _mod.publish(
"github-token",
"NousResearch/hermes-agent",
tmp_path,
"83202",
"image-token",
) is False
assert "no CI review comment" in capsys.readouterr().out
def test_find_review_comment_requires_the_evidence_marker():
pending = "<!-- hermes-ci-review-bot -->\n<!-- hermes-e2e-evidence:start -->\npending\n<!-- hermes-e2e-evidence:end -->"
assert _mod._find_review_comment([{"body": "<!-- hermes-ci-review-bot --> no evidence"}]) is None
assert _mod._find_review_comment([{"body": pending, "id": 123}]) == {"body": pending, "id": 123}
def test_replace_evidence_marker_requires_exactly_one_marker():
with pytest.raises(ValueError, match="does not contain one"):
_mod.replace_evidence_marker("no marker", "evidence")