From 53ececdcb2d54391006e7ef73b7e79baeeba86f4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=91=A8=E5=B0=8F=E8=88=9F?= Date: Thu, 1 Oct 2026 14:53:19 +0800 Subject: [PATCH] test(bench): fast-output regression set with per-stage timing - 17 fixed real inputs (owner's picks: long interviews, short sources, keynotes with screen demos, Chinese sources with burned captions, auto- caption-only and once-360p sources) in benchmarks/fast_output/cases.json. - scripts/fast_output_benchmark.py imports each through the product API and reports per-stage time, model calls, tokens and cost, clips and renders, source resolution, subtitle source, burned captions and packaging fallbacks; --baseline prints deltas. - Stage wall times (download, transcribe, clip finder, boundaries, framing, packaging, post copy, render) are recorded next to token usage. - Changelog: publish kit, outro, caption strategy, low-resolution fix. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 16 ++ backend/core/llm_usage.py | 38 +++++ backend/services/simple_pipeline_adapter.py | 6 +- backend/services/studio/jobs.py | 20 ++- backend/tests/test_llm_usage.py | 12 ++ benchmarks/fast_output/cases.json | 41 +++++ docs/FAST_OUTPUT_BENCHMARK.md | 44 ++++++ scripts/fast_output_benchmark.py | 157 ++++++++++++++++++++ 8 files changed, 325 insertions(+), 9 deletions(-) create mode 100644 benchmarks/fast_output/cases.json create mode 100644 docs/FAST_OUTPUT_BENCHMARK.md create mode 100644 scripts/fast_output_benchmark.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 6e359c62..cda04c90 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,13 @@ ## [未发布] ### 新增 +- **一键出片**:贴一个视频链接、选好要发的平台(抖音、小红书、TikTok、Reels、YouTube Shorts、B 站、YouTube),自动挑出值得发的片段,按每个平台的画幅、时长和包装直接生成可发布成片;可随时追加平台版本,失败的单条可重试。 +- **两套包装模板**:抖音 / 小红书用「访谈式」(上方两行标题、4:3 说话人窗口、中英双语字幕、左下名牌、编辑点评标签);TikTok / Reels / Shorts 用「播客式」(全屏跟随说话人、逐词高亮字幕、开头一句 hook、名牌)。外语素材自动翻成目标平台的语言;原片自带字幕时保留完整画面,字幕放在画面下方。 +- **按内容情绪变换风格**:模型判断每段内容的情绪(冷静、严肃、强观点、真诚、轻松),据此在 7 套配色和多种字幕动效里挑选,不同片段风格各异,同一片段在各平台保持一致。 +- **只先生成最好的 10 条**:长视频会挑出 20–40 个片段,默认自动生成评分最高的 10 条,其余列为「备选片段」,点「生成这条」再制作,省时间也省模型费用。 +- **每条成片自带发布包**:按目标平台写好标题、简介和话题(抖音钩子式、小红书笔记式、B 站信息完整、TikTok / Reels / Shorts 英文文案),按平台字数与话题数限制;封面在成片完成后自动设计——说话人画面、成片同款标题与配色、嘉宾名牌,按平台比例(竖屏 9:16、小红书 3:4、B 站 16:10、YouTube 16:9)。成片卡片上可直接改文案、复制、导出「发布包」(视频 + 封面 + 文案),也可以一键用 AI 重新设计封面;去发布时自动带上这套文案与封面。 +- **新片尾动画**:1.8 秒的 AutoClip 标志动画与提示音,竖屏、横屏各一版,其他比例自动居中适配。 +- 竖屏成片按说话人自动取景,主持人与嘉宾切换时画面跟着说话的人走。 - 新增赞助合作伙伴 88API:八语 README、首页与设置页推荐入口、独立 API Key、模型发现、兼容封面与 Whisper 转写;CLI / MCP 使用 `api88`。 - 模型设置新增赞助合作伙伴「Infistar 无限星河」:接口地址已预设,填好 Key 后自动列出账号可用的模型;设置页提供专属注册链接(可领 $5 体验额度)和接入说明。CLI / MCP 支持 `--provider infistar` 与 `INFISTAR_API_KEY`。 - 设置 → AI 模型重新整理为「AI 服务 / 字幕转写 / 封面 / 高级」四段:选一家服务、填好 Key,分析模型自动选好;画面识别和 AI 封面首次配置默认开启,画面识别注明适合游戏画面、口播较少的内容。供应商分组中的赞助伙伴改为「推荐」。 @@ -17,9 +24,18 @@ - 竖屏「满屏」构图新增按镜头自动取景:先识别镜头切换,再对准正在说话的人;引用卡、PPT 等没有人物的镜头自动改为完整画面 + 模糊背景,不再被裁掉两侧。首次使用需下载约 45 MB 的人物识别组件;每个镜头都可以单独改为「对准人物 / 完整画面」并微调位置。 ### 改进 +- **出片速度大幅提升**:一条 2 小时访谈从链接到成片,由约 65 分钟缩短到 7–8 分钟(视频有作者字幕时)或约 30 分钟(需要本地语音识别时)。片段挑选改为一次通读全文,原来几十次模型调用、半小时以上,现在一分钟以内;各步骤的模型调用并行执行;视频有作者上传的字幕时直接使用,不再本地语音识别。 +- **模型费用更低**:一条 2–3 小时访谈的模型费用由约 ¥0.6 降到 ¥0.1–0.2(qwen-plus 估算),模型调用由近 200 次降到 30 次左右;同语言包装不再让模型复述整段字幕。 +- **更安静**:成片用电脑自带的硬件编码器(macOS VideoToolbox、Windows NVENC / QSV / AMF),CPU 占用约降到原来的四分之一,风扇不再狂转;硬件编码不可用时自动改用软件编码。 +- **切点更自然**:片段从问题或观点的第一句开始,到回答讲完、说话人有明显停顿时才结束;按原片音频的真实停顿下刀,不再带进半句下一段话或主持人的下一个问题。 - 导入确认页去掉重复的分析方式提问,控件和文案与设置页统一;发布页和编辑器的「烧录」「标题卡」等术语换成直白说法,编辑器和导出对话框补上封面入口,发布页默认自动生成封面。 ### 修复 +- 外语素材原片烧有英文字幕时,抖音版也会配中文字幕。 +- 原片已有硬字幕时,按目标平台判断是否加字幕:中文硬字幕投抖音不再重复加中文字幕,投 TikTok 加英文;外语访谈配了英文硬字幕的,投 TikTok 不再重复。细小、无描边的硬字幕(常见于 B 站访谈)也能识别。 +- YouTube 偶尔只给 360p 画质时会自动换方式重新下载到 720p 以上,不再出模糊成片。 +- 同语言字幕与声音同步;电影感字幕改为一次 2–5 个词,不再一个词一个词跳。 +- 标题不再出现 Markdown 符号或 `&` 之类的转义字符;中文平台不会出现日文标题;包装偶发不合规时会自动重试一次,不再整段退回原字幕。 - 时间线重试只使用本次有效结果,避免失败后误用上次候选;原始响应缓存现在会正常解析,无效缓存不会自动触发模型请求。 - 智能导入后台任务提交失败后可明确重试,已有方案、草稿和成片保留;修改方案失败时同步恢复偏好。 - 时间线兼容常见时间戳格式;相邻短片段在丢弃前尝试合并,保留既有时长限制。 diff --git a/backend/core/llm_usage.py b/backend/core/llm_usage.py index e0cd2188..60a7e718 100644 --- a/backend/core/llm_usage.py +++ b/backend/core/llm_usage.py @@ -88,6 +88,8 @@ def summary(project_id: str) -> dict[str, Any]: row = json.loads(line) except ValueError: continue + if row.get('kind') == 'timing': + continue item = stages.setdefault(row.get('stage', 'other'), {'calls': 0, 'prompt_tokens': 0, 'completion_tokens': 0, 'estimated_calls': 0}) item['calls'] += 1 item['prompt_tokens'] += row.get('prompt_tokens') or 0 @@ -101,3 +103,39 @@ def run_in_context(fn): """Wrap `fn` so a worker thread records into the caller's sink and stage.""" context = contextvars.copy_context() return lambda *args, **kwargs: context.copy().run(fn, *args, **kwargs) + + +@contextmanager +def timed(stage_name: str): + """Record how long a step took (wall seconds) next to its token usage, for cost/time reports.""" + started = time.monotonic() + try: + yield + finally: + path = _sink.get() + if path is not None: + row = {'at': round(time.time(), 1), 'kind': 'timing', 'stage': stage_name, 'seconds': round(time.monotonic() - started, 2)} + try: + with _lock: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open('a', encoding='utf-8') as handle: + handle.write(json.dumps(row) + '\n') + except OSError: + pass + + +def timings(project_id: str) -> dict[str, float]: + """Total wall seconds per timed stage of one project.""" + out: dict[str, float] = {} + try: + lines = usage_path(project_id).read_text(encoding='utf-8').splitlines() + except OSError: + return out + for line in lines: + try: + row = json.loads(line) + except ValueError: + continue + if row.get('kind') == 'timing': + out[row['stage']] = round(out.get(row['stage'], 0) + float(row.get('seconds') or 0), 2) + return out diff --git a/backend/services/simple_pipeline_adapter.py b/backend/services/simple_pipeline_adapter.py index 1a41204d..4043bf5a 100644 --- a/backend/services/simple_pipeline_adapter.py +++ b/backend/services/simple_pipeline_adapter.py @@ -147,7 +147,8 @@ class SimplePipelineAdapter: emit_progress(self.project_id, "ANALYZE", "正在通读全文挑选片段") try: entries = TextProcessor.parse_srt(Path(srt_path)) - return find_clips(entries, text_json, threshold=resolve_min_score_threshold(), metadata_dir=metadata_dir) + with llm_usage.timed("clip_finder"): + return find_clips(entries, text_json, threshold=resolve_min_score_threshold(), metadata_dir=metadata_dir) except Exception as error: # noqa: BLE001 - the legacy steps still work with any model logger.warning("一次挑片不可用,改用分步分析: %s", error) return None @@ -199,7 +200,8 @@ class SimplePipelineAdapter: from backend.utils.speech_recognizer import SpeechRecognitionError logger.warning("没有SRT文件,尝试自动生成字幕") try: - srt_path = await self._generate_subtitle_automatically(input_video_path, metadata_dir) + with llm_usage.timed("transcribe"): + srt_path = await self._generate_subtitle_automatically(input_video_path, metadata_dir) except SpeechRecognitionError as e: raise failure_from_speech_error(str(e)) from e if not (srt_path and srt_path.exists()): diff --git a/backend/services/studio/jobs.py b/backend/services/studio/jobs.py index 862d6282..cebac039 100644 --- a/backend/services/studio/jobs.py +++ b/backend/services/studio/jobs.py @@ -74,7 +74,8 @@ def _render(project_id, draft, job_id, *, brand_outro=False): store.change(project_id, mutate) try: update(status='running', percent=5) - result = render_draft(project_id, source(project_id), draft, job_id, lambda p: update(percent=p), brand_outro=brand_outro) + with llm_usage.timed('render'): + result = render_draft(project_id, source(project_id), draft, job_id, lambda p: update(percent=p), brand_outro=brand_outro) update(status='completed', percent=100, result=result, duration_ms=round((monotonic() - started) * 1000)) _design_covers(project_id, draft, job_id) _sync_variant_status(project_id, job_id, 'completed') @@ -114,7 +115,8 @@ def _analyze(project_id, prefs, url, browser): try: mark_project(project_id, 'processing') if url: - download(project_id, url, browser) + with llm_usage.timed('download'): + download(project_id, url, browser) video = source(project_id) def stage(message): store.change(project_id, lambda data: data['analysis'].update(message=message)) @@ -399,7 +401,7 @@ def _complete_thought_bounds(project_id, clips): call = intelligence.text_json except Exception: # noqa: BLE001 call = None - with llm_usage.stage('boundaries'): + with llm_usage.stage('boundaries'), llm_usage.timed('boundaries'): return boundaries.refine_clips(rows, clips, call, boundaries.audio_silences(source(project_id)) if audio.has_audio(source(project_id)) else None) @@ -782,10 +784,13 @@ def _auto_generate(project_id, plan): now = base['id'] in automatic framed = None # on-demand versions are framed and packaged when the user asks (produce_variant) if now: - value, framed = _apply_framing(project_id, value, strategy_id, video, burned, framing_cache) + with llm_usage.timed('framing'): + value, framed = _apply_framing(project_id, value, strategy_id, video, burned, framing_cache) planned.append((strategy_id, value, trimmed, framed, now)) - _prefetch_packaging(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], burned, packaging_cache) - posts = _posts_for(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], packaging_cache) + with llm_usage.timed('packaging'): + _prefetch_packaging(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], burned, packaging_cache) + with llm_usage.timed('post_copy'): + posts = _posts_for(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], packaging_cache) for strategy_id, value, trimmed, framed, now in planned: if now: value = _apply_packaging(project_id, value, strategy_id, burned, packaging_cache) @@ -1062,7 +1067,8 @@ def _inspect(project_id, options, url, browser): from backend.services.studio.planning import recommend mark_project(project_id, 'processing', awaiting_confirmation=False) if url: - download(project_id, url, browser) + with llm_usage.timed('download'): + download(project_id, url, browser) store.change(project_id, lambda data:(data['analysis'].pop('percent', None), data['analysis'].update(message='快速判断适合的制作类型'))) plan = recommend(source(project_id), options) ensure_project_thumbnail(project_id) diff --git a/backend/tests/test_llm_usage.py b/backend/tests/test_llm_usage.py index b14b4faf..9a5eaca5 100644 --- a/backend/tests/test_llm_usage.py +++ b/backend/tests/test_llm_usage.py @@ -45,3 +45,15 @@ def test_manager_calls_record_provider_usage(monkeypatch, tmp_path): with llm_usage.tracking('p1'), llm_usage.stage('scoring'): assert manager.call('score', {'x': 1}) == '{"ok": true}' assert llm_usage.summary('p1')['stages']['scoring']['prompt_tokens'] == 42 + + +def test_stage_timings_add_up_and_do_not_count_as_model_calls(monkeypatch, tmp_path): + clock = iter([10.0, 12.5, 20.0, 21.0]) + monkeypatch.setattr(llm_usage.time, 'monotonic', lambda: next(clock)) + monkeypatch.setattr(llm_usage, 'usage_path', lambda _p: tmp_path / llm_usage.FILE) + with llm_usage.tracking('p1'): + for _ in range(2): + with llm_usage.timed('render'): + pass + assert llm_usage.timings('p1') == {'render': 3.5} + assert llm_usage.summary('p1')['total']['calls'] == 0 diff --git a/benchmarks/fast_output/cases.json b/benchmarks/fast_output/cases.json new file mode 100644 index 00000000..c27fc2e1 --- /dev/null +++ b/benchmarks/fast_output/cases.json @@ -0,0 +1,41 @@ +{ + "version": 1, + "note": "Fast-output regression set. Every optimisation is compared on these inputs (scripts/fast_output_benchmark.py). Picked 2026-10-01 by the owner; keep ids stable.", + "cases": [ + {"id": "neumann-doac", "url": "https://www.youtube.com/watch?v=IQ4JVWdj4Q0", "platforms": ["douyin", "tiktok"], "language": "en", + "traits": ["long interview", "founder story"], "watch": "WeWork rise and fall; story arcs should end on a complete answer"}, + {"id": "stallone-nyt", "url": "https://www.youtube.com/watch?v=ccs-B_nTfZs", "platforms": ["douyin"], "language": "en", + "traits": ["interview", "strong personality"], "watch": "Rocky, overnight fame, rivalry with Schwarzenegger"}, + {"id": "dafoe-hot-ones", "url": "https://www.youtube.com/watch?v=YqugY2zTIoI", "platforms": ["tiktok", "douyin"], "language": "en", + "traits": ["short source (23 min)", "reaction shots", "two speakers"], "watch": "hooks on reaction moments; short-tier clip lengths"}, + {"id": "druski-doac", "url": "https://www.youtube.com/watch?v=UhzI1fg8rCA", "platforms": ["tiktok"], "language": "en", + "traits": ["long interview", "creator economy"], "watch": "rejected by Netflix/Amazon, building his own audience"}, + {"id": "lidan-wangzuxian", "url": "https://www.bilibili.com/video/BV1XHhW6PEpf/", "platforms": ["douyin", "xiaohongshu"], "language": "zh", + "traits": ["Chinese source", "2 h", "likely burned Chinese captions"], "watch": "no duplicate Chinese captions; speech recognition time"}, + {"id": "rubin-doac", "url": "https://www.youtube.com/watch?v=a_GiFiHXJ6g", "platforms": ["douyin", "youtube_shorts"], "language": "en", + "traits": ["long interview", "creator audience"], "watch": "taste and judgement in the AI era; Shorts 180 s cap"}, + {"id": "openai-devday-2026", "url": "https://www.youtube.com/watch?v=Fls_onRviPM", "platforms": ["tiktok", "bilibili"], "language": "en", + "traits": ["keynote", "screen demos", "54 min"], "watch": "demo segments must keep the screen (full frame, no face crop)"}, + {"id": "dreamforce-keynote", "url": "https://www.youtube.com/watch?v=wYFt9NCYaVI", "platforms": ["douyin"], "language": "en", + "traits": ["keynote with guests", "1 h 48 m"], "watch": "chapters 45:10 Dario and 1:16:50 Jensen should be among the picks"}, + {"id": "luyu-guokeyu", "url": "https://www.bilibili.com/video/BV1LDYV6HEXR/", "platforms": ["douyin"], "language": "zh", + "traits": ["Chinese source", "story with a turn"], "watch": "complete turning-point stories, not fragments"}, + {"id": "hassabis-fry-rsa", "url": null, "platforms": ["douyin", "tiktok"], "language": "en", + "traits": ["talk plus conversation", "1 h"], "watch": "link to be added"}, + + {"id": "mrbeast-colin-samir", "url": "https://www.youtube.com/watch?v=9IQ_ldV9z_A", "platforms": ["tiktok"], "language": "en", + "traits": ["2 h", "auto captions only", "once downloaded at 360p"], "watch": "source must arrive at >=720p"}, + {"id": "karpathy-dwarkesh", "url": "https://www.youtube.com/watch?v=lXUZvyajciY", "platforms": ["douyin"], "language": "en", + "traits": ["2 h 26 m", "creator subtitles"], "watch": "no speech recognition (uploaded subtitles)"}, + {"id": "jensen-dwarkesh", "url": "https://www.youtube.com/watch?v=Hrbq66XqtCo", "platforms": ["xiaohongshu"], "language": "en", + "traits": ["1 h 43 m", "creator subtitles"], "watch": "baseline for the new pipeline: ~7.5 min"}, + {"id": "tim-luoyonghao", "url": "https://www.bilibili.com/video/BV1B5xkzPEhx/", "platforms": ["douyin", "tiktok"], "language": "zh", + "traits": ["2 h 52 m", "burned Chinese captions", "no subtitles"], "watch": "Douyin: no captions added; TikTok: English captions"}, + {"id": "dario-dwarkesh", "url": "https://www.youtube.com/watch?v=n1E9IZfvGMA", "platforms": ["douyin"], "language": "en", + "traits": ["2 h 22 m"], "watch": "golden samples for interview packaging"}, + {"id": "sam-altman-yc", "url": "https://www.youtube.com/watch?v=ZIaOBAjvc38", "platforms": ["xiaohongshu", "youtube_shorts"], "language": "en", + "traits": ["39 min", "two speakers on stage"], "watch": "speaker switches between Sam and Garry"}, + {"id": "kojima-wired", "url": "https://www.youtube.com/watch?v=02Ah5VQrzvA", "platforms": ["douyin", "tiktok"], "language": "ja", + "traits": ["17 min", "Japanese speech", "burned English captions"], "watch": "Douyin: Chinese captions; TikTok: no captions added"} + ] +} diff --git a/docs/FAST_OUTPUT_BENCHMARK.md b/docs/FAST_OUTPUT_BENCHMARK.md new file mode 100644 index 00000000..6e8ba4a6 --- /dev/null +++ b/docs/FAST_OUTPUT_BENCHMARK.md @@ -0,0 +1,44 @@ +# 快速出片回归测试集 + +每次优化切片、包装、渲染或成本,都用同一组真实素材对比:`benchmarks/fast_output/cases.json`(负责人 2026-10-01 选定;id 不改,新素材往后加)。 + +## 怎么跑 + +1. 用测试数据目录启动后端(不要用自己的真实数据目录),模型与语音识别按要测的配置设好: + + ```bash + AUTOCLIP_APP_DIR= AUTOCLIP_DATA_DIR= DATABASE_URL=sqlite:////autoclip.db \ + nice -n 15 python -m backend.main --port 18765 + ``` + +2. 跑测试集(逐条导入、等渲染完;YouTube 用 Chrome 的登录态下载): + + ```bash + python scripts/fast_output_benchmark.py --data-dir # 全部 + python scripts/fast_output_benchmark.py --data-dir --cases kojima-wired # 指定几条 + python scripts/fast_output_benchmark.py --data-dir --baseline benchmarks/fast_output/reports/<上次>.json + ``` + +3. 结果写在 `benchmarks/fast_output/reports/<时间>.json` 与 `.md`;带 `--baseline` 时表格里给出与上次的差值。 + +## 报告里有什么 + +| 字段 | 来源 | +|---|---| +| 各阶段用时(下载、语音识别、挑片、切点、取景、包装、文案、渲染) | 项目 `metadata/llm_usage.jsonl` 的 timing 记录 | +| 模型调用次数、tokens、估算费用(qwen-plus 价) | 同一文件的用量记录 | +| 片段数 → 自动出片数、备选数、失败数、片段时长 | Studio 状态 | +| 原片分辨率、字幕来源(作者字幕 / 语音识别)、是否有硬字幕及其语言、包装降级数 | Studio 状态 | +| 前几条标题与发布标题 | 用于人工抽查 | + +## 人工检查清单(每条抽 2 支) + +- 开头从问题或观点的第一句开始,结尾等到回答讲完、有停顿,不带进下一个问题。 +- 原片有硬字幕时不重复加同语言字幕;外语受众有对应语言字幕;字幕与声音同步。 +- 竖屏对着说话人,换人时跟着切;屏幕演示段(如 OpenAI DevDay)保留完整画面。 +- 画面清晰(原片 ≥ 720p);封面人脸不被标题遮挡,名牌是嘉宾;文案符合平台风格与长度。 +- 片尾动画完整、声音不突兀。 + +## 素材特点(为什么选它们) + +见 `cases.json` 里每条的 `traits` 与 `watch`:长访谈、短素材、双人对谈、主题演讲与屏幕演示、中文无字幕素材、硬字幕(中文 / 外语配英文)、只有自动字幕、曾经只下到 360p 的源。 diff --git a/scripts/fast_output_benchmark.py b/scripts/fast_output_benchmark.py new file mode 100644 index 00000000..8df0f16c --- /dev/null +++ b/scripts/fast_output_benchmark.py @@ -0,0 +1,157 @@ +"""Run the fast-output regression set through the product API and report time, cost and output. + + python scripts/fast_output_benchmark.py --server http://127.0.0.1:18765 --data-dir \ + [--cases id1,id2] [--baseline benchmarks/fast_output/reports/.json] + +Cases come from benchmarks/fast_output/cases.json and run one at a time (renders queue anyway). +Each run writes benchmarks/fast_output/reports/.json and .md. Numbers come from the +project's own records: stage timings and model tokens (metadata/llm_usage.jsonl), the studio state +(clips, variants, burned captions, source height). Nothing is published. +""" +from __future__ import annotations + +import argparse +import json +import subprocess +import sys +import time +import urllib.request +from datetime import datetime +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +CASES = ROOT / 'benchmarks' / 'fast_output' / 'cases.json' +REPORTS = ROOT / 'benchmarks' / 'fast_output' / 'reports' +PRICE_IN, PRICE_OUT = 0.8e-6, 2e-6 # qwen-plus, ¥ per token (estimate; check the provider's price page) +TERMINAL = ('completed', 'partial', 'failed') + + +def api(server: str, path: str): + with urllib.request.urlopen(f'{server}/api/v1/studio{path}', timeout=30) as response: + return json.load(response) + + +def start(server: str, case: dict) -> str: + args = ['curl', '-s', '-X', 'POST', f'{server}/api/v1/studio/import', '-F', f"url={case['url']}", '-F', f"name=bench-{case['id']}", + '-F', 'auto_start=true', '-F', 'brand_outro_enabled=true'] + if 'youtube' in case['url']: + args += ['-F', 'browser=chrome'] + for platform in case['platforms']: + args += ['-F', f'platforms={platform}'] + reply = json.loads(subprocess.run(args, capture_output=True, text=True, check=True).stdout) + if not reply.get('project_id'): + raise RuntimeError(f'import refused: {reply}') + return reply['project_id'] + + +def wait(server: str, project_id: str, timeout_min: float) -> dict: + deadline = time.time() + timeout_min * 60 + while time.time() < deadline: + state = api(server, f'/{project_id}') + generation, analysis = state.get('generation') or {}, state.get('analysis') or {} + if generation.get('status') in TERMINAL or analysis.get('status') == 'failed': + return state + time.sleep(20) + raise TimeoutError(f'{project_id} did not finish in {timeout_min} min') + + +def _seconds(start: str | None, end: str | None) -> float | None: + if not start or not end: + return None + parse = lambda value: datetime.fromisoformat(value.replace('Z', '+00:00')).timestamp() # noqa: E731 + return round(parse(end) - parse(start), 1) + + +def measure(case: dict, project_id: str, state: dict, data_dir: Path, wall: float) -> dict: + metadata = data_dir / 'projects' / project_id / 'metadata' + stages, timings = {}, {} + usage = metadata / 'llm_usage.jsonl' + for line in usage.read_text(encoding='utf-8').splitlines() if usage.exists() else []: + row = json.loads(line) + if row.get('kind') == 'timing': + timings[row['stage']] = round(timings.get(row['stage'], 0) + row['seconds'], 1) + continue + item = stages.setdefault(row['stage'], {'calls': 0, 'in': 0, 'out': 0}) + item['calls'] += 1 + item['in'] += row.get('prompt_tokens') or 0 + item['out'] += row.get('completion_tokens') or 0 + tokens_in, tokens_out = sum(s['in'] for s in stages.values()), sum(s['out'] for s in stages.values()) + drafts = {d['id']: d for d in state.get('drafts', [])} + variants = state.get('output_variants', []) + lengths = sorted(round(sum(sc['end'] - sc['start'] for sc in drafts[v['draft_id']]['scenes'])) for v in variants if v['draft_id'] in drafts) + generation, meta = state.get('generation') or {}, state.get('source_meta') or {} + packaged = [drafts[v['draft_id']].get('packaging') for v in variants if v['draft_id'] in drafts and drafts[v['draft_id']].get('packaging')] + return { + 'id': case['id'], 'project_id': project_id, 'platforms': case['platforms'], + 'outcome': generation.get('status') or (state.get('analysis') or {}).get('status'), + 'wall_min': round(wall / 60, 1), 'generation_min': round((_seconds(generation.get('started_at'), generation.get('finished_at')) or 0) / 60, 1), + 'timings_sec': timings, + 'calls': sum(s['calls'] for s in stages.values()), 'tokens_in': tokens_in, 'tokens_out': tokens_out, + 'yuan': round(tokens_in * PRICE_IN + tokens_out * PRICE_OUT, 3), 'stages': stages, + 'variants': len(variants), 'rendered': sum(v['status'] == 'completed' for v in variants), + 'on_demand': sum(v['status'] == 'on_demand' for v in variants), 'failed': sum(v['status'] == 'failed' for v in variants), + 'clip_sec': {'min': lengths[0], 'median': lengths[len(lengths) // 2], 'max': lengths[-1]} if lengths else None, + 'source_height': meta.get('source_height'), 'subtitle_source': meta.get('subtitle_source', 'speech'), + 'burned_captions': generation.get('source_has_burned_subtitles'), 'burned_caption_language': generation.get('burned_caption_language'), + 'packaging_fallbacks': sum(bool(p.get('fallback')) for p in packaged), + 'captioned_variants': sum(bool(p.get('cues')) for p in packaged), + 'titles': [' / '.join(drafts[v['draft_id']]['packaging']['title_lines']) for v in variants + if v['draft_id'] in drafts and (drafts[v['draft_id']].get('packaging') or {}).get('title_lines')][:6], + 'post_titles': [v['post']['title'] for v in variants if v.get('post')][:6], + } + + +def markdown(results: list[dict], baseline: dict | None) -> str: + old = {row['id']: row for row in (baseline or {}).get('results', [])} + + def delta(row, key): + before = old.get(row['id'], {}).get(key) + return f" ({row[key] - before:+.1f})" if isinstance(before, (int, float)) and isinstance(row.get(key), (int, float)) else '' + + lines = ['| case | outcome | minutes | transcribe s | clip finder s | render s | calls | tokens in/out | ¥ | clips → rendered | median clip s | source | burned | fallbacks |', + '|---|---|---|---|---|---|---|---|---|---|---|---|---|---|'] + for row in results: + t = row.get('timings_sec', {}) + lines.append(f"| {row['id']} | {row['outcome']} | {row['generation_min']}{delta(row, 'generation_min')} | {t.get('transcribe', 0)} | {t.get('clip_finder', 0)} | " + f"{t.get('render', 0)} | {row['calls']} | {row['tokens_in']}/{row['tokens_out']} | {row['yuan']}{delta(row, 'yuan')} | " + f"{row['variants']} → {row['rendered']} | {(row['clip_sec'] or {}).get('median', '-')} | {row['source_height']}p {row['subtitle_source']} | " + f"{row['burned_captions']} {row['burned_caption_language'] or ''} | {row['packaging_fallbacks']} |") + lines += ['', '人工检查:每条抽 2 支成片,看开头与结尾是否完整、字幕是否重复或不同步、取景是否对着说话人、封面与文案是否合适。'] + return '\n'.join(lines) + '\n' + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument('--server', default='http://127.0.0.1:18765') + parser.add_argument('--data-dir', type=Path, required=True) + parser.add_argument('--cases', default='') + parser.add_argument('--baseline', type=Path) + parser.add_argument('--timeout-min', type=float, default=120) + args = parser.parse_args() + cases = [c for c in json.loads(CASES.read_text(encoding='utf-8'))['cases'] if c.get('url')] + if args.cases: + wanted = args.cases.split(',') + cases = [c for c in cases if c['id'] in wanted] + baseline = json.loads(args.baseline.read_text(encoding='utf-8')) if args.baseline else None + REPORTS.mkdir(parents=True, exist_ok=True) + stamp = datetime.now().strftime('%Y%m%d-%H%M') + results = [] + for case in cases: + began = time.time() + try: + project_id = start(args.server, case) + state = wait(args.server, project_id, args.timeout_min) + results.append(measure(case, project_id, state, args.data_dir, time.time() - began)) + except Exception as error: # noqa: BLE001 - one bad case must not lose the others + results.append({'id': case['id'], 'outcome': f'error: {error}'[:200], 'generation_min': None, 'calls': 0, 'tokens_in': 0, 'tokens_out': 0, + 'yuan': 0, 'variants': 0, 'rendered': 0, 'clip_sec': None, 'source_height': None, 'subtitle_source': None, + 'burned_captions': None, 'burned_caption_language': None, 'packaging_fallbacks': 0, 'timings_sec': {}}) + print(json.dumps(results[-1], ensure_ascii=False), flush=True) + report = {'created': stamp, 'results': results} + (REPORTS / f'{stamp}.json').write_text(json.dumps(report, ensure_ascii=False, indent=1), encoding='utf-8') + (REPORTS / f'{stamp}.md').write_text(markdown(results, baseline), encoding='utf-8') + return 0 + + +if __name__ == '__main__': + sys.exit(main())