diff --git a/CHANGELOG.md b/CHANGELOG.md index 6e359c62..cda04c90 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,13 @@ ## [未发布] ### 新增 +- **一键出片**:贴一个视频链接、选好要发的平台(抖音、小红书、TikTok、Reels、YouTube Shorts、B 站、YouTube),自动挑出值得发的片段,按每个平台的画幅、时长和包装直接生成可发布成片;可随时追加平台版本,失败的单条可重试。 +- **两套包装模板**:抖音 / 小红书用「访谈式」(上方两行标题、4:3 说话人窗口、中英双语字幕、左下名牌、编辑点评标签);TikTok / Reels / Shorts 用「播客式」(全屏跟随说话人、逐词高亮字幕、开头一句 hook、名牌)。外语素材自动翻成目标平台的语言;原片自带字幕时保留完整画面,字幕放在画面下方。 +- **按内容情绪变换风格**:模型判断每段内容的情绪(冷静、严肃、强观点、真诚、轻松),据此在 7 套配色和多种字幕动效里挑选,不同片段风格各异,同一片段在各平台保持一致。 +- **只先生成最好的 10 条**:长视频会挑出 20–40 个片段,默认自动生成评分最高的 10 条,其余列为「备选片段」,点「生成这条」再制作,省时间也省模型费用。 +- **每条成片自带发布包**:按目标平台写好标题、简介和话题(抖音钩子式、小红书笔记式、B 站信息完整、TikTok / Reels / Shorts 英文文案),按平台字数与话题数限制;封面在成片完成后自动设计——说话人画面、成片同款标题与配色、嘉宾名牌,按平台比例(竖屏 9:16、小红书 3:4、B 站 16:10、YouTube 16:9)。成片卡片上可直接改文案、复制、导出「发布包」(视频 + 封面 + 文案),也可以一键用 AI 重新设计封面;去发布时自动带上这套文案与封面。 +- **新片尾动画**:1.8 秒的 AutoClip 标志动画与提示音,竖屏、横屏各一版,其他比例自动居中适配。 +- 竖屏成片按说话人自动取景,主持人与嘉宾切换时画面跟着说话的人走。 - 新增赞助合作伙伴 88API:八语 README、首页与设置页推荐入口、独立 API Key、模型发现、兼容封面与 Whisper 转写;CLI / MCP 使用 `api88`。 - 模型设置新增赞助合作伙伴「Infistar 无限星河」:接口地址已预设,填好 Key 后自动列出账号可用的模型;设置页提供专属注册链接(可领 $5 体验额度)和接入说明。CLI / MCP 支持 `--provider infistar` 与 `INFISTAR_API_KEY`。 - 设置 → AI 模型重新整理为「AI 服务 / 字幕转写 / 封面 / 高级」四段:选一家服务、填好 Key,分析模型自动选好;画面识别和 AI 封面首次配置默认开启,画面识别注明适合游戏画面、口播较少的内容。供应商分组中的赞助伙伴改为「推荐」。 @@ -17,9 +24,18 @@ - 竖屏「满屏」构图新增按镜头自动取景:先识别镜头切换,再对准正在说话的人;引用卡、PPT 等没有人物的镜头自动改为完整画面 + 模糊背景,不再被裁掉两侧。首次使用需下载约 45 MB 的人物识别组件;每个镜头都可以单独改为「对准人物 / 完整画面」并微调位置。 ### 改进 +- **出片速度大幅提升**:一条 2 小时访谈从链接到成片,由约 65 分钟缩短到 7–8 分钟(视频有作者字幕时)或约 30 分钟(需要本地语音识别时)。片段挑选改为一次通读全文,原来几十次模型调用、半小时以上,现在一分钟以内;各步骤的模型调用并行执行;视频有作者上传的字幕时直接使用,不再本地语音识别。 +- **模型费用更低**:一条 2–3 小时访谈的模型费用由约 ¥0.6 降到 ¥0.1–0.2(qwen-plus 估算),模型调用由近 200 次降到 30 次左右;同语言包装不再让模型复述整段字幕。 +- **更安静**:成片用电脑自带的硬件编码器(macOS VideoToolbox、Windows NVENC / QSV / AMF),CPU 占用约降到原来的四分之一,风扇不再狂转;硬件编码不可用时自动改用软件编码。 +- **切点更自然**:片段从问题或观点的第一句开始,到回答讲完、说话人有明显停顿时才结束;按原片音频的真实停顿下刀,不再带进半句下一段话或主持人的下一个问题。 - 导入确认页去掉重复的分析方式提问,控件和文案与设置页统一;发布页和编辑器的「烧录」「标题卡」等术语换成直白说法,编辑器和导出对话框补上封面入口,发布页默认自动生成封面。 ### 修复 +- 外语素材原片烧有英文字幕时,抖音版也会配中文字幕。 +- 原片已有硬字幕时,按目标平台判断是否加字幕:中文硬字幕投抖音不再重复加中文字幕,投 TikTok 加英文;外语访谈配了英文硬字幕的,投 TikTok 不再重复。细小、无描边的硬字幕(常见于 B 站访谈)也能识别。 +- YouTube 偶尔只给 360p 画质时会自动换方式重新下载到 720p 以上,不再出模糊成片。 +- 同语言字幕与声音同步;电影感字幕改为一次 2–5 个词,不再一个词一个词跳。 +- 标题不再出现 Markdown 符号或 `&` 之类的转义字符;中文平台不会出现日文标题;包装偶发不合规时会自动重试一次,不再整段退回原字幕。 - 时间线重试只使用本次有效结果,避免失败后误用上次候选;原始响应缓存现在会正常解析,无效缓存不会自动触发模型请求。 - 智能导入后台任务提交失败后可明确重试,已有方案、草稿和成片保留;修改方案失败时同步恢复偏好。 - 时间线兼容常见时间戳格式;相邻短片段在丢弃前尝试合并,保留既有时长限制。 diff --git a/backend/core/llm_usage.py b/backend/core/llm_usage.py index e0cd2188..60a7e718 100644 --- a/backend/core/llm_usage.py +++ b/backend/core/llm_usage.py @@ -88,6 +88,8 @@ def summary(project_id: str) -> dict[str, Any]: row = json.loads(line) except ValueError: continue + if row.get('kind') == 'timing': + continue item = stages.setdefault(row.get('stage', 'other'), {'calls': 0, 'prompt_tokens': 0, 'completion_tokens': 0, 'estimated_calls': 0}) item['calls'] += 1 item['prompt_tokens'] += row.get('prompt_tokens') or 0 @@ -101,3 +103,39 @@ def run_in_context(fn): """Wrap `fn` so a worker thread records into the caller's sink and stage.""" context = contextvars.copy_context() return lambda *args, **kwargs: context.copy().run(fn, *args, **kwargs) + + +@contextmanager +def timed(stage_name: str): + """Record how long a step took (wall seconds) next to its token usage, for cost/time reports.""" + started = time.monotonic() + try: + yield + finally: + path = _sink.get() + if path is not None: + row = {'at': round(time.time(), 1), 'kind': 'timing', 'stage': stage_name, 'seconds': round(time.monotonic() - started, 2)} + try: + with _lock: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open('a', encoding='utf-8') as handle: + handle.write(json.dumps(row) + '\n') + except OSError: + pass + + +def timings(project_id: str) -> dict[str, float]: + """Total wall seconds per timed stage of one project.""" + out: dict[str, float] = {} + try: + lines = usage_path(project_id).read_text(encoding='utf-8').splitlines() + except OSError: + return out + for line in lines: + try: + row = json.loads(line) + except ValueError: + continue + if row.get('kind') == 'timing': + out[row['stage']] = round(out.get(row['stage'], 0) + float(row.get('seconds') or 0), 2) + return out diff --git a/backend/services/simple_pipeline_adapter.py b/backend/services/simple_pipeline_adapter.py index 1a41204d..4043bf5a 100644 --- a/backend/services/simple_pipeline_adapter.py +++ b/backend/services/simple_pipeline_adapter.py @@ -147,7 +147,8 @@ class SimplePipelineAdapter: emit_progress(self.project_id, "ANALYZE", "正在通读全文挑选片段") try: entries = TextProcessor.parse_srt(Path(srt_path)) - return find_clips(entries, text_json, threshold=resolve_min_score_threshold(), metadata_dir=metadata_dir) + with llm_usage.timed("clip_finder"): + return find_clips(entries, text_json, threshold=resolve_min_score_threshold(), metadata_dir=metadata_dir) except Exception as error: # noqa: BLE001 - the legacy steps still work with any model logger.warning("一次挑片不可用,改用分步分析: %s", error) return None @@ -199,7 +200,8 @@ class SimplePipelineAdapter: from backend.utils.speech_recognizer import SpeechRecognitionError logger.warning("没有SRT文件,尝试自动生成字幕") try: - srt_path = await self._generate_subtitle_automatically(input_video_path, metadata_dir) + with llm_usage.timed("transcribe"): + srt_path = await self._generate_subtitle_automatically(input_video_path, metadata_dir) except SpeechRecognitionError as e: raise failure_from_speech_error(str(e)) from e if not (srt_path and srt_path.exists()): diff --git a/backend/services/studio/jobs.py b/backend/services/studio/jobs.py index 862d6282..cebac039 100644 --- a/backend/services/studio/jobs.py +++ b/backend/services/studio/jobs.py @@ -74,7 +74,8 @@ def _render(project_id, draft, job_id, *, brand_outro=False): store.change(project_id, mutate) try: update(status='running', percent=5) - result = render_draft(project_id, source(project_id), draft, job_id, lambda p: update(percent=p), brand_outro=brand_outro) + with llm_usage.timed('render'): + result = render_draft(project_id, source(project_id), draft, job_id, lambda p: update(percent=p), brand_outro=brand_outro) update(status='completed', percent=100, result=result, duration_ms=round((monotonic() - started) * 1000)) _design_covers(project_id, draft, job_id) _sync_variant_status(project_id, job_id, 'completed') @@ -114,7 +115,8 @@ def _analyze(project_id, prefs, url, browser): try: mark_project(project_id, 'processing') if url: - download(project_id, url, browser) + with llm_usage.timed('download'): + download(project_id, url, browser) video = source(project_id) def stage(message): store.change(project_id, lambda data: data['analysis'].update(message=message)) @@ -399,7 +401,7 @@ def _complete_thought_bounds(project_id, clips): call = intelligence.text_json except Exception: # noqa: BLE001 call = None - with llm_usage.stage('boundaries'): + with llm_usage.stage('boundaries'), llm_usage.timed('boundaries'): return boundaries.refine_clips(rows, clips, call, boundaries.audio_silences(source(project_id)) if audio.has_audio(source(project_id)) else None) @@ -782,10 +784,13 @@ def _auto_generate(project_id, plan): now = base['id'] in automatic framed = None # on-demand versions are framed and packaged when the user asks (produce_variant) if now: - value, framed = _apply_framing(project_id, value, strategy_id, video, burned, framing_cache) + with llm_usage.timed('framing'): + value, framed = _apply_framing(project_id, value, strategy_id, video, burned, framing_cache) planned.append((strategy_id, value, trimmed, framed, now)) - _prefetch_packaging(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], burned, packaging_cache) - posts = _posts_for(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], packaging_cache) + with llm_usage.timed('packaging'): + _prefetch_packaging(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], burned, packaging_cache) + with llm_usage.timed('post_copy'): + posts = _posts_for(project_id, [(strategy_id, value) for strategy_id, value, _, _, now in planned if now], packaging_cache) for strategy_id, value, trimmed, framed, now in planned: if now: value = _apply_packaging(project_id, value, strategy_id, burned, packaging_cache) @@ -1062,7 +1067,8 @@ def _inspect(project_id, options, url, browser): from backend.services.studio.planning import recommend mark_project(project_id, 'processing', awaiting_confirmation=False) if url: - download(project_id, url, browser) + with llm_usage.timed('download'): + download(project_id, url, browser) store.change(project_id, lambda data:(data['analysis'].pop('percent', None), data['analysis'].update(message='快速判断适合的制作类型'))) plan = recommend(source(project_id), options) ensure_project_thumbnail(project_id) diff --git a/backend/tests/test_llm_usage.py b/backend/tests/test_llm_usage.py index b14b4faf..9a5eaca5 100644 --- a/backend/tests/test_llm_usage.py +++ b/backend/tests/test_llm_usage.py @@ -45,3 +45,15 @@ def test_manager_calls_record_provider_usage(monkeypatch, tmp_path): with llm_usage.tracking('p1'), llm_usage.stage('scoring'): assert manager.call('score', {'x': 1}) == '{"ok": true}' assert llm_usage.summary('p1')['stages']['scoring']['prompt_tokens'] == 42 + + +def test_stage_timings_add_up_and_do_not_count_as_model_calls(monkeypatch, tmp_path): + clock = iter([10.0, 12.5, 20.0, 21.0]) + monkeypatch.setattr(llm_usage.time, 'monotonic', lambda: next(clock)) + monkeypatch.setattr(llm_usage, 'usage_path', lambda _p: tmp_path / llm_usage.FILE) + with llm_usage.tracking('p1'): + for _ in range(2): + with llm_usage.timed('render'): + pass + assert llm_usage.timings('p1') == {'render': 3.5} + assert llm_usage.summary('p1')['total']['calls'] == 0 diff --git a/benchmarks/fast_output/cases.json b/benchmarks/fast_output/cases.json new file mode 100644 index 00000000..c27fc2e1 --- /dev/null +++ b/benchmarks/fast_output/cases.json @@ -0,0 +1,41 @@ +{ + "version": 1, + "note": "Fast-output regression set. Every optimisation is compared on these inputs (scripts/fast_output_benchmark.py). Picked 2026-10-01 by the owner; keep ids stable.", + "cases": [ + {"id": "neumann-doac", "url": "https://www.youtube.com/watch?v=IQ4JVWdj4Q0", "platforms": ["douyin", "tiktok"], "language": "en", + "traits": ["long interview", "founder story"], "watch": "WeWork rise and fall; story arcs should end on a complete answer"}, + {"id": "stallone-nyt", "url": "https://www.youtube.com/watch?v=ccs-B_nTfZs", "platforms": ["douyin"], "language": "en", + "traits": ["interview", "strong personality"], "watch": "Rocky, overnight fame, rivalry with Schwarzenegger"}, + {"id": "dafoe-hot-ones", "url": "https://www.youtube.com/watch?v=YqugY2zTIoI", "platforms": ["tiktok", "douyin"], "language": "en", + "traits": ["short source (23 min)", "reaction shots", "two speakers"], "watch": "hooks on reaction moments; short-tier clip lengths"}, + {"id": "druski-doac", "url": "https://www.youtube.com/watch?v=UhzI1fg8rCA", "platforms": ["tiktok"], "language": "en", + "traits": ["long interview", "creator economy"], "watch": "rejected by Netflix/Amazon, building his own audience"}, + {"id": "lidan-wangzuxian", "url": "https://www.bilibili.com/video/BV1XHhW6PEpf/", "platforms": ["douyin", "xiaohongshu"], "language": "zh", + "traits": ["Chinese source", "2 h", "likely burned Chinese captions"], "watch": "no duplicate Chinese captions; speech recognition time"}, + {"id": "rubin-doac", "url": "https://www.youtube.com/watch?v=a_GiFiHXJ6g", "platforms": ["douyin", "youtube_shorts"], "language": "en", + "traits": ["long interview", "creator audience"], "watch": "taste and judgement in the AI era; Shorts 180 s cap"}, + {"id": "openai-devday-2026", "url": "https://www.youtube.com/watch?v=Fls_onRviPM", "platforms": ["tiktok", "bilibili"], "language": "en", + "traits": ["keynote", "screen demos", "54 min"], "watch": "demo segments must keep the screen (full frame, no face crop)"}, + {"id": "dreamforce-keynote", "url": "https://www.youtube.com/watch?v=wYFt9NCYaVI", "platforms": ["douyin"], "language": "en", + "traits": ["keynote with guests", "1 h 48 m"], "watch": "chapters 45:10 Dario and 1:16:50 Jensen should be among the picks"}, + {"id": "luyu-guokeyu", "url": "https://www.bilibili.com/video/BV1LDYV6HEXR/", "platforms": ["douyin"], "language": "zh", + "traits": ["Chinese source", "story with a turn"], "watch": "complete turning-point stories, not fragments"}, + {"id": "hassabis-fry-rsa", "url": null, "platforms": ["douyin", "tiktok"], "language": "en", + "traits": ["talk plus conversation", "1 h"], "watch": "link to be added"}, + + {"id": "mrbeast-colin-samir", "url": "https://www.youtube.com/watch?v=9IQ_ldV9z_A", "platforms": ["tiktok"], "language": "en", + "traits": ["2 h", "auto captions only", "once downloaded at 360p"], "watch": "source must arrive at >=720p"}, + {"id": "karpathy-dwarkesh", "url": "https://www.youtube.com/watch?v=lXUZvyajciY", "platforms": ["douyin"], "language": "en", + "traits": ["2 h 26 m", "creator subtitles"], "watch": "no speech recognition (uploaded subtitles)"}, + {"id": "jensen-dwarkesh", "url": "https://www.youtube.com/watch?v=Hrbq66XqtCo", "platforms": ["xiaohongshu"], "language": "en", + "traits": ["1 h 43 m", "creator subtitles"], "watch": "baseline for the new pipeline: ~7.5 min"}, + {"id": "tim-luoyonghao", "url": "https://www.bilibili.com/video/BV1B5xkzPEhx/", "platforms": ["douyin", "tiktok"], "language": "zh", + "traits": ["2 h 52 m", "burned Chinese captions", "no subtitles"], "watch": "Douyin: no captions added; TikTok: English captions"}, + {"id": "dario-dwarkesh", "url": "https://www.youtube.com/watch?v=n1E9IZfvGMA", "platforms": ["douyin"], "language": "en", + "traits": ["2 h 22 m"], "watch": "golden samples for interview packaging"}, + {"id": "sam-altman-yc", "url": "https://www.youtube.com/watch?v=ZIaOBAjvc38", "platforms": ["xiaohongshu", "youtube_shorts"], "language": "en", + "traits": ["39 min", "two speakers on stage"], "watch": "speaker switches between Sam and Garry"}, + {"id": "kojima-wired", "url": "https://www.youtube.com/watch?v=02Ah5VQrzvA", "platforms": ["douyin", "tiktok"], "language": "ja", + "traits": ["17 min", "Japanese speech", "burned English captions"], "watch": "Douyin: Chinese captions; TikTok: no captions added"} + ] +} diff --git a/docs/FAST_OUTPUT_BENCHMARK.md b/docs/FAST_OUTPUT_BENCHMARK.md new file mode 100644 index 00000000..6e8ba4a6 --- /dev/null +++ b/docs/FAST_OUTPUT_BENCHMARK.md @@ -0,0 +1,44 @@ +# 快速出片回归测试集 + +每次优化切片、包装、渲染或成本,都用同一组真实素材对比:`benchmarks/fast_output/cases.json`(负责人 2026-10-01 选定;id 不改,新素材往后加)。 + +## 怎么跑 + +1. 用测试数据目录启动后端(不要用自己的真实数据目录),模型与语音识别按要测的配置设好: + + ```bash + AUTOCLIP_APP_DIR=