mirror of
https://github.com/zhouxiaoka/autoclip.git
synced 2026-10-02 02:34:34 +08:00
feat(packaging): pick palette and style from the content mood
The packaging model names the clip's mood (calm/serious/bold/warm/playful); the mood decides which of seven content palettes and which template styles fit, and a seed from the clip picks among them, so clips vary while one clip keeps its look on every platform. Content colours are no longer tied to the product UI accent; light accents get dark text on filled pills. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -83,6 +83,9 @@ class Packaging(BaseModel):
|
||||
fallback: bool = False
|
||||
# Visual style inside the template; None = the golden default (classic / pop).
|
||||
style: Literal['classic', 'boxed', 'spotlight', 'pop', 'cinematic'] | None = None
|
||||
# Content mood chosen by the model; it picks the palette and style (packaging.choose_look).
|
||||
mood: Literal['calm', 'serious', 'bold', 'warm', 'playful'] | None = None
|
||||
palette: Literal['azure', 'amber', 'coral', 'mint', 'lemon', 'rose', 'lilac'] | None = None
|
||||
|
||||
@model_validator(mode='after')
|
||||
def short_title_lines(self):
|
||||
|
||||
@@ -39,8 +39,34 @@ PROMPT = (
|
||||
'tags 仅当 template 为 interview_zh 时给出 2–4 个编辑点评(中文,每个不超过 10 个字),必须具体点出这一句最有冲击力的内容,'
|
||||
'例如“七分钟干完三个月”“以退为进”;禁止“逻辑清晰”“直击核心”“干货满满”这类泛泛评价;line 指向被点评的行。'
|
||||
'highlights 仅当 template 为 podcast_en 时给出,每 3–4 行最多一个,word 必须是该行(翻译后)里出现的单个关键词。'
|
||||
'另外返回 "mood":按这段内容本身的情绪选一个——calm(冷静理性的分析)、serious(严肃、风险、警示)、'
|
||||
'bold(强观点、冲突、爆点)、warm(真诚、感动、个人经历)、playful(轻松、幽默、有趣)。'
|
||||
)
|
||||
|
||||
# Content looks: the mood decides which palettes and styles fit; a seed from the content picks
|
||||
# among them, so different clips look different while one clip looks the same on every platform.
|
||||
MOOD_LOOKS = {
|
||||
'calm': (('azure', 'mint', 'lilac'), {'interview_zh': ('classic', 'spotlight'), 'podcast_en': ('cinematic', 'boxed')}),
|
||||
'serious': (('azure', 'amber'), {'interview_zh': ('classic', 'boxed'), 'podcast_en': ('cinematic', 'boxed')}),
|
||||
'bold': (('lemon', 'coral'), {'interview_zh': ('boxed', 'classic'), 'podcast_en': ('pop', 'boxed')}),
|
||||
'warm': (('amber', 'rose'), {'interview_zh': ('spotlight', 'classic'), 'podcast_en': ('boxed', 'cinematic')}),
|
||||
'playful': (('rose', 'lemon', 'mint'), {'interview_zh': ('boxed', 'spotlight'), 'podcast_en': ('pop', 'boxed')}),
|
||||
}
|
||||
|
||||
|
||||
def choose_look(template: str, mood: object, seed: str) -> dict[str, str | None]:
|
||||
"""{'mood', 'palette', 'style'} for this content; no valid mood keeps the golden default look."""
|
||||
import random
|
||||
if mood not in MOOD_LOOKS:
|
||||
return {'mood': None, 'palette': None, 'style': None}
|
||||
palettes, styles = MOOD_LOOKS[mood]
|
||||
pick = random.Random(f'{seed}:{mood}')
|
||||
return {'mood': mood, 'palette': pick.choice(palettes), 'style': pick.choice(styles[template])}
|
||||
|
||||
|
||||
def _seed(lines: list[dict[str, Any]]) -> str:
|
||||
return f"{lines[0]['start']:.1f}-{lines[-1]['end']:.1f}" if lines else ''
|
||||
|
||||
|
||||
def source_language(texts: list[str]) -> str:
|
||||
joined = ''.join(texts)
|
||||
@@ -194,5 +220,6 @@ def _validated(result, lines, base, translate, burned, known_names, draft):
|
||||
highlights.append({'at': cue['start'], 'text': word})
|
||||
for cue in cues:
|
||||
cue.pop('lines', None)
|
||||
return {**base, 'title_lines': titles, 'title_accent_line': min(accent, max(0, len(titles) - 1)),
|
||||
look = choose_look(base['template'], result.get('mood'), _seed(lines))
|
||||
return {**base, **look, 'title_lines': titles, 'title_accent_line': min(accent, max(0, len(titles) - 1)),
|
||||
'cues': cues, 'speakers': speakers[:8], 'tags': tags, 'highlights': highlights[:40]}
|
||||
|
||||
@@ -2,7 +2,8 @@
|
||||
|
||||
Each scene is encoded on its own (render.py), so every overlay is expressed on the output
|
||||
timeline and clipped to the scene: the scene's ASS only carries events visible inside it,
|
||||
with times relative to the scene start. Colours follow DESIGN.md (dark theme).
|
||||
with times relative to the scene start. Colours come from the content palette (mood), not the
|
||||
product UI; the default palette matches DESIGN.md's accent.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -15,8 +16,29 @@ FONT_DIR = Path(__file__).resolve().parents[2] / 'assets' / 'fonts'
|
||||
FONT = 'Noto Sans SC' # bundled (OFL); covers CJK and Latin on every platform
|
||||
W, H = 1080, 1920
|
||||
WIN_Y, WIN_H = 560, 810
|
||||
BG, ACCENT_HEX = '0x1A1A19', '0x5A8BFF'
|
||||
ACCENT, WHITE, SUB, INK, INK_SOFT = '&H00FF8B5A', '&H00E6EAEC', '&H009BA2A6', '&H00191A1A', '&H26191A1A'
|
||||
WHITE, SUB, INK, INK_SOFT = '&H00E6EAEC', '&H009BA2A6', '&H00191A1A', '&H26191A1A'
|
||||
# Content palettes (accent, canvas). Content follows its mood, not the product UI: azure is the
|
||||
# golden default; the canvas stays a near-black tinted toward the accent.
|
||||
PALETTES = {
|
||||
'azure': ('5A8BFF', '1A1A19'), 'amber': ('F2B544', '1B1712'), 'coral': ('FF6F59', '1B1514'),
|
||||
'mint': ('46D3A6', '111917'), 'lemon': ('F4DC3C', '141413'), 'rose': ('FF7FA9', '1B1418'),
|
||||
'lilac': ('B69CFF', '17151C'),
|
||||
}
|
||||
|
||||
|
||||
def _ass(rgb: str) -> str:
|
||||
return f'&H00{rgb[4:6]}{rgb[2:4]}{rgb[0:2]}'.upper()
|
||||
|
||||
|
||||
def colours(palette: str | None) -> dict[str, str]:
|
||||
"""ASS and ffmpeg colours for a palette; light accents get dark text on filled pills."""
|
||||
accent, canvas = PALETTES.get(palette or 'azure', PALETTES['azure'])
|
||||
r, g, b = (int(accent[i:i + 2], 16) for i in (0, 2, 4))
|
||||
light = 0.2126 * r + 0.7152 * g + 0.0722 * b > 150
|
||||
return {'accent': _ass(accent), 'on_accent': INK if light else WHITE, 'accent_hex': f'0x{accent}', 'bg': f'0x{canvas}'}
|
||||
|
||||
|
||||
BG, ACCENT_HEX, ACCENT = colours(None)['bg'], colours(None)['accent_hex'], colours(None)['accent']
|
||||
NAMEPLATE_SEC = 3.2
|
||||
|
||||
|
||||
@@ -58,16 +80,18 @@ def _limit(font_size: int) -> float:
|
||||
return 920 / font_size
|
||||
|
||||
|
||||
def _emphasize(text: str, phrases: list[str]) -> str:
|
||||
def _emphasize(text: str, phrases: list[str], accent: str = ACCENT) -> str:
|
||||
"""Colour phrases (already escaped) in the accent blue inside an escaped caption."""
|
||||
for phrase in sorted({p for p in phrases if p}, key=len, reverse=True):
|
||||
safe = _esc(phrase)
|
||||
if safe in text:
|
||||
text = text.replace(safe, f'{{\\c{ACCENT}}}{safe}{{\\c{WHITE}}}', 1)
|
||||
text = text.replace(safe, f'{{\\c{accent}}}{safe}{{\\c{WHITE}}}', 1)
|
||||
return text
|
||||
|
||||
|
||||
def _header() -> str:
|
||||
def _header(look: dict[str, str] | None = None) -> str:
|
||||
look = look or colours(None)
|
||||
ACCENT, ON_ACCENT = look['accent'], look['on_accent'] # noqa: N806 - keeps the style table readable
|
||||
return f"""[Script Info]
|
||||
ScriptType: v4.00+
|
||||
PlayResX: {W}
|
||||
@@ -88,7 +112,7 @@ Style: PlateName,{FONT},46,{WHITE},{WHITE},{INK_SOFT},{INK_SOFT},1,0,0,0,100,100
|
||||
Style: PlateRole,{FONT},32,{SUB},{SUB},{INK_SOFT},{INK_SOFT},0,0,0,0,100,100,0,0,3,12,0,7,0,0,0,1
|
||||
Style: PlateBar,{FONT},10,{ACCENT},{ACCENT},{ACCENT},{ACCENT},0,0,0,0,100,100,0,0,1,0,0,7,0,0,0,1
|
||||
Style: CaptionBox,{FONT},58,{WHITE},{WHITE},{INK_SOFT},{INK_SOFT},1,0,0,0,100,100,1,0,3,14,0,2,48,48,0,1
|
||||
Style: TagPill,{FONT},48,{WHITE},{WHITE},{ACCENT},{ACCENT},1,0,0,0,100,100,2,0,3,14,0,5,40,40,0,1
|
||||
Style: TagPill,{FONT},48,{ON_ACCENT},{ON_ACCENT},{ACCENT},{ACCENT},1,0,0,0,100,100,2,0,3,14,0,5,40,40,0,1
|
||||
Style: Cine,{FONT},64,{WHITE},{WHITE},&H00000000,&H00000000,1,0,0,0,100,100,2,0,1,0,0,2,60,60,0,1
|
||||
Style: CineGlow,{FONT},64,{ACCENT},{ACCENT},{ACCENT},&H00000000,1,0,0,0,100,100,2,0,1,3,0,2,60,60,0,1
|
||||
|
||||
@@ -162,6 +186,8 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
|
||||
out = _Scene(offset, length)
|
||||
interview = packaging.template == 'interview_zh'
|
||||
style = packaging.style or ('classic' if interview else 'pop')
|
||||
look = colours(packaging.palette)
|
||||
accent = look['accent']
|
||||
tag_texts = [t.text for t in packaging.tags] if packaging.tags_enabled else []
|
||||
|
||||
lines = packaging.title_lines
|
||||
@@ -191,7 +217,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
|
||||
for s, e, text, original in timed_screens(cue.text, start, end, _limit(size), cue.original, _limit(36)):
|
||||
body = _esc_lines(text)
|
||||
if style == 'spotlight':
|
||||
body = _emphasize(body, tag_texts)
|
||||
body = _emphasize(body, tag_texts, accent)
|
||||
motion = f'\\move(540,{y + 24},540,{y},0,180)\\fad(120,60)'
|
||||
else:
|
||||
motion = f'\\pos(540,{y})\\fad(60,60)' if style == 'classic' else f'\\pos(540,{y})\\fad(90,90)'
|
||||
@@ -204,7 +230,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
|
||||
for s, e, text, _ in timed_screens(cue.text, start, end, _limit(58) / 0.95):
|
||||
body = _esc_lines(text)
|
||||
for word in highlights:
|
||||
body = re.sub(rf'(?i)\b({re.escape(_esc(word))})\b', lambda m: f'{{\\c{ACCENT}}}{m.group(1)}{{\\c{WHITE}}}', body, count=1)
|
||||
body = re.sub(rf'(?i)\b({re.escape(_esc(word))})\b', lambda m: f'{{\\c{accent}}}{m.group(1)}{{\\c{WHITE}}}', body, count=1)
|
||||
out.add(2, s, e, 'CaptionBox', f'{{\\an2\\pos(540,1420)\\fad(90,90)}}{body}')
|
||||
continue
|
||||
words = _words(cue.text, start, end, None)
|
||||
@@ -224,7 +250,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
|
||||
parts = []
|
||||
for k, (_, _, token) in enumerate(chunk):
|
||||
active = k == j or token.strip('.,?!;:').lower() in highlights
|
||||
parts.append(f'{{\\c{ACCENT}}}{_esc(token)}{{\\c{WHITE}}}' if active else _esc(token))
|
||||
parts.append(f'{{\\c{accent}}}{_esc(token)}{{\\c{WHITE}}}' if active else _esc(token))
|
||||
stop = chunk[j + 1][0] if j + 1 < len(chunk) else chunk_end
|
||||
pop = '{\\fscx86\\fscy86\\t(0,110,\\fscx100\\fscy100)}' if j == 0 else ''
|
||||
out.add(2, w0, stop, 'Words', f'{{\\an2\\pos(540,1400)}}{pop}' + ' '.join(parts))
|
||||
@@ -245,7 +271,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
|
||||
continue
|
||||
motion = '\\fscx50\\fscy50\\t(0,140,\\fscx112\\fscy112)\\t(140,260,\\fscx100\\fscy100)\\fad(0,240)'
|
||||
out.add(4, at, at + 1.8, 'Tag', f'{{\\an5\\pos(540,{WIN_Y + WIN_H - 120}){motion}}}{_esc(tag.text)}', carry=False)
|
||||
return _header() + '\n'.join(out.lines) + '\n'
|
||||
return _header(look) + '\n'.join(out.lines) + '\n'
|
||||
|
||||
|
||||
def scene_video_graph(draft: Draft, scene: Scene, index: int, ass_path: Path, w: int, h: int) -> tuple[str, str]:
|
||||
@@ -261,6 +287,8 @@ def scene_video_graph(draft: Draft, scene: Scene, index: int, ass_path: Path, w:
|
||||
total = rows[-1][2] + scene_duration(draft.scenes[-1])
|
||||
offset, length = rows[index][2], scene_duration(scene)
|
||||
ass = f"ass='{_escape_filter_path(ass_path)}':fontsdir='{_escape_filter_path(FONT_DIR)}'"
|
||||
look = colours(draft.packaging.palette if draft.packaging else None)
|
||||
BG, ACCENT_HEX = look['bg'], look['accent_hex'] # noqa: N806
|
||||
if draft.layout == 'window':
|
||||
if scene.crop_track or scene.crop_x is not None:
|
||||
x = crop_expression(scene, draft.crop_x)
|
||||
|
||||
@@ -125,3 +125,18 @@ def test_paragraph_sized_segments_are_rejected_for_translation():
|
||||
def test_source_language_detection():
|
||||
assert packaging.source_language(['这是一个中文字幕,内容比较长一些']) == 'zh'
|
||||
assert packaging.source_language(['This is an English subtitle line with words']) == 'en'
|
||||
|
||||
|
||||
def test_the_content_mood_picks_a_matching_look_that_is_stable_per_clip():
|
||||
result = packaging.build_packaging(DRAFT, LINES, platform_strategy('douyin'), call=lambda *_: good_response(mood='bold'))
|
||||
palettes, styles = packaging.MOOD_LOOKS['bold']
|
||||
assert result['mood'] == 'bold' and result['palette'] in palettes and result['style'] in styles['interview_zh']
|
||||
again = packaging.build_packaging(DRAFT, LINES, platform_strategy('xiaohongshu'), call=lambda *_: good_response(mood='bold'))
|
||||
assert (again['palette'], again['style']) == (result['palette'], result['style']) # same clip, same look
|
||||
looks = {tuple(packaging.choose_look('podcast_en', 'calm', f'{n}.0-{n + 60}.0').values()) for n in range(40)}
|
||||
assert len(looks) > 2 # different clips vary within the mood
|
||||
|
||||
|
||||
def test_an_unknown_mood_keeps_the_golden_default_look():
|
||||
result = packaging.build_packaging(DRAFT, LINES, platform_strategy('douyin'), call=lambda *_: good_response(mood='angry'))
|
||||
assert (result['mood'], result['palette'], result['style']) == (None, None, None)
|
||||
|
||||
@@ -124,3 +124,15 @@ def test_interview_window_renders_title_canvas_and_window(tmp_path):
|
||||
assert max(max(px) for px in title_region.getdata()) > 180 # bright title text drawn on the dark canvas
|
||||
window_row = [image.getpixel((x, pr.WIN_Y + 200)) for x in range(0, 1080, 60)]
|
||||
assert len(set(window_row)) > 3 # the source picture fills the 4:3 window
|
||||
|
||||
|
||||
def test_the_palette_colours_accents_and_keeps_pill_text_readable():
|
||||
lemon = pr.scene_ass(_packaging(palette='lemon', style='boxed'), SCENES, 1)
|
||||
assert '&H003CDCF4' in lemon # lemon F4DC3C in ASS BGR order
|
||||
pill = next(line for line in lemon.splitlines() if line.startswith('Style: TagPill'))
|
||||
assert pill.split(',')[3] == pr.INK # dark text on a light accent
|
||||
azure = pr.scene_ass(_packaging(), SCENES, 1)
|
||||
assert '&H00FF8B5A' in azure and '&H003CDCF4' not in azure
|
||||
draft = Draft(id='d', title='T', scenes=SCENES[:1], aspect='portrait', layout='window', packaging=_packaging(palette='mint'))
|
||||
graph, _ = pr.scene_video_graph(draft, SCENES[0], 0, pr.FONT_DIR / 'x.ass', 1080, 1920)
|
||||
assert 'color=c=0x111917' in graph and 'color=c=0x46D3A6' in graph
|
||||
|
||||
@@ -47,6 +47,8 @@ export interface Packaging {
|
||||
highlights: { at: number; text: string }[]
|
||||
burned_captions: boolean; fallback: boolean
|
||||
style?: 'classic' | 'boxed' | 'spotlight' | 'pop' | 'cinematic' | null
|
||||
mood?: 'calm' | 'serious' | 'bold' | 'warm' | 'playful' | null
|
||||
palette?: 'azure' | 'amber' | 'coral' | 'mint' | 'lemon' | 'rose' | 'lilac' | null
|
||||
}
|
||||
export interface Draft {
|
||||
id: string; title: string; hook: string; scenes: Scene[]; language: Language
|
||||
|
||||
Reference in New Issue
Block a user