feat(packaging): pick palette and style from the content mood

The packaging model names the clip's mood (calm/serious/bold/warm/playful);
the mood decides which of seven content palettes and which template styles
fit, and a seed from the clip picks among them, so clips vary while one clip
keeps its look on every platform. Content colours are no longer tied to the
product UI accent; light accents get dark text on filled pills.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
周小舟
2026-10-01 03:49:09 +08:00
co-authored by Claude Opus 5.5
parent 206d7421e7
commit a7b6b57264
6 changed files with 99 additions and 12 deletions
+3
View File
@@ -83,6 +83,9 @@ class Packaging(BaseModel):
fallback: bool = False
# Visual style inside the template; None = the golden default (classic / pop).
style: Literal['classic', 'boxed', 'spotlight', 'pop', 'cinematic'] | None = None
# Content mood chosen by the model; it picks the palette and style (packaging.choose_look).
mood: Literal['calm', 'serious', 'bold', 'warm', 'playful'] | None = None
palette: Literal['azure', 'amber', 'coral', 'mint', 'lemon', 'rose', 'lilac'] | None = None
@model_validator(mode='after')
def short_title_lines(self):
+28 -1
View File
@@ -39,8 +39,34 @@ PROMPT = (
'tags 仅当 template 为 interview_zh 时给出 2–4 个编辑点评(中文,每个不超过 10 个字),必须具体点出这一句最有冲击力的内容,'
'例如“七分钟干完三个月”“以退为进”;禁止“逻辑清晰”“直击核心”“干货满满”这类泛泛评价;line 指向被点评的行。'
'highlights 仅当 template 为 podcast_en 时给出,每 3–4 行最多一个,word 必须是该行(翻译后)里出现的单个关键词。'
'另外返回 "mood":按这段内容本身的情绪选一个——calm(冷静理性的分析)、serious(严肃、风险、警示)、'
'bold(强观点、冲突、爆点)、warm(真诚、感动、个人经历)、playful(轻松、幽默、有趣)。'
)
# Content looks: the mood decides which palettes and styles fit; a seed from the content picks
# among them, so different clips look different while one clip looks the same on every platform.
MOOD_LOOKS = {
'calm': (('azure', 'mint', 'lilac'), {'interview_zh': ('classic', 'spotlight'), 'podcast_en': ('cinematic', 'boxed')}),
'serious': (('azure', 'amber'), {'interview_zh': ('classic', 'boxed'), 'podcast_en': ('cinematic', 'boxed')}),
'bold': (('lemon', 'coral'), {'interview_zh': ('boxed', 'classic'), 'podcast_en': ('pop', 'boxed')}),
'warm': (('amber', 'rose'), {'interview_zh': ('spotlight', 'classic'), 'podcast_en': ('boxed', 'cinematic')}),
'playful': (('rose', 'lemon', 'mint'), {'interview_zh': ('boxed', 'spotlight'), 'podcast_en': ('pop', 'boxed')}),
}
def choose_look(template: str, mood: object, seed: str) -> dict[str, str | None]:
"""{'mood', 'palette', 'style'} for this content; no valid mood keeps the golden default look."""
import random
if mood not in MOOD_LOOKS:
return {'mood': None, 'palette': None, 'style': None}
palettes, styles = MOOD_LOOKS[mood]
pick = random.Random(f'{seed}:{mood}')
return {'mood': mood, 'palette': pick.choice(palettes), 'style': pick.choice(styles[template])}
def _seed(lines: list[dict[str, Any]]) -> str:
return f"{lines[0]['start']:.1f}-{lines[-1]['end']:.1f}" if lines else ''
def source_language(texts: list[str]) -> str:
joined = ''.join(texts)
@@ -194,5 +220,6 @@ def _validated(result, lines, base, translate, burned, known_names, draft):
highlights.append({'at': cue['start'], 'text': word})
for cue in cues:
cue.pop('lines', None)
return {**base, 'title_lines': titles, 'title_accent_line': min(accent, max(0, len(titles) - 1)),
look = choose_look(base['template'], result.get('mood'), _seed(lines))
return {**base, **look, 'title_lines': titles, 'title_accent_line': min(accent, max(0, len(titles) - 1)),
'cues': cues, 'speakers': speakers[:8], 'tags': tags, 'highlights': highlights[:40]}
+39 -11
View File
@@ -2,7 +2,8 @@
Each scene is encoded on its own (render.py), so every overlay is expressed on the output
timeline and clipped to the scene: the scene's ASS only carries events visible inside it,
with times relative to the scene start. Colours follow DESIGN.md (dark theme).
with times relative to the scene start. Colours come from the content palette (mood), not the
product UI; the default palette matches DESIGN.md's accent.
"""
from __future__ import annotations
@@ -15,8 +16,29 @@ FONT_DIR = Path(__file__).resolve().parents[2] / 'assets' / 'fonts'
FONT = 'Noto Sans SC' # bundled (OFL); covers CJK and Latin on every platform
W, H = 1080, 1920
WIN_Y, WIN_H = 560, 810
BG, ACCENT_HEX = '0x1A1A19', '0x5A8BFF'
ACCENT, WHITE, SUB, INK, INK_SOFT = '&H00FF8B5A', '&H00E6EAEC', '&H009BA2A6', '&H00191A1A', '&H26191A1A'
WHITE, SUB, INK, INK_SOFT = '&H00E6EAEC', '&H009BA2A6', '&H00191A1A', '&H26191A1A'
# Content palettes (accent, canvas). Content follows its mood, not the product UI: azure is the
# golden default; the canvas stays a near-black tinted toward the accent.
PALETTES = {
'azure': ('5A8BFF', '1A1A19'), 'amber': ('F2B544', '1B1712'), 'coral': ('FF6F59', '1B1514'),
'mint': ('46D3A6', '111917'), 'lemon': ('F4DC3C', '141413'), 'rose': ('FF7FA9', '1B1418'),
'lilac': ('B69CFF', '17151C'),
}
def _ass(rgb: str) -> str:
return f'&H00{rgb[4:6]}{rgb[2:4]}{rgb[0:2]}'.upper()
def colours(palette: str | None) -> dict[str, str]:
"""ASS and ffmpeg colours for a palette; light accents get dark text on filled pills."""
accent, canvas = PALETTES.get(palette or 'azure', PALETTES['azure'])
r, g, b = (int(accent[i:i + 2], 16) for i in (0, 2, 4))
light = 0.2126 * r + 0.7152 * g + 0.0722 * b > 150
return {'accent': _ass(accent), 'on_accent': INK if light else WHITE, 'accent_hex': f'0x{accent}', 'bg': f'0x{canvas}'}
BG, ACCENT_HEX, ACCENT = colours(None)['bg'], colours(None)['accent_hex'], colours(None)['accent']
NAMEPLATE_SEC = 3.2
@@ -58,16 +80,18 @@ def _limit(font_size: int) -> float:
return 920 / font_size
def _emphasize(text: str, phrases: list[str]) -> str:
def _emphasize(text: str, phrases: list[str], accent: str = ACCENT) -> str:
"""Colour phrases (already escaped) in the accent blue inside an escaped caption."""
for phrase in sorted({p for p in phrases if p}, key=len, reverse=True):
safe = _esc(phrase)
if safe in text:
text = text.replace(safe, f'{{\\c{ACCENT}}}{safe}{{\\c{WHITE}}}', 1)
text = text.replace(safe, f'{{\\c{accent}}}{safe}{{\\c{WHITE}}}', 1)
return text
def _header() -> str:
def _header(look: dict[str, str] | None = None) -> str:
look = look or colours(None)
ACCENT, ON_ACCENT = look['accent'], look['on_accent'] # noqa: N806 - keeps the style table readable
return f"""[Script Info]
ScriptType: v4.00+
PlayResX: {W}
@@ -88,7 +112,7 @@ Style: PlateName,{FONT},46,{WHITE},{WHITE},{INK_SOFT},{INK_SOFT},1,0,0,0,100,100
Style: PlateRole,{FONT},32,{SUB},{SUB},{INK_SOFT},{INK_SOFT},0,0,0,0,100,100,0,0,3,12,0,7,0,0,0,1
Style: PlateBar,{FONT},10,{ACCENT},{ACCENT},{ACCENT},{ACCENT},0,0,0,0,100,100,0,0,1,0,0,7,0,0,0,1
Style: CaptionBox,{FONT},58,{WHITE},{WHITE},{INK_SOFT},{INK_SOFT},1,0,0,0,100,100,1,0,3,14,0,2,48,48,0,1
Style: TagPill,{FONT},48,{WHITE},{WHITE},{ACCENT},{ACCENT},1,0,0,0,100,100,2,0,3,14,0,5,40,40,0,1
Style: TagPill,{FONT},48,{ON_ACCENT},{ON_ACCENT},{ACCENT},{ACCENT},1,0,0,0,100,100,2,0,3,14,0,5,40,40,0,1
Style: Cine,{FONT},64,{WHITE},{WHITE},&H00000000,&H00000000,1,0,0,0,100,100,2,0,1,0,0,2,60,60,0,1
Style: CineGlow,{FONT},64,{ACCENT},{ACCENT},{ACCENT},&H00000000,1,0,0,0,100,100,2,0,1,3,0,2,60,60,0,1
@@ -162,6 +186,8 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
out = _Scene(offset, length)
interview = packaging.template == 'interview_zh'
style = packaging.style or ('classic' if interview else 'pop')
look = colours(packaging.palette)
accent = look['accent']
tag_texts = [t.text for t in packaging.tags] if packaging.tags_enabled else []
lines = packaging.title_lines
@@ -191,7 +217,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
for s, e, text, original in timed_screens(cue.text, start, end, _limit(size), cue.original, _limit(36)):
body = _esc_lines(text)
if style == 'spotlight':
body = _emphasize(body, tag_texts)
body = _emphasize(body, tag_texts, accent)
motion = f'\\move(540,{y + 24},540,{y},0,180)\\fad(120,60)'
else:
motion = f'\\pos(540,{y})\\fad(60,60)' if style == 'classic' else f'\\pos(540,{y})\\fad(90,90)'
@@ -204,7 +230,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
for s, e, text, _ in timed_screens(cue.text, start, end, _limit(58) / 0.95):
body = _esc_lines(text)
for word in highlights:
body = re.sub(rf'(?i)\b({re.escape(_esc(word))})\b', lambda m: f'{{\\c{ACCENT}}}{m.group(1)}{{\\c{WHITE}}}', body, count=1)
body = re.sub(rf'(?i)\b({re.escape(_esc(word))})\b', lambda m: f'{{\\c{accent}}}{m.group(1)}{{\\c{WHITE}}}', body, count=1)
out.add(2, s, e, 'CaptionBox', f'{{\\an2\\pos(540,1420)\\fad(90,90)}}{body}')
continue
words = _words(cue.text, start, end, None)
@@ -224,7 +250,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
parts = []
for k, (_, _, token) in enumerate(chunk):
active = k == j or token.strip('.,?!;:').lower() in highlights
parts.append(f'{{\\c{ACCENT}}}{_esc(token)}{{\\c{WHITE}}}' if active else _esc(token))
parts.append(f'{{\\c{accent}}}{_esc(token)}{{\\c{WHITE}}}' if active else _esc(token))
stop = chunk[j + 1][0] if j + 1 < len(chunk) else chunk_end
pop = '{\\fscx86\\fscy86\\t(0,110,\\fscx100\\fscy100)}' if j == 0 else ''
out.add(2, w0, stop, 'Words', f'{{\\an2\\pos(540,1400)}}{pop}' + ' '.join(parts))
@@ -245,7 +271,7 @@ def scene_ass(packaging: Packaging, scenes: list[Scene], index: int, word_timing
continue
motion = '\\fscx50\\fscy50\\t(0,140,\\fscx112\\fscy112)\\t(140,260,\\fscx100\\fscy100)\\fad(0,240)'
out.add(4, at, at + 1.8, 'Tag', f'{{\\an5\\pos(540,{WIN_Y + WIN_H - 120}){motion}}}{_esc(tag.text)}', carry=False)
return _header() + '\n'.join(out.lines) + '\n'
return _header(look) + '\n'.join(out.lines) + '\n'
def scene_video_graph(draft: Draft, scene: Scene, index: int, ass_path: Path, w: int, h: int) -> tuple[str, str]:
@@ -261,6 +287,8 @@ def scene_video_graph(draft: Draft, scene: Scene, index: int, ass_path: Path, w:
total = rows[-1][2] + scene_duration(draft.scenes[-1])
offset, length = rows[index][2], scene_duration(scene)
ass = f"ass='{_escape_filter_path(ass_path)}':fontsdir='{_escape_filter_path(FONT_DIR)}'"
look = colours(draft.packaging.palette if draft.packaging else None)
BG, ACCENT_HEX = look['bg'], look['accent_hex'] # noqa: N806
if draft.layout == 'window':
if scene.crop_track or scene.crop_x is not None:
x = crop_expression(scene, draft.crop_x)
+15
View File
@@ -125,3 +125,18 @@ def test_paragraph_sized_segments_are_rejected_for_translation():
def test_source_language_detection():
assert packaging.source_language(['这是一个中文字幕,内容比较长一些']) == 'zh'
assert packaging.source_language(['This is an English subtitle line with words']) == 'en'
def test_the_content_mood_picks_a_matching_look_that_is_stable_per_clip():
result = packaging.build_packaging(DRAFT, LINES, platform_strategy('douyin'), call=lambda *_: good_response(mood='bold'))
palettes, styles = packaging.MOOD_LOOKS['bold']
assert result['mood'] == 'bold' and result['palette'] in palettes and result['style'] in styles['interview_zh']
again = packaging.build_packaging(DRAFT, LINES, platform_strategy('xiaohongshu'), call=lambda *_: good_response(mood='bold'))
assert (again['palette'], again['style']) == (result['palette'], result['style']) # same clip, same look
looks = {tuple(packaging.choose_look('podcast_en', 'calm', f'{n}.0-{n + 60}.0').values()) for n in range(40)}
assert len(looks) > 2 # different clips vary within the mood
def test_an_unknown_mood_keeps_the_golden_default_look():
result = packaging.build_packaging(DRAFT, LINES, platform_strategy('douyin'), call=lambda *_: good_response(mood='angry'))
assert (result['mood'], result['palette'], result['style']) == (None, None, None)
+12
View File
@@ -124,3 +124,15 @@ def test_interview_window_renders_title_canvas_and_window(tmp_path):
assert max(max(px) for px in title_region.getdata()) > 180 # bright title text drawn on the dark canvas
window_row = [image.getpixel((x, pr.WIN_Y + 200)) for x in range(0, 1080, 60)]
assert len(set(window_row)) > 3 # the source picture fills the 4:3 window
def test_the_palette_colours_accents_and_keeps_pill_text_readable():
lemon = pr.scene_ass(_packaging(palette='lemon', style='boxed'), SCENES, 1)
assert '&H003CDCF4' in lemon # lemon F4DC3C in ASS BGR order
pill = next(line for line in lemon.splitlines() if line.startswith('Style: TagPill'))
assert pill.split(',')[3] == pr.INK # dark text on a light accent
azure = pr.scene_ass(_packaging(), SCENES, 1)
assert '&H00FF8B5A' in azure and '&H003CDCF4' not in azure
draft = Draft(id='d', title='T', scenes=SCENES[:1], aspect='portrait', layout='window', packaging=_packaging(palette='mint'))
graph, _ = pr.scene_video_graph(draft, SCENES[0], 0, pr.FONT_DIR / 'x.ass', 1080, 1920)
assert 'color=c=0x111917' in graph and 'color=c=0x46D3A6' in graph
+2
View File
@@ -47,6 +47,8 @@ export interface Packaging {
highlights: { at: number; text: string }[]
burned_captions: boolean; fallback: boolean
style?: 'classic' | 'boxed' | 'spotlight' | 'pop' | 'cinematic' | null
mood?: 'calm' | 'serious' | 'bold' | 'warm' | 'playful' | null
palette?: 'azure' | 'amber' | 'coral' | 'mint' | 'lemon' | 'rose' | 'lilac' | null
}
export interface Draft {
id: string; title: string; hook: string; scenes: Scene[]; language: Language