mirror of
https://github.com/zhouxiaoka/autoclip.git
synced 2026-10-02 02:34:34 +08:00
feat(cover): the vision model picks the guest's frame; AI covers upgrade the design
- Cover frames: candidates are the sharpest face frame of each long shot; the vision model picks the guest's best one (never the host). Without a vision model the sharpest face is used and no nameplate is drawn, since colour or screen-time rules cannot tell a host's camera from a guest's. Tight crops are lightly sharpened after upscaling. - AI covers: after the designed cover, an image model (when set up) generates an editorial magazine-style cover from that frame, in the platform's ratio (qwen-image sizes now cover 3:4/1:1/4:3) and the clip's palette, in the audience's language; the vision model checks the headline and a wrong one is retried once, else the designed cover stays. Runs on its own pool so renders never wait; 'AI 重新设计封面' uses the same path. - Image requests wait up to 180 s (high-quality models are slow). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -312,6 +312,18 @@ def output_variant_cover(project_id: str, variant_id: str, db: Session = Depends
|
||||
return FileResponse(path, media_type='image/jpeg', headers={'Cache-Control': 'no-store'})
|
||||
|
||||
|
||||
@router.post('/{project_id}/output-variants/{variant_id}/cover/ai')
|
||||
def redesign_output_variant_cover(project_id: str, variant_id: str, db: Session = Depends(get_db)):
|
||||
from backend.services import cover
|
||||
project_or_404(project_id, db)
|
||||
_variant_job(project_id, variant_id)
|
||||
cfg = cover.load_config()
|
||||
if not (cfg.enabled and cfg.configured):
|
||||
raise HTTPException(409, '请先在设置里开启 AI 封面并选择图像模型')
|
||||
call(jobs.request_ai_cover, project_id, variant_id)
|
||||
return {'ok': True}
|
||||
|
||||
|
||||
@router.put('/{project_id}/output-variants/{variant_id}/post')
|
||||
def update_output_variant_post(project_id: str, variant_id: str, body: PostCopy, db: Session = Depends(get_db)):
|
||||
project_or_404(project_id, db)
|
||||
|
||||
@@ -25,6 +25,7 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
OPENAI_ROOT = "https://api.openai.com/v1"
|
||||
DASHSCOPE_ROOT = "https://dashscope.aliyuncs.com/api/v1"
|
||||
IMAGE_TIMEOUT = 180 # high-quality image models (gpt-image, qwen-image) often take 60–120 s per cover
|
||||
SEEDREAM_ROOT = "https://ark.cn-beijing.volces.com/api/v3"
|
||||
SEEDREAM_DEFAULT_MODEL = "doubao-seedream-5-0-260128"
|
||||
SEEDREAM_DEFAULT_OCR = "doubao-1.5-vision-pro-32k"
|
||||
@@ -182,7 +183,7 @@ def _post_json(
|
||||
*,
|
||||
seedream: bool = False,
|
||||
) -> Any:
|
||||
resp = session.post(url, headers={**headers, "Content-Type": "application/json"}, json=body, timeout=90)
|
||||
resp = session.post(url, headers={**headers, "Content-Type": "application/json"}, json=body, timeout=IMAGE_TIMEOUT)
|
||||
if int(getattr(resp, "status_code", 200) or 200) == 400:
|
||||
message = _error_message(resp).lower()
|
||||
retry = dict(body)
|
||||
@@ -207,7 +208,7 @@ def _post_json(
|
||||
retry.pop("watermark", None)
|
||||
changed = True
|
||||
if changed:
|
||||
resp = session.post(url, headers={**headers, "Content-Type": "application/json"}, json=retry, timeout=90)
|
||||
resp = session.post(url, headers={**headers, "Content-Type": "application/json"}, json=retry, timeout=IMAGE_TIMEOUT)
|
||||
return resp
|
||||
|
||||
|
||||
@@ -248,7 +249,7 @@ def generate_openai(
|
||||
headers=headers,
|
||||
data={"model": model, "prompt": request.prompt, "size": size, "n": "1"},
|
||||
files={"image": ("frame.jpg", request.reference, "image/jpeg")},
|
||||
timeout=90,
|
||||
timeout=IMAGE_TIMEOUT,
|
||||
)
|
||||
try:
|
||||
data = _raise_for_status(resp, edit=True)
|
||||
@@ -329,13 +330,14 @@ def generate_dashscope(
|
||||
if modern_qwen:
|
||||
headers.pop('X-DashScope-Async', None)
|
||||
route = 'image-generation' if modern_wan else 'multimodal-generation'
|
||||
size = '928*1664' if request.height > request.width else '1664*928'
|
||||
sizes = {'928*1664': 928 / 1664, '1140*1472': 1140 / 1472, '1328*1328': 1.0, '1472*1140': 1472 / 1140, '1664*928': 1664 / 928}
|
||||
size = min(sizes, key=lambda key: abs(sizes[key] - request.width / max(1, request.height)))
|
||||
if model.startswith('wan2.6-t2i'):
|
||||
size = '960*1696' if request.height > request.width else '1696*960'
|
||||
resp = http.post(f'{root}/services/aigc/{route}/generation',
|
||||
headers={**headers, 'Content-Type': 'application/json'},
|
||||
json={'model': model, 'input': {'messages': [{'role': 'user', 'content': content}]},
|
||||
'parameters': {'size': size, 'n': 1}}, timeout=90)
|
||||
'parameters': {'size': size, 'n': 1}}, timeout=IMAGE_TIMEOUT)
|
||||
else:
|
||||
return _generate_legacy_dashscope(api_key=api_key, base_url=base_url, request=request, session=http,
|
||||
cancel=cancel, poll_interval=poll_interval)
|
||||
|
||||
@@ -74,6 +74,186 @@ def cover_speaker(packaging: dict[str, Any]) -> tuple[str, str] | None:
|
||||
return None
|
||||
|
||||
|
||||
QUESTION_END = ('?', '?', '吗', '呢', '么')
|
||||
CANDIDATE_SHOTS = 3
|
||||
|
||||
|
||||
def _question_times(rows: list[dict[str, Any]]) -> list[tuple[float, float]]:
|
||||
"""Source intervals of rows that end a question: in an interview, those are the host speaking."""
|
||||
return [(row['start'], row['end']) for row in rows if row['text'].rstrip(' "”」』').endswith(QUESTION_END)]
|
||||
|
||||
|
||||
def _face_and_sharpness(frame_jpeg: bytes) -> tuple[float | None, float, bool]:
|
||||
"""(face centre x, sharpness of the face area or frame centre, face found). Works without OpenCV."""
|
||||
from PIL import Image, ImageFilter, ImageStat
|
||||
image = Image.open(io.BytesIO(frame_jpeg)).convert('L')
|
||||
box, centre, found = None, None, False
|
||||
try:
|
||||
from backend.services.studio import framing
|
||||
if framing.is_installed():
|
||||
framing.ensure_on_path()
|
||||
import cv2
|
||||
import numpy as np
|
||||
pixels = cv2.imdecode(np.frombuffer(frame_jpeg, dtype=np.uint8), cv2.IMREAD_COLOR)
|
||||
height, width = pixels.shape[:2]
|
||||
_, faces = framing._detector(cv2, width, height).detect(pixels)
|
||||
faces = [f for f in (faces if faces is not None else []) if f[2] >= framing.MIN_FACE * width]
|
||||
if faces:
|
||||
x, y, w, h = (float(v) for v in max(faces, key=lambda f: f[2] * f[3])[:4])
|
||||
box, centre, found = (int(x), int(y), int(x + w), int(y + h)), (x + w / 2) / width, True
|
||||
except Exception: # noqa: BLE001 - no detector: judge the frame centre
|
||||
box = None
|
||||
if box is None:
|
||||
width, height = image.size
|
||||
box = (width // 3, height // 6, width * 2 // 3, height * 2 // 3)
|
||||
edges = image.crop(box).filter(ImageFilter.FIND_EDGES)
|
||||
return centre, ImageStat.Stat(edges).var[0], found
|
||||
|
||||
|
||||
def _signature(video, at: float) -> list[float] | None:
|
||||
"""Colour profile of a frame (coarse RGB histogram): camera angles of an interview differ clearly."""
|
||||
from PIL import Image
|
||||
from backend.services.cover import extract_frame_jpeg
|
||||
try:
|
||||
image = Image.open(io.BytesIO(extract_frame_jpeg(video, at_sec=at, max_width=320))).convert('RGB').resize((64, 36))
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
histogram = image.quantize(colors=64, method=Image.Quantize.FASTOCTREE, kmeans=0).convert('RGB').histogram()
|
||||
bins = [sum(histogram[c * 256 + i * 32:c * 256 + (i + 1) * 32]) for c in range(3) for i in range(8)]
|
||||
total = sum(bins) or 1
|
||||
return [b / total for b in bins]
|
||||
|
||||
|
||||
def _similarity(a: list[float], b: list[float]) -> float:
|
||||
return sum(min(x, y) for x, y in zip(a, b))
|
||||
|
||||
|
||||
GUEST_SAMPLES = 24
|
||||
SAME_LOOK = 0.75
|
||||
_guest_looks: dict[str, list[float] | None] = {}
|
||||
|
||||
|
||||
def guest_look(video) -> list[float] | None:
|
||||
"""Colour profile of the camera the video shows most: in an interview, the guest's.
|
||||
|
||||
Frames across the whole video are grouped by look; the largest group wins. Cached per file.
|
||||
"""
|
||||
key = f'{video}:{Path(video).stat().st_size}'
|
||||
if key not in _guest_looks:
|
||||
from backend.services.publish_export import _probe
|
||||
duration = float(_probe(video).get('duration') or 0)
|
||||
looks = [sig for sig in (_signature(video, duration * (i + 0.5) / GUEST_SAMPLES) for i in range(GUEST_SAMPLES)) if sig]
|
||||
groups: list[list[list[float]]] = []
|
||||
for sig in looks:
|
||||
home = next((group for group in groups if _similarity(sig, group[0]) >= SAME_LOOK), None)
|
||||
(home.append(sig) if home else groups.append([sig]))
|
||||
big = max(groups, key=len) if groups else None
|
||||
# A single dominant look only: when no camera holds the screen, there is no guest look.
|
||||
_guest_looks[key] = big[0] if big and len(big) >= max(4, len(looks) * 0.35) else None
|
||||
return _guest_looks[key]
|
||||
|
||||
|
||||
def _guest_shots(video, start: float, shots) -> tuple[list, bool]:
|
||||
"""Shots that look like the guest's camera, best match first; ([], False) without a guest look."""
|
||||
look = guest_look(video)
|
||||
if look is None:
|
||||
return [], False
|
||||
scored = [(shot, _similarity(sig, look)) for shot in shots
|
||||
if (sig := _signature(video, start + (shot[0] + shot[1]) / 2)) is not None]
|
||||
matches = [shot for shot, score in sorted(scored, key=lambda item: item[1], reverse=True) if score >= SAME_LOOK]
|
||||
return matches, bool(matches)
|
||||
|
||||
|
||||
VISION_CANDIDATES = 6
|
||||
CHOOSE_PROMPT = (
|
||||
'这是一段访谈视频里的 {n} 张候选画面,编号 1–{n}。{who}'
|
||||
'请选出最适合做短视频封面的一张:必须是受访嘉宾本人(不是主持人、不是观众),正脸或清晰侧脸,'
|
||||
'不模糊、不闭眼、表情自然有感染力。如果没有一张是嘉宾本人,best 返回 0。'
|
||||
'只返回 JSON:{{"best":编号,"guest":[是嘉宾本人的编号]}}'
|
||||
)
|
||||
|
||||
|
||||
def _thumb(frame_jpeg: bytes, width: int = 384) -> str:
|
||||
import base64
|
||||
from PIL import Image
|
||||
image = Image.open(io.BytesIO(frame_jpeg)).convert('RGB')
|
||||
image.thumbnail((width, width))
|
||||
out = io.BytesIO()
|
||||
image.save(out, format='JPEG', quality=80)
|
||||
return 'data:image/jpeg;base64,' + base64.b64encode(out.getvalue()).decode()
|
||||
|
||||
|
||||
def _candidates(video, start: float, shots, limit: int) -> list[tuple[bytes, float | None, float, bool]]:
|
||||
"""Sharpest frame of each of the longest shots: (frame, face centre, sharpness, face found)."""
|
||||
from backend.services.cover import extract_frame_jpeg
|
||||
out = []
|
||||
for shot_start, shot_end in sorted(shots, key=lambda shot: shot[1] - shot[0], reverse=True)[:limit]:
|
||||
best = None
|
||||
for share in (0.35, 0.65):
|
||||
try:
|
||||
frame = extract_frame_jpeg(video, at_sec=start + shot_start + (shot_end - shot_start) * share, max_width=1920)
|
||||
except Exception: # noqa: BLE001
|
||||
continue
|
||||
centre, sharpness, found = _face_and_sharpness(frame)
|
||||
if best is None or (found, sharpness) > (best[3], best[2]):
|
||||
best = (frame, centre, sharpness, found)
|
||||
if best:
|
||||
out.append(best)
|
||||
return out
|
||||
|
||||
|
||||
def _choose_with_vision(candidates, guest: str, host: str) -> tuple[int, bool] | None:
|
||||
"""(index, is the guest) chosen by the vision model, or None when it is not set up or fails."""
|
||||
from backend.services.studio import intelligence
|
||||
if not intelligence.ready() or not candidates:
|
||||
return None
|
||||
who = (f'受访嘉宾是 {guest}。' if guest else '') + (f'主持人是 {host}。' if host else '')
|
||||
content = [{'type': 'text', 'text': CHOOSE_PROMPT.format(n=len(candidates), who=who)}]
|
||||
for index, (frame, *_rest) in enumerate(candidates, 1):
|
||||
content += [{'type': 'text', 'text': f'画面 {index}'}, {'type': 'image_url', 'image_url': {'url': _thumb(frame)}}]
|
||||
try:
|
||||
from backend.core import llm_usage
|
||||
from backend.services.studio.vision_settings import effective
|
||||
with llm_usage.stage('cover_frame'):
|
||||
answer = intelligence.vision_call(content, {**effective(), 'quick_screening': True}) or {}
|
||||
best = int(answer.get('best') or 0)
|
||||
except Exception: # noqa: BLE001 - fall back to the local choice
|
||||
return None
|
||||
if 1 <= best <= len(candidates):
|
||||
return best - 1, True
|
||||
return None if not answer else (max(range(len(candidates)), key=lambda i: (candidates[i][3], candidates[i][2])), False)
|
||||
|
||||
|
||||
def pick_frame(video, scene: dict[str, Any], rows: list[dict[str, Any]], packaging: dict[str, Any] | None = None,
|
||||
guest_hint: str = '') -> tuple[bytes, float | None, bool]:
|
||||
"""(frame, speaker centre, guest on screen) for the cover.
|
||||
|
||||
A fixed point in the clip often caught a hand in motion or the host's camera, labelled with the
|
||||
guest's name. Candidates are the sharpest frame of each of the longest shots. The vision model
|
||||
(when set up) picks the guest's best frame; colour and screen-time rules cannot tell a host's
|
||||
camera from a guest's, so without it the sharpest face is used and no nameplate is drawn.
|
||||
"""
|
||||
from backend.services.cover import extract_frame_jpeg
|
||||
from backend.services.studio import framing
|
||||
start, length = scene['start'], scene['end'] - scene['start']
|
||||
shots = framing.split_shots(length, framing.detect_cuts(video, start, length))
|
||||
candidates = _candidates(video, start, shots, VISION_CANDIDATES)
|
||||
if not candidates:
|
||||
at, _ = frame_time(scene)
|
||||
return extract_frame_jpeg(video, at_sec=at, max_width=1920), None, False
|
||||
speakers = (packaging or {}).get('speakers') or []
|
||||
guest = next((sp['name'] for sp in speakers if not any(w in (sp.get('role') or '').lower() for w in HOST)), '') or guest_hint
|
||||
host = next((sp['name'] for sp in speakers if any(w in (sp.get('role') or '').lower() for w in HOST)), '')
|
||||
chosen = _choose_with_vision(candidates, guest, host)
|
||||
if chosen is not None:
|
||||
index, is_guest = chosen
|
||||
frame, centre, _, _ = candidates[index]
|
||||
return frame, centre, is_guest
|
||||
questions = _question_times(rows)
|
||||
frame, centre, _, _ = max(candidates, key=lambda c: (c[3], c[2]))
|
||||
return frame, centre, False
|
||||
|
||||
|
||||
def _crop(image, width: int, height: int, centre: float | None):
|
||||
from PIL import Image
|
||||
src_w, src_h = image.size
|
||||
@@ -87,7 +267,10 @@ def _crop(image, width: int, height: int, centre: float | None):
|
||||
crop_h = round(src_w / aspect)
|
||||
top = max(0, min(src_h - crop_h, round((src_h - crop_h) * 0.3)))
|
||||
box = (0, top, src_w, top + crop_h)
|
||||
return image.crop(box).resize((width, height), Image.Resampling.LANCZOS)
|
||||
from PIL import ImageFilter
|
||||
scaled = image.crop(box).resize((width, height), Image.Resampling.LANCZOS)
|
||||
upscale = width / max(1, box[2] - box[0])
|
||||
return scaled.filter(ImageFilter.UnsharpMask(radius=2, percent=70, threshold=2)) if upscale > 1.05 else scaled
|
||||
|
||||
|
||||
def _fit_size(draw, lines: list[str], max_width: int, start: int, weight: str) -> int:
|
||||
|
||||
@@ -647,6 +647,31 @@ def _design_covers(project_id, draft, job_id):
|
||||
if item['id'] in designed and item.get('cover') != 'ai':
|
||||
item['cover'] = 'design'
|
||||
store.change(project_id, mark)
|
||||
for variant in targets:
|
||||
if variant['id'] in designed:
|
||||
request_ai_cover(project_id, variant['id'])
|
||||
|
||||
|
||||
cover_executor = ThreadPoolExecutor(max_workers=3, thread_name_prefix='studio-cover')
|
||||
|
||||
|
||||
def request_ai_cover(project_id, variant_id):
|
||||
"""Upgrade a variant's designed cover with the AI image model in the background (when one is set up)."""
|
||||
cover_executor.submit(_ai_cover_job, project_id, variant_id)
|
||||
|
||||
|
||||
@_tracked('ai_cover')
|
||||
def _ai_cover_job(project_id, variant_id):
|
||||
from backend.services.studio import publish_kit
|
||||
try:
|
||||
variant = next(item for item in store.read(project_id).get('output_variants', []) if item['id'] == variant_id)
|
||||
with llm_usage.timed('ai_cover'):
|
||||
made = publish_kit.ai_cover(project_id, variant['render_job_id'], variant['strategy_id'])
|
||||
except Exception as error: # noqa: BLE001 - the designed cover stays
|
||||
logger.warning('AI cover job failed: %s', type(error).__name__)
|
||||
made = False
|
||||
if made:
|
||||
store.change(project_id, lambda data: next(item for item in data['output_variants'] if item['id'] == variant_id).update(cover='ai'))
|
||||
|
||||
|
||||
def update_post(project_id, variant_id, post):
|
||||
|
||||
@@ -7,6 +7,7 @@ precedence wherever a cover is used — publishing reads the same files.
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import zipfile
|
||||
@@ -53,15 +54,26 @@ def design_cover(project_id: str, video: Path, draft: dict[str, Any], job_id: st
|
||||
scene = draft['scenes'][0]
|
||||
info = _probe(video)
|
||||
source_w, source_h = int(info.get('width') or 1920), int(info.get('height') or 1080)
|
||||
at, position = cd.frame_time(scene)
|
||||
centre = cd.speaker_centre(position, draft.get('layout') or 'crop', source_w, source_h)
|
||||
frame = cover.extract_frame_jpeg(video, at_sec=at, max_width=1920)
|
||||
from backend.services.studio import store
|
||||
listing = (store.read(project_id).get('source_meta') or {}).get('title', '')
|
||||
frame, centre, guest_on_screen = cd.pick_frame(video, scene, _scene_rows(project_id, scene), draft.get('packaging') or {},
|
||||
guest_hint=f'视频标题「{listing}」里的主角' if listing else '')
|
||||
if centre is None:
|
||||
_, position = cd.frame_time(scene)
|
||||
centre = cd.speaker_centre(position, draft.get('layout') or 'crop', source_w, source_h)
|
||||
packaging = draft.get('packaging') or {}
|
||||
lines, accent = cd.title_lines_for(draft, post_title or draft.get('title', ''), strategy_id)
|
||||
width, height = cd.size_for(strategy_id, source_w, source_h)
|
||||
# Name the guest only when the chosen frame is the guest's shot, never on a host frame.
|
||||
speaker = cd.cover_speaker(packaging) if guest_on_screen else None
|
||||
data = cd.design(frame, width=width, height=height, title_lines=lines, accent_line=accent,
|
||||
palette=packaging.get('palette'), speaker=cd.cover_speaker(packaging), crop_centre=centre)
|
||||
palette=packaging.get('palette'), speaker=speaker, crop_centre=centre)
|
||||
kit_cover_path(project_id, job_id, strategy_id).write_bytes(data)
|
||||
# The chosen frame is the AI cover's reference (same person, same moment).
|
||||
frame_path(project_id, job_id, strategy_id).write_bytes(frame)
|
||||
cd_meta_path(project_id, job_id, strategy_id).write_text(json.dumps({'guest': guest_on_screen, 'name': (speaker or ('', ''))[0],
|
||||
'lines': lines, 'accent': accent, 'palette': packaging.get('palette')},
|
||||
ensure_ascii=False), encoding='utf-8')
|
||||
slot = _publish_slot(strategy_id)
|
||||
meta = cover.read_cover_meta(project_id, clip_id(job_id), slot) or {}
|
||||
if meta.get('method') not in AI_METHODS and cd.size_for(strategy_id, source_w, source_h)[1] != 1440:
|
||||
@@ -71,6 +83,120 @@ def design_cover(project_id: str, video: Path, draft: dict[str, Any], job_id: st
|
||||
return True
|
||||
|
||||
|
||||
def _scene_rows(project_id: str, scene: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
try:
|
||||
from backend.pipeline.quality import to_seconds
|
||||
from backend.services.publish_export import _load_srt_entries
|
||||
rows = [{'start': to_seconds(e['start_time']), 'end': to_seconds(e['end_time']), 'text': str(e.get('text') or '')}
|
||||
for e in _load_srt_entries(project_id)]
|
||||
except Exception: # noqa: BLE001 - without rows every shot counts as the guest's
|
||||
return []
|
||||
return [row for row in rows if row['end'] > scene['start'] and row['start'] < scene['end']]
|
||||
|
||||
|
||||
def frame_path(project_id: str, job_id: str, strategy_id: str) -> Path:
|
||||
from backend.services import cover
|
||||
return cover.cover_dir(project_id, clip_id(job_id)) / f'frame-{strategy_id}.jpg'
|
||||
|
||||
|
||||
def cd_meta_path(project_id: str, job_id: str, strategy_id: str) -> Path:
|
||||
from backend.services import cover
|
||||
return cover.cover_dir(project_id, clip_id(job_id)) / f'kit-{strategy_id}.json'
|
||||
|
||||
|
||||
# One house style for every AI cover: an editorial magazine cover for interview content.
|
||||
# Layout per platform slot; colour from the clip's own palette so cover and video match.
|
||||
AI_LAYOUT = {
|
||||
'xiaohongshu': ('3:4 vertical Xiaohongshu note cover', 'title stacked in the top 40%, subject below and slightly off-centre'),
|
||||
'douyin': ('9:16 vertical Douyin video cover', 'title stacked in the upper third, subject filling the lower two thirds'),
|
||||
'tiktok': ('9:16 vertical TikTok cover', 'title stacked in the upper third, subject filling the lower two thirds'),
|
||||
'instagram_reels': ('9:16 vertical Instagram Reels cover', 'title stacked in the upper third, subject filling the lower two thirds'),
|
||||
'youtube_shorts': ('9:16 vertical YouTube Shorts cover', 'title stacked in the upper third, subject filling the lower two thirds'),
|
||||
'youtube_long': ('16:9 YouTube thumbnail', 'subject on one side (head and shoulders, large), title stacked on the other side'),
|
||||
'bilibili': ('16:10 Bilibili video cover', 'subject on one side (head and shoulders, large), title stacked on the other side'),
|
||||
}
|
||||
|
||||
|
||||
def ai_prompt(strategy_id: str, lines: list[str], accent: int, name: str, palette: str | None = None) -> str:
|
||||
from backend.services.platform_strategy import platform_strategy
|
||||
from backend.services.studio.packaging_render import PALETTES
|
||||
slot, layout = AI_LAYOUT.get(strategy_id, AI_LAYOUT['douyin'])
|
||||
colour = '#' + PALETTES.get(palette or 'azure', PALETTES['azure'])[0]
|
||||
keyword = lines[accent] if 0 <= accent < len(lines) and len(lines) > 1 else lines[-1]
|
||||
english = platform_strategy(strategy_id).audience_language == 'en'
|
||||
title = ' / '.join(lines)
|
||||
return (
|
||||
f'Design a {slot} in a premium editorial magazine-cover style for an interview clip. '
|
||||
f'Subject: the person in the reference image{" (" + name + ")" if name else ""} — keep their face, hair, age and clothing exactly; '
|
||||
f'never replace, restyle or beautify them. Cut the subject out cleanly, head and shoulders large and sharp, with a soft rim light; '
|
||||
f'background replaced by a deep, near-black gradient with subtle film grain, no clutter. Layout: {layout}. '
|
||||
f'Typography: a heavy condensed sans-serif headline, {"set in English" if english else "set in Simplified Chinese"}, '
|
||||
f'stacked in short lines exactly reading "{title}"; the words "{keyword}" sit on a solid {colour} colour block (or in {colour}), '
|
||||
f'the rest in off-white; strong hierarchy, generous margins, nothing touching the edges, the face never covered. '
|
||||
f'Optional: a small off-white name tag "{name}" near the subject. '
|
||||
f'Spell every character exactly as given; no other words, no watermark, no logo, no subtitles, no UI elements. '
|
||||
f'High contrast, crisp and clean, designed to stand out in a feed and earn the click.'
|
||||
)
|
||||
|
||||
|
||||
def _title_ok(image: bytes, lines: list[str]) -> bool | None:
|
||||
"""Whether the generated cover shows the title exactly (vision model), None when it cannot check."""
|
||||
import base64
|
||||
from backend.services.studio import intelligence
|
||||
if not intelligence.ready():
|
||||
return None
|
||||
try:
|
||||
from backend.core import llm_usage
|
||||
from backend.services.studio.vision_settings import effective
|
||||
with llm_usage.stage('cover_check'):
|
||||
answer = intelligence.vision_call([
|
||||
{'type': 'text', 'text': '读出这张封面上的标题文字,原样返回,只返回 JSON:{"text":"..."}'},
|
||||
{'type': 'image_url', 'image_url': {'url': 'data:image/jpeg;base64,' + base64.b64encode(image).decode()}}],
|
||||
{**effective(), 'quick_screening': True}) or {}
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
squash = lambda text: re.sub(r'[\s\W_]+', '', str(text)).lower() # noqa: E731
|
||||
return squash(''.join(lines)) in squash(answer.get('text', ''))
|
||||
|
||||
|
||||
def ai_cover(project_id: str, job_id: str, strategy_id: str) -> bool:
|
||||
"""Generate the AI cover for one variant from its designed cover's frame and title; True when stored."""
|
||||
from PIL import Image, ImageOps
|
||||
from backend.core.image_providers import ImageRequest, generate_image
|
||||
from backend.services import cover
|
||||
from backend.services.studio import cover_design as cd
|
||||
cfg = cover.load_config()
|
||||
frame_file, meta_file = frame_path(project_id, job_id, strategy_id), cd_meta_path(project_id, job_id, strategy_id)
|
||||
if not (cfg.enabled and cfg.configured) or not frame_file.is_file() or not meta_file.is_file():
|
||||
return False
|
||||
meta = json.loads(meta_file.read_text(encoding='utf-8'))
|
||||
width, height = cd.size_for(strategy_id)
|
||||
request = ImageRequest(prompt=ai_prompt(strategy_id, meta['lines'], meta['accent'], meta.get('name', ''), meta.get('palette')), width=width, height=height,
|
||||
reference=frame_file.read_bytes() if cfg.allow_send_frame else None, model=cfg.model)
|
||||
image = None
|
||||
for attempt in range(2):
|
||||
try:
|
||||
image = generate_image(provider=cfg.provider, api_key=cfg.api_key, base_url=cfg.base_url, request=request)
|
||||
except Exception as error: # noqa: BLE001 - the designed cover stays
|
||||
logger.warning('AI cover failed: %s', type(error).__name__)
|
||||
return False
|
||||
if _title_ok(image, meta['lines']) is not False:
|
||||
break
|
||||
request = ImageRequest(prompt=request.prompt + ' The previous attempt misspelled the headline: render it character by character exactly.', width=width, height=height,
|
||||
reference=request.reference, model=cfg.model)
|
||||
else:
|
||||
return False # the title stayed wrong twice: keep the designed cover
|
||||
fitted = ImageOps.fit(Image.open(io.BytesIO(image)).convert('RGB'), (width, height), method=Image.Resampling.LANCZOS)
|
||||
out = io.BytesIO()
|
||||
fitted.save(out, format='JPEG', quality=92)
|
||||
kit_cover_path(project_id, job_id, strategy_id).write_bytes(out.getvalue())
|
||||
if height != 1440: # the publish slot is 9:16 / 16:9; Xiaohongshu's 3:4 stays kit-only
|
||||
slot = _publish_slot(strategy_id)
|
||||
cover.cover_path(project_id, clip_id(job_id), slot).write_bytes(out.getvalue())
|
||||
cover.write_cover_meta(project_id, clip_id(job_id), slot, {'method': 'model', 'title': ' '.join(meta['lines'])})
|
||||
return True
|
||||
|
||||
|
||||
def caption_text(post: dict[str, Any]) -> str:
|
||||
tags = ' '.join(f'#{tag}' for tag in post.get('tags') or [])
|
||||
return '\n\n'.join(part for part in (post.get('title', ''), post.get('description', ''), tags) if part)
|
||||
|
||||
Reference in New Issue
Block a user