fix(video-export): real audio durations + burned-in subtitles (PR #937 review)

Two export-fidelity issues found in wyuc's deeper E2E:

A. Narration was scheduled from estimated durations, cutting audio off mid-
   sentence and advancing the timeline early. The scheduler trusted
   AudioFileRecord.duration (recorded only since #861), so the many existing
   classrooms without it fell back to text-length estimates — measured 4.35s
   average / 10.23s max underestimate across 47 clips. timeline-deps now probes
   the real duration from each narration blob via an off-document <audio>
   (symmetric to the existing video probe), preferring it over the stored
   duration, then the estimate only when no audio asset exists. Everything
   downstream (narration starts, scene/total duration, subtitle cues) re-derives
   from the corrected value in the pure compiler — no compiler change needed.

B. The final MP4 had no subtitles (only H.264+AAC), and the ZIP's SRT/VTT used
   the same estimated boundaries. The emitter now renders a burned-in subtitle
   overlay: one caption box + a hidden div per cue, revealed/hidden by the paused
   GSAP timeline at each cue's start/end (corrected timings from A), so Chromium's
   frame capture bakes them in. The producer has no subtitle track of its own, so
   burn-in is the v1 approach.

Verified: emitter unit tests + snapshot updated (subtitle overlay + toggle
statements, escaped text, hidden-by-default); 82 video-export tests pass incl.
the determinism red-line proxy. Rendered a synthetic subtitle project through the
container and confirmed by pixel analysis that captions appear only within their
cue window (2429 near-white px in the caption band at t=1.5s vs 0 at t=0.05s).
tsc / lint / prettier / i18n pass.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
cosarah
2026-07-16 14:42:45 +08:00
co-authored by Claude Opus 4.8
parent f717fa8ba6
commit 6a585c2921
4 changed files with 119 additions and 2 deletions
+50 -2
View File
@@ -73,6 +73,35 @@ function probeVideoDurationMs(blob: Blob): Promise<number | null> {
});
}
/**
* Probe a narration audio blob's natural duration (ms) via an off-document
* `<audio>`. Symmetric to {@link probeVideoDurationMs}. Resolves `null` when
* metadata never loads.
*
* This is the source of truth for narration timing: the TTS-time
* `AudioFileRecord.duration` was only recorded for classrooms generated after
* #861, so most existing courses have it unset and would otherwise fall back to
* text-length *estimates* — which run short and truncate the narration / advance
* the timeline early. Reading the real bytes makes the scheduled dwell match the
* clip for every classroom that actually has audio.
*/
function probeAudioDurationMs(blob: Blob): Promise<number | null> {
return new Promise((resolve) => {
const url = URL.createObjectURL(blob);
const audio = document.createElement('audio');
audio.preload = 'metadata';
const done = (value: number | null) => {
URL.revokeObjectURL(url);
audio.removeAttribute('src');
resolve(value);
};
audio.onloadedmetadata = () =>
done(Number.isFinite(audio.duration) ? Math.round(audio.duration * 1000) : null);
audio.onerror = () => done(null);
audio.src = url;
});
}
/**
* Load the Dexie-backed records for a classroom and build the synchronous
* compiler deps over them. Audio durations come from the stored records
@@ -100,6 +129,18 @@ export async function createVideoTimelineDeps(input: {
if (record) audioById.set(audioId, record);
}
// Probe real audio durations from the local blobs up front, so the compiler's
// sync `audioDurationMs` is an accurate table lookup rather than a text-length
// estimate. Only local blobs can be probed here; an ossKey-only (evicted)
// record has no bytes to read, so it falls back to the stored duration (or
// estimate) — the same asymmetry the video probe accepts.
const audioDurationMsByAudioId = new Map<string, number>();
for (const [audioId, record] of audioById) {
if (record.blob.size === 0) continue;
const ms = await probeAudioDurationMs(record.blob);
if (ms !== null) audioDurationMsByAudioId.set(audioId, ms);
}
// Media: all generated media for this stage, keyed by elementId.
const mediaRecords = await db.mediaFiles.where('stageId').equals(stage.id).toArray();
const mediaByElementId = new Map<string, MediaFileRecord>();
@@ -121,7 +162,12 @@ export async function createVideoTimelineDeps(input: {
const timing: TimingProbe = {
audioDurationMs(action: SpeechAction): number | null {
const record = action.audioId ? audioById.get(action.audioId) : undefined;
if (!action.audioId) return null;
// Prefer the real probed duration; fall back to the stored TTS duration
// (older records), then null (→ compiler estimates from text length).
const probed = audioDurationMsByAudioId.get(action.audioId);
if (probed != null) return probed;
const record = audioById.get(action.audioId);
if (!record || typeof record.duration !== 'number') return null;
return Math.round(record.duration * 1000);
},
@@ -135,11 +181,13 @@ export async function createVideoTimelineDeps(input: {
if (!action.audioId) return null;
const record = audioById.get(action.audioId);
if (!record) return { id: action.audioId, present: false };
const probed = audioDurationMsByAudioId.get(action.audioId);
return {
id: action.audioId,
mimeType: record.blob.type || undefined,
format: record.format || 'mp3',
durationMs: typeof record.duration === 'number' ? record.duration * 1000 : undefined,
durationMs:
probed ?? (typeof record.duration === 'number' ? record.duration * 1000 : undefined),
// Present when locally held or fetchable from its CDN ossKey at collect time.
present: record.blob.size > 0 || !!record.ossKey,
};
@@ -124,6 +124,53 @@ function renderNarration(scene: VideoTimelineScene): string[] {
});
}
/**
* Subtitle overlay: one absolutely-positioned caption box at the bottom of the
* stage, plus one hidden `<div>` per cue. The captions are *burned in* — the
* paused GSAP timeline reveals each cue at its `startMs` and hides it at its
* `endMs`, so Chromium's frame capture bakes them into the video (the producer
* has no subtitle track of its own). Cue timings are the IR's, which now derive
* from real audio durations, so they stay aligned with the narration.
*
* Returns the overlay HTML and the `tl.set` statements that toggle visibility.
*/
function renderSubtitles(
ir: VideoTimeline,
height: number,
): { html: string; statements: string[] } {
const cues = ir.subtitles.filter((c) => c.text.trim());
if (cues.length === 0) return { html: '', statements: [] };
// Scale caption type to the render height so it reads at any resolution.
const fontPx = Math.max(16, Math.round(height * 0.033));
const padV = Math.round(fontPx * 0.35);
const padH = Math.round(fontPx * 0.7);
const bottom = Math.round(height * 0.055);
const cueDivs = cues
.map(
(c, i) =>
` <div id="subtitle-cue-${i}" style="display:inline-block;visibility:hidden;max-width:80%;margin:0 auto;padding:${padV}px ${padH}px;background:rgba(0,0,0,0.66);color:#fff;font-size:${fontPx}px;line-height:1.3;border-radius:${padV}px;white-space:pre-wrap;text-shadow:0 1px 2px rgba(0,0,0,0.9)">${escapeHtml(c.text)}</div>`,
)
.join('\n');
const html = [
`<div id="subtitles" style="position:absolute;left:0;right:0;bottom:${bottom}px;z-index:50;text-align:center;pointer-events:none;font-family:system-ui,sans-serif">`,
cueDivs,
`</div>`,
].join('\n');
// Toggle each cue with `visibility` (via GSAP autoAlpha would also flip
// opacity; visibility keeps it crisp and avoids cross-fade overlap).
const statements: string[] = [];
for (let i = 0; i < cues.length; i++) {
const c = cues[i];
statements.push(`tl.set('#subtitle-cue-${i}',{visibility:'visible'},${sec(c.startMs)});`);
statements.push(`tl.set('#subtitle-cue-${i}',{visibility:'hidden'},${sec(c.endMs)});`);
}
return { html, statements };
}
function renderReadme(project: {
compositionId: string;
width: number;
@@ -198,6 +245,10 @@ export function emitHyperframes(
}
}
// Burned-in subtitle overlay, driven by the same paused timeline.
const subtitles = renderSubtitles(ir, height);
statements.push(...subtitles.statements);
// Extend the timeline to the full composition length even if the last tween
// ends earlier, so clips (esp. video/audio) are not cut short.
statements.push(`tl.set({}, {}, ${totalSec});`);
@@ -218,6 +269,7 @@ export function emitHyperframes(
<div id="${compositionId}" data-composition-id="${compositionId}" data-start="0" data-duration="${totalSec}" data-width="${width}" data-height="${height}" style="position:relative;width:${width}px;height:${height}px;overflow:hidden;background:#000">
${sceneHtml.filter(Boolean).join('\n')}
${effectHtml.join('\n')}
${subtitles.html}
</div>
<script src="${escapeHtml(gsapVendorPath)}"></script>
<script>
@@ -40,6 +40,9 @@ exports[`emitHyperframes > matches the HTML snapshot 1`] = `
<div style="width:10px;height:10px;border-radius:9999px;background-color:#00ff88;box-shadow:0 0 8px 2px #00ff8860"></div>
</div>
</div>
<div id="subtitles" style="position:absolute;left:0;right:0;bottom:59px;z-index:50;text-align:center;pointer-events:none;font-family:system-ui,sans-serif">
<div id="subtitle-cue-0" style="display:inline-block;visibility:hidden;max-width:80%;margin:0 auto;padding:13px 25px;background:rgba(0,0,0,0.66);color:#fff;font-size:36px;line-height:1.3;border-radius:13px;white-space:pre-wrap;text-shadow:0 1px 2px rgba(0,0,0,0.9)">Welcome to the lesson</div>
</div>
</div>
<script src="assets/vendor/gsap.min.js"></script>
<script>
@@ -86,6 +89,8 @@ tl.fromTo('#fx-0-2-ring',{scale:1,opacity:0.6},{scale:2.8,opacity:0,duration:1.5
tl.to('#fx-0-2',{autoAlpha:0,x:-1056,y:-591.6,duration:0.25,ease:EASE_IN},4.75);
tl.set('#fx-0-2',{autoAlpha:0},5);
tl.set('#fx-0-2-ring',{opacity:0},5);
tl.set('#subtitle-cue-0',{visibility:'visible'},0);
tl.set('#subtitle-cue-0',{visibility:'hidden'},2);
tl.set({}, {}, 7);
window.__timelines = window.__timelines || {};
window.__timelines["openmaic"] = tl;
@@ -86,6 +86,18 @@ describe('emitHyperframes', () => {
expect(html).toContain('#00ff88'); // authored laser color survives into the DOM
});
it('burns in a subtitle overlay driven by the timeline', () => {
// A caption container plus one cue div per non-empty speech action.
expect(html).toContain('id="subtitles"');
expect(html).toContain('id="subtitle-cue-0"');
// Cues start hidden and are toggled visible/hidden by the paused timeline.
expect(html).toMatch(/id="subtitle-cue-0"[^>]*visibility:hidden/);
expect(html).toMatch(/tl\.set\('#subtitle-cue-0',\{visibility:'visible'\},[\d.]+\);/);
expect(html).toMatch(/tl\.set\('#subtitle-cue-0',\{visibility:'hidden'\},[\d.]+\);/);
// Narration text is rendered into the caption.
expect(html).toContain('Welcome to the lesson');
});
it('references vendored GSAP, never a CDN', () => {
expect(html).toContain('<script src="assets/vendor/gsap.min.js"></script>');
expect(project.gsapVendorPath).toBe('assets/vendor/gsap.min.js');