From 1d27c509238ff5d81403d9b3df6a8007f886739c Mon Sep 17 00:00:00 2001 From: Yi-111-a <> Date: Wed, 30 Sep 2026 00:43:53 +0800 Subject: [PATCH] fix(stories): keep a WebVTT cue whose identifier is a metadata word `captionBlocks` dropped every WebVTT block whose first line was `STYLE`, `REGION` or `NOTE`, so a cue whose *identifier* happened to be one of those words lost its dialogue. The remaining cues still imported, which made the omission easy to miss. WebVTT's block parser gives a timing line in position two precedence over the identifier, and `backend/services/srt_parser.py` already applies that rule, so the same file parsed correctly for dubbing. Apply it here too. Verified with the issue's reproduction: `importToText` on a file whose first cue identifier is `intro`, `STYLE` or `REGION` now returns `"Hello world.\nStill here."` for all three. Regression test fails before and passes after; 28/28 pass in `importStory.test.js`. The one frontend test that asserted the old divergent behaviour for `NOTE` is replaced by the same parametrization the backend suite uses in `test_webvtt_metadata_words_can_identify_a_cue`, so the two parsers can no longer disagree on this case. Fixes #2434 --- CHANGELOG.md | 1 + electron/src/shared/utils/importStory.js | 6 +++++- electron/src/shared/utils/importStory.test.js | 19 ++++++++++++++----- 3 files changed, 20 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1861e91c3..9c4f1d1dc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -86,6 +86,7 @@ metadata and the backend fallback mirror it. - The Twilio guide and integration directory describe the guided setup and in-app integration pages (#2304) ### Fixed +- Stories caption import keeps a WebVTT cue whose identifier is STYLE or REGION, matching the backend subtitle parser (#2434) — thanks @17lijunyi! - Dub assembly, cached segments and audio tools can read generated WAV files when TorchCodec is missing or cannot load (#2379) — thanks @tokutei58301-boop! - Remote API-key and share-PIN clients can record and read export history while native filesystem operations stay local (#2383, #2384) — thanks @sedatdagg! - Concurrent job events receive unique sequence numbers (#2384) — thanks @sedatdagg! diff --git a/electron/src/shared/utils/importStory.js b/electron/src/shared/utils/importStory.js index 2186d892a..e15a3f695 100644 --- a/electron/src/shared/utils/importStory.js +++ b/electron/src/shared/utils/importStory.js @@ -83,7 +83,11 @@ function captionBlocks(text, webvtt) { if (!lines.length) return false; const first = lines[0]; if (/^WEBVTT(?:[ \t]|$)/.test(first)) return false; - return !isWebVttMetadata(first); + if (!isWebVttMetadata(first)) return true; + // WebVTT's block parser gives a timing line in position two precedence + // over the identifier, so `STYLE`/`REGION`/`NOTE` can name a real cue. + // https://www.w3.org/TR/webvtt1/#file-parsing — mirrors parse_srt. + return lines.length > 1 && isTimingLine(lines[1]); }); } diff --git a/electron/src/shared/utils/importStory.test.js b/electron/src/shared/utils/importStory.test.js index b5d8c89e7..18a08bebe 100644 --- a/electron/src/shared/utils/importStory.test.js +++ b/electron/src/shared/utils/importStory.test.js @@ -109,11 +109,20 @@ describe('parseSrt', () => { '00:00:01.000 --> 00:00:02.000\nSpoken text\n'; expect(parseSrt(vtt)).toBe('Spoken text'); }); - it('never treats a NOTE block as a cue even when it contains a valid timing line', () => { - const vtt = - 'WEBVTT\n\nNOTE\n00:00:01.000 --> 00:00:02.000\nprivate note\n\n' + - '00:00:03.000 --> 00:00:04.000\nSpoken text\n'; - expect(parseSrt(vtt)).toBe('Spoken text'); + it('lets a timing line in position two outrank a metadata identifier', () => { + for (const identifier of ['STYLE', 'REGION', 'NOTE', 'NOTE identifier']) { + expect(parseSrt(`WEBVTT\n\n${identifier}\n00:01.000 --> 00:02.000\nSpoken text\n`)).toBe( + 'Spoken text', + ); + } + }); + it('keeps a cue whose identifier is STYLE or REGION in a mixed file', () => { + for (const identifier of ['intro', 'STYLE', 'REGION']) { + const vtt = + `WEBVTT\n\n${identifier}\n00:00:01.000 --> 00:00:02.000\nHello world.\n\n` + + 'next\n00:00:03.000 --> 00:00:04.000\nStill here.\n'; + expect(parseSrt(vtt)).toBe('Hello world.\nStill here.'); + } }); it('keeps NOTE when it is the spoken dialogue', () => { expect(