fix(playback): replay sentence visual cues after navigation (#1720)

Co-authored-by: sophietao20-star <283060850+sophietao20-star@users.noreply.github.com>
Co-authored-by: wyuc <wang-yc24@mails.tsinghua.edu.cn>
This commit is contained in:
sophietao20-star
2026-09-29 18:22:49 +08:00
committed by GitHub
co-authored by sophietao20-star wyuc
parent fa47efe1b3
commit 8f7d51e5ac
3 changed files with 226 additions and 8 deletions
+14
View File
@@ -45,6 +45,20 @@ export function isWhiteboardPlaybackAction(action: Action): boolean {
return WHITEBOARD_ACTION_TYPES.has(action.type);
}
/** Only the contiguous visual cues immediately before a speech belong to its replay. */
export function getSpeechVisualCueStartIndex(
actions: readonly Action[],
speechIndex: number,
): number {
let start = speechIndex;
while (start > 0) {
const type = actions[start - 1]?.type;
if (type !== 'spotlight' && type !== 'laser') break;
start--;
}
return start;
}
export function canReconstructPrefixForAction(
actions: readonly Action[],
actionIndex: number,
+34 -8
View File
@@ -43,6 +43,7 @@ import {
} from '@/lib/choreography';
import {
canJumpWithinReconstructablePrefix,
getSpeechVisualCueStartIndex,
isWhiteboardPlaybackAction,
} from '@/lib/playback/action-navigation';
import { useCanvasStore } from '@/lib/store/canvas';
@@ -88,6 +89,8 @@ export class PlaybackEngine {
private browserTTSPausedChunks: string[] = []; // remaining chunks saved on pause (for cancel+re-speak)
private speechTimerRemaining: number = 0; // remaining ms (set on pause)
private playbackGeneration: number = 0;
// Keep the cursor on speech for progress/persistence; defer its cues until playback starts.
private pendingNavigationSpeechIndex: number | null = null;
constructor(
scenes: Scene[],
@@ -135,6 +138,7 @@ export class PlaybackEngine {
/** Restore playback position from a snapshot */
restoreFromSnapshot(snapshot: PlaybackSnapshot): void {
this.pendingNavigationSpeechIndex = null;
this.sceneIndex = snapshot.sceneIndex;
this.actionIndex = snapshot.actionIndex;
this.consumedDiscussions = new Set(snapshot.consumedDiscussions);
@@ -149,6 +153,7 @@ export class PlaybackEngine {
this.sceneIndex = 0;
this.actionIndex = 0;
this.pendingNavigationSpeechIndex = null;
this.invalidatePlaybackGeneration();
this.setMode('playing');
this.processNext();
@@ -179,6 +184,7 @@ export class PlaybackEngine {
const autoplay = options.autoplay ?? this.mode === 'playing';
const generation = this.invalidatePlaybackGeneration();
this.pendingNavigationSpeechIndex = null;
this.cancelActivePlaybackWork();
this.sceneIndex = 0;
this.actionIndex = 0;
@@ -200,6 +206,7 @@ export class PlaybackEngine {
this.actionEngine.clearEffects();
this.sceneIndex = 0;
this.actionIndex = actionIndex;
this.pendingNavigationSpeechIndex = actionIndex;
this.callbacks.onProgress?.(this.getSnapshot());
if (autoplay) {
@@ -310,6 +317,7 @@ export class PlaybackEngine {
/** → idle */
stop(): void {
this.invalidatePlaybackGeneration();
this.pendingNavigationSpeechIndex = null;
// Set mode BEFORE stopping audio to prevent spurious processNext from
// synchronous onend callbacks (see handleUserInterrupt for details).
this.setMode('idle');
@@ -540,6 +548,17 @@ export class PlaybackEngine {
return { action: res.action, sceneId: res.sceneId };
}
private fireVisualCue(action: Extract<Action, { type: 'spotlight' | 'laser' }>): void {
this.actionEngine.execute(action);
this.callbacks.onEffectFire?.({
kind: action.type,
targetId: action.elementId,
...(action.type === 'spotlight'
? { dimOpacity: action.dimOpacity }
: { color: action.color }),
} as Effect);
}
/**
* Core processing loop: consume the next action.
*/
@@ -567,6 +586,20 @@ export class PlaybackEngine {
const { action } = current;
const replayCues =
this.sceneIndex === 0 && this.pendingNavigationSpeechIndex === this.actionIndex;
this.pendingNavigationSpeechIndex = null;
if (replayCues && action.type === 'speech') {
const actions = this.scenes[0].actions ?? [];
const targetIndex = this.actionIndex;
for (let i = getSpeechVisualCueStartIndex(actions, targetIndex); i < targetIndex; i++) {
if (this.mode !== 'playing' || !this.isCurrentGeneration(generation)) return;
const cue = actions[i];
if (cue.type === 'spotlight' || cue.type === 'laser') this.fireVisualCue(cue);
}
if (this.mode !== 'playing' || !this.isCurrentGeneration(generation)) return;
}
// Notify progress BEFORE advancing the cursor so the snapshot points at
// the current action. On restore the same action will be replayed — this
// is the desired behaviour for speech (user may have only heard half).
@@ -652,14 +685,7 @@ export class PlaybackEngine {
case 'spotlight':
case 'laser': {
// Fire-and-forget visual effects via ActionEngine
this.actionEngine.execute(action);
this.callbacks.onEffectFire?.({
kind: action.type,
targetId: action.elementId,
...(action.type === 'spotlight'
? { dimOpacity: action.dimOpacity }
: { color: action.color }),
} as Effect);
this.fireVisualCue(action);
// Don't block — continue immediately (use queueMicrotask to avoid
// stack overflow from deep synchronous recursion when many consecutive
// spotlight/laser actions appear in sequence)
+178
View File
@@ -258,6 +258,42 @@ describe('PlaybackEngine action navigation', () => {
vi.unstubAllGlobals();
});
it('replays the target speech visual cues on jump, just as sequential playback does', async () => {
const actions = [
speech('previous'),
{ id: 'spot', type: 'spotlight', elementId: 'box' } as Action,
{ id: 'laser', type: 'laser', elementId: 'label', color: '#f00' } as Action,
speech('target'),
];
const natural = createActionEngine();
const naturalAudio = createAudioPlayer(async () => true);
const naturalEngine = new PlaybackEngine([scene(actions)], natural.engine, naturalAudio.player);
naturalEngine.start();
naturalAudio.fireEnded();
await flushPromises();
expect(natural.executions.map(({ action }) => action.id)).toEqual(['spot', 'laser']);
const jumped = createActionEngine();
const { player } = createAudioPlayer(async () => true);
const onSpeechStart = vi.fn();
const onProgress = vi.fn();
const onEffectFire = vi.fn();
const engine = new PlaybackEngine([scene(actions)], jumped.engine, player, {
onSpeechStart,
onProgress,
onEffectFire,
});
expect(await engine.jumpToAction(3, { autoplay: true })).toBe(true);
await flushPromises();
expect(jumped.executions).toEqual(natural.executions);
expect(onEffectFire.mock.calls.map(([effect]) => effect.kind)).toEqual(['spotlight', 'laser']);
expect(onSpeechStart.mock.calls).toEqual([['target']]);
expect(onProgress.mock.calls.every(([snapshot]) => snapshot.actionIndex === 3)).toBe(true);
engine.stop();
naturalEngine.stop();
});
it('rejects invalid and unsafe jump targets', async () => {
const { engine: actionEngine } = createActionEngine();
const { player } = createAudioPlayer();
@@ -279,6 +315,148 @@ describe('PlaybackEngine action navigation', () => {
expect('seekTo' in engine).toBe(false);
});
it.each(['idle', 'paused'] as const)(
'defers cues while %s and keeps the selected speech in snapshots until playback',
async (mode) => {
const { engine: actionEngine, executions } = createActionEngine();
const { player } = createAudioPlayer(async () => true);
const onSpeechStart = vi.fn();
const onProgress = vi.fn();
const actions = [
speech('previous'),
{ id: 'spot', type: 'spotlight', elementId: 'box' } as Action,
speech('target'),
];
const engine = new PlaybackEngine([scene(actions)], actionEngine, player, {
onSpeechStart,
onProgress,
});
if (mode === 'paused') {
engine.start();
engine.pause();
}
onSpeechStart.mockClear();
onProgress.mockClear();
await engine.jumpToAction(2, { autoplay: false });
expect(engine.getSnapshot().actionIndex).toBe(2);
expect(engine.getMode()).toBe(mode);
expect(executions).toEqual([]);
expect(onSpeechStart).not.toHaveBeenCalled();
if (mode === 'paused') engine.resume();
else engine.continuePlayback();
await flushPromises();
expect(executions.map(({ action }) => action.id)).toEqual(['spot']);
expect(onSpeechStart.mock.calls).toEqual([['target']]);
expect(onProgress.mock.calls.every(([snapshot]) => snapshot.actionIndex === 2)).toBe(true);
engine.pause();
engine.resume();
expect(executions).toHaveLength(1);
engine.stop();
},
);
it('replays first-sentence cues without clearing them again at the scene boundary', async () => {
const { engine: actionEngine, executions } = createActionEngine();
const { player } = createAudioPlayer(async () => true);
const actions = [
{ id: 'spot', type: 'spotlight', elementId: 'box' } as Action,
speech('first'),
];
const engine = new PlaybackEngine([scene(actions)], actionEngine, player);
await engine.jumpToAction(1, { autoplay: false });
vi.mocked(actionEngine.clearEffects).mockClear();
engine.continuePlayback();
await flushPromises();
expect(executions.map(({ action }) => action.id)).toEqual(['spot']);
expect(actionEngine.clearEffects).not.toHaveBeenCalled();
engine.stop();
});
it('stops collecting cues at whiteboard actions and earlier speech', async () => {
const { engine: actionEngine, executions } = createActionEngine();
const { player } = createAudioPlayer(async () => true);
const actions = [
{ id: 'old-cue', type: 'spotlight', elementId: 'old' } as Action,
speech('previous'),
{ id: 'before-board', type: 'laser', elementId: 'old' } as Action,
{ id: 'board', type: 'wb_open' } as Action,
{ id: 'target-cue', type: 'spotlight', elementId: 'box' } as Action,
speech('target'),
speech('no-cue'),
];
const engine = new PlaybackEngine([scene(actions)], actionEngine, player);
await engine.jumpToAction(5, { autoplay: true });
expect(executions).toEqual([
{ action: actions[3], silent: true },
{ action: actions[4], silent: undefined },
]);
executions.length = 0;
await engine.jumpToAction(6, { autoplay: true });
expect(executions).toEqual([{ action: actions[3], silent: true }]);
engine.stop();
});
it('only replays cues from the last of multiple paused jumps', async () => {
const { engine: actionEngine, executions } = createActionEngine();
const { player } = createAudioPlayer(async () => true);
const actions = [
{ id: 'first-cue', type: 'spotlight', elementId: 'first' } as Action,
speech('first'),
{ id: 'second-cue', type: 'laser', elementId: 'second' } as Action,
speech('second'),
];
const engine = new PlaybackEngine([scene(actions)], actionEngine, player);
await engine.jumpToAction(1, { autoplay: false });
await engine.jumpToAction(3, { autoplay: false });
engine.continuePlayback();
await flushPromises();
expect(executions.map(({ action }) => action.id)).toEqual(['second-cue']);
engine.stop();
});
it('discards cues from a jump superseded during whiteboard reconstruction', async () => {
const { engine: actionEngine, executions } = createActionEngine();
const reconstruction = deferred<void>();
vi.mocked(actionEngine.execute).mockImplementationOnce(() => reconstruction.promise);
const { player } = createAudioPlayer(async () => true);
const onSpeechStart = vi.fn();
const actions = [
speech('first'),
{ id: 'board', type: 'wb_open' } as Action,
{ id: 'spot', type: 'spotlight', elementId: 'box' } as Action,
speech('target'),
];
const engine = new PlaybackEngine([scene(actions)], actionEngine, player, { onSpeechStart });
const staleJump = engine.jumpToAction(3, { autoplay: true });
expect(await engine.jumpToAction(0, { autoplay: true })).toBe(true);
reconstruction.resolve();
expect(await staleJump).toBe(false);
expect(executions).toEqual([]);
expect(onSpeechStart.mock.calls).toEqual([['first']]);
engine.stop();
});
it('drops deferred cues on stop and preserves ordinary playback after restarting', async () => {
const { engine: actionEngine, executions } = createActionEngine();
const { player, fireEnded } = createAudioPlayer(async () => true);
const actions = [
speech('first'),
{ id: 'spot', type: 'spotlight', elementId: 'box' } as Action,
speech('target'),
];
const engine = new PlaybackEngine([scene(actions)], actionEngine, player);
await engine.jumpToAction(2, { autoplay: false });
engine.stop();
engine.start();
expect(executions).toEqual([]);
fireEnded();
await flushPromises();
expect(executions.map(({ action }) => action.id)).toEqual(['spot']);
engine.stop();
});
it.each([
['widget_setState', { id: 'u', type: 'widget_setState', state: {} }],
['discussion', { id: 'u', type: 'discussion', topic: 'Discuss' }],