feat: add SenseVoiceSmall as a local subtitle transcription option (#244)

* feat: add optional isolated SenseVoice local transcription

* test: verify local transcription selector switches SenseVoice and Whisper

* docs: set realistic disk space expectation for SenseVoice preparation

* docs: reconcile issue 67 delivery with historical ASR plans
This commit is contained in:
Kris K
2026-09-30 21:16:00 +08:00
committed by GitHub
parent 35758311f2
commit e9e6c90b42
31 changed files with 876 additions and 30 deletions
+45
View File
@@ -0,0 +1,45 @@
name: Local SenseVoice acceptance
on:
workflow_dispatch:
pull_request:
paths:
- 'backend/services/sensevoice*.py'
- 'backend/tests/test_sensevoice.py'
- 'scripts/verify_sensevoice.py'
- '.github/workflows/sensevoice.yml'
permissions:
contents: read
jobs:
real-transcription:
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, windows-latest, macos-14]
runs-on: ${{ matrix.os }}
timeout-minutes: 35
env:
PYTHONUTF8: '1'
PYTHONIOENCODING: utf-8
HF_HUB_DISABLE_PROGRESS_BARS: '1'
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: '3.13'
cache: pip
- run: python -m pip install -r requirements.txt
- if: runner.os == 'Linux'
run: sudo apt-get update && sudo apt-get install -y ffmpeg
- if: runner.os == 'Windows'
run: choco install ffmpeg --yes --no-progress
- if: runner.os == 'macOS'
run: brew install ffmpeg
- run: python -m pytest backend/tests/test_sensevoice.py -q
- name: Install optional runtime, download models, transcribe real speech offline
run: python scripts/verify_sensevoice.py
- uses: actions/upload-artifact@v4
if: always()
with:
name: sensevoice-${{ matrix.os }}
path: sensevoice-acceptance.json
if-no-files-found: ignore
+2
View File
@@ -5,6 +5,8 @@
## 当前交付与验证范围
- **2026-09-30 本地 SenseVoice 增补(PR #244,未发布)**:维护者要求接入 #67 后,设置页增加 SenseVoiceSmall 按需准备与选择;保存后由现有转写入口路由,独立 CPU 进程运行 FunASR,不上传音频、不改变 Whisper 默认值。三系统 Python 3.13 的真实 45 秒访谈离线验收通过,均生成 21 条字幕、157 个词级时间戳,最长约 3.36 秒;窗口安装包尚未做这项新增能力的真机验收。Windows 实装组件与模型约 5.8 GB,准备前建议预留 8 GB。详见 [配置与边界](docs/AI_MODEL_CONFIGURATION.md#本地-sensevoice67)。下方“#67 仍调研、先等通用协议”的文字是历史排期;通用插件框架仍是后续工作,本次具体本地转写能力已经实现。
- 1.4 已包含统一 Studio 导入→推荐/确认→制作→编辑→导出;默认字幕分析,视觉调用须遵守用户选择与确认边界。未确认不开始正式制作。
- macOS 原生导入至保存、受控旧版本升级/回退已有证据,见 [RC 验收](docs/RC_ACCEPTANCE_1_4.md) 与 [发布记录](docs/RELEASE_1_4.md)。这些记录不等于第二台全新 Mac 或真实用户数据库迁移验收。
- Windows 构建已有记录,真机安装/导入/保存仍待验收。多游戏完整观感、事件边界和推广差异化也仍待验收;不承诺广告投放效果。
+24 -1
View File
@@ -43,6 +43,29 @@ def get_speech_recognizer() -> SpeechRecognizer:
# ===== Whisper 运行时(按需安装)=====
@router.get("/sensevoice/status")
async def sensevoice_status():
from backend.services import sensevoice_runtime
return sensevoice_runtime.status()
@router.post("/sensevoice/prepare")
async def sensevoice_prepare():
from backend.services import sensevoice_runtime
try:
return sensevoice_runtime.start_prepare()
except RuntimeError as exc:
raise HTTPException(status_code=409, detail=str(exc)) from exc
@router.delete("/sensevoice")
async def sensevoice_uninstall():
from backend.services import sensevoice_runtime
try:
return sensevoice_runtime.uninstall()
except RuntimeError as exc:
raise HTTPException(status_code=409, detail=str(exc)) from exc
@router.get("/whisper/runtime-status")
async def whisper_runtime_status():
"""Whisper 运行时安装状态(前端轮询)。"""
@@ -598,4 +621,4 @@ async def get_speech_recommendations():
except Exception as e:
logger.error(f"获取配置建议失败: {e}")
raise HTTPException(status_code=500, detail=f"获取建议失败: {str(e)}")
raise HTTPException(status_code=500, detail=f"获取建议失败: {str(e)}")
+10
View File
@@ -168,6 +168,12 @@ def timeline_failure_from_report(topic_count: int, report: dict) -> PipelineFail
def missing_subtitle_failure() -> PipelineFailure:
"""视频没有字幕,自动转写也没留下 srt。按当前 Whisper 状态区分下一步。"""
from backend.services import whisper_runtime
from backend.services.ai_model_settings import load
settings = load()
if settings and settings.transcription and settings.transcription.provider == 'sensevoice_local':
return PipelineFailure('SUBTITLE', '没有字幕可分析:SenseVoice 本次没有生成可用字幕。',
'到「设置 → 转写」检查 SenseVoiceSmall 是否就绪,或导入 .srt 字幕后重试。',
code='subtitle_setup')
status = whisper_runtime.get_status()
state = status.get("status")
@@ -196,6 +202,10 @@ def missing_subtitle_failure() -> PipelineFailure:
def failure_from_speech_error(message: str) -> PipelineFailure:
"""转写异常收成同一套失败码,不再把「设置 → 语音识别」和「设置 → 转写」叠在一起。"""
text = (message or "").strip().replace("设置 → 语音识别", "设置 → 转写")
if 'SenseVoice' in text:
return PipelineFailure('SUBTITLE', text,
'' if '设置 → 转写' in text else '到「设置 → 转写」检查 SenseVoiceSmall,或导入 .srt 字幕。',
code='subtitle_setup')
from backend.services import whisper_runtime
state = whisper_runtime.get_status().get("status")
+5 -1
View File
@@ -76,7 +76,7 @@ class Assignment(BaseModel):
class Transcription(BaseModel):
provider: Literal['whisper_local', 'cloud'] = 'whisper_local'
provider: Literal['whisper_local', 'sensevoice_local', 'cloud'] = 'whisper_local'
model: str = Field(default='base', min_length=1, max_length=200)
connection_id: str | None = None
@@ -89,6 +89,10 @@ class Transcription(BaseModel):
if self.model not in {'tiny', 'base', 'small', 'medium', 'large', 'large-v3'}:
raise ValueError('请选择本地 Whisper 模型')
self.connection_id = None
elif self.provider == 'sensevoice_local':
if self.model != 'SenseVoiceSmall':
raise ValueError('请选择本地 SenseVoiceSmall 模型')
self.connection_id = None
elif not self.connection_id:
raise ValueError('请选择转写供应商')
return self
+73
View File
@@ -0,0 +1,73 @@
"""SenseVoice CTC alignment, in milliseconds; never invent subtitle timing.
Adapted from LauraGPT's MIT-licensed timestamp fix for 123mlly/autoclip PR #1.
We require complete CTC alignment instead of its coarse/proportional fallbacks.
"""
import math
import re
def clean(text):
return re.sub(r"<\|[^|]*\|>", "", text).replace("▁", " ").strip()
def aligned_cues(result, duration_ms, max_chars=42, max_duration_ms=8000):
"""Preserve all source text and real word times, across VAD segments."""
if not isinstance(result, list) or not result or duration_ms <= 0:
raise ValueError('SenseVoice 未返回可用的词级时间戳')
cues, current = [], []
previous_end = 0
def flush():
if current:
cues.append({'start': current[0]['start'], 'end': current[-1]['end'],
'text': ''.join(w['text'] for w in current).strip(),
'words': list(current)})
current.clear()
for item in result:
if not isinstance(item, dict) or not isinstance(item.get('text'), str):
raise ValueError('SenseVoice 返回了无效的转写结果')
source = clean(item['text'])
if not source:
continue
words, times = item.get('words'), item.get('timestamp')
if not isinstance(words, list) or not isinstance(times, list) or not words or len(words) != len(times):
raise ValueError('SenseVoice 词与时间戳未完整配对,请重新准备模型或改用 Whisper')
cursor, aligned = 0, []
for word, time in zip(words, times):
if not isinstance(word, str) or not clean(word) or not isinstance(time, (list, tuple)) or len(time) != 2:
raise ValueError('SenseVoice 返回了无效的词级时间戳')
if not all(isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t) for t in time):
raise ValueError('SenseVoice 返回了无效的词级时间戳')
start, end = (int(t) for t in time) # FunASR lists are always milliseconds.
# CTC has a 60ms frame step; only tolerate final-frame rounding.
if not 0 <= previous_end <= start < end <= duration_ms + 60:
raise ValueError('SenseVoice 时间戳倒退、重叠或超出音频范围')
end = min(end, duration_ms)
if end <= start:
raise ValueError('SenseVoice 时间戳超出音频范围')
token = clean(word)
position = source.find(token, cursor)
if position < 0 or any(c.isalnum() for c in source[cursor:position]):
raise ValueError('SenseVoice 原文与词级时间戳无法完整对齐')
stop = position + len(token)
aligned.append({'text': source[cursor:stop], 'start': start / 1000, 'end': end / 1000})
cursor, previous_end = stop, end
if any(c.isalnum() for c in source[cursor:]):
raise ValueError('SenseVoice 原文与词级时间戳无法完整对齐')
aligned[-1]['text'] += source[cursor:]
for word in aligned:
if current and (len(''.join(w['text'] for w in current) + word['text']) > max_chars
or (word['end'] - current[0]['start']) * 1000 > max_duration_ms
or word['start'] - current[-1]['end'] > 1.2):
flush()
if len(word['text'].strip()) > max_chars or (word['end'] - word['start']) * 1000 > max_duration_ms:
raise ValueError('SenseVoice 单词时间戳异常,无法生成可读字幕')
current.append(word)
if re.search(r'[。!?;!?.;]$', word['text'].strip()):
flush()
flush()
if not cues:
raise ValueError('SenseVoice 未识别出可用人声')
return cues
+212
View File
@@ -0,0 +1,212 @@
"""Optional, isolated CPU SenseVoice runtime. No FunASR imports in this process."""
from contextlib import contextmanager
import json
import logging
import os
from pathlib import Path
import shutil
import subprocess
import sys
import tempfile
import threading
import wave
from backend.core.path_utils import get_data_directory
from backend.utils.ffmpeg_utils import get_ffmpeg_path
logger = logging.getLogger(__name__)
MODEL = 'SenseVoiceSmall'
VERSION = 1
def root():
return get_data_directory() / 'sensevoice'
def packages():
cpu = '+cpu' if sys.platform != 'darwin' else ''
return ['funasr==1.3.14', f'torch==2.8.0{cpu}', f'torchaudio==2.8.0{cpu}',
'setuptools==80.9.0', 'huggingface-hub<1', 'numpy<3']
def status():
try:
value = json.loads((root() / 'status.json').read_text(encoding='utf-8'))
except (OSError, ValueError):
value = {'status': 'not_installed', 'message': ''}
# Validate paths without importing a large optional runtime on every poll.
try:
ready = json.loads((root() / 'ready.json').read_text(encoding='utf-8'))
valid = (ready['version'] == VERSION and (root() / 'runtime' / 'funasr' / '__init__.py').exists()
and all(Path(p).resolve().is_relative_to((root() / 'models').resolve())
and (Path(p) / 'model.pt').exists() for p in ready['paths'].values())
and set(ready['paths']) == {'model', 'vad_model'})
except (OSError, ValueError, KeyError, TypeError):
valid = False
if valid and value['status'] != 'installing':
value = {'status': 'ready', 'message': ''}
elif value.get('status') == 'ready':
value = {'status': 'not_installed', 'message': '模型文件缺失,请重新准备模型'}
if value.get('status') == 'installing':
try:
with operation():
value = {'status': 'error', 'message': '上次模型准备被中断,请重试'}
except RuntimeError:
pass
return {**value, 'model': MODEL}
def _state(state, message=''):
root().mkdir(parents=True, exist_ok=True)
target = root() / 'status.json'
pending = target.with_suffix('.tmp')
pending.write_text(json.dumps({'status': state, 'message': message}, ensure_ascii=False), encoding='utf-8')
os.replace(pending, target)
@contextmanager
def operation():
"""Cross-process lock: API installs and Celery inference share the data dir."""
base = root()
base.mkdir(parents=True, exist_ok=True)
with open(base.parent / 'sensevoice.lock', 'a+b') as handle:
handle.seek(0)
try:
if sys.platform == 'win32':
import msvcrt
if handle.read(1) == b'':
handle.write(b'0')
handle.flush()
handle.seek(0)
msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1)
else:
import fcntl
fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
except OSError:
raise RuntimeError('SenseVoice 正在准备或转写,请完成后重试') from None
try:
yield
finally:
if sys.platform == 'win32':
handle.seek(0)
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
else:
fcntl.flock(handle, fcntl.LOCK_UN)
def worker(action, result, audio=None, language='auto', timeout=None, deny_network=False):
command = [sys.executable, '-S', str(Path(__file__).with_name('sensevoice_worker.py')),
action, '--root', str(root()), '--result', str(result), '--language', language]
if audio:
command += ['--audio', str(audio)]
if deny_network:
command += ['--deny-network']
# Logs can contain private text. Keep them in the local diagnostic log only.
with tempfile.TemporaryFile() as log:
completed = subprocess.run(command, stdout=log, stderr=log, timeout=timeout,
env={**os.environ, 'PYTHONUTF8': '1', 'PYTHONIOENCODING': 'utf-8'})
if completed.returncode:
log.seek(max(0, log.tell() - 12000))
logger.error('SenseVoice worker failed: %s', log.read().decode('utf-8', errors='replace'))
raise RuntimeError('SenseVoice 模型运行失败,请到「设置 → 转写」重新准备模型;仍失败时附上脱敏日志')
def _prepare():
_state('installing', '正在安装组件,首次下载可能需要几分钟')
try:
(root() / 'ready.json').unlink(missing_ok=True)
runtime = root() / 'runtime'
shutil.rmtree(runtime, ignore_errors=True)
command = [sys.executable, '-m', 'pip', 'install', '--target', str(runtime),
'--only-binary', 'torch,torchaudio', *packages()]
if sys.platform != 'darwin':
command += ['--extra-index-url', 'https://download.pytorch.org/whl/cpu']
with tempfile.TemporaryFile() as log:
completed = subprocess.run(command, stdout=log, stderr=log, timeout=1800,
env={**os.environ, 'PIP_PROGRESS_BAR': 'off', 'PYTHONIOENCODING': 'utf-8'})
if completed.returncode:
log.seek(max(0, log.tell() - 12000))
logger.error('SenseVoice pip failed: %s', log.read().decode('utf-8', errors='replace'))
raise RuntimeError('SenseVoice 组件安装失败,请检查网络和磁盘空间后重试')
_state('installing', '正在下载并检查 SenseVoiceSmall 模型')
result = root() / 'prepare-result.json'
worker('prepare', result, timeout=1800)
value = json.loads(result.read_text(encoding='utf-8'))
value['version'] = VERSION
ready = root() / 'ready.json'
pending = ready.with_suffix('.tmp')
pending.write_text(json.dumps(value), encoding='utf-8')
os.replace(pending, ready)
result.unlink(missing_ok=True)
_state('ready')
except Exception as exc:
_state('error', str(exc) if isinstance(exc, RuntimeError) else 'SenseVoice 准备失败,请检查网络后重试')
raise
def prepare():
with operation():
_prepare()
def start_prepare():
lock = operation()
lock.__enter__()
def run():
try:
_prepare()
except Exception:
logger.exception('SenseVoice preparation failed')
finally:
lock.__exit__(None, None, None)
try:
_state('installing', '正在准备组件…')
threading.Thread(target=run, daemon=True).start()
except Exception:
lock.__exit__(None, None, None)
raise
return status()
def uninstall():
with operation():
shutil.rmtree(root())
return status()
def transcribe(video, output=None, language='auto', timeout=0, *, deny_network=False):
from backend.services.sensevoice_alignment import aligned_cues
from backend.utils.speech_recognizer import SpeechRecognitionError, SpeechRecognizer
from backend.utils.word_timing import write_word_timing
language = {'zh-TW': 'zh', 'en-US': 'en', 'en-GB': 'en'}.get(language, language)
if language not in {'auto', 'zh', 'en', 'ja', 'ko', 'yue'}:
raise SpeechRecognitionError('SenseVoiceSmall 支持中文、粤语、英语、日语和韩语,请选择支持的语言或自动检测')
output = Path(output) if output else Path(video).with_suffix('.srt')
try:
with operation(), tempfile.TemporaryDirectory(prefix='autoclip-sensevoice-') as temporary:
if status()['status'] != 'ready':
raise RuntimeError('SenseVoice 尚未就绪,请到「设置 → 转写」准备 SenseVoiceSmall 模型')
audio, result = Path(temporary) / 'audio.wav', Path(temporary) / 'result.json'
extracted = subprocess.run([get_ffmpeg_path(), '-nostdin', '-v', 'error', '-y', '-i', str(video),
'-vn', '-ac', '1', '-ar', '16000', '-c:a', 'pcm_s16le', str(audio)],
capture_output=True, timeout=timeout or 300)
if extracted.returncode:
raise RuntimeError('SenseVoice 无法读取视频音轨,请确认视频包含可播放的人声')
with wave.open(str(audio), 'rb') as source:
duration_ms = source.getnframes() * 1000 // source.getframerate()
worker('transcribe', result, audio, language, timeout or None, deny_network)
cues = aligned_cues(json.loads(result.read_text(encoding='utf-8')), duration_ms)
output.parent.mkdir(parents=True, exist_ok=True)
with tempfile.NamedTemporaryFile(mode='w', encoding='utf-8', dir=output.parent, delete=False) as stream:
pending = Path(stream.name)
stream.write(SpeechRecognizer._segments_to_srt(cues))
try:
os.replace(pending, output)
write_word_timing(output, cues, language, source='sensevoice')
finally:
pending.unlink(missing_ok=True)
return output
except subprocess.TimeoutExpired:
raise SpeechRecognitionError('SenseVoice 转写超时,请缩短视频或增加转写超时时间') from None
except (RuntimeError, ValueError, OSError) as exc:
raise SpeechRecognitionError(str(exc)) from exc
+61
View File
@@ -0,0 +1,61 @@
"""Standalone worker: launched with -S, importing only the optional runtime.
Do not import backend here. PyTorch/NumPy must never enter the API process.
"""
import argparse
import json
import os
from pathlib import Path
import sys
def main():
parser = argparse.ArgumentParser()
parser.add_argument('action', choices=['prepare', 'transcribe'])
parser.add_argument('--root', type=Path, required=True)
parser.add_argument('--result', type=Path, required=True)
parser.add_argument('--audio', type=Path)
parser.add_argument('--language', default='auto')
parser.add_argument('--deny-network', action='store_true') # acceptance only
args = parser.parse_args()
runtime = args.root / 'runtime'
sys.path.insert(0, str(runtime))
os.environ.update(OMP_NUM_THREADS='2', MKL_NUM_THREADS='2', NUMBA_NUM_THREADS='2',
HF_HOME=str(args.root / 'models'), HF_HUB_DISABLE_PROGRESS_BARS='1')
if args.action == 'transcribe':
os.environ.update(HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1', MODELSCOPE_OFFLINE='1')
if args.deny_network:
import socket
def denied(*a, **kw):
raise RuntimeError('offline acceptance forbids network access')
socket.socket.connect = denied
socket.create_connection = denied
import torch
import funasr
assert Path(torch.__file__).resolve().is_relative_to(runtime.resolve())
assert Path(funasr.__file__).resolve().is_relative_to(runtime.resolve())
torch.set_num_threads(2)
from funasr import AutoModel
if args.action == 'prepare':
from huggingface_hub import snapshot_download
paths = {key: snapshot_download(repo_id=repo, cache_dir=str(args.root / 'models' / 'hub'),
allow_patterns=['*.json', '*.yaml', '*.txt', '*.model', '*.mvn', 'model.pt'])
for key, repo in [('model', 'FunAudioLLM/SenseVoiceSmall'), ('vad_model', 'funasr/fsmn-vad')]}
else:
paths = json.loads((args.root / 'ready.json').read_text(encoding='utf-8'))['paths']
if any(not Path(p).resolve().is_relative_to((args.root / 'models').resolve()) for p in paths.values()):
raise ValueError('Invalid model cache path')
model = AutoModel(**paths, hub='hf', device='cpu', ncpu=2, disable_update=True,
disable_pbar=True, trust_remote_code=False,
vad_kwargs={'max_single_segment_time': 30000})
if args.action == 'prepare':
result = {'paths': paths}
else:
result = model.generate(input=str(args.audio), cache={}, language=args.language,
use_itn=True, output_timestamp=True, batch_size_s=30,
merge_vad=False, disable_pbar=True)
args.result.write_text(json.dumps(result, ensure_ascii=False, allow_nan=False), encoding='utf-8')
if __name__ == '__main__':
main()
+4 -2
View File
@@ -216,8 +216,10 @@ def _generate_import_subtitle(task, project_id: str, video_path: str):
logger.info(f"使用语音转写配置 - 方法: {speech_config.method}")
if models and models.transcription and models.transcription.provider == "cloud":
generated_subtitle = generate_subtitle_for_video(Path(video_path), method="auto")
if models and models.transcription and models.transcription.provider in {"cloud", "sensevoice_local"}:
generated_subtitle = generate_subtitle_for_video(
Path(video_path), method="auto", language=speech_config.whisper_config.language,
timeout=speech_config.whisper_config.timeout)
elif speech_config.method == "whisper_local":
model = configured_whisper_model(speech_config.whisper_config.model_name)
language = speech_config.whisper_config.language
@@ -29,6 +29,11 @@ def test_missing_mandatory_dependency_still_blocks_bundle(tmp_path):
assert "- openai" in result.stdout
def test_isolated_sensevoice_worker_dependencies_are_not_bundled(tmp_path):
result = run_guard(tmp_path, "def worker():\n import torch\n import funasr\n import huggingface_hub\n")
assert result.returncode == 0, result.stdout + result.stderr
def test_bundle_local_modules_resolve(tmp_path):
result = run_guard(tmp_path, "import backend\nimport json\n")
assert result.returncode == 0, result.stdout + result.stderr
+157
View File
@@ -0,0 +1,157 @@
"""CTC timing, complete text, route selection, dependency isolation and locking."""
import json
from pathlib import Path
import subprocess
import sys
from types import SimpleNamespace
import wave
import pytest
from backend.services.sensevoice_alignment import aligned_cues
from backend.services import sensevoice_runtime as runtime
from backend.services.ai_model_settings import ModelSettings, Transcription
def fixture(text='这是第一段。这里是第二段。最后一段用于验证音频末尾。'):
words = list(text)
return [{'text': '<|zh|><|NEUTRAL|>' + text, 'words': words,
'timestamp': [[1050 + i * 800, 1550 + i * 800] for i in range(len(words))],
'sentence_info': [{'sentence': text, 'start': 0, 'end': 30010}]}]
def test_ctc_avoids_coarse_vad_cues_and_preserves_all_text():
raw = fixture()
cues = aligned_cues(raw, 45100)
assert len(cues) > 2
assert ''.join(c['text'] for c in cues) == raw[0]['text'].split('>')[-1]
assert cues[0]['start'] == 1.05
assert cues[-1]['end'] == raw[0]['timestamp'][-1][1] / 1000
from backend.utils.word_timing import valid_words
assert all(valid_words(c) and c['end'] - c['start'] <= 8 for c in cues)
def test_spaces_numbers_and_punctuation_are_preserved():
raw = [{'text': 'Hello, world. It was 19 years ago.',
'words': ['Hello', ',', 'world', '.', 'It', 'was', '1', '9', 'years', 'ago', '.'],
'timestamp': [[i * 500, i * 500 + 400] for i in range(11)]}]
cues = aligned_cues(raw, 6000)
assert [c['text'] for c in cues] == ['Hello, world.', 'It was 19 years ago.']
assert ''.join(w['text'] for c in cues for w in c['words']) == raw[0]['text']
@pytest.mark.parametrize('broken', ['missing', 'partial', 'reverse', 'overlap', 'nan', 'beyond', 'text'])
def test_bad_alignment_never_fabricates_or_partially_writes(broken):
raw = fixture()
if broken == 'missing': raw[0].pop('timestamp')
if broken == 'partial': raw[0]['words'].pop()
if broken == 'reverse': raw[0]['timestamp'][0] = [2000, 1000]
if broken == 'overlap': raw[0]['timestamp'][1] = [1100, 1600]
if broken == 'nan': raw[0]['timestamp'][0][0] = float('nan')
if broken == 'beyond': raw[0]['timestamp'][-1] = [46000, 47000]
if broken == 'text': raw[0]['text'] += '丢失的正文'
with pytest.raises(ValueError):
aligned_cues(raw, 45100)
def test_multiple_vad_items_must_be_complete_and_monotonic():
first = fixture('你好。')
second = fixture('再见。')
for time in second[0]['timestamp']:
time[0] += 5000; time[1] += 5000
cues = aligned_cues(first + second, 10000)
assert [c['text'] for c in cues] == ['你好。', '再见。']
with pytest.raises(ValueError): aligned_cues(second + first, 10000)
def test_local_selection_has_no_connection_and_validates_model(monkeypatch, tmp_path):
selection = Transcription(provider='sensevoice_local', model='SenseVoiceSmall', connection_id='obsolete')
assert selection.connection_id is None
with pytest.raises(ValueError): Transcription(provider='sensevoice_local', model='base')
from backend.services import ai_model_settings
from backend.utils import speech_recognizer
monkeypatch.setattr(ai_model_settings, 'load', lambda: ModelSettings(transcription=selection))
calls = []
def transcribe(*args):
calls.append(args)
return tmp_path / 'out.srt'
monkeypatch.setattr(runtime, 'transcribe', transcribe)
monkeypatch.setattr(speech_recognizer, 'SpeechRecognizer', lambda: pytest.fail('must not load Whisper'))
assert speech_recognizer.generate_subtitle_for_video(tmp_path / 'video.mp4', model='tiny') == tmp_path / 'out.srt'
assert calls[0][2:] == ('auto', 0)
def test_cross_process_lock_rejects_uninstall_and_releases(tmp_path, monkeypatch):
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'sensevoice')
with runtime.operation():
with pytest.raises(RuntimeError, match='正在准备或转写'): runtime.uninstall()
assert runtime.uninstall()['status'] == 'not_installed'
def test_worker_uses_isolated_interpreter_not_parent_import_path(tmp_path, monkeypatch):
monkeypatch.setattr(runtime, 'root', lambda: tmp_path)
calls = []
def run(cmd, **kw):
calls.append(cmd)
return SimpleNamespace(returncode=0)
monkeypatch.setattr(runtime.subprocess, 'run', run)
runtime.worker('transcribe', tmp_path / 'r.json', tmp_path / 'a.wav')
assert calls[0][:2] == [sys.executable, '-S']
assert calls[0][2].endswith('sensevoice_worker.py')
assert '-m' not in calls[0]
def test_interrupted_install_is_retryable_after_restart(tmp_path, monkeypatch):
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'sensevoice')
runtime._state('installing')
assert runtime.status()['status'] == 'error'
with runtime.operation():
assert runtime.status()['status'] == 'installing'
def test_status_api_is_lightweight_and_install_conflict_is_explicit(tmp_path, monkeypatch):
from fastapi import FastAPI
from fastapi.testclient import TestClient
from backend.api.v1.speech_recognition import router
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'sensevoice')
app = FastAPI()
app.include_router(router)
with TestClient(app) as client:
assert client.get('/sensevoice/status').json()['status'] == 'not_installed'
with runtime.operation():
assert client.post('/sensevoice/prepare').status_code == 409
assert client.delete('/sensevoice').status_code == 409
def test_transcribe_srt_and_real_word_sidecar_without_optional_imports(tmp_path, monkeypatch):
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'runtime')
monkeypatch.setattr(runtime, 'status', lambda: {'status': 'ready'})
raw = fixture()
def extract(cmd, **kw):
with wave.open(cmd[-1], 'wb') as target:
target.setparams((1, 2, 16000, 0, 'NONE', 'not compressed'))
target.writeframes(b'\0\0' * (16000 * 45))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(runtime.subprocess, 'run', extract)
def worker(action, result, *args): result.write_text(json.dumps(raw), encoding='utf-8')
monkeypatch.setattr(runtime, 'worker', worker)
output = tmp_path / '字幕.srt'
runtime.transcribe(tmp_path / '源.mp4', output)
from backend.utils.word_timing import load_word_timing
assert load_word_timing(output)
before = output.read_bytes()
raw[0]['words'].pop()
from backend.utils.speech_recognizer import SpeechRecognitionError
with pytest.raises(SpeechRecognitionError): runtime.transcribe(tmp_path / '源.mp4', output)
assert output.read_bytes() == before
output.write_text('edited', encoding='utf-8')
assert load_word_timing(output) is None
def test_sensevoice_error_does_not_point_to_whisper(monkeypatch):
from backend.pipeline.failures import failure_from_speech_error
from backend.services import whisper_runtime
monkeypatch.setattr(whisper_runtime, 'get_status', lambda: pytest.fail('unrelated Whisper probe'))
failure = failure_from_speech_error('SenseVoice 词与时间戳未完整配对')
assert failure.code == 'subtitle_setup' and 'SenseVoice' in failure.hint
assert 'Whisper' not in failure.user_message()
+3
View File
@@ -830,6 +830,9 @@ def generate_subtitle_for_video(video_path: Path, output_path: Optional[Path] =
from backend.services.ai_model_settings import load as load_model_settings
model_settings = load_model_settings()
selection = model_settings.transcription if model_settings else None
if method == 'sensevoice_local' or method == 'auto' and selection and selection.provider == 'sensevoice_local':
from backend.services.sensevoice_runtime import transcribe
return transcribe(Path(video_path), output_path, language, timeout)
if method == 'auto' and selection and selection.provider == 'cloud':
from backend.services.cloud_transcription import transcribe
return transcribe(Path(video_path), output_path, model_settings, language, timeout)
+5 -3
View File
@@ -43,8 +43,10 @@ def valid_words(segment):
return False
def write_word_timing(srt: Path, segments, language=None):
payload = {'schema_version': 1, 'source': 'faster-whisper', 'language': language,
def write_word_timing(srt: Path, segments, language=None, source='faster-whisper'):
if source not in {'faster-whisper', 'sensevoice'}:
raise ValueError('Unsupported ASR word timing source')
payload = {'schema_version': 1, 'source': source, 'language': language,
'srt_sha256': hashlib.sha256(srt.read_bytes()).hexdigest(),
'segments': [s for s in segments if s.get('text', '').strip()]}
# A bad segment keeps sentence captions usable, but must not become karaoke.
@@ -65,7 +67,7 @@ def load_word_timing(srt: Path):
"""Return only complete, unchanged source captions; otherwise safe fallback."""
try:
payload = json.loads(sidecar_path(srt).read_text(encoding='utf-8'))
if payload.get('schema_version') != 1 or payload.get('source') != 'faster-whisper':
if payload.get('schema_version') != 1 or payload.get('source') not in {'faster-whisper', 'sensevoice'}:
return None
if payload.get('srt_sha256') != hashlib.sha256(srt.read_bytes()).hexdigest():
return None
+18 -1
View File
@@ -5,7 +5,7 @@
- 高光分析:获取可用模型后自动预选,也可手动更换。
- 画面理解:默认复用高光分析模型;高级设置中可指定另一个连接和模型。
- 封面生成:默认复用供应商和密钥,预选已适配的生图模型;没有适合的型号时使用视频截帧。高级设置中可以配置其他供应商。
- 字幕转写:以 Whisper 本地服务和模型选择呈现。一次“准备模型”操作自动完成必要组件安装及所选模型下载;安装下载即时执行,选择随设置保存。新导入任务使用所选型号,本地服务不自动回退到云端。
- 字幕转写:可选择 Whisper 或 SenseVoiceSmall 本地转写,也可单独绑定云端服务。一次“准备模型”操作完成必要组件安装及模型下载;安装下载即时执行,选择随设置保存。新导入任务使用所选型号,本地服务不自动回退到云端。
普通配置不需要命名或创建连接,也不使用配置弹窗。内部通过连接引用共享地址和密钥,并保留各供应商已填写的配置。按用途提供封面供应商切换与可选的补充画面模型;预置服务不显示接口地址,地址仅在自定义或本地服务中出现。只有未知的自定义模型显示能力选择。分析方式自动判断,原有显式偏好保留,重新选择分析模型后恢复自动方式。改变服务地址时必须重新填写已保存的密钥,避免把旧密钥自动发送给新地址。空输入保留已保存密钥。
@@ -54,6 +54,7 @@
| 提供商 | 核查到的 ASR | 本次可生成带时间戳字幕 |
| --- | --- | --- |
| 本地 | Whisper tiny/base/small/medium/large-v3 | 是,沿用本地运行时 |
| 本地 FunASR | SenseVoiceSmall | 是,消费 CTC 词级时间戳;按需安装,独立进程运行 |
| OpenAI | whisper-1、gpt-4o-transcribe-diarize、gpt-4o-transcribe、gpt-4o-mini-transcribe | 前两项;普通 transcribe 的纯文字输出不能替代字幕时间轴 |
| 阿里云百炼 | Qwen-Audio-3.1/3.0-ASR-Flash、Fun-ASR-Flash、Qwen-ASR-Filetrans、Fun-ASR、Paraformer | 同步 Flash 系列;SSE 收集所有已完成句子,毫秒转秒。Filetrans 需公网音频地址,暂不代建对象存储 |
| 无限星河 | 公开目录当前列出 qwen-audio-3.0-asr-flash-streaming / filetrans | 目录可预览,但其具体上传/时间戳协议尚未确认,标为未接入,不能选为可用默认值 |
@@ -74,3 +75,19 @@
云端任务以 16 kHz 单声道 PCM 每 180 秒分块(原始 5.76 MB,Base64 后低于 10 MB),用准确样本偏移合并时间轴。固定分块边界可能截断词语,是当前分段方式的限制。只有全任务成功才发布 SRT;请求失败不偷偷切换供应商或本地模型,不把凭证或服务端原始错误体写入日志。
新手流程:没有任何可用连接时,首页弹出一次性的「连接 AI 服务」对话框(`FirstRunSetup.tsx`,本会话内「稍后再说」后不再弹),内容与设置页 AI 服务一节完全相同,也可点「更多选项」进入设置页。首次空白配置默认:画面识别开、封面「AI 生成」且「参考视频画面」开,供应商分组把赞助伙伴放在「推荐」并保留合作说明;已有配置不强制改动。未连接 AI 服务时导入视频会被拦下并提示先连接。封面模型缺失时回落到视频截帧。模型使用单选,搜索无匹配时允许显式选择自定义型号。关闭画面识别后的连接测试也不发送图片。
## 本地 SenseVoice(#67)
设置 → 字幕转写 → 转写方式选择 **SenseVoice · 本地**,点击“准备模型”,就绪后保存设置。首次需联网下载 FunASR/PyTorch 组件和 SenseVoiceSmall/FSMN-VAD 模型,请预留至少 8 GB 磁盘空间(Windows 的真实安装验收约 5.8 GB,另需安装临时空间)。转写时使用已下载的本地路径,不上传音频、不请求云端转写、不需要 API Key。默认 Whisper 和已有选择保持原样。
这是 FunASR 1.3.14 的 SenseVoiceSmall 适配器,不是把 FunASR 工具箱中的所有模型都加入列表。支持中文、粤语、英语、日语、韩语及自动语言检测。首版使用 CPU,限制为两条计算线程;不加载 CAM++ 说话人分离或额外标点模型,也不承诺 issue 中的通用速度数字。模型产生的空格和标点会保留。高光分析仍使用另行配置的分析模型。
运行时位于数据目录 `sensevoice/runtime`,模型缓存位于 `sensevoice/models`。独立 Python 子进程导入这些组件,避免 PyTorch/NumPy 与 Whisper、API 后端依赖冲突。后台准备失败后可以重试;重启后可识别中断的准备操作。安装、删除和转写共享跨进程锁,忙碌时禁止修改组件。高级项可以删除组件及模型。
显式请求 `output_timestamp=True`,严格配对 `words` 与毫秒 `timestamp`,先恢复原文空格和标点,再按标点、42 字符、8 秒上限以及停顿生成字幕。词级信息写入带 SRT 内容校验的 `.words.json`,用于已有字幕对齐流程。缺失、不完整、倒退、重叠或明显超出音频的时间戳会中断任务并显示 SenseVoice 的处理建议;不会按文字比例伪造时间轴,也不会把一条粗 VAD 段当作精确字幕。失败不会覆盖已存在的字幕。
接入参考 [#67 讨论](https://github.com/zhouxiaoka/autoclip/issues/67)、[123mlly 的 feature/dev 分支](https://github.com/123mlly/autoclip/tree/feature/dev) 和 [LauraGPT 的时间戳修复 PR](https://github.com/123mlly/autoclip/pull/1)。词级聚合思路依据该 MIT 许可实现改写,保留项目 [MIT 许可证](../LICENSE);不搬入 fork 的其他改动和比例铺字幕兜底。
FunASR/SenseVoice 源码 MIT 与模型权重许可不同。权重使用前请阅读 [SenseVoiceSmall 模型卡及其许可链接](https://huggingface.co/FunAudioLLM/SenseVoiceSmall);安装包不内置或重新分发这些权重。硅基流动的云端 SenseVoice 型号仍需独立的云端时间戳适配,不能与这个本地实现混用。
结构化输出回归测试:`backend/tests/test_sensevoice.py`。真实语音验收:`.github/workflows/sensevoice.yml` 在三种系统的 Python 3.13 环境中实际安装、下载模型,然后在阻断网络的子进程内转写仓库公开访谈前 45 秒,检查完整词级信息、字幕顺序和边界、依赖隔离及卸载。该测试属于真实 ASR 验收,不代表长视频质量对照或桌面安装包真机验收。
+2 -2
View File
@@ -118,7 +118,7 @@ Building 和 Testing 现在是空的:更新日志里没有「已写完、还
**Researching**
- ASR 后端可插拔(#67,协议先于具体模型)
- 本地 ASR 模型增补(#67:SenseVoiceSmall 按需接入,Whisper 保持默认;通用插件协议继续评估)
- 切片质量对照(5 分钟和 60 分钟真视频)
- Step 3 评分后端可插拔
@@ -149,7 +149,7 @@ Building 和 Testing 现在是空的:更新日志里没有「已写完、还
- Apple 公证和 Windows 代码签名。应用内更新和崩溃上报已经在 v1.3.1,未签名仍会拦住安装。
- 首页和项目卡按设计系统重做。八语界面已经上线,这两处还是 Ant Design 默认件。
- CLI 真视频,以及竖屏三条预设的人工看片。
3. **识别和评分的可替换接口留在 Researching。** #67 在协议写下来之前不接具体厂商。账号仍等公证和代码签名完成,不因为界面语言已经上线就提前开。
3. **通用识别和评分插件接口继续评估。** #67 已按维护者 2026-09-30 的接入要求实现可选本地 SenseVoiceSmall,见 [PR #244](https://github.com/zhouxiaoka/autoclip/pull/244) 和 [模型配置](AI_MODEL_CONFIGURATION.md#本地-sensevoice67);现有转写入口负责统一路由,词级时间戳必须完整、单调且不超出音频。此交付不代表全部 FunASR 模型或通用插件框架已完成。账号仍等公证和代码签名完成,不因为界面语言已经上线就提前开。
4. **新的社区需求不插到上面这几件前面**,除非它就是其中一件的复现,例如切片结果为空。
Quackback 同时满足下面三条再立项,缺一条就继续用 GitHub:
+1 -1
View File
@@ -14,7 +14,7 @@ const enums: Record<string, readonly string[]> = {
provider: ['dashscope', 'openai', 'gemini', 'deepseek', 'seed', 'kimi', 'glm', 'grok', 'infistar', 'api88', 'compatible', 'ollama', 'lmstudio', 'other'],
source: ['live', 'cache', 'catalog'], outcome: ['completed', 'failed', 'blocked', 'unknown'],
resolution: ['created', 'reused'], material_origin: ['sample', 'user', 'unknown'],
analysis_mode: ['auto', 'subtitle', 'visual'], transcription_mode: ['whisper_local', 'cloud', 'unknown'],
analysis_mode: ['auto', 'subtitle', 'visual'], transcription_mode: ['whisper_local', 'sensevoice_local', 'cloud', 'unknown'],
cover_mode: ['ai', 'frame'], capability: ['text', 'multimodal', 'auto'],
}
export function safeExperience(value: Record<string, unknown>): Properties {
+1 -1
View File
@@ -78,7 +78,7 @@ export function buildFeedbackDraft(input: FeedbackDraftInput): FeedbackDraft {
stage: scrubText(input.stage, 80) || undefined,
errorMessage: scrubText(input.errorMessage, 500) || undefined,
errorCode: safeFeedbackErrorCode(input.errorCode),
transcriptionProvider: ['whisper_local', 'cloud'].includes(input.transcriptionProvider || '') ? input.transcriptionProvider : undefined,
transcriptionProvider: ['whisper_local', 'sensevoice_local', 'cloud'].includes(input.transcriptionProvider || '') ? input.transcriptionProvider : undefined,
transcriptionModel: scrubText(input.transcriptionModel, 80) || undefined,
version: scrubText(input.version, 40),
os: scrubText(input.os, 40),
@@ -1,4 +1,4 @@
import { useEffect } from 'react'
import { useEffect, useState } from 'react'
import { useTranslation } from 'react-i18next'
import { Select, Switch } from 'antd'
import { t } from '../../i18n'
@@ -9,6 +9,53 @@ import { useModelSettings, type ModelSettingsStore } from './useModelSettings'
import { connectionOf, isTextOnly, mainConnection, presetKey, type SaveIssue } from './modelSettingsLogic'
import ProviderFields from './ProviderFields'
import ModelPicker from './ModelPicker'
import api from '../../services/api'
type SenseVoiceStatus = { status: 'not_installed' | 'installing' | 'ready' | 'error'; message: string }
function SenseVoiceConfig() {
const [runtime, setRuntime] = useState<SenseVoiceStatus | null>(null)
const [error, setError] = useState('')
const [busy, setBusy] = useState(false)
useEffect(() => {
let active = true
const refresh = async () => {
try {
const value = await api.get<unknown, SenseVoiceStatus>('/sensevoice/status')
if (active) { setRuntime(value); setError('') }
} catch { if (active) setError(t('暂时无法读取本地模型状态。')) }
}
void refresh()
const timer = window.setInterval(() => void refresh(), 3000)
return () => { active = false; window.clearInterval(timer) }
}, [])
const act = async (remove = false) => {
setBusy(true)
try {
const value = remove ? await api.delete<unknown, SenseVoiceStatus>('/sensevoice')
: await api.post<unknown, SenseVoiceStatus>('/sensevoice/prepare')
setRuntime(value); setError('')
} catch { setError(t('操作失败,请稍后重试。')) }
finally { setBusy(false) }
}
const preparing = runtime?.status === 'installing'
return <div className="ac-rows">
<Row label={t('转写模型')} hint={t('支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。')}><span>SenseVoiceSmall</span></Row>
<Row label={runtime?.status === 'ready' ? t('模型已就绪') : t('准备本地模型')}
hint={runtime?.status === 'ready' ? t('保存设置后,新任务将使用这个模型。') : t('首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。')}>
{runtime?.status === 'ready' ? <StatusDot tone="ok" label={t('已就绪')} />
: preparing ? <StatusDot tone="accent" label={t('正在准备组件与模型…')} />
: <Btn size="sm" disabled={!runtime || busy} onClick={() => void act()}>{t('准备模型')}</Btn>}
</Row>
{(error || runtime?.status === 'error') && <p className="ac-note" role="alert">{error || t('模型准备失败,请检查网络和磁盘空间后重试。')}</p>}
<p className="ac-note">{t('仅安装所需模型,不启用说话人分离。')} <a href="https://huggingface.co/FunAudioLLM/SenseVoiceSmall" target="_blank" rel="noreferrer">{t('模型与许可说明')}</a></p>
{runtime?.status === 'ready' && <details className="ac-disclosure"><summary>{t('高级')}</summary>
<Row label={t('删除 SenseVoice 组件与模型')} hint={t('删除后需要重新下载才能使用。')}>
<Btn variant="danger" size="sm" disabled={busy} onClick={() => void act(true)}>{t('删除')}</Btn>
</Row>
</details>}
</div>
}
const ANCHORS: Record<string, string> = { model: 'ai-model', analysis: 'ai-model', vision: 'ai-model', speech: 'ai-speech', cover: 'ai-cover' }
const issueAnchor = (issue: SaveIssue) => issue.role === 'transcription' ? 'ai-speech' : issue.role === 'cover' ? 'ai-cover' : 'ai-model'
@@ -49,19 +96,20 @@ function TranscriptionSection({ m }: { m: ModelSettingsStore }) {
const { settings, lists, busy, listErrors } = m
if (!settings) return null
const cloud = settings.transcription?.provider === 'cloud'
const sensevoice = settings.transcription?.provider === 'sensevoice_local'
const connection = cloud ? connectionOf(settings, 'transcription') : undefined
const main = mainConnection(settings)
const options = [
{ label: t('本机运行'), options: [{ value: 'whisper_local', label: t('Whisper · 本地') }] },
{ label: t('本机运行'), options: [{ value: 'whisper_local', label: t('Whisper · 本地') }, { value: 'sensevoice_local', label: t('SenseVoice · 本地') }] },
...providerPickerOptions().map(group => ({ ...group, options: group.options.filter(p => ['openai', 'dashscope', 'infistar', 'api88', 'glm', 'compatible'].includes(p.value)) })).filter(group => group.options.length),
]
return <div id="ai-speech" className="ac-model-section">
<h3 className="ac-model-section-title">{t('字幕转写')}</h3>
<p className="ac-note">{t('视频没有字幕时,先把说话声转成字幕。默认在本机免费转写,不上传音频。')}</p>
<Row wide label={t('转写方式')} hint={t('本地转写免费,首次需下载模型;云端转写无需下载,按用量计费。')}>
<Select aria-label={t('转写方式')} style={{ width: '100%' }} value={cloud ? connection?.provider : 'whisper_local'} options={options}
onChange={value => value === 'whisper_local' ? m.setTranscriptionLocal(settings.transcription?.model && settings.transcription.provider === 'whisper_local' ? settings.transcription.model : 'base') : m.chooseProvider('transcription', value as ProviderKey)}
optionRender={option => <span>{option.value === 'whisper_local' ? t('Whisper · 本地') : PROVIDERS[option.value as ProviderKey]?.name}{PROVIDERS[option.value as ProviderKey]?.sponsor && <small style={{ marginLeft: 8, color: 'var(--sub)' }}>{t('赞助')} · {PROVIDERS[option.value as ProviderKey]?.sponsor?.offer}</small>}</span>} />
<Select aria-label={t('转写方式')} style={{ width: '100%' }} value={cloud ? connection?.provider : sensevoice ? 'sensevoice_local' : 'whisper_local'} options={options}
onChange={value => value === 'sensevoice_local' ? m.update({ transcription: { provider: 'sensevoice_local', model: 'SenseVoiceSmall' } }) : value === 'whisper_local' ? m.setTranscriptionLocal(settings.transcription?.model && settings.transcription.provider === 'whisper_local' ? settings.transcription.model : 'base') : m.chooseProvider('transcription', value as ProviderKey)}
optionRender={option => <span>{option.value === 'whisper_local' ? t('Whisper · 本地') : option.value === 'sensevoice_local' ? t('SenseVoice · 本地') : PROVIDERS[option.value as ProviderKey]?.name}{PROVIDERS[option.value as ProviderKey]?.sponsor && <small style={{ marginLeft: 8, color: 'var(--sub)' }}>{t('赞助')} · {PROVIDERS[option.value as ProviderKey]?.sponsor?.offer}</small>}</span>} />
</Row>
{cloud && connection ? <>
<ProviderFields connection={connection} ariaPrefix={t('转写')} hideProvider placement="settings_model"
@@ -71,7 +119,7 @@ function TranscriptionSection({ m }: { m: ModelSettingsStore }) {
<ModelPicker role="transcription" model={settings.transcription?.model || ''} connection={connection} list={lists[connection.id]} busy={!!busy[connection.id]} listError={listErrors[connection.id]} mode={settings.analysis_mode}
onChange={model => m.update({ transcription: { provider: 'cloud', connection_id: connection.id, model } })} onRefresh={() => void m.discover(connection, true)} />
</Row>
</> : <SpeechRecognitionConfig hideProvider selectedModel={settings.transcription?.model || 'base'} onModelChange={m.setTranscriptionLocal} />}
</> : sensevoice ? <SenseVoiceConfig /> : <SpeechRecognitionConfig hideProvider selectedModel={settings.transcription?.model || 'base'} onModelChange={m.setTranscriptionLocal} />}
</div>
}
@@ -20,7 +20,7 @@ export interface ModelSettings {
analysis: Assignment | null
vision: Assignment | null
cover: Assignment | null
transcription?: { provider: 'whisper_local' | 'cloud'; model: string; connection_id?: string; capability?: Capability } | null
transcription?: { provider: 'whisper_local' | 'sensevoice_local' | 'cloud'; model: string; connection_id?: string; capability?: Capability } | null
cover_enabled: boolean
allow_send_frame: boolean
analysis_mode: 'auto' | 'subtitle' | 'visual'
@@ -41,4 +41,4 @@ export const modelSettingsApi = {
export function bindingCapability(binding: Assignment | null, lists: Record<string, ModelList>): Capability {
if (!binding) return 'auto'
return binding.capability !== 'auto' ? binding.capability : lists[binding.connection_id]?.models.find(m => m.id === binding.model)?.capability || 'auto'
}
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "New users can receive trial credit via our referral link; live support is available on the platform.",
"待确认制作类型": "Waiting for production type confirmation",
"查看建议": "View suggestion",
"制作中 · 查看进度": "Creating · View progress"
"制作中 · 查看进度": "Creating · View progress",
"SenseVoice · 本地": "SenseVoice · Local",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Supports Mandarin, Cantonese, English, Japanese and Korean, with word timestamps for subtitles.",
"正在准备组件与模型…": "Preparing components and model…",
"模型准备失败,请检查网络和磁盘空间后重试。": "Model preparation failed. Check your network and disk space, then retry.",
"仅安装所需模型,不启用说话人分离。": "Installs only the required models; speaker diarization is disabled.",
"模型与许可说明": "Model and license details",
"删除 SenseVoice 组件与模型": "Delete SenseVoice components and models",
"删除后需要重新下载才能使用。": "Download again to use it after deletion.",
"操作失败,请稍后重试。": "Operation failed. Please try again later.",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "Internet access is required for the initial component and model download. Reserve at least 8 GB of disk space. Once ready, transcription does not upload audio."
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Los nuevos usuarios pueden recibir crédito de prueba mediante nuestro enlace; la plataforma ofrece atención humana.",
"待确认制作类型": "Pendiente de confirmar el tipo de producción",
"查看建议": "Ver sugerencia",
"制作中 · 查看进度": "Creando · Ver progreso"
"制作中 · 查看进度": "Creando · Ver progreso",
"SenseVoice · 本地": "SenseVoice · Local",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Admite mandarín, cantonés, inglés, japonés y coreano, con marcas de tiempo por palabra.",
"正在准备组件与模型…": "Preparando componentes y modelo…",
"模型准备失败,请检查网络和磁盘空间后重试。": "No se pudo preparar el modelo. Comprueba la conexión y el espacio en disco y reintenta.",
"仅安装所需模型,不启用说话人分离。": "Solo instala los modelos necesarios; la separación de hablantes está desactivada.",
"模型与许可说明": "Modelo y licencia",
"删除 SenseVoice 组件与模型": "Eliminar componentes y modelos de SenseVoice",
"删除后需要重新下载才能使用。": "Después de eliminarlo, tendrás que descargarlo de nuevo.",
"操作失败,请稍后重试。": "La operación falló. Inténtalo más tarde.",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "La descarga inicial de componentes y modelos requiere internet. Reserva al menos 8 GB de espacio en disco. Después, la transcripción no sube el audio."
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Les nouveaux utilisateurs peuvent recevoir un crédit d’essai via notre lien ; une assistance humaine est disponible sur la plateforme.",
"待确认制作类型": "Confirmation du type de production en attente",
"查看建议": "Voir la suggestion",
"制作中 · 查看进度": "Création · Voir la progression"
"制作中 · 查看进度": "Création · Voir la progression",
"SenseVoice · 本地": "SenseVoice · Local",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Prend en charge le mandarin, le cantonais, l’anglais, le japonais et le coréen, avec des horodatages par mot.",
"正在准备组件与模型…": "Préparation des composants et du modèle…",
"模型准备失败,请检查网络和磁盘空间后重试。": "La préparation du modèle a échoué. Vérifiez le réseau et l’espace disque, puis réessayez.",
"仅安装所需模型,不启用说话人分离。": "Seuls les modèles nécessaires sont installés ; la diarisation est désactivée.",
"模型与许可说明": "Modèle et licence",
"删除 SenseVoice 组件与模型": "Supprimer les composants et modèles SenseVoice",
"删除后需要重新下载才能使用。": "Après suppression, un nouveau téléchargement sera nécessaire.",
"操作失败,请稍后重试。": "L’opération a échoué. Réessayez plus tard.",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "Le téléchargement initial des composants et du modèle nécessite Internet. Prévoyez au moins 8 Go d’espace disque. Ensuite, la transcription ne téléverse pas l’audio."
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "専用リンクから新規登録すると体験クレジットを受け取れます。サイト内で有人サポートを利用できます。",
"待确认制作类型": "制作タイプの確認待ち",
"查看建议": "提案を見る",
"制作中 · 查看进度": "制作中 · 進捗を見る"
"制作中 · 查看进度": "制作中 · 進捗を見る",
"SenseVoice · 本地": "SenseVoice · ローカル",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "中国語、広東語、英語、日本語、韓国語に対応し、単語の時刻から字幕を生成します。",
"正在准备组件与模型…": "コンポーネントとモデルを準備中…",
"模型准备失败,请检查网络和磁盘空间后重试。": "モデルの準備に失敗しました。ネットワークと空き容量を確認して再試行してください。",
"仅安装所需模型,不启用说话人分离。": "必要なモデルのみをインストールし、話者分離は使用しません。",
"模型与许可说明": "モデルとライセンス",
"删除 SenseVoice 组件与模型": "SenseVoice のコンポーネントとモデルを削除",
"删除后需要重新下载才能使用。": "削除後は再ダウンロードが必要です。",
"操作失败,请稍后重试。": "操作に失敗しました。後でもう一度お試しください。",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "初回はコンポーネントとモデルのダウンロードにネット接続が必要です。8 GB 以上の空き容量を確保してください。準備後の文字起こしでは音声をアップロードしません。"
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "전용 링크로 신규 가입하면 체험 크레딧을 받을 수 있습니다. 사이트에서 상담원 지원을 제공합니다.",
"待确认制作类型": "제작 유형 확인 대기",
"查看建议": "제안 보기",
"制作中 · 查看进度": "제작 중 · 진행 상황 보기"
"制作中 · 查看进度": "제작 중 · 진행 상황 보기",
"SenseVoice · 本地": "SenseVoice · 로컬",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "중국어, 광둥어, 영어, 일본어, 한국어를 지원하며 단어 타임스탬프로 자막을 생성합니다.",
"正在准备组件与模型…": "구성 요소와 모델 준비 중…",
"模型准备失败,请检查网络和磁盘空间后重试。": "모델 준비에 실패했습니다. 네트워크와 디스크 공간을 확인한 후 다시 시도하세요.",
"仅安装所需模型,不启用说话人分离。": "필요한 모델만 설치하며 화자 분리는 사용하지 않습니다.",
"模型与许可说明": "모델 및 라이선스 정보",
"删除 SenseVoice 组件与模型": "SenseVoice 구성 요소와 모델 삭제",
"删除后需要重新下载才能使用。": "삭제 후 사용하려면 다시 다운로드해야 합니다.",
"操作失败,请稍后重试。": "작업에 실패했습니다. 나중에 다시 시도하세요.",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "처음에는 인터넷으로 구성 요소와 모델을 다운로드해야 합니다. 최소 8 GB의 디스크 공간을 확보하세요. 준비 후 전사 시 오디오를 업로드하지 않습니다."
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Novos usuários podem receber crédito de teste pelo nosso link; a plataforma oferece atendimento humano.",
"待确认制作类型": "Aguardando confirmação do tipo de produção",
"查看建议": "Ver sugestão",
"制作中 · 查看进度": "Criando · Ver progresso"
"制作中 · 查看进度": "Criando · Ver progresso",
"SenseVoice · 本地": "SenseVoice · Local",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Suporta mandarim, cantonês, inglês, japonês e coreano, com tempos por palavra.",
"正在准备组件与模型…": "Preparando componentes e modelo…",
"模型准备失败,请检查网络和磁盘空间后重试。": "Falha ao preparar o modelo. Verifique a rede e o espaço em disco e tente novamente.",
"仅安装所需模型,不启用说话人分离。": "Instala apenas os modelos necessários; a separação de falantes fica desativada.",
"模型与许可说明": "Modelo e licença",
"删除 SenseVoice 组件与模型": "Excluir componentes e modelos do SenseVoice",
"删除后需要重新下载才能使用。": "Após excluir, será necessário baixar novamente.",
"操作失败,请稍后重试。": "A operação falhou. Tente novamente mais tarde.",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "O download inicial dos componentes e modelos requer internet. Reserve pelo menos 8 GB de espaço em disco. Depois, a transcrição não envia o áudio."
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Новые пользователи могут получить пробный кредит по нашей ссылке; на платформе доступна поддержка операторов.",
"待确认制作类型": "Ожидается подтверждение типа монтажа",
"查看建议": "Посмотреть предложение",
"制作中 · 查看进度": "Создание · Посмотреть прогресс"
"制作中 · 查看进度": "Создание · Посмотреть прогресс",
"SenseVoice · 本地": "SenseVoice · Локально",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Поддерживает китайский, кантонский, английский, японский и корейский с временными метками слов.",
"正在准备组件与模型…": "Подготовка компонентов и модели…",
"模型准备失败,请检查网络和磁盘空间后重试。": "Не удалось подготовить модель. Проверьте сеть и место на диске и повторите попытку.",
"仅安装所需模型,不启用说话人分离。": "Устанавливаются только необходимые модели; разделение говорящих отключено.",
"模型与许可说明": "Модель и лицензия",
"删除 SenseVoice 组件与模型": "Удалить компоненты и модели SenseVoice",
"删除后需要重新下载才能使用。": "После удаления для использования потребуется повторная загрузка.",
"操作失败,请稍后重试。": "Операция не удалась. Повторите попытку позже.",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "Для первоначальной загрузки компонентов и модели нужен интернет. Освободите не менее 8 ГБ на диске. После подготовки аудио не загружается на сервер."
}
+11 -1
View File
@@ -1416,5 +1416,15 @@
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "新用户通过专属链接注册可获体验额度;站内提供人工客服。",
"待确认制作类型": "待确认制作类型",
"查看建议": "查看建议",
"制作中 · 查看进度": "制作中 · 查看进度"
"制作中 · 查看进度": "制作中 · 查看进度",
"SenseVoice · 本地": "SenseVoice · 本地",
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。",
"正在准备组件与模型…": "正在准备组件与模型…",
"模型准备失败,请检查网络和磁盘空间后重试。": "模型准备失败,请检查网络和磁盘空间后重试。",
"仅安装所需模型,不启用说话人分离。": "仅安装所需模型,不启用说话人分离。",
"模型与许可说明": "模型与许可说明",
"删除 SenseVoice 组件与模型": "删除 SenseVoice 组件与模型",
"删除后需要重新下载才能使用。": "删除后需要重新下载才能使用。",
"操作失败,请稍后重试。": "操作失败,请稍后重试。",
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。"
}
@@ -0,0 +1,54 @@
const { test } = require('node:test')
const assert = require('node:assert/strict')
const fs = require('node:fs')
const vm = require('node:vm')
const ts = require('typescript')
function loadSection() {
const jsx = (type, props) => ({ type, props })
const modules = {
react: { useEffect() {}, useState() {} },
'react/jsx-runtime': { jsx, jsxs: jsx },
'react-i18next': { useTranslation() {} },
antd: { Select: 'Select', Switch: 'Switch' },
'../../i18n': { t: s => s },
'../../ui': { Btn: 'Btn', Row: 'Row', Section: 'Section', Segmented: 'Segmented', StatusDot: 'StatusDot' },
'../../components/SpeechRecognitionConfig': { default: 'WhisperConfig' },
'./providers': { PROVIDERS: {}, providerPickerOptions: () => [] },
'./useModelSettings': {},
'./modelSettingsLogic': { connectionOf: () => undefined, mainConnection: () => undefined },
'./ProviderFields': {}, './ModelPicker': {}, '../../services/api': {},
}
const source = fs.readFileSync('src/features/settings/AIModelSettings.tsx', 'utf8') + '\nexport const testSection = TranscriptionSection;'
const code = ts.transpileModule(source, { compilerOptions: { module: ts.ModuleKind.CommonJS, jsx: ts.JsxEmit.ReactJSX } }).outputText
const module = { exports: {} }
vm.runInNewContext(code, { module, exports: module.exports, require: name => modules[name] || assert.fail(name) })
return module.exports.testSection
}
function findSelect(node) {
if (!node || typeof node !== 'object') return null
if (node.type === 'Select') return node
const children = node.props?.children
for (const child of Array.isArray(children) ? children.flat(5) : [children]) {
const found = findSelect(child)
if (found) return found
}
return null
}
test('the actual transcription selector offers SenseVoice and switches local engines without a cloud connection', () => {
const Section = loadSection()
let settings = { connections: [], transcription: { provider: 'whisper_local', model: 'small' } }
const m = { settings, lists: {}, busy: {}, listErrors: {},
update: patch => { settings = { ...settings, ...patch } },
setTranscriptionLocal: model => { settings.transcription = { provider: 'whisper_local', model } },
chooseProvider: () => assert.fail('must not route a local engine to a cloud provider') }
let select = findSelect(Section({ m }))
assert.ok(select.props.options[0].options.some(x => x.value === 'sensevoice_local'))
select.props.onChange('sensevoice_local')
assert.deepEqual(JSON.parse(JSON.stringify(settings.transcription)), { provider: 'sensevoice_local', model: 'SenseVoiceSmall' })
m.settings = settings
select = findSelect(Section({ m }))
assert.equal(select.props.value, 'sensevoice_local')
select.props.onChange('whisper_local')
assert.deepEqual(settings.transcription, { provider: 'whisper_local', model: 'base' })
})
+1 -1
View File
@@ -69,7 +69,7 @@ def normalize_event(raw: dict[str, Any]) -> dict[str, str] | None:
"stage": scrub(raw.get("stage"), 80),
"error_message": scrub(raw.get("error_message"), 500),
"error_code": error_code if ERROR_CODE_RE.fullmatch(error_code) else "",
"transcription_provider": transcription_provider if transcription_provider in {"whisper_local", "cloud"} else "",
"transcription_provider": transcription_provider if transcription_provider in {"whisper_local", "sensevoice_local", "cloud"} else "",
"transcription_model": scrub(raw.get("transcription_model"), 80),
"version": scrub(raw.get("app_version") or raw.get("version"), 40),
"os": scrub(raw.get("os"), 40),
+1 -1
View File
@@ -172,7 +172,7 @@ stdlib = set(sys.stdlib_module_names)
# cv2: speaker framing (backend/services/studio/framing.py), installed on demand like Whisper.
# numpy: imported only after loading Whisper, for the PyAV decoding fallback;
# faster-whisper's runtime installation provides it through its dependencies.
runtime_optional = {"faster_whisper", "ctranslate2", "huggingface_hub", "cv2", "numpy"}
runtime_optional = {"faster_whisper", "ctranslate2", "huggingface_hub", "cv2", "numpy", "funasr", "torch"}
mods = set()
for root, _, files in os.walk(backend_dir):
if '__pycache__' in root:
+48
View File
@@ -0,0 +1,48 @@
"""Remote-only acceptance: real runtime/model install and offline speech timing."""
import json
import os
from pathlib import Path
import subprocess
import sys
import tempfile
checkout = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(checkout))
base = Path(tempfile.mkdtemp(prefix='autoclip-sensevoice-acceptance-'))
os.environ['AUTOCLIP_DATA_DIR'] = str(base)
os.environ['AUTOCLIP_APP_DIR'] = str(base)
from backend.services import ai_model_settings as settings, sensevoice_runtime as runtime
from backend.utils.ffmpeg_utils import get_ffmpeg_path
from backend.utils.speech_recognizer import generate_subtitle_for_video
from backend.utils.word_timing import load_word_timing
runtime.prepare()
assert runtime.status()['status'] == 'ready', runtime.status()
video = base / '真实语音 sample.mp4'
subprocess.run([get_ffmpeg_path(), '-v', 'error', '-i', str(checkout / 'backend/assets/example/source.mp4'),
'-t', '45', '-c', 'copy', str(video)], check=True, timeout=60)
models = settings.ModelSettings(transcription=settings.Transcription(provider='sensevoice_local', model='SenseVoiceSmall'))
settings.path().write_text(models.model_dump_json(), encoding='utf-8')
# Prove the normal saved-selection entry point reaches isolated offline inference.
original_transcribe = runtime.transcribe
def offline(*args, **kw):
return original_transcribe(*args, **kw, deny_network=True)
runtime.transcribe = offline
subtitle = generate_subtitle_for_video(video, language='en')
cues = load_word_timing(subtitle)
assert cues and len(cues) > 2, cues
assert 'months' in subtitle.read_text(encoding='utf-8').lower()
assert all(0 <= c['start'] < c['end'] <= 45.2 and c['end'] - c['start'] <= 8.001 for c in cues)
assert all(a['end'] <= b['start'] for a, b in zip(cues, cues[1:]))
assert 'funasr' not in sys.modules and 'torch' not in sys.modules
size = sum(p.stat().st_size for p in runtime.root().rglob('*') if p.is_file())
Path('sensevoice-acceptance.json').write_text(json.dumps({
'platform': sys.platform, 'python': sys.version.split()[0], 'model': 'SenseVoiceSmall',
'source': 'public interview, first 45 seconds', 'install': 'passed',
'offline_network_denied': 'passed', 'normal_selection_route': 'passed',
'parent_dependency_isolation': 'passed', 'cues': len(cues),
'words': sum(len(c['words']) for c in cues), 'final_seconds': cues[-1]['end'],
'maximum_cue_seconds': max(c['end'] - c['start'] for c in cues), 'disk_bytes': size,
}, indent=2), encoding='utf-8')
runtime.uninstall()
assert runtime.status()['status'] == 'not_installed'