mirror of
https://github.com/zhouxiaoka/autoclip.git
synced 2026-10-02 02:34:34 +08:00
feat: add SenseVoiceSmall as a local subtitle transcription option (#244)
* feat: add optional isolated SenseVoice local transcription * test: verify local transcription selector switches SenseVoice and Whisper * docs: set realistic disk space expectation for SenseVoice preparation * docs: reconcile issue 67 delivery with historical ASR plans
This commit is contained in:
@@ -0,0 +1,45 @@
|
||||
name: Local SenseVoice acceptance
|
||||
on:
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'backend/services/sensevoice*.py'
|
||||
- 'backend/tests/test_sensevoice.py'
|
||||
- 'scripts/verify_sensevoice.py'
|
||||
- '.github/workflows/sensevoice.yml'
|
||||
permissions:
|
||||
contents: read
|
||||
jobs:
|
||||
real-transcription:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, windows-latest, macos-14]
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 35
|
||||
env:
|
||||
PYTHONUTF8: '1'
|
||||
PYTHONIOENCODING: utf-8
|
||||
HF_HUB_DISABLE_PROGRESS_BARS: '1'
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.13'
|
||||
cache: pip
|
||||
- run: python -m pip install -r requirements.txt
|
||||
- if: runner.os == 'Linux'
|
||||
run: sudo apt-get update && sudo apt-get install -y ffmpeg
|
||||
- if: runner.os == 'Windows'
|
||||
run: choco install ffmpeg --yes --no-progress
|
||||
- if: runner.os == 'macOS'
|
||||
run: brew install ffmpeg
|
||||
- run: python -m pytest backend/tests/test_sensevoice.py -q
|
||||
- name: Install optional runtime, download models, transcribe real speech offline
|
||||
run: python scripts/verify_sensevoice.py
|
||||
- uses: actions/upload-artifact@v4
|
||||
if: always()
|
||||
with:
|
||||
name: sensevoice-${{ matrix.os }}
|
||||
path: sensevoice-acceptance.json
|
||||
if-no-files-found: ignore
|
||||
@@ -5,6 +5,8 @@
|
||||
|
||||
## 当前交付与验证范围
|
||||
|
||||
- **2026-09-30 本地 SenseVoice 增补(PR #244,未发布)**:维护者要求接入 #67 后,设置页增加 SenseVoiceSmall 按需准备与选择;保存后由现有转写入口路由,独立 CPU 进程运行 FunASR,不上传音频、不改变 Whisper 默认值。三系统 Python 3.13 的真实 45 秒访谈离线验收通过,均生成 21 条字幕、157 个词级时间戳,最长约 3.36 秒;窗口安装包尚未做这项新增能力的真机验收。Windows 实装组件与模型约 5.8 GB,准备前建议预留 8 GB。详见 [配置与边界](docs/AI_MODEL_CONFIGURATION.md#本地-sensevoice67)。下方“#67 仍调研、先等通用协议”的文字是历史排期;通用插件框架仍是后续工作,本次具体本地转写能力已经实现。
|
||||
|
||||
- 1.4 已包含统一 Studio 导入→推荐/确认→制作→编辑→导出;默认字幕分析,视觉调用须遵守用户选择与确认边界。未确认不开始正式制作。
|
||||
- macOS 原生导入至保存、受控旧版本升级/回退已有证据,见 [RC 验收](docs/RC_ACCEPTANCE_1_4.md) 与 [发布记录](docs/RELEASE_1_4.md)。这些记录不等于第二台全新 Mac 或真实用户数据库迁移验收。
|
||||
- Windows 构建已有记录,真机安装/导入/保存仍待验收。多游戏完整观感、事件边界和推广差异化也仍待验收;不承诺广告投放效果。
|
||||
|
||||
@@ -43,6 +43,29 @@ def get_speech_recognizer() -> SpeechRecognizer:
|
||||
|
||||
# ===== Whisper 运行时(按需安装)=====
|
||||
|
||||
@router.get("/sensevoice/status")
|
||||
async def sensevoice_status():
|
||||
from backend.services import sensevoice_runtime
|
||||
return sensevoice_runtime.status()
|
||||
|
||||
|
||||
@router.post("/sensevoice/prepare")
|
||||
async def sensevoice_prepare():
|
||||
from backend.services import sensevoice_runtime
|
||||
try:
|
||||
return sensevoice_runtime.start_prepare()
|
||||
except RuntimeError as exc:
|
||||
raise HTTPException(status_code=409, detail=str(exc)) from exc
|
||||
|
||||
|
||||
@router.delete("/sensevoice")
|
||||
async def sensevoice_uninstall():
|
||||
from backend.services import sensevoice_runtime
|
||||
try:
|
||||
return sensevoice_runtime.uninstall()
|
||||
except RuntimeError as exc:
|
||||
raise HTTPException(status_code=409, detail=str(exc)) from exc
|
||||
|
||||
@router.get("/whisper/runtime-status")
|
||||
async def whisper_runtime_status():
|
||||
"""Whisper 运行时安装状态(前端轮询)。"""
|
||||
|
||||
@@ -168,6 +168,12 @@ def timeline_failure_from_report(topic_count: int, report: dict) -> PipelineFail
|
||||
def missing_subtitle_failure() -> PipelineFailure:
|
||||
"""视频没有字幕,自动转写也没留下 srt。按当前 Whisper 状态区分下一步。"""
|
||||
from backend.services import whisper_runtime
|
||||
from backend.services.ai_model_settings import load
|
||||
settings = load()
|
||||
if settings and settings.transcription and settings.transcription.provider == 'sensevoice_local':
|
||||
return PipelineFailure('SUBTITLE', '没有字幕可分析:SenseVoice 本次没有生成可用字幕。',
|
||||
'到「设置 → 转写」检查 SenseVoiceSmall 是否就绪,或导入 .srt 字幕后重试。',
|
||||
code='subtitle_setup')
|
||||
|
||||
status = whisper_runtime.get_status()
|
||||
state = status.get("status")
|
||||
@@ -196,6 +202,10 @@ def missing_subtitle_failure() -> PipelineFailure:
|
||||
def failure_from_speech_error(message: str) -> PipelineFailure:
|
||||
"""转写异常收成同一套失败码,不再把「设置 → 语音识别」和「设置 → 转写」叠在一起。"""
|
||||
text = (message or "").strip().replace("设置 → 语音识别", "设置 → 转写")
|
||||
if 'SenseVoice' in text:
|
||||
return PipelineFailure('SUBTITLE', text,
|
||||
'' if '设置 → 转写' in text else '到「设置 → 转写」检查 SenseVoiceSmall,或导入 .srt 字幕。',
|
||||
code='subtitle_setup')
|
||||
from backend.services import whisper_runtime
|
||||
|
||||
state = whisper_runtime.get_status().get("status")
|
||||
|
||||
@@ -76,7 +76,7 @@ class Assignment(BaseModel):
|
||||
|
||||
|
||||
class Transcription(BaseModel):
|
||||
provider: Literal['whisper_local', 'cloud'] = 'whisper_local'
|
||||
provider: Literal['whisper_local', 'sensevoice_local', 'cloud'] = 'whisper_local'
|
||||
model: str = Field(default='base', min_length=1, max_length=200)
|
||||
connection_id: str | None = None
|
||||
|
||||
@@ -89,6 +89,10 @@ class Transcription(BaseModel):
|
||||
if self.model not in {'tiny', 'base', 'small', 'medium', 'large', 'large-v3'}:
|
||||
raise ValueError('请选择本地 Whisper 模型')
|
||||
self.connection_id = None
|
||||
elif self.provider == 'sensevoice_local':
|
||||
if self.model != 'SenseVoiceSmall':
|
||||
raise ValueError('请选择本地 SenseVoiceSmall 模型')
|
||||
self.connection_id = None
|
||||
elif not self.connection_id:
|
||||
raise ValueError('请选择转写供应商')
|
||||
return self
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
"""SenseVoice CTC alignment, in milliseconds; never invent subtitle timing.
|
||||
|
||||
Adapted from LauraGPT's MIT-licensed timestamp fix for 123mlly/autoclip PR #1.
|
||||
We require complete CTC alignment instead of its coarse/proportional fallbacks.
|
||||
"""
|
||||
import math
|
||||
import re
|
||||
|
||||
|
||||
def clean(text):
|
||||
return re.sub(r"<\|[^|]*\|>", "", text).replace("▁", " ").strip()
|
||||
|
||||
|
||||
def aligned_cues(result, duration_ms, max_chars=42, max_duration_ms=8000):
|
||||
"""Preserve all source text and real word times, across VAD segments."""
|
||||
if not isinstance(result, list) or not result or duration_ms <= 0:
|
||||
raise ValueError('SenseVoice 未返回可用的词级时间戳')
|
||||
cues, current = [], []
|
||||
previous_end = 0
|
||||
|
||||
def flush():
|
||||
if current:
|
||||
cues.append({'start': current[0]['start'], 'end': current[-1]['end'],
|
||||
'text': ''.join(w['text'] for w in current).strip(),
|
||||
'words': list(current)})
|
||||
current.clear()
|
||||
|
||||
for item in result:
|
||||
if not isinstance(item, dict) or not isinstance(item.get('text'), str):
|
||||
raise ValueError('SenseVoice 返回了无效的转写结果')
|
||||
source = clean(item['text'])
|
||||
if not source:
|
||||
continue
|
||||
words, times = item.get('words'), item.get('timestamp')
|
||||
if not isinstance(words, list) or not isinstance(times, list) or not words or len(words) != len(times):
|
||||
raise ValueError('SenseVoice 词与时间戳未完整配对,请重新准备模型或改用 Whisper')
|
||||
cursor, aligned = 0, []
|
||||
for word, time in zip(words, times):
|
||||
if not isinstance(word, str) or not clean(word) or not isinstance(time, (list, tuple)) or len(time) != 2:
|
||||
raise ValueError('SenseVoice 返回了无效的词级时间戳')
|
||||
if not all(isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t) for t in time):
|
||||
raise ValueError('SenseVoice 返回了无效的词级时间戳')
|
||||
start, end = (int(t) for t in time) # FunASR lists are always milliseconds.
|
||||
# CTC has a 60ms frame step; only tolerate final-frame rounding.
|
||||
if not 0 <= previous_end <= start < end <= duration_ms + 60:
|
||||
raise ValueError('SenseVoice 时间戳倒退、重叠或超出音频范围')
|
||||
end = min(end, duration_ms)
|
||||
if end <= start:
|
||||
raise ValueError('SenseVoice 时间戳超出音频范围')
|
||||
token = clean(word)
|
||||
position = source.find(token, cursor)
|
||||
if position < 0 or any(c.isalnum() for c in source[cursor:position]):
|
||||
raise ValueError('SenseVoice 原文与词级时间戳无法完整对齐')
|
||||
stop = position + len(token)
|
||||
aligned.append({'text': source[cursor:stop], 'start': start / 1000, 'end': end / 1000})
|
||||
cursor, previous_end = stop, end
|
||||
if any(c.isalnum() for c in source[cursor:]):
|
||||
raise ValueError('SenseVoice 原文与词级时间戳无法完整对齐')
|
||||
aligned[-1]['text'] += source[cursor:]
|
||||
for word in aligned:
|
||||
if current and (len(''.join(w['text'] for w in current) + word['text']) > max_chars
|
||||
or (word['end'] - current[0]['start']) * 1000 > max_duration_ms
|
||||
or word['start'] - current[-1]['end'] > 1.2):
|
||||
flush()
|
||||
if len(word['text'].strip()) > max_chars or (word['end'] - word['start']) * 1000 > max_duration_ms:
|
||||
raise ValueError('SenseVoice 单词时间戳异常,无法生成可读字幕')
|
||||
current.append(word)
|
||||
if re.search(r'[。!?;!?.;]$', word['text'].strip()):
|
||||
flush()
|
||||
flush()
|
||||
if not cues:
|
||||
raise ValueError('SenseVoice 未识别出可用人声')
|
||||
return cues
|
||||
@@ -0,0 +1,212 @@
|
||||
"""Optional, isolated CPU SenseVoice runtime. No FunASR imports in this process."""
|
||||
from contextlib import contextmanager
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import threading
|
||||
import wave
|
||||
|
||||
from backend.core.path_utils import get_data_directory
|
||||
from backend.utils.ffmpeg_utils import get_ffmpeg_path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
MODEL = 'SenseVoiceSmall'
|
||||
VERSION = 1
|
||||
|
||||
|
||||
def root():
|
||||
return get_data_directory() / 'sensevoice'
|
||||
|
||||
|
||||
def packages():
|
||||
cpu = '+cpu' if sys.platform != 'darwin' else ''
|
||||
return ['funasr==1.3.14', f'torch==2.8.0{cpu}', f'torchaudio==2.8.0{cpu}',
|
||||
'setuptools==80.9.0', 'huggingface-hub<1', 'numpy<3']
|
||||
|
||||
|
||||
def status():
|
||||
try:
|
||||
value = json.loads((root() / 'status.json').read_text(encoding='utf-8'))
|
||||
except (OSError, ValueError):
|
||||
value = {'status': 'not_installed', 'message': ''}
|
||||
# Validate paths without importing a large optional runtime on every poll.
|
||||
try:
|
||||
ready = json.loads((root() / 'ready.json').read_text(encoding='utf-8'))
|
||||
valid = (ready['version'] == VERSION and (root() / 'runtime' / 'funasr' / '__init__.py').exists()
|
||||
and all(Path(p).resolve().is_relative_to((root() / 'models').resolve())
|
||||
and (Path(p) / 'model.pt').exists() for p in ready['paths'].values())
|
||||
and set(ready['paths']) == {'model', 'vad_model'})
|
||||
except (OSError, ValueError, KeyError, TypeError):
|
||||
valid = False
|
||||
if valid and value['status'] != 'installing':
|
||||
value = {'status': 'ready', 'message': ''}
|
||||
elif value.get('status') == 'ready':
|
||||
value = {'status': 'not_installed', 'message': '模型文件缺失,请重新准备模型'}
|
||||
if value.get('status') == 'installing':
|
||||
try:
|
||||
with operation():
|
||||
value = {'status': 'error', 'message': '上次模型准备被中断,请重试'}
|
||||
except RuntimeError:
|
||||
pass
|
||||
return {**value, 'model': MODEL}
|
||||
|
||||
|
||||
def _state(state, message=''):
|
||||
root().mkdir(parents=True, exist_ok=True)
|
||||
target = root() / 'status.json'
|
||||
pending = target.with_suffix('.tmp')
|
||||
pending.write_text(json.dumps({'status': state, 'message': message}, ensure_ascii=False), encoding='utf-8')
|
||||
os.replace(pending, target)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def operation():
|
||||
"""Cross-process lock: API installs and Celery inference share the data dir."""
|
||||
base = root()
|
||||
base.mkdir(parents=True, exist_ok=True)
|
||||
with open(base.parent / 'sensevoice.lock', 'a+b') as handle:
|
||||
handle.seek(0)
|
||||
try:
|
||||
if sys.platform == 'win32':
|
||||
import msvcrt
|
||||
if handle.read(1) == b'':
|
||||
handle.write(b'0')
|
||||
handle.flush()
|
||||
handle.seek(0)
|
||||
msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1)
|
||||
else:
|
||||
import fcntl
|
||||
fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
except OSError:
|
||||
raise RuntimeError('SenseVoice 正在准备或转写,请完成后重试') from None
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
if sys.platform == 'win32':
|
||||
handle.seek(0)
|
||||
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
|
||||
else:
|
||||
fcntl.flock(handle, fcntl.LOCK_UN)
|
||||
|
||||
|
||||
def worker(action, result, audio=None, language='auto', timeout=None, deny_network=False):
|
||||
command = [sys.executable, '-S', str(Path(__file__).with_name('sensevoice_worker.py')),
|
||||
action, '--root', str(root()), '--result', str(result), '--language', language]
|
||||
if audio:
|
||||
command += ['--audio', str(audio)]
|
||||
if deny_network:
|
||||
command += ['--deny-network']
|
||||
# Logs can contain private text. Keep them in the local diagnostic log only.
|
||||
with tempfile.TemporaryFile() as log:
|
||||
completed = subprocess.run(command, stdout=log, stderr=log, timeout=timeout,
|
||||
env={**os.environ, 'PYTHONUTF8': '1', 'PYTHONIOENCODING': 'utf-8'})
|
||||
if completed.returncode:
|
||||
log.seek(max(0, log.tell() - 12000))
|
||||
logger.error('SenseVoice worker failed: %s', log.read().decode('utf-8', errors='replace'))
|
||||
raise RuntimeError('SenseVoice 模型运行失败,请到「设置 → 转写」重新准备模型;仍失败时附上脱敏日志')
|
||||
|
||||
|
||||
def _prepare():
|
||||
_state('installing', '正在安装组件,首次下载可能需要几分钟')
|
||||
try:
|
||||
(root() / 'ready.json').unlink(missing_ok=True)
|
||||
runtime = root() / 'runtime'
|
||||
shutil.rmtree(runtime, ignore_errors=True)
|
||||
command = [sys.executable, '-m', 'pip', 'install', '--target', str(runtime),
|
||||
'--only-binary', 'torch,torchaudio', *packages()]
|
||||
if sys.platform != 'darwin':
|
||||
command += ['--extra-index-url', 'https://download.pytorch.org/whl/cpu']
|
||||
with tempfile.TemporaryFile() as log:
|
||||
completed = subprocess.run(command, stdout=log, stderr=log, timeout=1800,
|
||||
env={**os.environ, 'PIP_PROGRESS_BAR': 'off', 'PYTHONIOENCODING': 'utf-8'})
|
||||
if completed.returncode:
|
||||
log.seek(max(0, log.tell() - 12000))
|
||||
logger.error('SenseVoice pip failed: %s', log.read().decode('utf-8', errors='replace'))
|
||||
raise RuntimeError('SenseVoice 组件安装失败,请检查网络和磁盘空间后重试')
|
||||
_state('installing', '正在下载并检查 SenseVoiceSmall 模型')
|
||||
result = root() / 'prepare-result.json'
|
||||
worker('prepare', result, timeout=1800)
|
||||
value = json.loads(result.read_text(encoding='utf-8'))
|
||||
value['version'] = VERSION
|
||||
ready = root() / 'ready.json'
|
||||
pending = ready.with_suffix('.tmp')
|
||||
pending.write_text(json.dumps(value), encoding='utf-8')
|
||||
os.replace(pending, ready)
|
||||
result.unlink(missing_ok=True)
|
||||
_state('ready')
|
||||
except Exception as exc:
|
||||
_state('error', str(exc) if isinstance(exc, RuntimeError) else 'SenseVoice 准备失败,请检查网络后重试')
|
||||
raise
|
||||
|
||||
|
||||
def prepare():
|
||||
with operation():
|
||||
_prepare()
|
||||
|
||||
|
||||
def start_prepare():
|
||||
lock = operation()
|
||||
lock.__enter__()
|
||||
def run():
|
||||
try:
|
||||
_prepare()
|
||||
except Exception:
|
||||
logger.exception('SenseVoice preparation failed')
|
||||
finally:
|
||||
lock.__exit__(None, None, None)
|
||||
try:
|
||||
_state('installing', '正在准备组件…')
|
||||
threading.Thread(target=run, daemon=True).start()
|
||||
except Exception:
|
||||
lock.__exit__(None, None, None)
|
||||
raise
|
||||
return status()
|
||||
|
||||
|
||||
def uninstall():
|
||||
with operation():
|
||||
shutil.rmtree(root())
|
||||
return status()
|
||||
|
||||
|
||||
def transcribe(video, output=None, language='auto', timeout=0, *, deny_network=False):
|
||||
from backend.services.sensevoice_alignment import aligned_cues
|
||||
from backend.utils.speech_recognizer import SpeechRecognitionError, SpeechRecognizer
|
||||
from backend.utils.word_timing import write_word_timing
|
||||
language = {'zh-TW': 'zh', 'en-US': 'en', 'en-GB': 'en'}.get(language, language)
|
||||
if language not in {'auto', 'zh', 'en', 'ja', 'ko', 'yue'}:
|
||||
raise SpeechRecognitionError('SenseVoiceSmall 支持中文、粤语、英语、日语和韩语,请选择支持的语言或自动检测')
|
||||
output = Path(output) if output else Path(video).with_suffix('.srt')
|
||||
try:
|
||||
with operation(), tempfile.TemporaryDirectory(prefix='autoclip-sensevoice-') as temporary:
|
||||
if status()['status'] != 'ready':
|
||||
raise RuntimeError('SenseVoice 尚未就绪,请到「设置 → 转写」准备 SenseVoiceSmall 模型')
|
||||
audio, result = Path(temporary) / 'audio.wav', Path(temporary) / 'result.json'
|
||||
extracted = subprocess.run([get_ffmpeg_path(), '-nostdin', '-v', 'error', '-y', '-i', str(video),
|
||||
'-vn', '-ac', '1', '-ar', '16000', '-c:a', 'pcm_s16le', str(audio)],
|
||||
capture_output=True, timeout=timeout or 300)
|
||||
if extracted.returncode:
|
||||
raise RuntimeError('SenseVoice 无法读取视频音轨,请确认视频包含可播放的人声')
|
||||
with wave.open(str(audio), 'rb') as source:
|
||||
duration_ms = source.getnframes() * 1000 // source.getframerate()
|
||||
worker('transcribe', result, audio, language, timeout or None, deny_network)
|
||||
cues = aligned_cues(json.loads(result.read_text(encoding='utf-8')), duration_ms)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
with tempfile.NamedTemporaryFile(mode='w', encoding='utf-8', dir=output.parent, delete=False) as stream:
|
||||
pending = Path(stream.name)
|
||||
stream.write(SpeechRecognizer._segments_to_srt(cues))
|
||||
try:
|
||||
os.replace(pending, output)
|
||||
write_word_timing(output, cues, language, source='sensevoice')
|
||||
finally:
|
||||
pending.unlink(missing_ok=True)
|
||||
return output
|
||||
except subprocess.TimeoutExpired:
|
||||
raise SpeechRecognitionError('SenseVoice 转写超时,请缩短视频或增加转写超时时间') from None
|
||||
except (RuntimeError, ValueError, OSError) as exc:
|
||||
raise SpeechRecognitionError(str(exc)) from exc
|
||||
@@ -0,0 +1,61 @@
|
||||
"""Standalone worker: launched with -S, importing only the optional runtime.
|
||||
|
||||
Do not import backend here. PyTorch/NumPy must never enter the API process.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument('action', choices=['prepare', 'transcribe'])
|
||||
parser.add_argument('--root', type=Path, required=True)
|
||||
parser.add_argument('--result', type=Path, required=True)
|
||||
parser.add_argument('--audio', type=Path)
|
||||
parser.add_argument('--language', default='auto')
|
||||
parser.add_argument('--deny-network', action='store_true') # acceptance only
|
||||
args = parser.parse_args()
|
||||
runtime = args.root / 'runtime'
|
||||
sys.path.insert(0, str(runtime))
|
||||
os.environ.update(OMP_NUM_THREADS='2', MKL_NUM_THREADS='2', NUMBA_NUM_THREADS='2',
|
||||
HF_HOME=str(args.root / 'models'), HF_HUB_DISABLE_PROGRESS_BARS='1')
|
||||
if args.action == 'transcribe':
|
||||
os.environ.update(HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1', MODELSCOPE_OFFLINE='1')
|
||||
if args.deny_network:
|
||||
import socket
|
||||
def denied(*a, **kw):
|
||||
raise RuntimeError('offline acceptance forbids network access')
|
||||
socket.socket.connect = denied
|
||||
socket.create_connection = denied
|
||||
import torch
|
||||
import funasr
|
||||
assert Path(torch.__file__).resolve().is_relative_to(runtime.resolve())
|
||||
assert Path(funasr.__file__).resolve().is_relative_to(runtime.resolve())
|
||||
torch.set_num_threads(2)
|
||||
from funasr import AutoModel
|
||||
if args.action == 'prepare':
|
||||
from huggingface_hub import snapshot_download
|
||||
paths = {key: snapshot_download(repo_id=repo, cache_dir=str(args.root / 'models' / 'hub'),
|
||||
allow_patterns=['*.json', '*.yaml', '*.txt', '*.model', '*.mvn', 'model.pt'])
|
||||
for key, repo in [('model', 'FunAudioLLM/SenseVoiceSmall'), ('vad_model', 'funasr/fsmn-vad')]}
|
||||
else:
|
||||
paths = json.loads((args.root / 'ready.json').read_text(encoding='utf-8'))['paths']
|
||||
if any(not Path(p).resolve().is_relative_to((args.root / 'models').resolve()) for p in paths.values()):
|
||||
raise ValueError('Invalid model cache path')
|
||||
model = AutoModel(**paths, hub='hf', device='cpu', ncpu=2, disable_update=True,
|
||||
disable_pbar=True, trust_remote_code=False,
|
||||
vad_kwargs={'max_single_segment_time': 30000})
|
||||
if args.action == 'prepare':
|
||||
result = {'paths': paths}
|
||||
else:
|
||||
result = model.generate(input=str(args.audio), cache={}, language=args.language,
|
||||
use_itn=True, output_timestamp=True, batch_size_s=30,
|
||||
merge_vad=False, disable_pbar=True)
|
||||
args.result.write_text(json.dumps(result, ensure_ascii=False, allow_nan=False), encoding='utf-8')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -216,8 +216,10 @@ def _generate_import_subtitle(task, project_id: str, video_path: str):
|
||||
|
||||
logger.info(f"使用语音转写配置 - 方法: {speech_config.method}")
|
||||
|
||||
if models and models.transcription and models.transcription.provider == "cloud":
|
||||
generated_subtitle = generate_subtitle_for_video(Path(video_path), method="auto")
|
||||
if models and models.transcription and models.transcription.provider in {"cloud", "sensevoice_local"}:
|
||||
generated_subtitle = generate_subtitle_for_video(
|
||||
Path(video_path), method="auto", language=speech_config.whisper_config.language,
|
||||
timeout=speech_config.whisper_config.timeout)
|
||||
elif speech_config.method == "whisper_local":
|
||||
model = configured_whisper_model(speech_config.whisper_config.model_name)
|
||||
language = speech_config.whisper_config.language
|
||||
|
||||
@@ -29,6 +29,11 @@ def test_missing_mandatory_dependency_still_blocks_bundle(tmp_path):
|
||||
assert "- openai" in result.stdout
|
||||
|
||||
|
||||
def test_isolated_sensevoice_worker_dependencies_are_not_bundled(tmp_path):
|
||||
result = run_guard(tmp_path, "def worker():\n import torch\n import funasr\n import huggingface_hub\n")
|
||||
assert result.returncode == 0, result.stdout + result.stderr
|
||||
|
||||
|
||||
def test_bundle_local_modules_resolve(tmp_path):
|
||||
result = run_guard(tmp_path, "import backend\nimport json\n")
|
||||
assert result.returncode == 0, result.stdout + result.stderr
|
||||
|
||||
@@ -0,0 +1,157 @@
|
||||
"""CTC timing, complete text, route selection, dependency isolation and locking."""
|
||||
import json
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
import wave
|
||||
|
||||
import pytest
|
||||
|
||||
from backend.services.sensevoice_alignment import aligned_cues
|
||||
from backend.services import sensevoice_runtime as runtime
|
||||
from backend.services.ai_model_settings import ModelSettings, Transcription
|
||||
|
||||
|
||||
def fixture(text='这是第一段。这里是第二段。最后一段用于验证音频末尾。'):
|
||||
words = list(text)
|
||||
return [{'text': '<|zh|><|NEUTRAL|>' + text, 'words': words,
|
||||
'timestamp': [[1050 + i * 800, 1550 + i * 800] for i in range(len(words))],
|
||||
'sentence_info': [{'sentence': text, 'start': 0, 'end': 30010}]}]
|
||||
|
||||
|
||||
def test_ctc_avoids_coarse_vad_cues_and_preserves_all_text():
|
||||
raw = fixture()
|
||||
cues = aligned_cues(raw, 45100)
|
||||
assert len(cues) > 2
|
||||
assert ''.join(c['text'] for c in cues) == raw[0]['text'].split('>')[-1]
|
||||
assert cues[0]['start'] == 1.05
|
||||
assert cues[-1]['end'] == raw[0]['timestamp'][-1][1] / 1000
|
||||
from backend.utils.word_timing import valid_words
|
||||
assert all(valid_words(c) and c['end'] - c['start'] <= 8 for c in cues)
|
||||
|
||||
|
||||
def test_spaces_numbers_and_punctuation_are_preserved():
|
||||
raw = [{'text': 'Hello, world. It was 19 years ago.',
|
||||
'words': ['Hello', ',', 'world', '.', 'It', 'was', '1', '9', 'years', 'ago', '.'],
|
||||
'timestamp': [[i * 500, i * 500 + 400] for i in range(11)]}]
|
||||
cues = aligned_cues(raw, 6000)
|
||||
assert [c['text'] for c in cues] == ['Hello, world.', 'It was 19 years ago.']
|
||||
assert ''.join(w['text'] for c in cues for w in c['words']) == raw[0]['text']
|
||||
|
||||
|
||||
@pytest.mark.parametrize('broken', ['missing', 'partial', 'reverse', 'overlap', 'nan', 'beyond', 'text'])
|
||||
def test_bad_alignment_never_fabricates_or_partially_writes(broken):
|
||||
raw = fixture()
|
||||
if broken == 'missing': raw[0].pop('timestamp')
|
||||
if broken == 'partial': raw[0]['words'].pop()
|
||||
if broken == 'reverse': raw[0]['timestamp'][0] = [2000, 1000]
|
||||
if broken == 'overlap': raw[0]['timestamp'][1] = [1100, 1600]
|
||||
if broken == 'nan': raw[0]['timestamp'][0][0] = float('nan')
|
||||
if broken == 'beyond': raw[0]['timestamp'][-1] = [46000, 47000]
|
||||
if broken == 'text': raw[0]['text'] += '丢失的正文'
|
||||
with pytest.raises(ValueError):
|
||||
aligned_cues(raw, 45100)
|
||||
|
||||
|
||||
def test_multiple_vad_items_must_be_complete_and_monotonic():
|
||||
first = fixture('你好。')
|
||||
second = fixture('再见。')
|
||||
for time in second[0]['timestamp']:
|
||||
time[0] += 5000; time[1] += 5000
|
||||
cues = aligned_cues(first + second, 10000)
|
||||
assert [c['text'] for c in cues] == ['你好。', '再见。']
|
||||
with pytest.raises(ValueError): aligned_cues(second + first, 10000)
|
||||
|
||||
|
||||
def test_local_selection_has_no_connection_and_validates_model(monkeypatch, tmp_path):
|
||||
selection = Transcription(provider='sensevoice_local', model='SenseVoiceSmall', connection_id='obsolete')
|
||||
assert selection.connection_id is None
|
||||
with pytest.raises(ValueError): Transcription(provider='sensevoice_local', model='base')
|
||||
from backend.services import ai_model_settings
|
||||
from backend.utils import speech_recognizer
|
||||
monkeypatch.setattr(ai_model_settings, 'load', lambda: ModelSettings(transcription=selection))
|
||||
calls = []
|
||||
def transcribe(*args):
|
||||
calls.append(args)
|
||||
return tmp_path / 'out.srt'
|
||||
monkeypatch.setattr(runtime, 'transcribe', transcribe)
|
||||
monkeypatch.setattr(speech_recognizer, 'SpeechRecognizer', lambda: pytest.fail('must not load Whisper'))
|
||||
assert speech_recognizer.generate_subtitle_for_video(tmp_path / 'video.mp4', model='tiny') == tmp_path / 'out.srt'
|
||||
assert calls[0][2:] == ('auto', 0)
|
||||
|
||||
|
||||
def test_cross_process_lock_rejects_uninstall_and_releases(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'sensevoice')
|
||||
with runtime.operation():
|
||||
with pytest.raises(RuntimeError, match='正在准备或转写'): runtime.uninstall()
|
||||
assert runtime.uninstall()['status'] == 'not_installed'
|
||||
|
||||
|
||||
def test_worker_uses_isolated_interpreter_not_parent_import_path(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(runtime, 'root', lambda: tmp_path)
|
||||
calls = []
|
||||
def run(cmd, **kw):
|
||||
calls.append(cmd)
|
||||
return SimpleNamespace(returncode=0)
|
||||
monkeypatch.setattr(runtime.subprocess, 'run', run)
|
||||
runtime.worker('transcribe', tmp_path / 'r.json', tmp_path / 'a.wav')
|
||||
assert calls[0][:2] == [sys.executable, '-S']
|
||||
assert calls[0][2].endswith('sensevoice_worker.py')
|
||||
assert '-m' not in calls[0]
|
||||
|
||||
|
||||
def test_interrupted_install_is_retryable_after_restart(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'sensevoice')
|
||||
runtime._state('installing')
|
||||
assert runtime.status()['status'] == 'error'
|
||||
with runtime.operation():
|
||||
assert runtime.status()['status'] == 'installing'
|
||||
|
||||
|
||||
def test_status_api_is_lightweight_and_install_conflict_is_explicit(tmp_path, monkeypatch):
|
||||
from fastapi import FastAPI
|
||||
from fastapi.testclient import TestClient
|
||||
from backend.api.v1.speech_recognition import router
|
||||
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'sensevoice')
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
with TestClient(app) as client:
|
||||
assert client.get('/sensevoice/status').json()['status'] == 'not_installed'
|
||||
with runtime.operation():
|
||||
assert client.post('/sensevoice/prepare').status_code == 409
|
||||
assert client.delete('/sensevoice').status_code == 409
|
||||
|
||||
|
||||
def test_transcribe_srt_and_real_word_sidecar_without_optional_imports(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(runtime, 'root', lambda: tmp_path / 'runtime')
|
||||
monkeypatch.setattr(runtime, 'status', lambda: {'status': 'ready'})
|
||||
raw = fixture()
|
||||
def extract(cmd, **kw):
|
||||
with wave.open(cmd[-1], 'wb') as target:
|
||||
target.setparams((1, 2, 16000, 0, 'NONE', 'not compressed'))
|
||||
target.writeframes(b'\0\0' * (16000 * 45))
|
||||
return SimpleNamespace(returncode=0)
|
||||
monkeypatch.setattr(runtime.subprocess, 'run', extract)
|
||||
def worker(action, result, *args): result.write_text(json.dumps(raw), encoding='utf-8')
|
||||
monkeypatch.setattr(runtime, 'worker', worker)
|
||||
output = tmp_path / '字幕.srt'
|
||||
runtime.transcribe(tmp_path / '源.mp4', output)
|
||||
from backend.utils.word_timing import load_word_timing
|
||||
assert load_word_timing(output)
|
||||
before = output.read_bytes()
|
||||
raw[0]['words'].pop()
|
||||
from backend.utils.speech_recognizer import SpeechRecognitionError
|
||||
with pytest.raises(SpeechRecognitionError): runtime.transcribe(tmp_path / '源.mp4', output)
|
||||
assert output.read_bytes() == before
|
||||
output.write_text('edited', encoding='utf-8')
|
||||
assert load_word_timing(output) is None
|
||||
|
||||
|
||||
def test_sensevoice_error_does_not_point_to_whisper(monkeypatch):
|
||||
from backend.pipeline.failures import failure_from_speech_error
|
||||
from backend.services import whisper_runtime
|
||||
monkeypatch.setattr(whisper_runtime, 'get_status', lambda: pytest.fail('unrelated Whisper probe'))
|
||||
failure = failure_from_speech_error('SenseVoice 词与时间戳未完整配对')
|
||||
assert failure.code == 'subtitle_setup' and 'SenseVoice' in failure.hint
|
||||
assert 'Whisper' not in failure.user_message()
|
||||
@@ -830,6 +830,9 @@ def generate_subtitle_for_video(video_path: Path, output_path: Optional[Path] =
|
||||
from backend.services.ai_model_settings import load as load_model_settings
|
||||
model_settings = load_model_settings()
|
||||
selection = model_settings.transcription if model_settings else None
|
||||
if method == 'sensevoice_local' or method == 'auto' and selection and selection.provider == 'sensevoice_local':
|
||||
from backend.services.sensevoice_runtime import transcribe
|
||||
return transcribe(Path(video_path), output_path, language, timeout)
|
||||
if method == 'auto' and selection and selection.provider == 'cloud':
|
||||
from backend.services.cloud_transcription import transcribe
|
||||
return transcribe(Path(video_path), output_path, model_settings, language, timeout)
|
||||
|
||||
@@ -43,8 +43,10 @@ def valid_words(segment):
|
||||
return False
|
||||
|
||||
|
||||
def write_word_timing(srt: Path, segments, language=None):
|
||||
payload = {'schema_version': 1, 'source': 'faster-whisper', 'language': language,
|
||||
def write_word_timing(srt: Path, segments, language=None, source='faster-whisper'):
|
||||
if source not in {'faster-whisper', 'sensevoice'}:
|
||||
raise ValueError('Unsupported ASR word timing source')
|
||||
payload = {'schema_version': 1, 'source': source, 'language': language,
|
||||
'srt_sha256': hashlib.sha256(srt.read_bytes()).hexdigest(),
|
||||
'segments': [s for s in segments if s.get('text', '').strip()]}
|
||||
# A bad segment keeps sentence captions usable, but must not become karaoke.
|
||||
@@ -65,7 +67,7 @@ def load_word_timing(srt: Path):
|
||||
"""Return only complete, unchanged source captions; otherwise safe fallback."""
|
||||
try:
|
||||
payload = json.loads(sidecar_path(srt).read_text(encoding='utf-8'))
|
||||
if payload.get('schema_version') != 1 or payload.get('source') != 'faster-whisper':
|
||||
if payload.get('schema_version') != 1 or payload.get('source') not in {'faster-whisper', 'sensevoice'}:
|
||||
return None
|
||||
if payload.get('srt_sha256') != hashlib.sha256(srt.read_bytes()).hexdigest():
|
||||
return None
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
- 高光分析:获取可用模型后自动预选,也可手动更换。
|
||||
- 画面理解:默认复用高光分析模型;高级设置中可指定另一个连接和模型。
|
||||
- 封面生成:默认复用供应商和密钥,预选已适配的生图模型;没有适合的型号时使用视频截帧。高级设置中可以配置其他供应商。
|
||||
- 字幕转写:以 Whisper 本地服务和模型选择呈现。一次“准备模型”操作自动完成必要组件安装及所选模型下载;安装下载即时执行,选择随设置保存。新导入任务使用所选型号,本地服务不自动回退到云端。
|
||||
- 字幕转写:可选择 Whisper 或 SenseVoiceSmall 本地转写,也可单独绑定云端服务。一次“准备模型”操作完成必要组件安装及模型下载;安装下载即时执行,选择随设置保存。新导入任务使用所选型号,本地服务不自动回退到云端。
|
||||
|
||||
普通配置不需要命名或创建连接,也不使用配置弹窗。内部通过连接引用共享地址和密钥,并保留各供应商已填写的配置。按用途提供封面供应商切换与可选的补充画面模型;预置服务不显示接口地址,地址仅在自定义或本地服务中出现。只有未知的自定义模型显示能力选择。分析方式自动判断,原有显式偏好保留,重新选择分析模型后恢复自动方式。改变服务地址时必须重新填写已保存的密钥,避免把旧密钥自动发送给新地址。空输入保留已保存密钥。
|
||||
|
||||
@@ -54,6 +54,7 @@
|
||||
| 提供商 | 核查到的 ASR | 本次可生成带时间戳字幕 |
|
||||
| --- | --- | --- |
|
||||
| 本地 | Whisper tiny/base/small/medium/large-v3 | 是,沿用本地运行时 |
|
||||
| 本地 FunASR | SenseVoiceSmall | 是,消费 CTC 词级时间戳;按需安装,独立进程运行 |
|
||||
| OpenAI | whisper-1、gpt-4o-transcribe-diarize、gpt-4o-transcribe、gpt-4o-mini-transcribe | 前两项;普通 transcribe 的纯文字输出不能替代字幕时间轴 |
|
||||
| 阿里云百炼 | Qwen-Audio-3.1/3.0-ASR-Flash、Fun-ASR-Flash、Qwen-ASR-Filetrans、Fun-ASR、Paraformer | 同步 Flash 系列;SSE 收集所有已完成句子,毫秒转秒。Filetrans 需公网音频地址,暂不代建对象存储 |
|
||||
| 无限星河 | 公开目录当前列出 qwen-audio-3.0-asr-flash-streaming / filetrans | 目录可预览,但其具体上传/时间戳协议尚未确认,标为未接入,不能选为可用默认值 |
|
||||
@@ -74,3 +75,19 @@
|
||||
云端任务以 16 kHz 单声道 PCM 每 180 秒分块(原始 5.76 MB,Base64 后低于 10 MB),用准确样本偏移合并时间轴。固定分块边界可能截断词语,是当前分段方式的限制。只有全任务成功才发布 SRT;请求失败不偷偷切换供应商或本地模型,不把凭证或服务端原始错误体写入日志。
|
||||
|
||||
新手流程:没有任何可用连接时,首页弹出一次性的「连接 AI 服务」对话框(`FirstRunSetup.tsx`,本会话内「稍后再说」后不再弹),内容与设置页 AI 服务一节完全相同,也可点「更多选项」进入设置页。首次空白配置默认:画面识别开、封面「AI 生成」且「参考视频画面」开,供应商分组把赞助伙伴放在「推荐」并保留合作说明;已有配置不强制改动。未连接 AI 服务时导入视频会被拦下并提示先连接。封面模型缺失时回落到视频截帧。模型使用单选,搜索无匹配时允许显式选择自定义型号。关闭画面识别后的连接测试也不发送图片。
|
||||
|
||||
## 本地 SenseVoice(#67)
|
||||
|
||||
设置 → 字幕转写 → 转写方式选择 **SenseVoice · 本地**,点击“准备模型”,就绪后保存设置。首次需联网下载 FunASR/PyTorch 组件和 SenseVoiceSmall/FSMN-VAD 模型,请预留至少 8 GB 磁盘空间(Windows 的真实安装验收约 5.8 GB,另需安装临时空间)。转写时使用已下载的本地路径,不上传音频、不请求云端转写、不需要 API Key。默认 Whisper 和已有选择保持原样。
|
||||
|
||||
这是 FunASR 1.3.14 的 SenseVoiceSmall 适配器,不是把 FunASR 工具箱中的所有模型都加入列表。支持中文、粤语、英语、日语、韩语及自动语言检测。首版使用 CPU,限制为两条计算线程;不加载 CAM++ 说话人分离或额外标点模型,也不承诺 issue 中的通用速度数字。模型产生的空格和标点会保留。高光分析仍使用另行配置的分析模型。
|
||||
|
||||
运行时位于数据目录 `sensevoice/runtime`,模型缓存位于 `sensevoice/models`。独立 Python 子进程导入这些组件,避免 PyTorch/NumPy 与 Whisper、API 后端依赖冲突。后台准备失败后可以重试;重启后可识别中断的准备操作。安装、删除和转写共享跨进程锁,忙碌时禁止修改组件。高级项可以删除组件及模型。
|
||||
|
||||
显式请求 `output_timestamp=True`,严格配对 `words` 与毫秒 `timestamp`,先恢复原文空格和标点,再按标点、42 字符、8 秒上限以及停顿生成字幕。词级信息写入带 SRT 内容校验的 `.words.json`,用于已有字幕对齐流程。缺失、不完整、倒退、重叠或明显超出音频的时间戳会中断任务并显示 SenseVoice 的处理建议;不会按文字比例伪造时间轴,也不会把一条粗 VAD 段当作精确字幕。失败不会覆盖已存在的字幕。
|
||||
|
||||
接入参考 [#67 讨论](https://github.com/zhouxiaoka/autoclip/issues/67)、[123mlly 的 feature/dev 分支](https://github.com/123mlly/autoclip/tree/feature/dev) 和 [LauraGPT 的时间戳修复 PR](https://github.com/123mlly/autoclip/pull/1)。词级聚合思路依据该 MIT 许可实现改写,保留项目 [MIT 许可证](../LICENSE);不搬入 fork 的其他改动和比例铺字幕兜底。
|
||||
|
||||
FunASR/SenseVoice 源码 MIT 与模型权重许可不同。权重使用前请阅读 [SenseVoiceSmall 模型卡及其许可链接](https://huggingface.co/FunAudioLLM/SenseVoiceSmall);安装包不内置或重新分发这些权重。硅基流动的云端 SenseVoice 型号仍需独立的云端时间戳适配,不能与这个本地实现混用。
|
||||
|
||||
结构化输出回归测试:`backend/tests/test_sensevoice.py`。真实语音验收:`.github/workflows/sensevoice.yml` 在三种系统的 Python 3.13 环境中实际安装、下载模型,然后在阻断网络的子进程内转写仓库公开访谈前 45 秒,检查完整词级信息、字幕顺序和边界、依赖隔离及卸载。该测试属于真实 ASR 验收,不代表长视频质量对照或桌面安装包真机验收。
|
||||
|
||||
@@ -118,7 +118,7 @@ Building 和 Testing 现在是空的:更新日志里没有「已写完、还
|
||||
|
||||
**Researching**
|
||||
|
||||
- ASR 后端可插拔(#67,协议先于具体模型)
|
||||
- 本地 ASR 模型增补(#67:SenseVoiceSmall 按需接入,Whisper 保持默认;通用插件协议继续评估)
|
||||
- 切片质量对照(5 分钟和 60 分钟真视频)
|
||||
- Step 3 评分后端可插拔
|
||||
|
||||
@@ -149,7 +149,7 @@ Building 和 Testing 现在是空的:更新日志里没有「已写完、还
|
||||
- Apple 公证和 Windows 代码签名。应用内更新和崩溃上报已经在 v1.3.1,未签名仍会拦住安装。
|
||||
- 首页和项目卡按设计系统重做。八语界面已经上线,这两处还是 Ant Design 默认件。
|
||||
- CLI 真视频,以及竖屏三条预设的人工看片。
|
||||
3. **识别和评分的可替换接口留在 Researching。** #67 在协议写下来之前不接具体厂商。账号仍等公证和代码签名完成,不因为界面语言已经上线就提前开。
|
||||
3. **通用识别和评分插件接口继续评估。** #67 已按维护者 2026-09-30 的接入要求实现可选本地 SenseVoiceSmall,见 [PR #244](https://github.com/zhouxiaoka/autoclip/pull/244) 和 [模型配置](AI_MODEL_CONFIGURATION.md#本地-sensevoice67);现有转写入口负责统一路由,词级时间戳必须完整、单调且不超出音频。此交付不代表全部 FunASR 模型或通用插件框架已完成。账号仍等公证和代码签名完成,不因为界面语言已经上线就提前开。
|
||||
4. **新的社区需求不插到上面这几件前面**,除非它就是其中一件的复现,例如切片结果为空。
|
||||
|
||||
Quackback 同时满足下面三条再立项,缺一条就继续用 GitHub:
|
||||
|
||||
@@ -14,7 +14,7 @@ const enums: Record<string, readonly string[]> = {
|
||||
provider: ['dashscope', 'openai', 'gemini', 'deepseek', 'seed', 'kimi', 'glm', 'grok', 'infistar', 'api88', 'compatible', 'ollama', 'lmstudio', 'other'],
|
||||
source: ['live', 'cache', 'catalog'], outcome: ['completed', 'failed', 'blocked', 'unknown'],
|
||||
resolution: ['created', 'reused'], material_origin: ['sample', 'user', 'unknown'],
|
||||
analysis_mode: ['auto', 'subtitle', 'visual'], transcription_mode: ['whisper_local', 'cloud', 'unknown'],
|
||||
analysis_mode: ['auto', 'subtitle', 'visual'], transcription_mode: ['whisper_local', 'sensevoice_local', 'cloud', 'unknown'],
|
||||
cover_mode: ['ai', 'frame'], capability: ['text', 'multimodal', 'auto'],
|
||||
}
|
||||
export function safeExperience(value: Record<string, unknown>): Properties {
|
||||
|
||||
@@ -78,7 +78,7 @@ export function buildFeedbackDraft(input: FeedbackDraftInput): FeedbackDraft {
|
||||
stage: scrubText(input.stage, 80) || undefined,
|
||||
errorMessage: scrubText(input.errorMessage, 500) || undefined,
|
||||
errorCode: safeFeedbackErrorCode(input.errorCode),
|
||||
transcriptionProvider: ['whisper_local', 'cloud'].includes(input.transcriptionProvider || '') ? input.transcriptionProvider : undefined,
|
||||
transcriptionProvider: ['whisper_local', 'sensevoice_local', 'cloud'].includes(input.transcriptionProvider || '') ? input.transcriptionProvider : undefined,
|
||||
transcriptionModel: scrubText(input.transcriptionModel, 80) || undefined,
|
||||
version: scrubText(input.version, 40),
|
||||
os: scrubText(input.os, 40),
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { useEffect } from 'react'
|
||||
import { useEffect, useState } from 'react'
|
||||
import { useTranslation } from 'react-i18next'
|
||||
import { Select, Switch } from 'antd'
|
||||
import { t } from '../../i18n'
|
||||
@@ -9,6 +9,53 @@ import { useModelSettings, type ModelSettingsStore } from './useModelSettings'
|
||||
import { connectionOf, isTextOnly, mainConnection, presetKey, type SaveIssue } from './modelSettingsLogic'
|
||||
import ProviderFields from './ProviderFields'
|
||||
import ModelPicker from './ModelPicker'
|
||||
import api from '../../services/api'
|
||||
|
||||
type SenseVoiceStatus = { status: 'not_installed' | 'installing' | 'ready' | 'error'; message: string }
|
||||
|
||||
function SenseVoiceConfig() {
|
||||
const [runtime, setRuntime] = useState<SenseVoiceStatus | null>(null)
|
||||
const [error, setError] = useState('')
|
||||
const [busy, setBusy] = useState(false)
|
||||
useEffect(() => {
|
||||
let active = true
|
||||
const refresh = async () => {
|
||||
try {
|
||||
const value = await api.get<unknown, SenseVoiceStatus>('/sensevoice/status')
|
||||
if (active) { setRuntime(value); setError('') }
|
||||
} catch { if (active) setError(t('暂时无法读取本地模型状态。')) }
|
||||
}
|
||||
void refresh()
|
||||
const timer = window.setInterval(() => void refresh(), 3000)
|
||||
return () => { active = false; window.clearInterval(timer) }
|
||||
}, [])
|
||||
const act = async (remove = false) => {
|
||||
setBusy(true)
|
||||
try {
|
||||
const value = remove ? await api.delete<unknown, SenseVoiceStatus>('/sensevoice')
|
||||
: await api.post<unknown, SenseVoiceStatus>('/sensevoice/prepare')
|
||||
setRuntime(value); setError('')
|
||||
} catch { setError(t('操作失败,请稍后重试。')) }
|
||||
finally { setBusy(false) }
|
||||
}
|
||||
const preparing = runtime?.status === 'installing'
|
||||
return <div className="ac-rows">
|
||||
<Row label={t('转写模型')} hint={t('支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。')}><span>SenseVoiceSmall</span></Row>
|
||||
<Row label={runtime?.status === 'ready' ? t('模型已就绪') : t('准备本地模型')}
|
||||
hint={runtime?.status === 'ready' ? t('保存设置后,新任务将使用这个模型。') : t('首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。')}>
|
||||
{runtime?.status === 'ready' ? <StatusDot tone="ok" label={t('已就绪')} />
|
||||
: preparing ? <StatusDot tone="accent" label={t('正在准备组件与模型…')} />
|
||||
: <Btn size="sm" disabled={!runtime || busy} onClick={() => void act()}>{t('准备模型')}</Btn>}
|
||||
</Row>
|
||||
{(error || runtime?.status === 'error') && <p className="ac-note" role="alert">{error || t('模型准备失败,请检查网络和磁盘空间后重试。')}</p>}
|
||||
<p className="ac-note">{t('仅安装所需模型,不启用说话人分离。')} <a href="https://huggingface.co/FunAudioLLM/SenseVoiceSmall" target="_blank" rel="noreferrer">{t('模型与许可说明')}</a></p>
|
||||
{runtime?.status === 'ready' && <details className="ac-disclosure"><summary>{t('高级')}</summary>
|
||||
<Row label={t('删除 SenseVoice 组件与模型')} hint={t('删除后需要重新下载才能使用。')}>
|
||||
<Btn variant="danger" size="sm" disabled={busy} onClick={() => void act(true)}>{t('删除')}</Btn>
|
||||
</Row>
|
||||
</details>}
|
||||
</div>
|
||||
}
|
||||
|
||||
const ANCHORS: Record<string, string> = { model: 'ai-model', analysis: 'ai-model', vision: 'ai-model', speech: 'ai-speech', cover: 'ai-cover' }
|
||||
const issueAnchor = (issue: SaveIssue) => issue.role === 'transcription' ? 'ai-speech' : issue.role === 'cover' ? 'ai-cover' : 'ai-model'
|
||||
@@ -49,19 +96,20 @@ function TranscriptionSection({ m }: { m: ModelSettingsStore }) {
|
||||
const { settings, lists, busy, listErrors } = m
|
||||
if (!settings) return null
|
||||
const cloud = settings.transcription?.provider === 'cloud'
|
||||
const sensevoice = settings.transcription?.provider === 'sensevoice_local'
|
||||
const connection = cloud ? connectionOf(settings, 'transcription') : undefined
|
||||
const main = mainConnection(settings)
|
||||
const options = [
|
||||
{ label: t('本机运行'), options: [{ value: 'whisper_local', label: t('Whisper · 本地') }] },
|
||||
{ label: t('本机运行'), options: [{ value: 'whisper_local', label: t('Whisper · 本地') }, { value: 'sensevoice_local', label: t('SenseVoice · 本地') }] },
|
||||
...providerPickerOptions().map(group => ({ ...group, options: group.options.filter(p => ['openai', 'dashscope', 'infistar', 'api88', 'glm', 'compatible'].includes(p.value)) })).filter(group => group.options.length),
|
||||
]
|
||||
return <div id="ai-speech" className="ac-model-section">
|
||||
<h3 className="ac-model-section-title">{t('字幕转写')}</h3>
|
||||
<p className="ac-note">{t('视频没有字幕时,先把说话声转成字幕。默认在本机免费转写,不上传音频。')}</p>
|
||||
<Row wide label={t('转写方式')} hint={t('本地转写免费,首次需下载模型;云端转写无需下载,按用量计费。')}>
|
||||
<Select aria-label={t('转写方式')} style={{ width: '100%' }} value={cloud ? connection?.provider : 'whisper_local'} options={options}
|
||||
onChange={value => value === 'whisper_local' ? m.setTranscriptionLocal(settings.transcription?.model && settings.transcription.provider === 'whisper_local' ? settings.transcription.model : 'base') : m.chooseProvider('transcription', value as ProviderKey)}
|
||||
optionRender={option => <span>{option.value === 'whisper_local' ? t('Whisper · 本地') : PROVIDERS[option.value as ProviderKey]?.name}{PROVIDERS[option.value as ProviderKey]?.sponsor && <small style={{ marginLeft: 8, color: 'var(--sub)' }}>{t('赞助')} · {PROVIDERS[option.value as ProviderKey]?.sponsor?.offer}</small>}</span>} />
|
||||
<Select aria-label={t('转写方式')} style={{ width: '100%' }} value={cloud ? connection?.provider : sensevoice ? 'sensevoice_local' : 'whisper_local'} options={options}
|
||||
onChange={value => value === 'sensevoice_local' ? m.update({ transcription: { provider: 'sensevoice_local', model: 'SenseVoiceSmall' } }) : value === 'whisper_local' ? m.setTranscriptionLocal(settings.transcription?.model && settings.transcription.provider === 'whisper_local' ? settings.transcription.model : 'base') : m.chooseProvider('transcription', value as ProviderKey)}
|
||||
optionRender={option => <span>{option.value === 'whisper_local' ? t('Whisper · 本地') : option.value === 'sensevoice_local' ? t('SenseVoice · 本地') : PROVIDERS[option.value as ProviderKey]?.name}{PROVIDERS[option.value as ProviderKey]?.sponsor && <small style={{ marginLeft: 8, color: 'var(--sub)' }}>{t('赞助')} · {PROVIDERS[option.value as ProviderKey]?.sponsor?.offer}</small>}</span>} />
|
||||
</Row>
|
||||
{cloud && connection ? <>
|
||||
<ProviderFields connection={connection} ariaPrefix={t('转写')} hideProvider placement="settings_model"
|
||||
@@ -71,7 +119,7 @@ function TranscriptionSection({ m }: { m: ModelSettingsStore }) {
|
||||
<ModelPicker role="transcription" model={settings.transcription?.model || ''} connection={connection} list={lists[connection.id]} busy={!!busy[connection.id]} listError={listErrors[connection.id]} mode={settings.analysis_mode}
|
||||
onChange={model => m.update({ transcription: { provider: 'cloud', connection_id: connection.id, model } })} onRefresh={() => void m.discover(connection, true)} />
|
||||
</Row>
|
||||
</> : <SpeechRecognitionConfig hideProvider selectedModel={settings.transcription?.model || 'base'} onModelChange={m.setTranscriptionLocal} />}
|
||||
</> : sensevoice ? <SenseVoiceConfig /> : <SpeechRecognitionConfig hideProvider selectedModel={settings.transcription?.model || 'base'} onModelChange={m.setTranscriptionLocal} />}
|
||||
</div>
|
||||
}
|
||||
|
||||
|
||||
@@ -20,7 +20,7 @@ export interface ModelSettings {
|
||||
analysis: Assignment | null
|
||||
vision: Assignment | null
|
||||
cover: Assignment | null
|
||||
transcription?: { provider: 'whisper_local' | 'cloud'; model: string; connection_id?: string; capability?: Capability } | null
|
||||
transcription?: { provider: 'whisper_local' | 'sensevoice_local' | 'cloud'; model: string; connection_id?: string; capability?: Capability } | null
|
||||
cover_enabled: boolean
|
||||
allow_send_frame: boolean
|
||||
analysis_mode: 'auto' | 'subtitle' | 'visual'
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "New users can receive trial credit via our referral link; live support is available on the platform.",
|
||||
"待确认制作类型": "Waiting for production type confirmation",
|
||||
"查看建议": "View suggestion",
|
||||
"制作中 · 查看进度": "Creating · View progress"
|
||||
"制作中 · 查看进度": "Creating · View progress",
|
||||
"SenseVoice · 本地": "SenseVoice · Local",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Supports Mandarin, Cantonese, English, Japanese and Korean, with word timestamps for subtitles.",
|
||||
"正在准备组件与模型…": "Preparing components and model…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "Model preparation failed. Check your network and disk space, then retry.",
|
||||
"仅安装所需模型,不启用说话人分离。": "Installs only the required models; speaker diarization is disabled.",
|
||||
"模型与许可说明": "Model and license details",
|
||||
"删除 SenseVoice 组件与模型": "Delete SenseVoice components and models",
|
||||
"删除后需要重新下载才能使用。": "Download again to use it after deletion.",
|
||||
"操作失败,请稍后重试。": "Operation failed. Please try again later.",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "Internet access is required for the initial component and model download. Reserve at least 8 GB of disk space. Once ready, transcription does not upload audio."
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Los nuevos usuarios pueden recibir crédito de prueba mediante nuestro enlace; la plataforma ofrece atención humana.",
|
||||
"待确认制作类型": "Pendiente de confirmar el tipo de producción",
|
||||
"查看建议": "Ver sugerencia",
|
||||
"制作中 · 查看进度": "Creando · Ver progreso"
|
||||
"制作中 · 查看进度": "Creando · Ver progreso",
|
||||
"SenseVoice · 本地": "SenseVoice · Local",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Admite mandarín, cantonés, inglés, japonés y coreano, con marcas de tiempo por palabra.",
|
||||
"正在准备组件与模型…": "Preparando componentes y modelo…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "No se pudo preparar el modelo. Comprueba la conexión y el espacio en disco y reintenta.",
|
||||
"仅安装所需模型,不启用说话人分离。": "Solo instala los modelos necesarios; la separación de hablantes está desactivada.",
|
||||
"模型与许可说明": "Modelo y licencia",
|
||||
"删除 SenseVoice 组件与模型": "Eliminar componentes y modelos de SenseVoice",
|
||||
"删除后需要重新下载才能使用。": "Después de eliminarlo, tendrás que descargarlo de nuevo.",
|
||||
"操作失败,请稍后重试。": "La operación falló. Inténtalo más tarde.",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "La descarga inicial de componentes y modelos requiere internet. Reserva al menos 8 GB de espacio en disco. Después, la transcripción no sube el audio."
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Les nouveaux utilisateurs peuvent recevoir un crédit d’essai via notre lien ; une assistance humaine est disponible sur la plateforme.",
|
||||
"待确认制作类型": "Confirmation du type de production en attente",
|
||||
"查看建议": "Voir la suggestion",
|
||||
"制作中 · 查看进度": "Création · Voir la progression"
|
||||
"制作中 · 查看进度": "Création · Voir la progression",
|
||||
"SenseVoice · 本地": "SenseVoice · Local",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Prend en charge le mandarin, le cantonais, l’anglais, le japonais et le coréen, avec des horodatages par mot.",
|
||||
"正在准备组件与模型…": "Préparation des composants et du modèle…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "La préparation du modèle a échoué. Vérifiez le réseau et l’espace disque, puis réessayez.",
|
||||
"仅安装所需模型,不启用说话人分离。": "Seuls les modèles nécessaires sont installés ; la diarisation est désactivée.",
|
||||
"模型与许可说明": "Modèle et licence",
|
||||
"删除 SenseVoice 组件与模型": "Supprimer les composants et modèles SenseVoice",
|
||||
"删除后需要重新下载才能使用。": "Après suppression, un nouveau téléchargement sera nécessaire.",
|
||||
"操作失败,请稍后重试。": "L’opération a échoué. Réessayez plus tard.",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "Le téléchargement initial des composants et du modèle nécessite Internet. Prévoyez au moins 8 Go d’espace disque. Ensuite, la transcription ne téléverse pas l’audio."
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "専用リンクから新規登録すると体験クレジットを受け取れます。サイト内で有人サポートを利用できます。",
|
||||
"待确认制作类型": "制作タイプの確認待ち",
|
||||
"查看建议": "提案を見る",
|
||||
"制作中 · 查看进度": "制作中 · 進捗を見る"
|
||||
"制作中 · 查看进度": "制作中 · 進捗を見る",
|
||||
"SenseVoice · 本地": "SenseVoice · ローカル",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "中国語、広東語、英語、日本語、韓国語に対応し、単語の時刻から字幕を生成します。",
|
||||
"正在准备组件与模型…": "コンポーネントとモデルを準備中…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "モデルの準備に失敗しました。ネットワークと空き容量を確認して再試行してください。",
|
||||
"仅安装所需模型,不启用说话人分离。": "必要なモデルのみをインストールし、話者分離は使用しません。",
|
||||
"模型与许可说明": "モデルとライセンス",
|
||||
"删除 SenseVoice 组件与模型": "SenseVoice のコンポーネントとモデルを削除",
|
||||
"删除后需要重新下载才能使用。": "削除後は再ダウンロードが必要です。",
|
||||
"操作失败,请稍后重试。": "操作に失敗しました。後でもう一度お試しください。",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "初回はコンポーネントとモデルのダウンロードにネット接続が必要です。8 GB 以上の空き容量を確保してください。準備後の文字起こしでは音声をアップロードしません。"
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "전용 링크로 신규 가입하면 체험 크레딧을 받을 수 있습니다. 사이트에서 상담원 지원을 제공합니다.",
|
||||
"待确认制作类型": "제작 유형 확인 대기",
|
||||
"查看建议": "제안 보기",
|
||||
"制作中 · 查看进度": "제작 중 · 진행 상황 보기"
|
||||
"制作中 · 查看进度": "제작 중 · 진행 상황 보기",
|
||||
"SenseVoice · 本地": "SenseVoice · 로컬",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "중국어, 광둥어, 영어, 일본어, 한국어를 지원하며 단어 타임스탬프로 자막을 생성합니다.",
|
||||
"正在准备组件与模型…": "구성 요소와 모델 준비 중…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "모델 준비에 실패했습니다. 네트워크와 디스크 공간을 확인한 후 다시 시도하세요.",
|
||||
"仅安装所需模型,不启用说话人分离。": "필요한 모델만 설치하며 화자 분리는 사용하지 않습니다.",
|
||||
"模型与许可说明": "모델 및 라이선스 정보",
|
||||
"删除 SenseVoice 组件与模型": "SenseVoice 구성 요소와 모델 삭제",
|
||||
"删除后需要重新下载才能使用。": "삭제 후 사용하려면 다시 다운로드해야 합니다.",
|
||||
"操作失败,请稍后重试。": "작업에 실패했습니다. 나중에 다시 시도하세요.",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "처음에는 인터넷으로 구성 요소와 모델을 다운로드해야 합니다. 최소 8 GB의 디스크 공간을 확보하세요. 준비 후 전사 시 오디오를 업로드하지 않습니다."
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Novos usuários podem receber crédito de teste pelo nosso link; a plataforma oferece atendimento humano.",
|
||||
"待确认制作类型": "Aguardando confirmação do tipo de produção",
|
||||
"查看建议": "Ver sugestão",
|
||||
"制作中 · 查看进度": "Criando · Ver progresso"
|
||||
"制作中 · 查看进度": "Criando · Ver progresso",
|
||||
"SenseVoice · 本地": "SenseVoice · Local",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Suporta mandarim, cantonês, inglês, japonês e coreano, com tempos por palavra.",
|
||||
"正在准备组件与模型…": "Preparando componentes e modelo…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "Falha ao preparar o modelo. Verifique a rede e o espaço em disco e tente novamente.",
|
||||
"仅安装所需模型,不启用说话人分离。": "Instala apenas os modelos necessários; a separação de falantes fica desativada.",
|
||||
"模型与许可说明": "Modelo e licença",
|
||||
"删除 SenseVoice 组件与模型": "Excluir componentes e modelos do SenseVoice",
|
||||
"删除后需要重新下载才能使用。": "Após excluir, será necessário baixar novamente.",
|
||||
"操作失败,请稍后重试。": "A operação falhou. Tente novamente mais tarde.",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "O download inicial dos componentes e modelos requer internet. Reserve pelo menos 8 GB de espaço em disco. Depois, a transcrição não envia o áudio."
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "Новые пользователи могут получить пробный кредит по нашей ссылке; на платформе доступна поддержка операторов.",
|
||||
"待确认制作类型": "Ожидается подтверждение типа монтажа",
|
||||
"查看建议": "Посмотреть предложение",
|
||||
"制作中 · 查看进度": "Создание · Посмотреть прогресс"
|
||||
"制作中 · 查看进度": "Создание · Посмотреть прогресс",
|
||||
"SenseVoice · 本地": "SenseVoice · Локально",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "Поддерживает китайский, кантонский, английский, японский и корейский с временными метками слов.",
|
||||
"正在准备组件与模型…": "Подготовка компонентов и модели…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "Не удалось подготовить модель. Проверьте сеть и место на диске и повторите попытку.",
|
||||
"仅安装所需模型,不启用说话人分离。": "Устанавливаются только необходимые модели; разделение говорящих отключено.",
|
||||
"模型与许可说明": "Модель и лицензия",
|
||||
"删除 SenseVoice 组件与模型": "Удалить компоненты и модели SenseVoice",
|
||||
"删除后需要重新下载才能使用。": "После удаления для использования потребуется повторная загрузка.",
|
||||
"操作失败,请稍后重试。": "Операция не удалась. Повторите попытку позже.",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "Для первоначальной загрузки компонентов и модели нужен интернет. Освободите не менее 8 ГБ на диске. После подготовки аудио не загружается на сервер."
|
||||
}
|
||||
|
||||
@@ -1416,5 +1416,15 @@
|
||||
"新用户通过专属链接注册可获体验额度;站内提供人工客服。": "新用户通过专属链接注册可获体验额度;站内提供人工客服。",
|
||||
"待确认制作类型": "待确认制作类型",
|
||||
"查看建议": "查看建议",
|
||||
"制作中 · 查看进度": "制作中 · 查看进度"
|
||||
"制作中 · 查看进度": "制作中 · 查看进度",
|
||||
"SenseVoice · 本地": "SenseVoice · 本地",
|
||||
"支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。": "支持中文、粤语、英语、日语和韩语,使用词级时间戳生成字幕。",
|
||||
"正在准备组件与模型…": "正在准备组件与模型…",
|
||||
"模型准备失败,请检查网络和磁盘空间后重试。": "模型准备失败,请检查网络和磁盘空间后重试。",
|
||||
"仅安装所需模型,不启用说话人分离。": "仅安装所需模型,不启用说话人分离。",
|
||||
"模型与许可说明": "模型与许可说明",
|
||||
"删除 SenseVoice 组件与模型": "删除 SenseVoice 组件与模型",
|
||||
"删除后需要重新下载才能使用。": "删除后需要重新下载才能使用。",
|
||||
"操作失败,请稍后重试。": "操作失败,请稍后重试。",
|
||||
"首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。": "首次需联网下载组件和模型,请预留至少 8 GB 磁盘空间;准备完成后转写不上传音频。"
|
||||
}
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
const { test } = require('node:test')
|
||||
const assert = require('node:assert/strict')
|
||||
const fs = require('node:fs')
|
||||
const vm = require('node:vm')
|
||||
const ts = require('typescript')
|
||||
|
||||
function loadSection() {
|
||||
const jsx = (type, props) => ({ type, props })
|
||||
const modules = {
|
||||
react: { useEffect() {}, useState() {} },
|
||||
'react/jsx-runtime': { jsx, jsxs: jsx },
|
||||
'react-i18next': { useTranslation() {} },
|
||||
antd: { Select: 'Select', Switch: 'Switch' },
|
||||
'../../i18n': { t: s => s },
|
||||
'../../ui': { Btn: 'Btn', Row: 'Row', Section: 'Section', Segmented: 'Segmented', StatusDot: 'StatusDot' },
|
||||
'../../components/SpeechRecognitionConfig': { default: 'WhisperConfig' },
|
||||
'./providers': { PROVIDERS: {}, providerPickerOptions: () => [] },
|
||||
'./useModelSettings': {},
|
||||
'./modelSettingsLogic': { connectionOf: () => undefined, mainConnection: () => undefined },
|
||||
'./ProviderFields': {}, './ModelPicker': {}, '../../services/api': {},
|
||||
}
|
||||
const source = fs.readFileSync('src/features/settings/AIModelSettings.tsx', 'utf8') + '\nexport const testSection = TranscriptionSection;'
|
||||
const code = ts.transpileModule(source, { compilerOptions: { module: ts.ModuleKind.CommonJS, jsx: ts.JsxEmit.ReactJSX } }).outputText
|
||||
const module = { exports: {} }
|
||||
vm.runInNewContext(code, { module, exports: module.exports, require: name => modules[name] || assert.fail(name) })
|
||||
return module.exports.testSection
|
||||
}
|
||||
function findSelect(node) {
|
||||
if (!node || typeof node !== 'object') return null
|
||||
if (node.type === 'Select') return node
|
||||
const children = node.props?.children
|
||||
for (const child of Array.isArray(children) ? children.flat(5) : [children]) {
|
||||
const found = findSelect(child)
|
||||
if (found) return found
|
||||
}
|
||||
return null
|
||||
}
|
||||
test('the actual transcription selector offers SenseVoice and switches local engines without a cloud connection', () => {
|
||||
const Section = loadSection()
|
||||
let settings = { connections: [], transcription: { provider: 'whisper_local', model: 'small' } }
|
||||
const m = { settings, lists: {}, busy: {}, listErrors: {},
|
||||
update: patch => { settings = { ...settings, ...patch } },
|
||||
setTranscriptionLocal: model => { settings.transcription = { provider: 'whisper_local', model } },
|
||||
chooseProvider: () => assert.fail('must not route a local engine to a cloud provider') }
|
||||
let select = findSelect(Section({ m }))
|
||||
assert.ok(select.props.options[0].options.some(x => x.value === 'sensevoice_local'))
|
||||
select.props.onChange('sensevoice_local')
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(settings.transcription)), { provider: 'sensevoice_local', model: 'SenseVoiceSmall' })
|
||||
m.settings = settings
|
||||
select = findSelect(Section({ m }))
|
||||
assert.equal(select.props.value, 'sensevoice_local')
|
||||
select.props.onChange('whisper_local')
|
||||
assert.deepEqual(settings.transcription, { provider: 'whisper_local', model: 'base' })
|
||||
})
|
||||
@@ -69,7 +69,7 @@ def normalize_event(raw: dict[str, Any]) -> dict[str, str] | None:
|
||||
"stage": scrub(raw.get("stage"), 80),
|
||||
"error_message": scrub(raw.get("error_message"), 500),
|
||||
"error_code": error_code if ERROR_CODE_RE.fullmatch(error_code) else "",
|
||||
"transcription_provider": transcription_provider if transcription_provider in {"whisper_local", "cloud"} else "",
|
||||
"transcription_provider": transcription_provider if transcription_provider in {"whisper_local", "sensevoice_local", "cloud"} else "",
|
||||
"transcription_model": scrub(raw.get("transcription_model"), 80),
|
||||
"version": scrub(raw.get("app_version") or raw.get("version"), 40),
|
||||
"os": scrub(raw.get("os"), 40),
|
||||
|
||||
@@ -172,7 +172,7 @@ stdlib = set(sys.stdlib_module_names)
|
||||
# cv2: speaker framing (backend/services/studio/framing.py), installed on demand like Whisper.
|
||||
# numpy: imported only after loading Whisper, for the PyAV decoding fallback;
|
||||
# faster-whisper's runtime installation provides it through its dependencies.
|
||||
runtime_optional = {"faster_whisper", "ctranslate2", "huggingface_hub", "cv2", "numpy"}
|
||||
runtime_optional = {"faster_whisper", "ctranslate2", "huggingface_hub", "cv2", "numpy", "funasr", "torch"}
|
||||
mods = set()
|
||||
for root, _, files in os.walk(backend_dir):
|
||||
if '__pycache__' in root:
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Remote-only acceptance: real runtime/model install and offline speech timing."""
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
checkout = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(checkout))
|
||||
base = Path(tempfile.mkdtemp(prefix='autoclip-sensevoice-acceptance-'))
|
||||
os.environ['AUTOCLIP_DATA_DIR'] = str(base)
|
||||
os.environ['AUTOCLIP_APP_DIR'] = str(base)
|
||||
from backend.services import ai_model_settings as settings, sensevoice_runtime as runtime
|
||||
from backend.utils.ffmpeg_utils import get_ffmpeg_path
|
||||
from backend.utils.speech_recognizer import generate_subtitle_for_video
|
||||
from backend.utils.word_timing import load_word_timing
|
||||
|
||||
runtime.prepare()
|
||||
assert runtime.status()['status'] == 'ready', runtime.status()
|
||||
video = base / '真实语音 sample.mp4'
|
||||
subprocess.run([get_ffmpeg_path(), '-v', 'error', '-i', str(checkout / 'backend/assets/example/source.mp4'),
|
||||
'-t', '45', '-c', 'copy', str(video)], check=True, timeout=60)
|
||||
models = settings.ModelSettings(transcription=settings.Transcription(provider='sensevoice_local', model='SenseVoiceSmall'))
|
||||
settings.path().write_text(models.model_dump_json(), encoding='utf-8')
|
||||
# Prove the normal saved-selection entry point reaches isolated offline inference.
|
||||
original_transcribe = runtime.transcribe
|
||||
def offline(*args, **kw):
|
||||
return original_transcribe(*args, **kw, deny_network=True)
|
||||
runtime.transcribe = offline
|
||||
subtitle = generate_subtitle_for_video(video, language='en')
|
||||
cues = load_word_timing(subtitle)
|
||||
assert cues and len(cues) > 2, cues
|
||||
assert 'months' in subtitle.read_text(encoding='utf-8').lower()
|
||||
assert all(0 <= c['start'] < c['end'] <= 45.2 and c['end'] - c['start'] <= 8.001 for c in cues)
|
||||
assert all(a['end'] <= b['start'] for a, b in zip(cues, cues[1:]))
|
||||
assert 'funasr' not in sys.modules and 'torch' not in sys.modules
|
||||
size = sum(p.stat().st_size for p in runtime.root().rglob('*') if p.is_file())
|
||||
Path('sensevoice-acceptance.json').write_text(json.dumps({
|
||||
'platform': sys.platform, 'python': sys.version.split()[0], 'model': 'SenseVoiceSmall',
|
||||
'source': 'public interview, first 45 seconds', 'install': 'passed',
|
||||
'offline_network_denied': 'passed', 'normal_selection_route': 'passed',
|
||||
'parent_dependency_isolation': 'passed', 'cues': len(cues),
|
||||
'words': sum(len(c['words']) for c in cues), 'final_seconds': cues[-1]['end'],
|
||||
'maximum_cue_seconds': max(c['end'] - c['start'] for c in cues), 'disk_bytes': size,
|
||||
}, indent=2), encoding='utf-8')
|
||||
runtime.uninstall()
|
||||
assert runtime.status()['status'] == 'not_installed'
|
||||
Reference in New Issue
Block a user