fix(tts): derive Azure SSML locale from selected voice (#1566)

* fix(tts): derive Azure SSML locale from voice

* fix(tts): preserve Azure voice script locales

---------

Co-authored-by: wyuc <wang-yc24@mails.tsinghua.edu.cn>
This commit is contained in:
LeoParkerOu
2026-09-19 22:31:33 +08:00
committed by GitHub
co-authored by wyuc
parent 8de7ec2870
commit 19f3eff6ac
2 changed files with 111 additions and 2 deletions
+19 -2
View File
@@ -790,12 +790,13 @@ async function generateAzureTTS(
signal: AbortSignal,
): Promise<TTSGenerationResult> {
const baseUrl = config.baseUrl || TTS_PROVIDERS['azure-tts'].defaultBaseUrl;
const voiceLocale = resolveAzureVoiceLocale(config.voice);
// Build SSML
const rate = config.speed ? `${((config.speed - 1) * 100).toFixed(0)}%` : '0%';
const ssml = `
<speak version='1.0' xml:lang='zh-CN'>
<voice xml:lang='zh-CN' name='${config.voice}'>
<speak version='1.0' xml:lang='${voiceLocale}'>
<voice xml:lang='${voiceLocale}' name='${config.voice}'>
<prosody rate='${rate}'>${escapeXml(text)}</prosody>
</voice>
</speak>
@@ -820,6 +821,22 @@ async function generateAzureTTS(
return await validateTTSAudioResponse(response, 'Azure', 'mp3');
}
/** Resolve the BCP-47 locale encoded by an Azure voice identifier. */
function resolveAzureVoiceLocale(voice: string): string {
const configuredVoice = TTS_PROVIDERS['azure-tts'].voices.find(({ id }) => id === voice);
if (configuredVoice?.language) return configuredVoice.language;
// Azure voice IDs conventionally start with a BCP-47 locale (for example,
// `en-US-JennyNeural` or `sr-Latn-RS-SophieNeural`). Preserve optional
// script and variant subtags for voices outside the small configured list
// while retaining the existing Chinese default for an unrecognised ID.
return (
voice.match(
/^[a-z]{2,3}(?:-[A-Z][a-z]{3})?-(?:[A-Z]{2}|\d{3})(?:-(?:[a-z0-9]{5,8}|\d[a-z0-9]{3}))*(?=-[A-Z]|$)/,
)?.[0] ?? 'zh-CN'
);
}
/**
* GLM TTS implementation (GLM API)
*/
+92
View File
@@ -0,0 +1,92 @@
import { beforeEach, describe, expect, it, vi, type Mock } from 'vitest';
import { generateTTS } from '@/lib/audio/tts-providers';
const mockFetch = vi.hoisted(() => vi.fn() as Mock);
vi.mock('undici', async (importOriginal) => {
const actual = await importOriginal<typeof import('undici')>();
return { ...actual, fetch: mockFetch };
});
function audioResponse() {
return {
ok: true,
status: 200,
headers: { get: () => 'audio/mpeg' },
arrayBuffer: async () => new Uint8Array([0xff, 0xfb, 0x90, 0x64]).buffer,
};
}
describe('Azure TTS SSML locale', () => {
beforeEach(() => {
mockFetch.mockReset();
mockFetch.mockResolvedValue(audioResponse());
});
it('uses the selected non-Chinese voice locale in both SSML language attributes', async () => {
await generateTTS(
{
providerId: 'azure-tts',
apiKey: 'azure-key',
baseUrl: 'https://eastus.tts.speech.microsoft.com',
voice: 'en-US-JennyNeural',
},
'Hello',
);
const ssml = mockFetch.mock.calls[0][1].body as string;
expect(ssml).toContain("<speak version='1.0' xml:lang='en-US'>");
expect(ssml).toContain("<voice xml:lang='en-US' name='en-US-JennyNeural'>");
expect(ssml).not.toContain("xml:lang='zh-CN'");
});
it('keeps zh-CN for the default Chinese voice', async () => {
await generateTTS(
{
providerId: 'azure-tts',
apiKey: 'azure-key',
baseUrl: 'https://eastus.tts.speech.microsoft.com',
voice: 'zh-CN-XiaoxiaoNeural',
},
'你好',
);
const ssml = mockFetch.mock.calls[0][1].body as string;
expect(ssml).toContain("<speak version='1.0' xml:lang='zh-CN'>");
expect(ssml).toContain("<voice xml:lang='zh-CN' name='zh-CN-XiaoxiaoNeural'>");
});
it('derives the locale from an otherwise unlisted Azure voice ID', async () => {
await generateTTS(
{
providerId: 'azure-tts',
apiKey: 'azure-key',
baseUrl: 'https://eastus.tts.speech.microsoft.com',
voice: 'fr-FR-DeniseNeural',
},
'Bonjour',
);
const ssml = mockFetch.mock.calls[0][1].body as string;
expect(ssml).toContain("xml:lang='fr-FR'");
});
it.each([
['sr-Latn-RS-SophieNeural', 'sr-Latn-RS'],
['iu-Cans-CA-SiqiniqNeural', 'iu-Cans-CA'],
])('preserves the script subtag for %s', async (voice, locale) => {
await generateTTS(
{
providerId: 'azure-tts',
apiKey: 'azure-key',
baseUrl: 'https://eastus.tts.speech.microsoft.com',
voice,
},
'Hello',
);
const ssml = mockFetch.mock.calls[0][1].body as string;
expect(ssml).toContain(`<speak version='1.0' xml:lang='${locale}'>`);
expect(ssml).toContain(`<voice xml:lang='${locale}' name='${voice}'>`);
});
});