chore: prepare local speech context for isolated tts work

This commit is contained in:
2026-07-23 22:12:42 +08:00
parent dfaa198ac8
commit bf82ff9691
2 changed files with 18 additions and 10 deletions
+11 -10
View File
@@ -1,4 +1,4 @@
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api'; import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import type { SpeechCapture } from './speech-capture'; import type { SpeechCapture } from './speech-capture';
@@ -103,11 +103,11 @@ export const transcribeSpeechCapture = (
? transcribeSpeechBlob(capture.blob, filename, registerAbort) ? transcribeSpeechBlob(capture.blob, filename, registerAbort)
: transcribeSpeechFile(capture.file, registerAbort); : transcribeSpeechFile(capture.file, registerAbort);
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) => export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
apiRequest<TtsResponse>({ apiRequest<TtsResponse>({
url: '/api/ai/tts', url: '/api/ai/tts',
method: 'POST', method: 'POST',
data: { text: text.trim().slice(0, 300), voiceProfile }, data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
timeout: 60000 timeout: 60000
}); });
@@ -133,7 +133,8 @@ export const createSpeechPlaybackController = (
let generation = 0; let generation = 0;
const sourceCache = new Map<string, string>(); const sourceCache = new Map<string, string>();
const maxCachedSources = 4; const maxCachedSources = 4;
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]); const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
JSON.stringify([text, profile || null, practiceContext || null]);
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => { const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
activeKey = key; activeKey = key;
@@ -212,7 +213,7 @@ export const createSpeechPlaybackController = (
context.play(); context.play();
}; };
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => { const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => {
if (activeKey === key) { if (activeKey === key) {
stop(); stop();
return; return;
@@ -221,7 +222,7 @@ export const createSpeechPlaybackController = (
const value = text.trim(); const value = text.trim();
if (!value) return; if (!value) return;
const profile = voiceProfile || defaultVoiceProfile; const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile); const sourceKey = cacheKey(value, profile, practiceContext);
const currentGeneration = generation; const currentGeneration = generation;
publish(key, 'loading'); publish(key, 'loading');
try { try {
@@ -232,7 +233,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, cachedSource, sourceKey, currentGeneration); playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return; return;
} }
const result = await synthesizeSpeech(value, profile); const result = await synthesizeSpeech(value, profile, practiceContext);
if (currentGeneration !== generation || activeKey !== key) return; if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl; const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频'); if (!source) throw new Error('语音合成未返回音频');
@@ -257,15 +258,15 @@ export const createSpeechPlaybackController = (
* WeChat-style voice bubble can show duration before the first tap. * WeChat-style voice bubble can show duration before the first tap.
* Returns the audio source URL, or null when remote TTS is unavailable. * Returns the audio source URL, or null when remote TTS is unavailable.
*/ */
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => { const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
const value = text.trim(); const value = text.trim();
if (!value) return null; if (!value) return null;
const profile = voiceProfile || defaultVoiceProfile; const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile); const sourceKey = cacheKey(value, profile, practiceContext);
const cached = sourceCache.get(sourceKey); const cached = sourceCache.get(sourceKey);
if (cached) return cached; if (cached) return cached;
try { try {
const result = await synthesizeSpeech(value, profile); const result = await synthesizeSpeech(value, profile, practiceContext);
const source = result.inlineAudioUrl || result.audioUrl; const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null; if (!source) return null;
sourceCache.set(sourceKey, source); sourceCache.set(sourceKey, source);
+7
View File
@@ -31,6 +31,13 @@ export interface TtsResponse {
ossId?: number | string; ossId?: number | string;
} }
/** 可选的服务端训练回合上下文;仅用于关联本次新生成的业主 TTS 录音。 */
export interface PracticeTtsContext {
sessionId: string;
turnIndex: number;
role: 'customer';
}
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing'; export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
export interface SpeechVoiceProfile { export interface SpeechVoiceProfile {