chore: prepare local speech context for isolated tts work

This commit is contained in:
2026-07-23 22:12:42 +08:00
parent dfaa198ac8
commit bf82ff9691
2 changed files with 18 additions and 10 deletions
+11 -10
View File
@@ -1,4 +1,4 @@
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import type { SpeechCapture } from './speech-capture';
@@ -103,11 +103,11 @@ export const transcribeSpeechCapture = (
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
: transcribeSpeechFile(capture.file, registerAbort);
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
apiRequest<TtsResponse>({
url: '/api/ai/tts',
method: 'POST',
data: { text: text.trim().slice(0, 300), voiceProfile },
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
timeout: 60000
});
@@ -133,7 +133,8 @@ export const createSpeechPlaybackController = (
let generation = 0;
const sourceCache = new Map<string, string>();
const maxCachedSources = 4;
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
JSON.stringify([text, profile || null, practiceContext || null]);
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
activeKey = key;
@@ -212,7 +213,7 @@ export const createSpeechPlaybackController = (
context.play();
};
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => {
if (activeKey === key) {
stop();
return;
@@ -221,7 +222,7 @@ export const createSpeechPlaybackController = (
const value = text.trim();
if (!value) return;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const sourceKey = cacheKey(value, profile, practiceContext);
const currentGeneration = generation;
publish(key, 'loading');
try {
@@ -232,7 +233,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return;
}
const result = await synthesizeSpeech(value, profile);
const result = await synthesizeSpeech(value, profile, practiceContext);
if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频');
@@ -257,15 +258,15 @@ export const createSpeechPlaybackController = (
* WeChat-style voice bubble can show duration before the first tap.
* Returns the audio source URL, or null when remote TTS is unavailable.
*/
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
const value = text.trim();
if (!value) return null;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const sourceKey = cacheKey(value, profile, practiceContext);
const cached = sourceCache.get(sourceKey);
if (cached) return cached;
try {
const result = await synthesizeSpeech(value, profile);
const result = await synthesizeSpeech(value, profile, practiceContext);
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null;
sourceCache.set(sourceKey, source);
+7
View File
@@ -31,6 +31,13 @@ export interface TtsResponse {
ossId?: number | string;
}
/** 可选的服务端训练回合上下文;仅用于关联本次新生成的业主 TTS 录音。 */
export interface PracticeTtsContext {
sessionId: string;
turnIndex: number;
role: 'customer';
}
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
export interface SpeechVoiceProfile {