chore: prepare local speech context for isolated tts work
This commit is contained in:
@@ -1,4 +1,4 @@
|
||||
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
||||
import type { SpeechCapture } from './speech-capture';
|
||||
|
||||
@@ -103,11 +103,11 @@ export const transcribeSpeechCapture = (
|
||||
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
|
||||
: transcribeSpeechFile(capture.file, registerAbort);
|
||||
|
||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
|
||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
||||
apiRequest<TtsResponse>({
|
||||
url: '/api/ai/tts',
|
||||
method: 'POST',
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile },
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
|
||||
timeout: 60000
|
||||
});
|
||||
|
||||
@@ -133,7 +133,8 @@ export const createSpeechPlaybackController = (
|
||||
let generation = 0;
|
||||
const sourceCache = new Map<string, string>();
|
||||
const maxCachedSources = 4;
|
||||
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
|
||||
const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
||||
JSON.stringify([text, profile || null, practiceContext || null]);
|
||||
|
||||
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
|
||||
activeKey = key;
|
||||
@@ -212,7 +213,7 @@ export const createSpeechPlaybackController = (
|
||||
context.play();
|
||||
};
|
||||
|
||||
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
|
||||
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => {
|
||||
if (activeKey === key) {
|
||||
stop();
|
||||
return;
|
||||
@@ -221,7 +222,7 @@ export const createSpeechPlaybackController = (
|
||||
const value = text.trim();
|
||||
if (!value) return;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile);
|
||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
||||
const currentGeneration = generation;
|
||||
publish(key, 'loading');
|
||||
try {
|
||||
@@ -232,7 +233,7 @@ export const createSpeechPlaybackController = (
|
||||
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
||||
return;
|
||||
}
|
||||
const result = await synthesizeSpeech(value, profile);
|
||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
||||
if (currentGeneration !== generation || activeKey !== key) return;
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) throw new Error('语音合成未返回音频');
|
||||
@@ -257,15 +258,15 @@ export const createSpeechPlaybackController = (
|
||||
* WeChat-style voice bubble can show duration before the first tap.
|
||||
* Returns the audio source URL, or null when remote TTS is unavailable.
|
||||
*/
|
||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
|
||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
|
||||
const value = text.trim();
|
||||
if (!value) return null;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile);
|
||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
||||
const cached = sourceCache.get(sourceKey);
|
||||
if (cached) return cached;
|
||||
try {
|
||||
const result = await synthesizeSpeech(value, profile);
|
||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) return null;
|
||||
sourceCache.set(sourceKey, source);
|
||||
|
||||
@@ -31,6 +31,13 @@ export interface TtsResponse {
|
||||
ossId?: number | string;
|
||||
}
|
||||
|
||||
/** 可选的服务端训练回合上下文;仅用于关联本次新生成的业主 TTS 录音。 */
|
||||
export interface PracticeTtsContext {
|
||||
sessionId: string;
|
||||
turnIndex: number;
|
||||
role: 'customer';
|
||||
}
|
||||
|
||||
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
|
||||
|
||||
export interface SpeechVoiceProfile {
|
||||
|
||||
Reference in New Issue
Block a user