feat(speech): add role-aware emotional TTS profiles

This commit is contained in:
2026-07-18 18:27:14 +08:00
parent c0538ad8c3
commit 68d5600534
14 changed files with 228 additions and 45 deletions
+39 -19
View File
@@ -1,8 +1,16 @@
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, TtsResponse } from '@/types/api';
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.92 };
export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96 };
export const customerVoiceProfile = (emotion = 0): SpeechVoiceProfile => {
if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12 };
if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04 };
return { role: 'customer', emotion: 'calm', speed: 0.98 };
};
// OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。
export const normalizeOssId = (value: unknown): string | undefined => {
const text = String(value ?? '').trim();
@@ -83,11 +91,11 @@ export const transcribeSpeechBlob = (
xhr.send(form);
});
export const synthesizeSpeech = (text: string) =>
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
apiRequest<TtsResponse>({
url: '/api/ai/tts',
method: 'POST',
data: { text: text.trim().slice(0, 300) },
data: { text: text.trim().slice(0, 300), voiceProfile },
timeout: 60000
});
@@ -98,7 +106,8 @@ export interface SpeechPlaybackSnapshot {
export const createSpeechPlaybackController = (
onState: (snapshot: SpeechPlaybackSnapshot) => void,
onError: (message: string) => void
onError: (message: string) => void,
defaultVoiceProfile?: SpeechVoiceProfile
) => {
let audio: ReturnType<typeof uni.createInnerAudioContext> | null = null;
let browserUtterance: SpeechSynthesisUtterance | null = null;
@@ -106,6 +115,7 @@ export const createSpeechPlaybackController = (
let generation = 0;
const sourceCache = new Map<string, string>();
const maxCachedSources = 4;
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
activeKey = key;
@@ -126,7 +136,12 @@ export const createSpeechPlaybackController = (
publish();
};
const playWithBrowserSpeech = (key: string, value: string, currentGeneration: number) => {
const playWithBrowserSpeech = (
key: string,
value: string,
currentGeneration: number,
profile?: SpeechVoiceProfile
) => {
if (
typeof window === 'undefined'
|| !window.speechSynthesis
@@ -135,7 +150,8 @@ export const createSpeechPlaybackController = (
const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600));
utterance.lang = 'zh-CN';
utterance.rate = 0.95;
utterance.rate = profile?.speed || 0.95;
utterance.pitch = profile?.role === 'mentor' ? 0.85 : profile?.role === 'customer' ? 1.05 : 1;
utterance.onend = () => {
if (currentGeneration === generation && activeKey === key) stop();
};
@@ -173,7 +189,7 @@ export const createSpeechPlaybackController = (
context.play();
};
const toggle = async (key: string, text: string) => {
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
if (activeKey === key) {
stop();
return;
@@ -181,30 +197,32 @@ export const createSpeechPlaybackController = (
stop();
const value = text.trim();
if (!value) return;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const currentGeneration = generation;
publish(key, 'loading');
try {
const cachedSource = sourceCache.get(value);
const cachedSource = sourceCache.get(sourceKey);
if (cachedSource) {
sourceCache.delete(value);
sourceCache.set(value, cachedSource);
playAudioSource(key, cachedSource, value, currentGeneration);
sourceCache.delete(sourceKey);
sourceCache.set(sourceKey, cachedSource);
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return;
}
const result = await synthesizeSpeech(value);
const result = await synthesizeSpeech(value, profile);
if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频');
sourceCache.set(value, source);
sourceCache.set(sourceKey, source);
while (sourceCache.size > maxCachedSources) {
const oldestKey = sourceCache.keys().next().value as string | undefined;
if (!oldestKey) break;
sourceCache.delete(oldestKey);
}
playAudioSource(key, source, value, currentGeneration);
playAudioSource(key, source, sourceKey, currentGeneration);
} catch (error) {
if (currentGeneration !== generation) return;
if (playWithBrowserSpeech(key, value, currentGeneration)) return;
if (playWithBrowserSpeech(key, value, currentGeneration, profile)) return;
stop();
onError(error instanceof Error ? error.message : '语音生成失败');
}
@@ -215,16 +233,18 @@ export const createSpeechPlaybackController = (
* WeChat-style voice bubble can show duration before the first tap.
* Returns the audio source URL, or null when remote TTS is unavailable.
*/
const preload = async (text: string): Promise<string | null> => {
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
const value = text.trim();
if (!value) return null;
const cached = sourceCache.get(value);
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const cached = sourceCache.get(sourceKey);
if (cached) return cached;
try {
const result = await synthesizeSpeech(value);
const result = await synthesizeSpeech(value, profile);
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null;
sourceCache.set(value, source);
sourceCache.set(sourceKey, source);
while (sourceCache.size > maxCachedSources) {
const oldestKey = sourceCache.keys().next().value as string | undefined;
if (!oldestKey) break;