feat(speech): add role-aware emotional TTS profiles
This commit is contained in:
@@ -156,6 +156,7 @@ import { resetPageScroll } from '@/services/navigation';
|
||||
import {
|
||||
chooseSpeechAudio,
|
||||
createSpeechPlaybackController,
|
||||
INTERVIEWER_VOICE_PROFILE,
|
||||
normalizeOssId,
|
||||
transcribeSpeechBlob,
|
||||
transcribeSpeechFile
|
||||
@@ -186,7 +187,8 @@ const speechPlayback = createSpeechPlaybackController(
|
||||
(text) => {
|
||||
message.value = text;
|
||||
uni.showToast({ title: text, icon: 'none' });
|
||||
}
|
||||
},
|
||||
INTERVIEWER_VOICE_PROFILE
|
||||
);
|
||||
|
||||
const interviewBusy = computed(() => interviewStatus.value === 'starting' || interviewStatus.value === 'submitting');
|
||||
|
||||
@@ -348,6 +348,7 @@ import type {
|
||||
PracticePrepCard,
|
||||
PracticeAssignment,
|
||||
PracticeRole,
|
||||
SpeechVoiceProfile,
|
||||
SpeechPlaybackStatus,
|
||||
PracticeTurn,
|
||||
PracticeRecord
|
||||
@@ -357,7 +358,7 @@ import { getSelectedPosition, isLoggedIn, rememberLoginRedirect } from '@/servic
|
||||
import { searchKnowledge } from '@/services/knowledge';
|
||||
import { resetPageScroll } from '@/services/navigation';
|
||||
import { ensureEmployeePosition } from '@/services/position';
|
||||
import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
|
||||
import { chooseSpeechAudio, createSpeechPlaybackController, customerVoiceProfile, measureAudioDuration, MENTOR_VOICE_PROFILE, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
|
||||
import { createPageRequestScope, currentAccountKey, type RequestScopeTicket } from '@/services/request-scope';
|
||||
import { stripInternalCodes } from '@/services/text';
|
||||
|
||||
@@ -433,7 +434,8 @@ const speechPlayback = createSpeechPlaybackController(
|
||||
(text) => {
|
||||
message.value = text;
|
||||
uni.showToast({ title: text, icon: 'none' });
|
||||
}
|
||||
},
|
||||
MENTOR_VOICE_PROFILE
|
||||
);
|
||||
const busy = computed(() => status.value === 'starting' || status.value === 'submitting');
|
||||
const practiceView = computed<'prep' | 'active' | 'result'>(() => {
|
||||
@@ -517,12 +519,19 @@ interface TurnVoice {
|
||||
duration: number | null;
|
||||
unread: boolean;
|
||||
open: boolean;
|
||||
profile: SpeechVoiceProfile;
|
||||
}
|
||||
|
||||
const turnVoices = ref<Record<number, TurnVoice>>({});
|
||||
|
||||
const turnVoice = (index: number): TurnVoice =>
|
||||
turnVoices.value[index] || { status: 'unavailable', duration: null, unread: false, open: false };
|
||||
turnVoices.value[index] || {
|
||||
status: 'unavailable',
|
||||
duration: null,
|
||||
unread: false,
|
||||
open: false,
|
||||
profile: customerVoiceProfile()
|
||||
};
|
||||
|
||||
const patchTurnVoice = (index: number, patch: Partial<TurnVoice>) => {
|
||||
turnVoices.value = { ...turnVoices.value, [index]: { ...turnVoice(index), ...patch } };
|
||||
@@ -530,11 +539,12 @@ const patchTurnVoice = (index: number, patch: Partial<TurnVoice>) => {
|
||||
|
||||
/** Customer lines arrive as WeChat-style voice messages: pre-synthesize so the
|
||||
* bubble shows a duration and an unread dot; fall back to text when TTS is off. */
|
||||
const prepareTurnVoice = async (index: number, text: string) => {
|
||||
const prepareTurnVoice = async (index: number, text: string, emotion = 0) => {
|
||||
const spoken = (text || '').trim();
|
||||
if (!spoken) return;
|
||||
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false });
|
||||
const source = await speechPlayback.preload(spoken);
|
||||
const profile = customerVoiceProfile(emotion);
|
||||
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile });
|
||||
const source = await speechPlayback.preload(spoken, profile);
|
||||
if (!turnVoices.value[index]) return;
|
||||
if (!source) {
|
||||
patchTurnVoice(index, { status: 'unavailable', unread: false });
|
||||
@@ -560,7 +570,7 @@ const playTurnVoice = (index: number, text: string) => {
|
||||
return;
|
||||
}
|
||||
patchTurnVoice(index, { unread: false });
|
||||
void speechPlayback.toggle(`turn-${index}`, (text || '').trim());
|
||||
void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile);
|
||||
};
|
||||
|
||||
const toggleTurnTranscript = (index: number) => {
|
||||
@@ -748,11 +758,11 @@ const requireLogin = () => {
|
||||
return false;
|
||||
};
|
||||
|
||||
const appendTurn = (role: PracticeRole, text?: string) => {
|
||||
const appendTurn = (role: PracticeRole, text?: string, emotion = 0) => {
|
||||
if (!text) return;
|
||||
turns.value.push({ role, text });
|
||||
if (role === 'customer') {
|
||||
void prepareTurnVoice(turns.value.length - 1, text);
|
||||
void prepareTurnVoice(turns.value.length - 1, text, emotion);
|
||||
}
|
||||
scrollThreadToBottom();
|
||||
};
|
||||
@@ -1146,7 +1156,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => {
|
||||
emotionScore.value = data.emotion ?? 0;
|
||||
trustScore.value = data.trust ?? 0;
|
||||
sessionId.value = data.sessionId;
|
||||
appendTurn('customer', data.customerText);
|
||||
appendTurn('customer', data.customerText, data.emotion);
|
||||
status.value = 'active';
|
||||
} catch (error) {
|
||||
if (!requests.isCurrent(request)) return;
|
||||
@@ -1267,7 +1277,7 @@ const submit = async () => {
|
||||
await finish(true);
|
||||
return;
|
||||
}
|
||||
appendTurn('customer', data.customerText);
|
||||
appendTurn('customer', data.customerText, data.emotion);
|
||||
roundIndex.value = data.roundIndex;
|
||||
status.value = 'active';
|
||||
} catch (error) {
|
||||
|
||||
@@ -261,7 +261,7 @@ import {
|
||||
} from '@/services/auth';
|
||||
import { resetPageScroll } from '@/services/navigation';
|
||||
import { ensureEmployeePosition } from '@/services/position';
|
||||
import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
|
||||
import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, MENTOR_VOICE_PROFILE, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
|
||||
import { downloadSummaryCardImage } from '@/services/summary-card-image';
|
||||
import { computed } from 'vue';
|
||||
|
||||
@@ -337,7 +337,8 @@ const defaultToolCode = computed<'MY_PRACTICE_SUMMARY' | 'TEAM_PRACTICE_SUMMARY'
|
||||
|
||||
const speechPlayback = createSpeechPlaybackController(
|
||||
(state) => { speechState.value = state; },
|
||||
(text) => { uni.showToast({ title: text, icon: 'none' }); }
|
||||
(text) => { uni.showToast({ title: text, icon: 'none' }); },
|
||||
MENTOR_VOICE_PROFILE
|
||||
);
|
||||
|
||||
const scrollToBottom = () => {
|
||||
|
||||
@@ -117,7 +117,7 @@ import { onShow, onUnload } from '@dcloudio/uni-app';
|
||||
import type { SpeechPlaybackStatus, WebAiCapabilities, WebAiResponse } from '@/types/api';
|
||||
import ChatComposer from '@/components/chat/ChatComposer.vue';
|
||||
import { isLoggedIn, rememberLoginRedirect } from '@/services/auth';
|
||||
import { chooseSpeechAudio, createSpeechPlaybackController, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
|
||||
import { chooseSpeechAudio, createSpeechPlaybackController, MENTOR_VOICE_PROFILE, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
|
||||
import { getWebAiCapabilities, queryWebAi } from '@/services/web-ai';
|
||||
|
||||
const path = '/pages/user/web-ai/index';
|
||||
@@ -164,7 +164,8 @@ let abortVoiceTranscription: (() => void) | null = null;
|
||||
|
||||
const speechPlayback = createSpeechPlaybackController(
|
||||
(state) => { speechState.value = state; },
|
||||
(text) => { uni.showToast({ title: text, icon: 'none' }); }
|
||||
(text) => { uni.showToast({ title: text, icon: 'none' }); },
|
||||
MENTOR_VOICE_PROFILE
|
||||
);
|
||||
|
||||
const scrollToBottom = () => {
|
||||
|
||||
@@ -1,8 +1,16 @@
|
||||
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, TtsResponse } from '@/types/api';
|
||||
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
||||
|
||||
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
|
||||
|
||||
export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.92 };
|
||||
export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96 };
|
||||
export const customerVoiceProfile = (emotion = 0): SpeechVoiceProfile => {
|
||||
if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12 };
|
||||
if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04 };
|
||||
return { role: 'customer', emotion: 'calm', speed: 0.98 };
|
||||
};
|
||||
|
||||
// OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。
|
||||
export const normalizeOssId = (value: unknown): string | undefined => {
|
||||
const text = String(value ?? '').trim();
|
||||
@@ -83,11 +91,11 @@ export const transcribeSpeechBlob = (
|
||||
xhr.send(form);
|
||||
});
|
||||
|
||||
export const synthesizeSpeech = (text: string) =>
|
||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
|
||||
apiRequest<TtsResponse>({
|
||||
url: '/api/ai/tts',
|
||||
method: 'POST',
|
||||
data: { text: text.trim().slice(0, 300) },
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile },
|
||||
timeout: 60000
|
||||
});
|
||||
|
||||
@@ -98,7 +106,8 @@ export interface SpeechPlaybackSnapshot {
|
||||
|
||||
export const createSpeechPlaybackController = (
|
||||
onState: (snapshot: SpeechPlaybackSnapshot) => void,
|
||||
onError: (message: string) => void
|
||||
onError: (message: string) => void,
|
||||
defaultVoiceProfile?: SpeechVoiceProfile
|
||||
) => {
|
||||
let audio: ReturnType<typeof uni.createInnerAudioContext> | null = null;
|
||||
let browserUtterance: SpeechSynthesisUtterance | null = null;
|
||||
@@ -106,6 +115,7 @@ export const createSpeechPlaybackController = (
|
||||
let generation = 0;
|
||||
const sourceCache = new Map<string, string>();
|
||||
const maxCachedSources = 4;
|
||||
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
|
||||
|
||||
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
|
||||
activeKey = key;
|
||||
@@ -126,7 +136,12 @@ export const createSpeechPlaybackController = (
|
||||
publish();
|
||||
};
|
||||
|
||||
const playWithBrowserSpeech = (key: string, value: string, currentGeneration: number) => {
|
||||
const playWithBrowserSpeech = (
|
||||
key: string,
|
||||
value: string,
|
||||
currentGeneration: number,
|
||||
profile?: SpeechVoiceProfile
|
||||
) => {
|
||||
if (
|
||||
typeof window === 'undefined'
|
||||
|| !window.speechSynthesis
|
||||
@@ -135,7 +150,8 @@ export const createSpeechPlaybackController = (
|
||||
|
||||
const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600));
|
||||
utterance.lang = 'zh-CN';
|
||||
utterance.rate = 0.95;
|
||||
utterance.rate = profile?.speed || 0.95;
|
||||
utterance.pitch = profile?.role === 'mentor' ? 0.85 : profile?.role === 'customer' ? 1.05 : 1;
|
||||
utterance.onend = () => {
|
||||
if (currentGeneration === generation && activeKey === key) stop();
|
||||
};
|
||||
@@ -173,7 +189,7 @@ export const createSpeechPlaybackController = (
|
||||
context.play();
|
||||
};
|
||||
|
||||
const toggle = async (key: string, text: string) => {
|
||||
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
|
||||
if (activeKey === key) {
|
||||
stop();
|
||||
return;
|
||||
@@ -181,30 +197,32 @@ export const createSpeechPlaybackController = (
|
||||
stop();
|
||||
const value = text.trim();
|
||||
if (!value) return;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile);
|
||||
const currentGeneration = generation;
|
||||
publish(key, 'loading');
|
||||
try {
|
||||
const cachedSource = sourceCache.get(value);
|
||||
const cachedSource = sourceCache.get(sourceKey);
|
||||
if (cachedSource) {
|
||||
sourceCache.delete(value);
|
||||
sourceCache.set(value, cachedSource);
|
||||
playAudioSource(key, cachedSource, value, currentGeneration);
|
||||
sourceCache.delete(sourceKey);
|
||||
sourceCache.set(sourceKey, cachedSource);
|
||||
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
||||
return;
|
||||
}
|
||||
const result = await synthesizeSpeech(value);
|
||||
const result = await synthesizeSpeech(value, profile);
|
||||
if (currentGeneration !== generation || activeKey !== key) return;
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) throw new Error('语音合成未返回音频');
|
||||
sourceCache.set(value, source);
|
||||
sourceCache.set(sourceKey, source);
|
||||
while (sourceCache.size > maxCachedSources) {
|
||||
const oldestKey = sourceCache.keys().next().value as string | undefined;
|
||||
if (!oldestKey) break;
|
||||
sourceCache.delete(oldestKey);
|
||||
}
|
||||
playAudioSource(key, source, value, currentGeneration);
|
||||
playAudioSource(key, source, sourceKey, currentGeneration);
|
||||
} catch (error) {
|
||||
if (currentGeneration !== generation) return;
|
||||
if (playWithBrowserSpeech(key, value, currentGeneration)) return;
|
||||
if (playWithBrowserSpeech(key, value, currentGeneration, profile)) return;
|
||||
stop();
|
||||
onError(error instanceof Error ? error.message : '语音生成失败');
|
||||
}
|
||||
@@ -215,16 +233,18 @@ export const createSpeechPlaybackController = (
|
||||
* WeChat-style voice bubble can show duration before the first tap.
|
||||
* Returns the audio source URL, or null when remote TTS is unavailable.
|
||||
*/
|
||||
const preload = async (text: string): Promise<string | null> => {
|
||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
|
||||
const value = text.trim();
|
||||
if (!value) return null;
|
||||
const cached = sourceCache.get(value);
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile);
|
||||
const cached = sourceCache.get(sourceKey);
|
||||
if (cached) return cached;
|
||||
try {
|
||||
const result = await synthesizeSpeech(value);
|
||||
const result = await synthesizeSpeech(value, profile);
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) return null;
|
||||
sourceCache.set(value, source);
|
||||
sourceCache.set(sourceKey, source);
|
||||
while (sourceCache.size > maxCachedSources) {
|
||||
const oldestKey = sourceCache.keys().next().value as string | undefined;
|
||||
if (!oldestKey) break;
|
||||
|
||||
@@ -27,6 +27,13 @@ export interface TtsResponse {
|
||||
|
||||
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
|
||||
|
||||
export interface SpeechVoiceProfile {
|
||||
role: 'mentor' | 'customer' | 'interviewer' | 'neutral';
|
||||
voice?: string;
|
||||
speed?: number;
|
||||
emotion?: 'warm' | 'professional' | 'serious' | 'intense' | 'calm';
|
||||
}
|
||||
|
||||
export interface ToolItem {
|
||||
title: string;
|
||||
desc: string;
|
||||
|
||||
Reference in New Issue
Block a user