feat(aihr): complete M3 protected practice replay

This commit is contained in:
2026-07-23 22:21:43 +08:00
parent 83f775faec
commit 067cc30d80
11 changed files with 413 additions and 47 deletions
+25 -9
View File
@@ -387,6 +387,7 @@ import type {
PracticePrepCard,
PracticeAssignment,
PracticeRole,
PracticeTtsContext,
SpeechVoiceProfile,
SpeechPlaybackStatus,
PracticeTurn,
@@ -573,9 +574,12 @@ interface TurnVoice {
unread: boolean;
open: boolean;
profile: SpeechVoiceProfile;
practiceContext?: PracticeTtsContext;
requestKey?: string;
}
const turnVoices = ref<Record<number, TurnVoice>>({});
let turnVoiceGeneration = 0;
const turnVoice = (index: number): TurnVoice =>
turnVoices.value[index] || {
@@ -592,18 +596,26 @@ const patchTurnVoice = (index: number, patch: Partial<TurnVoice>) => {
/** Customer lines arrive as WeChat-style voice messages: pre-synthesize so the
* bubble shows a duration and an unread dot; fall back to text when TTS is off. */
const prepareTurnVoice = async (index: number, text: string, emotion = 0) => {
const prepareTurnVoice = async (index: number, text: string, emotion = 0, customerTurnIndex?: number) => {
const spoken = (text || '').trim();
if (!spoken) return;
const generation = turnVoiceGeneration;
const profile = customerVoiceProfile(emotion);
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile });
const source = await speechPlayback.preload(spoken, profile);
if (!turnVoices.value[index]) return;
const practiceContext: PracticeTtsContext | undefined = sessionId.value && Number.isInteger(customerTurnIndex)
? { sessionId: sessionId.value, turnIndex: Number(customerTurnIndex), role: 'customer' }
: undefined;
const requestKey = `${generation}:${index}:${spoken}:${practiceContext?.sessionId || ''}:${practiceContext?.turnIndex ?? ''}`;
const isCurrentVoiceRequest = () =>
turnVoiceGeneration === generation && turnVoices.value[index]?.requestKey === requestKey;
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile, practiceContext, requestKey });
const source = await speechPlayback.preload(spoken, profile, practiceContext);
if (!isCurrentVoiceRequest()) return;
if (!source) {
patchTurnVoice(index, { status: 'unavailable', unread: false });
return;
}
const duration = await measureAudioDuration(source);
if (!isCurrentVoiceRequest()) return;
patchTurnVoice(index, { status: 'ready', duration });
};
@@ -623,7 +635,7 @@ const playTurnVoice = (index: number, text: string) => {
return;
}
patchTurnVoice(index, { unread: false });
void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile);
void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile, turnVoice(index).practiceContext);
};
const toggleTurnTranscript = (index: number) => {
@@ -762,6 +774,7 @@ const refreshAuthState = () => {
};
const clearPracticeData = () => {
turnVoiceGeneration += 1;
stopRealtime();
speechPlayback.stop();
cancelRecording();
@@ -773,6 +786,7 @@ const clearPracticeData = () => {
prepCard.value = null;
prepLoading.value = false;
turns.value = [];
turnVoices.value = {};
result.value = null;
satisfactionScore.value = 0;
satisfactionComment.value = '';
@@ -812,11 +826,11 @@ const requireLogin = () => {
return false;
};
const appendTurn = (role: PracticeRole, text?: string, emotion = 0, synthesizeCustomerVoice = true) => {
const appendTurn = (role: PracticeRole, text?: string, emotion = 0, synthesizeCustomerVoice = true, customerTurnIndex?: number) => {
if (!text) return;
turns.value.push({ role, text });
if (role === 'customer' && synthesizeCustomerVoice) {
void prepareTurnVoice(turns.value.length - 1, text, emotion);
void prepareTurnVoice(turns.value.length - 1, text, emotion, customerTurnIndex);
}
scrollThreadToBottom();
};
@@ -873,6 +887,7 @@ const startRealtime = async () => {
if (!canStartPractice.value || !requireLogin()) return;
speechPlayback.stop();
cancelRecording();
turnVoiceGeneration += 1;
turns.value = [];
turnVoices.value = {};
result.value = null;
@@ -1254,6 +1269,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => {
satisfactionScore.value = 0;
satisfactionComment.value = '';
satisfactionSubmitted.value = false;
turnVoiceGeneration += 1;
turns.value = [];
turnVoices.value = {};
threadScrollTop.value = 0;
@@ -1276,7 +1292,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => {
emotionScore.value = data.emotion ?? 0;
trustScore.value = data.trust ?? 0;
sessionId.value = data.sessionId;
appendTurn('customer', data.customerText, data.emotion);
appendTurn('customer', data.customerText, data.emotion, true, 0);
status.value = 'active';
} catch (error) {
if (!requests.isCurrent(request)) return;
@@ -1403,7 +1419,7 @@ const submit = async () => {
await finish(true);
return;
}
appendTurn('customer', data.customerText, data.emotion);
appendTurn('customer', data.customerText, data.emotion, true, data.roundIndex);
roundIndex.value = data.roundIndex;
status.value = 'active';
} catch (error) {
+11 -10
View File
@@ -1,4 +1,4 @@
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import type { SpeechCapture } from './speech-capture';
@@ -103,11 +103,11 @@ export const transcribeSpeechCapture = (
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
: transcribeSpeechFile(capture.file, registerAbort);
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
apiRequest<TtsResponse>({
url: '/api/ai/tts',
method: 'POST',
data: { text: text.trim().slice(0, 300), voiceProfile },
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
timeout: 60000
});
@@ -133,7 +133,8 @@ export const createSpeechPlaybackController = (
let generation = 0;
const sourceCache = new Map<string, string>();
const maxCachedSources = 4;
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
JSON.stringify([text, profile || null, practiceContext || null]);
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
activeKey = key;
@@ -212,7 +213,7 @@ export const createSpeechPlaybackController = (
context.play();
};
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => {
if (activeKey === key) {
stop();
return;
@@ -221,7 +222,7 @@ export const createSpeechPlaybackController = (
const value = text.trim();
if (!value) return;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const sourceKey = cacheKey(value, profile, practiceContext);
const currentGeneration = generation;
publish(key, 'loading');
try {
@@ -232,7 +233,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return;
}
const result = await synthesizeSpeech(value, profile);
const result = await synthesizeSpeech(value, profile, practiceContext);
if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频');
@@ -257,15 +258,15 @@ export const createSpeechPlaybackController = (
* WeChat-style voice bubble can show duration before the first tap.
* Returns the audio source URL, or null when remote TTS is unavailable.
*/
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
const value = text.trim();
if (!value) return null;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const sourceKey = cacheKey(value, profile, practiceContext);
const cached = sourceCache.get(sourceKey);
if (cached) return cached;
try {
const result = await synthesizeSpeech(value, profile);
const result = await synthesizeSpeech(value, profile, practiceContext);
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null;
sourceCache.set(sourceKey, source);
+7
View File
@@ -31,6 +31,13 @@ export interface TtsResponse {
ossId?: number | string;
}
/** 可选的服务端训练回合上下文;仅用于关联本次新生成的业主 TTS 录音。 */
export interface PracticeTtsContext {
sessionId: string;
turnIndex: number;
role: 'customer';
}
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
export interface SpeechVoiceProfile {
+21 -1
View File
@@ -68,6 +68,26 @@ test('语音 profile 透传方言枚举,并保持普通话默认值', async ()
assert.match(speechService, /dialect === 'sichuanese'[\s\S]*?当前设备语音不支持四川话/);
});
test('M3 将业主 TTS 绑定到当前训练会话与准确回合,供主管复盘回放', async () => {
const [practicePage, speechService, apiTypes] = await Promise.all([
pageSource('../src/pages/user/practice/index.vue'),
pageSource('../src/services/speech.ts'),
pageSource('../src/types/api.ts')
]);
assert.match(apiTypes, /export interface PracticeTtsContext/);
assert.match(speechService, /practiceContext\?: PracticeTtsContext/);
assert.match(speechService, /data: \{ text: text\.trim\(\)\.slice\(0, 300\), voiceProfile, practiceContext \}/);
assert.match(practicePage, /\{ sessionId: sessionId\.value, turnIndex: Number\(customerTurnIndex\), role: 'customer' \}/);
assert.match(practicePage, /speechPlayback\.preload\(spoken, profile, practiceContext\)/);
assert.match(practicePage, /turnVoice\(index\)\.profile, turnVoice\(index\)\.practiceContext/);
assert.match(practicePage, /let turnVoiceGeneration = 0/);
assert.match(practicePage, /turnVoiceGeneration === generation && turnVoices\.value\[index\]\?\.requestKey === requestKey/);
assert.match(practicePage, /turnVoiceGeneration \+= 1/);
assert.match(practicePage, /appendTurn\('customer', data\.customerText, data\.emotion, true, 0\)/);
assert.match(practicePage, /appendTurn\('customer', data\.customerText, data\.emotion, true, data\.roundIndex\)/);
});
test('训练结果回到顶部,SOP 答案生成后定位到回答区', async () => {
const [practiceSource, sopSource, navigationSource] = await Promise.all([
pageSource('../src/pages/user/practice/index.vue'),
@@ -560,7 +580,7 @@ test('播报复用已合成音频且总结图按内容高度生成并及时释
assert.match(speechService, /const sourceCache = new Map<string, string>\(\)/);
assert.match(speechService, /const maxCachedSources = 4/);
assert.match(speechService, /const sourceKey = cacheKey\(value, profile\)/);
assert.match(speechService, /const sourceKey = cacheKey\(value, profile, practiceContext\)/);
assert.match(speechService, /const cachedSource = sourceCache\.get\(sourceKey\)/);
assert.match(speechService, /sourceCache\.clear\(\)/);
assert.match(speechService, /registerAbort\?\.\(\(\) => task\.abort\(\)\)/);