feat(aihr): complete M3 protected practice replay
This commit is contained in:
@@ -387,6 +387,7 @@ import type {
|
||||
PracticePrepCard,
|
||||
PracticeAssignment,
|
||||
PracticeRole,
|
||||
PracticeTtsContext,
|
||||
SpeechVoiceProfile,
|
||||
SpeechPlaybackStatus,
|
||||
PracticeTurn,
|
||||
@@ -573,9 +574,12 @@ interface TurnVoice {
|
||||
unread: boolean;
|
||||
open: boolean;
|
||||
profile: SpeechVoiceProfile;
|
||||
practiceContext?: PracticeTtsContext;
|
||||
requestKey?: string;
|
||||
}
|
||||
|
||||
const turnVoices = ref<Record<number, TurnVoice>>({});
|
||||
let turnVoiceGeneration = 0;
|
||||
|
||||
const turnVoice = (index: number): TurnVoice =>
|
||||
turnVoices.value[index] || {
|
||||
@@ -592,18 +596,26 @@ const patchTurnVoice = (index: number, patch: Partial<TurnVoice>) => {
|
||||
|
||||
/** Customer lines arrive as WeChat-style voice messages: pre-synthesize so the
|
||||
* bubble shows a duration and an unread dot; fall back to text when TTS is off. */
|
||||
const prepareTurnVoice = async (index: number, text: string, emotion = 0) => {
|
||||
const prepareTurnVoice = async (index: number, text: string, emotion = 0, customerTurnIndex?: number) => {
|
||||
const spoken = (text || '').trim();
|
||||
if (!spoken) return;
|
||||
const generation = turnVoiceGeneration;
|
||||
const profile = customerVoiceProfile(emotion);
|
||||
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile });
|
||||
const source = await speechPlayback.preload(spoken, profile);
|
||||
if (!turnVoices.value[index]) return;
|
||||
const practiceContext: PracticeTtsContext | undefined = sessionId.value && Number.isInteger(customerTurnIndex)
|
||||
? { sessionId: sessionId.value, turnIndex: Number(customerTurnIndex), role: 'customer' }
|
||||
: undefined;
|
||||
const requestKey = `${generation}:${index}:${spoken}:${practiceContext?.sessionId || ''}:${practiceContext?.turnIndex ?? ''}`;
|
||||
const isCurrentVoiceRequest = () =>
|
||||
turnVoiceGeneration === generation && turnVoices.value[index]?.requestKey === requestKey;
|
||||
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile, practiceContext, requestKey });
|
||||
const source = await speechPlayback.preload(spoken, profile, practiceContext);
|
||||
if (!isCurrentVoiceRequest()) return;
|
||||
if (!source) {
|
||||
patchTurnVoice(index, { status: 'unavailable', unread: false });
|
||||
return;
|
||||
}
|
||||
const duration = await measureAudioDuration(source);
|
||||
if (!isCurrentVoiceRequest()) return;
|
||||
patchTurnVoice(index, { status: 'ready', duration });
|
||||
};
|
||||
|
||||
@@ -623,7 +635,7 @@ const playTurnVoice = (index: number, text: string) => {
|
||||
return;
|
||||
}
|
||||
patchTurnVoice(index, { unread: false });
|
||||
void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile);
|
||||
void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile, turnVoice(index).practiceContext);
|
||||
};
|
||||
|
||||
const toggleTurnTranscript = (index: number) => {
|
||||
@@ -762,6 +774,7 @@ const refreshAuthState = () => {
|
||||
};
|
||||
|
||||
const clearPracticeData = () => {
|
||||
turnVoiceGeneration += 1;
|
||||
stopRealtime();
|
||||
speechPlayback.stop();
|
||||
cancelRecording();
|
||||
@@ -773,6 +786,7 @@ const clearPracticeData = () => {
|
||||
prepCard.value = null;
|
||||
prepLoading.value = false;
|
||||
turns.value = [];
|
||||
turnVoices.value = {};
|
||||
result.value = null;
|
||||
satisfactionScore.value = 0;
|
||||
satisfactionComment.value = '';
|
||||
@@ -812,11 +826,11 @@ const requireLogin = () => {
|
||||
return false;
|
||||
};
|
||||
|
||||
const appendTurn = (role: PracticeRole, text?: string, emotion = 0, synthesizeCustomerVoice = true) => {
|
||||
const appendTurn = (role: PracticeRole, text?: string, emotion = 0, synthesizeCustomerVoice = true, customerTurnIndex?: number) => {
|
||||
if (!text) return;
|
||||
turns.value.push({ role, text });
|
||||
if (role === 'customer' && synthesizeCustomerVoice) {
|
||||
void prepareTurnVoice(turns.value.length - 1, text, emotion);
|
||||
void prepareTurnVoice(turns.value.length - 1, text, emotion, customerTurnIndex);
|
||||
}
|
||||
scrollThreadToBottom();
|
||||
};
|
||||
@@ -873,6 +887,7 @@ const startRealtime = async () => {
|
||||
if (!canStartPractice.value || !requireLogin()) return;
|
||||
speechPlayback.stop();
|
||||
cancelRecording();
|
||||
turnVoiceGeneration += 1;
|
||||
turns.value = [];
|
||||
turnVoices.value = {};
|
||||
result.value = null;
|
||||
@@ -1254,6 +1269,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => {
|
||||
satisfactionScore.value = 0;
|
||||
satisfactionComment.value = '';
|
||||
satisfactionSubmitted.value = false;
|
||||
turnVoiceGeneration += 1;
|
||||
turns.value = [];
|
||||
turnVoices.value = {};
|
||||
threadScrollTop.value = 0;
|
||||
@@ -1276,7 +1292,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => {
|
||||
emotionScore.value = data.emotion ?? 0;
|
||||
trustScore.value = data.trust ?? 0;
|
||||
sessionId.value = data.sessionId;
|
||||
appendTurn('customer', data.customerText, data.emotion);
|
||||
appendTurn('customer', data.customerText, data.emotion, true, 0);
|
||||
status.value = 'active';
|
||||
} catch (error) {
|
||||
if (!requests.isCurrent(request)) return;
|
||||
@@ -1403,7 +1419,7 @@ const submit = async () => {
|
||||
await finish(true);
|
||||
return;
|
||||
}
|
||||
appendTurn('customer', data.customerText, data.emotion);
|
||||
appendTurn('customer', data.customerText, data.emotion, true, data.roundIndex);
|
||||
roundIndex.value = data.roundIndex;
|
||||
status.value = 'active';
|
||||
} catch (error) {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
||||
import type { SpeechCapture } from './speech-capture';
|
||||
|
||||
@@ -103,11 +103,11 @@ export const transcribeSpeechCapture = (
|
||||
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
|
||||
: transcribeSpeechFile(capture.file, registerAbort);
|
||||
|
||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
|
||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
||||
apiRequest<TtsResponse>({
|
||||
url: '/api/ai/tts',
|
||||
method: 'POST',
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile },
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
|
||||
timeout: 60000
|
||||
});
|
||||
|
||||
@@ -133,7 +133,8 @@ export const createSpeechPlaybackController = (
|
||||
let generation = 0;
|
||||
const sourceCache = new Map<string, string>();
|
||||
const maxCachedSources = 4;
|
||||
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
|
||||
const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
||||
JSON.stringify([text, profile || null, practiceContext || null]);
|
||||
|
||||
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
|
||||
activeKey = key;
|
||||
@@ -212,7 +213,7 @@ export const createSpeechPlaybackController = (
|
||||
context.play();
|
||||
};
|
||||
|
||||
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
|
||||
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => {
|
||||
if (activeKey === key) {
|
||||
stop();
|
||||
return;
|
||||
@@ -221,7 +222,7 @@ export const createSpeechPlaybackController = (
|
||||
const value = text.trim();
|
||||
if (!value) return;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile);
|
||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
||||
const currentGeneration = generation;
|
||||
publish(key, 'loading');
|
||||
try {
|
||||
@@ -232,7 +233,7 @@ export const createSpeechPlaybackController = (
|
||||
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
||||
return;
|
||||
}
|
||||
const result = await synthesizeSpeech(value, profile);
|
||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
||||
if (currentGeneration !== generation || activeKey !== key) return;
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) throw new Error('语音合成未返回音频');
|
||||
@@ -257,15 +258,15 @@ export const createSpeechPlaybackController = (
|
||||
* WeChat-style voice bubble can show duration before the first tap.
|
||||
* Returns the audio source URL, or null when remote TTS is unavailable.
|
||||
*/
|
||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
|
||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
|
||||
const value = text.trim();
|
||||
if (!value) return null;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile);
|
||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
||||
const cached = sourceCache.get(sourceKey);
|
||||
if (cached) return cached;
|
||||
try {
|
||||
const result = await synthesizeSpeech(value, profile);
|
||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) return null;
|
||||
sourceCache.set(sourceKey, source);
|
||||
|
||||
@@ -31,6 +31,13 @@ export interface TtsResponse {
|
||||
ossId?: number | string;
|
||||
}
|
||||
|
||||
/** 可选的服务端训练回合上下文;仅用于关联本次新生成的业主 TTS 录音。 */
|
||||
export interface PracticeTtsContext {
|
||||
sessionId: string;
|
||||
turnIndex: number;
|
||||
role: 'customer';
|
||||
}
|
||||
|
||||
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
|
||||
|
||||
export interface SpeechVoiceProfile {
|
||||
|
||||
@@ -68,6 +68,26 @@ test('语音 profile 透传方言枚举,并保持普通话默认值', async ()
|
||||
assert.match(speechService, /dialect === 'sichuanese'[\s\S]*?当前设备语音不支持四川话/);
|
||||
});
|
||||
|
||||
test('M3 将业主 TTS 绑定到当前训练会话与准确回合,供主管复盘回放', async () => {
|
||||
const [practicePage, speechService, apiTypes] = await Promise.all([
|
||||
pageSource('../src/pages/user/practice/index.vue'),
|
||||
pageSource('../src/services/speech.ts'),
|
||||
pageSource('../src/types/api.ts')
|
||||
]);
|
||||
|
||||
assert.match(apiTypes, /export interface PracticeTtsContext/);
|
||||
assert.match(speechService, /practiceContext\?: PracticeTtsContext/);
|
||||
assert.match(speechService, /data: \{ text: text\.trim\(\)\.slice\(0, 300\), voiceProfile, practiceContext \}/);
|
||||
assert.match(practicePage, /\{ sessionId: sessionId\.value, turnIndex: Number\(customerTurnIndex\), role: 'customer' \}/);
|
||||
assert.match(practicePage, /speechPlayback\.preload\(spoken, profile, practiceContext\)/);
|
||||
assert.match(practicePage, /turnVoice\(index\)\.profile, turnVoice\(index\)\.practiceContext/);
|
||||
assert.match(practicePage, /let turnVoiceGeneration = 0/);
|
||||
assert.match(practicePage, /turnVoiceGeneration === generation && turnVoices\.value\[index\]\?\.requestKey === requestKey/);
|
||||
assert.match(practicePage, /turnVoiceGeneration \+= 1/);
|
||||
assert.match(practicePage, /appendTurn\('customer', data\.customerText, data\.emotion, true, 0\)/);
|
||||
assert.match(practicePage, /appendTurn\('customer', data\.customerText, data\.emotion, true, data\.roundIndex\)/);
|
||||
});
|
||||
|
||||
test('训练结果回到顶部,SOP 答案生成后定位到回答区', async () => {
|
||||
const [practiceSource, sopSource, navigationSource] = await Promise.all([
|
||||
pageSource('../src/pages/user/practice/index.vue'),
|
||||
@@ -560,7 +580,7 @@ test('播报复用已合成音频且总结图按内容高度生成并及时释
|
||||
|
||||
assert.match(speechService, /const sourceCache = new Map<string, string>\(\)/);
|
||||
assert.match(speechService, /const maxCachedSources = 4/);
|
||||
assert.match(speechService, /const sourceKey = cacheKey\(value, profile\)/);
|
||||
assert.match(speechService, /const sourceKey = cacheKey\(value, profile, practiceContext\)/);
|
||||
assert.match(speechService, /const cachedSource = sourceCache\.get\(sourceKey\)/);
|
||||
assert.match(speechService, /sourceCache\.clear\(\)/);
|
||||
assert.match(speechService, /registerAbort\?\.\(\(\) => task\.abort\(\)\)/);
|
||||
|
||||
Reference in New Issue
Block a user