feat(aihr): add asr/tts speech gateway and mobile voice practice

新增 /api/ai/asr(multipart转写)与 /api/ai/tts(mp3 base64 dataURL),走
OpenAI 兼容 audio 接口(硅基流动 SenseVoice/CosyVoice2,复用模型管理
配置,新增 asr/tts 类目);移动端对练页接语音输入(MediaRecorder)与
客户台词自动播报,未配置/失败降级文本。ASR 限 5MB、TTS 限 300 字,
multipart 头部字段去 CRLF 防注入。
This commit is contained in:
2026-07-03 20:55:38 +08:00
parent 96c70a618f
commit e0490b28cb
5 changed files with 335 additions and 0 deletions
+82
View File
@@ -208,6 +208,10 @@
</div>
<textarea v-model.trim="practiceDraft" rows="3" placeholder="输入你的回复" />
<div class="practice-actions">
<button type="button" :disabled="asrBusy" @click="toggleRecording">
{{ recording ? '停止录音' : asrBusy ? '识别中' : '语音输入' }}
</button>
<button type="button" @click="toggleVoice">{{ voiceEnabled ? '播报:开' : '播报:关' }}</button>
<button type="button" @click="fillPracticeReply">填入建议回复</button>
<button type="button" :disabled="practiceStatus === 'starting' || practiceStatus === 'submitting'" @click="submitMobilePractice">
{{ practiceStatus === 'submitting' ? '提交中' : '提交本轮' }}
@@ -498,6 +502,12 @@ const selectedReview = ref<PracticeReviewDetail | null>(null);
const reviewMessage = ref('');
const markingReviewed = ref(false);
const competencyProfile = ref<CompetencyProfile | null>(null);
const voiceEnabled = ref(true);
const recording = ref(false);
const asrBusy = ref(false);
let mediaRecorder: MediaRecorder | null = null;
let recordedChunks: Blob[] = [];
let customerAudio: HTMLAudioElement | null = null;
const syncPath = () => window.history.pushState({}, '', rolePaths[roleKey.value]);
const tap = (name: string) => {
@@ -531,10 +541,79 @@ const readApi = async <T,>(response: Response): Promise<T> => {
const apiHeaders = () => ({
'Content-Type': 'application/json',
...authOnlyHeaders()
});
// multipart 上传不能手动设置 Content-Type,浏览器需自动带 boundary
const authOnlyHeaders = () => ({
...(authToken.value ? { Authorization: `Bearer ${authToken.value}` } : {}),
...(clientId.value ? { clientid: clientId.value } : {})
});
const playCustomerVoice = async (text: string) => {
if (!voiceEnabled.value || !text) return;
try {
const data = await readApi<{ audioUrl: string }>(
await fetch('/dev-api/api/ai/tts', {
method: 'POST',
headers: apiHeaders(),
body: JSON.stringify({ text, voice: '' })
})
);
customerAudio?.pause();
customerAudio = new Audio(data.audioUrl);
void customerAudio.play();
} catch {
// TTS 未配置或失败时静默降级为纯文本
}
};
const toggleVoice = () => {
voiceEnabled.value = !voiceEnabled.value;
if (!voiceEnabled.value) customerAudio?.pause();
};
const toggleRecording = async () => {
if (recording.value) {
mediaRecorder?.stop();
return;
}
try {
const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
recordedChunks = [];
mediaRecorder = new MediaRecorder(stream);
mediaRecorder.ondataavailable = (event) => {
if (event.data.size > 0) recordedChunks.push(event.data);
};
mediaRecorder.onstop = () => {
stream.getTracks().forEach((track) => track.stop());
recording.value = false;
void transcribeRecording(new Blob(recordedChunks, { type: mediaRecorder?.mimeType || 'audio/webm' }));
};
mediaRecorder.start();
recording.value = true;
practiceMessage.value = '';
} catch {
practiceMessage.value = '无法访问麦克风,请改用文字输入';
}
};
const transcribeRecording = async (blob: Blob) => {
asrBusy.value = true;
try {
const form = new FormData();
form.append('file', blob, 'practice.webm');
const data = await readApi<{ text: string }>(
await fetch('/dev-api/api/ai/asr', { method: 'POST', headers: authOnlyHeaders(), body: form })
);
practiceDraft.value = data.text;
} catch (error) {
practiceMessage.value = error instanceof Error ? error.message : '语音识别失败,请改用文字输入';
} finally {
asrBusy.value = false;
}
};
const handlePrimaryAction = () => {
if (roleKey.value !== 'user') {
tap(workerRole.value.primary.cta);
@@ -550,6 +629,9 @@ const appendPracticeTurn = (role: PracticeRole, text: string) => {
coach: 'AI教练'
};
practiceTurns.value.push({ role, label: labels[role], text });
if (role === 'customer') {
void playCustomerVoice(text);
}
};
const fillPracticeReply = () => {