import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api'; import { getAuth } from './auth'; import type { SpeechCapture } from './speech-capture'; import { estimateMpegAudioDuration } from './mpeg-audio-duration'; import { dataUriToBlobUrl, decodeDataAudioUri, hasNativeAudioFileSupport, isDataAudioUri, removeNativeTempAudio, writeNativeTempAudio } from './speech-audio-source'; import { applyTtsVoicePreference, getTtsDialectPreference, getTtsVoicePreference } from './tts-voice-preference'; const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg']; export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.88, dialect: 'mandarin' }; export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96, dialect: 'mandarin' }; export const customerVoiceProfile = ( emotion = 0, dialect: SpeechVoiceProfile['dialect'] = 'mandarin' ): SpeechVoiceProfile => { if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12, dialect }; if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04, dialect }; return { role: 'customer', emotion: 'calm', speed: 0.98, dialect }; }; // OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。 export const normalizeOssId = (value: unknown): string | undefined => { const text = String(value ?? '').trim(); return /^\d+$/.test(text) ? text : undefined; }; export const chooseSpeechAudio = () => new Promise((resolve, reject) => { uni.chooseFile({ count: 1, type: 'all', extension: audioExtensions, success: (result) => { const tempFiles = Array.isArray(result.tempFiles) ? result.tempFiles : []; const file = (tempFiles[0] || {}) as Partial; const path = file.path || result.tempFilePaths?.[0] || ''; if (!path) { reject(new Error('未获取到音频文件')); return; } resolve({ path, name: file.name || path.split('/').pop() || 'practice-audio', size: Number(file.size || 0) }); }, fail: () => reject(new Error('未选择音频')) }); }); export type SpeechAbortRegistrar = (abort: (() => void) | null) => void; export const transcribeSpeechFile = (file: SpeechSelectedFile, registerAbort?: SpeechAbortRegistrar) => new Promise((resolve, reject) => { const task = uni.uploadFile({ url: apiUrl('/api/ai/asr'), filePath: file.path, name: 'file', header: authHeaders(false), success: (response) => { try { resolve(readTextPayload(response.statusCode, response.data)); } catch (error) { reject(error); } }, fail: () => reject(new Error('语音上传失败')), complete: () => registerAbort?.(null) }); registerAbort?.(() => task.abort()); }); export const transcribeSpeechBlob = ( blob: Blob, filename = 'practice-audio.webm', registerAbort?: SpeechAbortRegistrar ) => new Promise((resolve, reject) => { const form = new FormData(); form.append('file', blob, filename); const xhr = new XMLHttpRequest(); xhr.open('POST', apiUrl('/api/ai/asr')); xhr.timeout = 45000; Object.entries(authHeaders(false)).forEach(([key, value]) => xhr.setRequestHeader(key, value)); xhr.onload = () => { try { resolve(readTextPayload(xhr.status, xhr.responseText || '{}')); } catch (error) { reject(error); } }; xhr.onerror = () => reject(new Error('语音上传失败')); xhr.onabort = () => reject(new Error('语音转写已取消')); xhr.ontimeout = () => reject(new Error('语音转写超时')); xhr.onloadend = () => registerAbort?.(null); registerAbort?.(() => xhr.abort()); xhr.send(form); }); export const transcribeSpeechCapture = ( capture: SpeechCapture, filename: string, registerAbort?: SpeechAbortRegistrar ) => capture.kind === 'blob' ? transcribeSpeechBlob(capture.blob, filename, registerAbort) : transcribeSpeechFile(capture.file, registerAbort); export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => { const auth = getAuth(); return applyTtsVoicePreference( voiceProfile, getTtsVoicePreference(auth.tenantId, auth.phone), getTtsDialectPreference(auth.tenantId, auth.phone) ); }; export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => apiRequest({ url: '/api/ai/tts', method: 'POST', data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext }, timeout: 60000 }); export interface SpeechPlaybackSnapshot { key: string; status: SpeechPlaybackStatus; } const browserDialectLanguage = (dialect: SpeechVoiceProfile['dialect']) => { if (dialect === 'cantonese') return 'zh-HK'; if (dialect && dialect !== 'mandarin') return ''; return 'zh-CN'; }; export const createSpeechPlaybackController = ( onState: (snapshot: SpeechPlaybackSnapshot) => void, onError: (message: string) => void, defaultVoiceProfile?: SpeechVoiceProfile ) => { let audio: ReturnType | null = null; let browserUtterance: SpeechSynthesisUtterance | null = null; let activeKey = ''; let generation = 0; const sourceCache = new Map(); const maxCachedSources = 4; const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => JSON.stringify([text, profile || null, practiceContext || null]); const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => { activeKey = key; onState({ key, status }); }; const stop = () => { generation += 1; if (audio) { audio.stop(); audio.destroy(); audio = null; } if (typeof window !== 'undefined' && window.speechSynthesis && browserUtterance) { window.speechSynthesis.cancel(); browserUtterance = null; } publish(); }; const playWithBrowserSpeech = ( key: string, value: string, currentGeneration: number, profile?: SpeechVoiceProfile ): { played: boolean; message?: string } => { if ( typeof window === 'undefined' || !window.speechSynthesis || typeof window.SpeechSynthesisUtterance !== 'function' ) return { played: false }; const lang = browserDialectLanguage(profile?.dialect); if (!lang) { return { played: false, message: '当前设备语音不支持四川话,请使用服务端语音或文字查看' }; } const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600)); utterance.lang = lang; utterance.rate = profile?.speed || 0.95; utterance.pitch = profile?.role === 'mentor' ? 1.06 : profile?.role === 'customer' ? 1.05 : 1; utterance.onend = () => { if (currentGeneration === generation && activeKey === key) stop(); }; utterance.onerror = () => { if (currentGeneration !== generation || activeKey !== key) return; stop(); onError('当前设备语音播报不可用'); }; browserUtterance = utterance; publish(key, 'playing'); window.speechSynthesis.speak(utterance); return { played: true }; }; const playAudioSource = (key: string, source: string, cacheKey: string, currentGeneration: number) => { const isActivePlayback = () => currentGeneration === generation && activeKey === key; const failPlayback = () => { if (!isActivePlayback()) return; sourceCache.delete(cacheKey); stop(); onError('语音播放失败,请稍后重试'); }; const startPlayback = (src: string, cleanup?: () => void) => { const context = uni.createInnerAudioContext(); audio = context; const isCurrent = () => currentGeneration === generation && audio === context && activeKey === key; const finish = () => { if (cleanup) cleanup(); }; context.onPlay(() => { if (isCurrent()) publish(key, 'playing'); }); context.onEnded(() => { if (isCurrent()) { finish(); stop(); } }); context.onStop(() => { if (isCurrent()) { finish(); publish(); } }); context.onError(() => { if (!isCurrent()) return; finish(); failPlayback(); }); context.src = src; context.play(); }; // 后端 TTS 返回 data: URI base64;App-Plus 不支持 data: URI,iOS Safari/微信 // webview 也常静默失败,统一先转换:H5 转 Blob URL,App 端落临时文件。 if (isDataAudioUri(source)) { if (hasNativeAudioFileSupport()) { void writeNativeTempAudio(source).then((temp) => { if (!temp) { failPlayback(); return; } if (!isActivePlayback()) { removeNativeTempAudio(temp.plusPath); return; } startPlayback(temp.playbackUrl, () => removeNativeTempAudio(temp.plusPath)); }); return; } const blobUrl = dataUriToBlobUrl(source); if (blobUrl) { startPlayback(blobUrl, () => URL.revokeObjectURL(blobUrl)); return; } // Blob 不可用时退回原始 data: URI(桌面 Chrome 等环境可直接播放) } startPlayback(source); }; const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => { if (activeKey === key) { stop(); return; } stop(); const value = text.trim(); if (!value) return; const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile); const sourceKey = cacheKey(value, effectiveProfile, practiceContext); const currentGeneration = generation; publish(key, 'loading'); try { const cachedSource = sourceCache.get(sourceKey); if (cachedSource) { sourceCache.delete(sourceKey); sourceCache.set(sourceKey, cachedSource); playAudioSource(key, cachedSource, sourceKey, currentGeneration); return; } const result = await synthesizeSpeech(value, effectiveProfile, practiceContext); if (currentGeneration !== generation || activeKey !== key) return; const source = result.inlineAudioUrl || result.audioUrl; if (!source) throw new Error('语音合成未返回音频'); sourceCache.set(sourceKey, source); while (sourceCache.size > maxCachedSources) { const oldestKey = sourceCache.keys().next().value as string | undefined; if (!oldestKey) break; sourceCache.delete(oldestKey); } playAudioSource(key, source, sourceKey, currentGeneration); } catch (error) { if (currentGeneration !== generation) return; const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile); if (browserFallback.played) return; stop(); onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败')); } }; /** * Synthesize (and cache) audio for `text` without playing it, so a * WeChat-style voice bubble can show duration before the first tap. * Returns the audio source URL, or null when remote TTS is unavailable. */ const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise => { const value = text.trim(); if (!value) return null; const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile); const sourceKey = cacheKey(value, effectiveProfile, practiceContext); const cached = sourceCache.get(sourceKey); if (cached) return cached; try { const result = await synthesizeSpeech(value, effectiveProfile, practiceContext); const source = result.inlineAudioUrl || result.audioUrl; if (!source) return null; sourceCache.set(sourceKey, source); while (sourceCache.size > maxCachedSources) { const oldestKey = sourceCache.keys().next().value as string | undefined; if (!oldestKey) break; sourceCache.delete(oldestKey); } return source; } catch { return null; } }; const destroy = () => { stop(); sourceCache.clear(); }; return { toggle, stop, destroy, preload }; }; /** * data: URI 直接数帧,不经宿主音频 API。 * * App-Plus 的逻辑层是 JsCore,没有 `Audio`,原先在这里直接返回 null, * 导致语音条只显示「语音」而没有秒数。字节解析在 H5 上同样可用且更快, * 所以两端都优先走它,`Audio` 只作为远端 URL 的兜底。 */ const measureFromBytes = (source: string): number | null => { if (!isDataAudioUri(source)) return null; const decoded = decodeDataAudioUri(source); if (!decoded) return null; const seconds = estimateMpegAudioDuration(decoded.bytes); return seconds ? Math.max(1, Math.round(seconds)) : null; }; export const measureAudioDuration = (source: string): Promise => new Promise((resolve) => { const fromBytes = measureFromBytes(source); if (fromBytes !== null) { resolve(fromBytes); return; } if (typeof Audio === 'undefined') { resolve(null); return; } const probe = new Audio(); let settled = false; const finish = (value: number | null) => { if (settled) return; settled = true; probe.src = ''; resolve(value); }; const timer = setTimeout(() => finish(null), 8000); probe.preload = 'metadata'; probe.onloadedmetadata = () => { clearTimeout(timer); const duration = probe.duration; finish(Number.isFinite(duration) && duration > 0 ? Math.max(1, Math.round(duration)) : null); }; probe.onerror = () => { clearTimeout(timer); finish(null); }; probe.src = source; });