后端 TTS 以 data: URI 返回 base64 mp3(生产实证为合法 LAME mp3),但 App-Plus 的 InnerAudioContext 不支持 data: URI,iOS Safari/微信 webview 也常静默失败,统一报"语音播放失败"。播放前按平台转换:H5 转 Blob URL, App 端经 plus.io 落 _doc/ 临时文件后播放 file:// 路径,播完清理;Blob 不可 用时退回原始 data: URI。竞态守卫与缓存语义保持不变,158 单测全绿。
374 lines
13 KiB
TypeScript
374 lines
13 KiB
TypeScript
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
|
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
|
import { getAuth } from './auth';
|
|
import type { SpeechCapture } from './speech-capture';
|
|
import { dataUriToBlobUrl, hasNativeAudioFileSupport, isDataAudioUri, removeNativeTempAudio, writeNativeTempAudio } from './speech-audio-source';
|
|
import { applyTtsVoicePreference, getTtsDialectPreference, getTtsVoicePreference } from './tts-voice-preference';
|
|
|
|
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
|
|
|
|
export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.88, dialect: 'mandarin' };
|
|
export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96, dialect: 'mandarin' };
|
|
export const customerVoiceProfile = (
|
|
emotion = 0,
|
|
dialect: SpeechVoiceProfile['dialect'] = 'mandarin'
|
|
): SpeechVoiceProfile => {
|
|
if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12, dialect };
|
|
if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04, dialect };
|
|
return { role: 'customer', emotion: 'calm', speed: 0.98, dialect };
|
|
};
|
|
|
|
// OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。
|
|
export const normalizeOssId = (value: unknown): string | undefined => {
|
|
const text = String(value ?? '').trim();
|
|
return /^\d+$/.test(text) ? text : undefined;
|
|
};
|
|
|
|
export const chooseSpeechAudio = () =>
|
|
new Promise<SpeechSelectedFile>((resolve, reject) => {
|
|
uni.chooseFile({
|
|
count: 1,
|
|
type: 'all',
|
|
extension: audioExtensions,
|
|
success: (result) => {
|
|
const tempFiles = Array.isArray(result.tempFiles) ? result.tempFiles : [];
|
|
const file = (tempFiles[0] || {}) as Partial<SpeechSelectedFile>;
|
|
const path = file.path || result.tempFilePaths?.[0] || '';
|
|
if (!path) {
|
|
reject(new Error('未获取到音频文件'));
|
|
return;
|
|
}
|
|
resolve({
|
|
path,
|
|
name: file.name || path.split('/').pop() || 'practice-audio',
|
|
size: Number(file.size || 0)
|
|
});
|
|
},
|
|
fail: () => reject(new Error('未选择音频'))
|
|
});
|
|
});
|
|
|
|
export type SpeechAbortRegistrar = (abort: (() => void) | null) => void;
|
|
|
|
export const transcribeSpeechFile = (file: SpeechSelectedFile, registerAbort?: SpeechAbortRegistrar) =>
|
|
new Promise<AsrResponse>((resolve, reject) => {
|
|
const task = uni.uploadFile({
|
|
url: apiUrl('/api/ai/asr'),
|
|
filePath: file.path,
|
|
name: 'file',
|
|
header: authHeaders(false),
|
|
success: (response) => {
|
|
try {
|
|
resolve(readTextPayload<AsrResponse>(response.statusCode, response.data));
|
|
} catch (error) {
|
|
reject(error);
|
|
}
|
|
},
|
|
fail: () => reject(new Error('语音上传失败')),
|
|
complete: () => registerAbort?.(null)
|
|
});
|
|
registerAbort?.(() => task.abort());
|
|
});
|
|
|
|
export const transcribeSpeechBlob = (
|
|
blob: Blob,
|
|
filename = 'practice-audio.webm',
|
|
registerAbort?: SpeechAbortRegistrar
|
|
) =>
|
|
new Promise<AsrResponse>((resolve, reject) => {
|
|
const form = new FormData();
|
|
form.append('file', blob, filename);
|
|
|
|
const xhr = new XMLHttpRequest();
|
|
xhr.open('POST', apiUrl('/api/ai/asr'));
|
|
xhr.timeout = 45000;
|
|
Object.entries(authHeaders(false)).forEach(([key, value]) => xhr.setRequestHeader(key, value));
|
|
xhr.onload = () => {
|
|
try {
|
|
resolve(readTextPayload<AsrResponse>(xhr.status, xhr.responseText || '{}'));
|
|
} catch (error) {
|
|
reject(error);
|
|
}
|
|
};
|
|
xhr.onerror = () => reject(new Error('语音上传失败'));
|
|
xhr.onabort = () => reject(new Error('语音转写已取消'));
|
|
xhr.ontimeout = () => reject(new Error('语音转写超时'));
|
|
xhr.onloadend = () => registerAbort?.(null);
|
|
registerAbort?.(() => xhr.abort());
|
|
xhr.send(form);
|
|
});
|
|
|
|
export const transcribeSpeechCapture = (
|
|
capture: SpeechCapture,
|
|
filename: string,
|
|
registerAbort?: SpeechAbortRegistrar
|
|
) => capture.kind === 'blob'
|
|
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
|
|
: transcribeSpeechFile(capture.file, registerAbort);
|
|
|
|
export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => {
|
|
const auth = getAuth();
|
|
return applyTtsVoicePreference(
|
|
voiceProfile,
|
|
getTtsVoicePreference(auth.tenantId, auth.phone),
|
|
getTtsDialectPreference(auth.tenantId, auth.phone)
|
|
);
|
|
};
|
|
|
|
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
|
apiRequest<TtsResponse>({
|
|
url: '/api/ai/tts',
|
|
method: 'POST',
|
|
data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext },
|
|
timeout: 60000
|
|
});
|
|
|
|
export interface SpeechPlaybackSnapshot {
|
|
key: string;
|
|
status: SpeechPlaybackStatus;
|
|
}
|
|
|
|
const browserDialectLanguage = (dialect: SpeechVoiceProfile['dialect']) => {
|
|
if (dialect === 'cantonese') return 'zh-HK';
|
|
if (dialect && dialect !== 'mandarin') return '';
|
|
return 'zh-CN';
|
|
};
|
|
|
|
export const createSpeechPlaybackController = (
|
|
onState: (snapshot: SpeechPlaybackSnapshot) => void,
|
|
onError: (message: string) => void,
|
|
defaultVoiceProfile?: SpeechVoiceProfile
|
|
) => {
|
|
let audio: ReturnType<typeof uni.createInnerAudioContext> | null = null;
|
|
let browserUtterance: SpeechSynthesisUtterance | null = null;
|
|
let activeKey = '';
|
|
let generation = 0;
|
|
const sourceCache = new Map<string, string>();
|
|
const maxCachedSources = 4;
|
|
const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
|
JSON.stringify([text, profile || null, practiceContext || null]);
|
|
|
|
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
|
|
activeKey = key;
|
|
onState({ key, status });
|
|
};
|
|
|
|
const stop = () => {
|
|
generation += 1;
|
|
if (audio) {
|
|
audio.stop();
|
|
audio.destroy();
|
|
audio = null;
|
|
}
|
|
if (typeof window !== 'undefined' && window.speechSynthesis && browserUtterance) {
|
|
window.speechSynthesis.cancel();
|
|
browserUtterance = null;
|
|
}
|
|
publish();
|
|
};
|
|
|
|
const playWithBrowserSpeech = (
|
|
key: string,
|
|
value: string,
|
|
currentGeneration: number,
|
|
profile?: SpeechVoiceProfile
|
|
): { played: boolean; message?: string } => {
|
|
if (
|
|
typeof window === 'undefined'
|
|
|| !window.speechSynthesis
|
|
|| typeof window.SpeechSynthesisUtterance !== 'function'
|
|
) return { played: false };
|
|
|
|
const lang = browserDialectLanguage(profile?.dialect);
|
|
if (!lang) {
|
|
return { played: false, message: '当前设备语音不支持四川话,请使用服务端语音或文字查看' };
|
|
}
|
|
|
|
const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600));
|
|
utterance.lang = lang;
|
|
utterance.rate = profile?.speed || 0.95;
|
|
utterance.pitch = profile?.role === 'mentor' ? 1.06 : profile?.role === 'customer' ? 1.05 : 1;
|
|
utterance.onend = () => {
|
|
if (currentGeneration === generation && activeKey === key) stop();
|
|
};
|
|
utterance.onerror = () => {
|
|
if (currentGeneration !== generation || activeKey !== key) return;
|
|
stop();
|
|
onError('当前设备语音播报不可用');
|
|
};
|
|
browserUtterance = utterance;
|
|
publish(key, 'playing');
|
|
window.speechSynthesis.speak(utterance);
|
|
return { played: true };
|
|
};
|
|
|
|
const playAudioSource = (key: string, source: string, cacheKey: string, currentGeneration: number) => {
|
|
const isActivePlayback = () => currentGeneration === generation && activeKey === key;
|
|
const failPlayback = () => {
|
|
if (!isActivePlayback()) return;
|
|
sourceCache.delete(cacheKey);
|
|
stop();
|
|
onError('语音播放失败,请稍后重试');
|
|
};
|
|
const startPlayback = (src: string, cleanup?: () => void) => {
|
|
const context = uni.createInnerAudioContext();
|
|
audio = context;
|
|
const isCurrent = () => currentGeneration === generation && audio === context && activeKey === key;
|
|
const finish = () => {
|
|
if (cleanup) cleanup();
|
|
};
|
|
context.onPlay(() => {
|
|
if (isCurrent()) publish(key, 'playing');
|
|
});
|
|
context.onEnded(() => {
|
|
if (isCurrent()) {
|
|
finish();
|
|
stop();
|
|
}
|
|
});
|
|
context.onStop(() => {
|
|
if (isCurrent()) {
|
|
finish();
|
|
publish();
|
|
}
|
|
});
|
|
context.onError(() => {
|
|
if (!isCurrent()) return;
|
|
finish();
|
|
failPlayback();
|
|
});
|
|
context.src = src;
|
|
context.play();
|
|
};
|
|
|
|
// 后端 TTS 返回 data: URI base64;App-Plus 不支持 data: URI,iOS Safari/微信
|
|
// webview 也常静默失败,统一先转换:H5 转 Blob URL,App 端落临时文件。
|
|
if (isDataAudioUri(source)) {
|
|
if (hasNativeAudioFileSupport()) {
|
|
void writeNativeTempAudio(source).then((fileUrl) => {
|
|
if (!fileUrl) {
|
|
failPlayback();
|
|
return;
|
|
}
|
|
if (!isActivePlayback()) {
|
|
removeNativeTempAudio(fileUrl);
|
|
return;
|
|
}
|
|
startPlayback(fileUrl, () => removeNativeTempAudio(fileUrl));
|
|
});
|
|
return;
|
|
}
|
|
const blobUrl = dataUriToBlobUrl(source);
|
|
if (blobUrl) {
|
|
startPlayback(blobUrl, () => URL.revokeObjectURL(blobUrl));
|
|
return;
|
|
}
|
|
// Blob 不可用时退回原始 data: URI(桌面 Chrome 等环境可直接播放)
|
|
}
|
|
startPlayback(source);
|
|
};
|
|
|
|
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => {
|
|
if (activeKey === key) {
|
|
stop();
|
|
return;
|
|
}
|
|
stop();
|
|
const value = text.trim();
|
|
if (!value) return;
|
|
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
|
|
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
|
|
const currentGeneration = generation;
|
|
publish(key, 'loading');
|
|
try {
|
|
const cachedSource = sourceCache.get(sourceKey);
|
|
if (cachedSource) {
|
|
sourceCache.delete(sourceKey);
|
|
sourceCache.set(sourceKey, cachedSource);
|
|
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
|
return;
|
|
}
|
|
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
|
|
if (currentGeneration !== generation || activeKey !== key) return;
|
|
const source = result.inlineAudioUrl || result.audioUrl;
|
|
if (!source) throw new Error('语音合成未返回音频');
|
|
sourceCache.set(sourceKey, source);
|
|
while (sourceCache.size > maxCachedSources) {
|
|
const oldestKey = sourceCache.keys().next().value as string | undefined;
|
|
if (!oldestKey) break;
|
|
sourceCache.delete(oldestKey);
|
|
}
|
|
playAudioSource(key, source, sourceKey, currentGeneration);
|
|
} catch (error) {
|
|
if (currentGeneration !== generation) return;
|
|
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile);
|
|
if (browserFallback.played) return;
|
|
stop();
|
|
onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败'));
|
|
}
|
|
};
|
|
|
|
/**
|
|
* Synthesize (and cache) audio for `text` without playing it, so a
|
|
* WeChat-style voice bubble can show duration before the first tap.
|
|
* Returns the audio source URL, or null when remote TTS is unavailable.
|
|
*/
|
|
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
|
|
const value = text.trim();
|
|
if (!value) return null;
|
|
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
|
|
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
|
|
const cached = sourceCache.get(sourceKey);
|
|
if (cached) return cached;
|
|
try {
|
|
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
|
|
const source = result.inlineAudioUrl || result.audioUrl;
|
|
if (!source) return null;
|
|
sourceCache.set(sourceKey, source);
|
|
while (sourceCache.size > maxCachedSources) {
|
|
const oldestKey = sourceCache.keys().next().value as string | undefined;
|
|
if (!oldestKey) break;
|
|
sourceCache.delete(oldestKey);
|
|
}
|
|
return source;
|
|
} catch {
|
|
return null;
|
|
}
|
|
};
|
|
|
|
const destroy = () => {
|
|
stop();
|
|
sourceCache.clear();
|
|
};
|
|
|
|
return { toggle, stop, destroy, preload };
|
|
};
|
|
|
|
export const measureAudioDuration = (source: string): Promise<number | null> =>
|
|
new Promise((resolve) => {
|
|
if (typeof Audio === 'undefined') {
|
|
resolve(null);
|
|
return;
|
|
}
|
|
const probe = new Audio();
|
|
let settled = false;
|
|
const finish = (value: number | null) => {
|
|
if (settled) return;
|
|
settled = true;
|
|
probe.src = '';
|
|
resolve(value);
|
|
};
|
|
const timer = setTimeout(() => finish(null), 8000);
|
|
probe.preload = 'metadata';
|
|
probe.onloadedmetadata = () => {
|
|
clearTimeout(timer);
|
|
const duration = probe.duration;
|
|
finish(Number.isFinite(duration) && duration > 0 ? Math.max(1, Math.round(duration)) : null);
|
|
};
|
|
probe.onerror = () => {
|
|
clearTimeout(timer);
|
|
finish(null);
|
|
};
|
|
probe.src = source;
|
|
});
|