Files
prop-ai-hr/mobile-uni/src/services/speech.ts
T

303 lines
10 KiB
TypeScript

import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import type { SpeechCapture } from './speech-capture';
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.88 };
export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96 };
export const customerVoiceProfile = (emotion = 0): SpeechVoiceProfile => {
if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12 };
if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04 };
return { role: 'customer', emotion: 'calm', speed: 0.98 };
};
// OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。
export const normalizeOssId = (value: unknown): string | undefined => {
const text = String(value ?? '').trim();
return /^\d+$/.test(text) ? text : undefined;
};
export const chooseSpeechAudio = () =>
new Promise<SpeechSelectedFile>((resolve, reject) => {
uni.chooseFile({
count: 1,
type: 'all',
extension: audioExtensions,
success: (result) => {
const tempFiles = Array.isArray(result.tempFiles) ? result.tempFiles : [];
const file = (tempFiles[0] || {}) as Partial<SpeechSelectedFile>;
const path = file.path || result.tempFilePaths?.[0] || '';
if (!path) {
reject(new Error('未获取到音频文件'));
return;
}
resolve({
path,
name: file.name || path.split('/').pop() || 'practice-audio',
size: Number(file.size || 0)
});
},
fail: () => reject(new Error('未选择音频'))
});
});
export type SpeechAbortRegistrar = (abort: (() => void) | null) => void;
export const transcribeSpeechFile = (file: SpeechSelectedFile, registerAbort?: SpeechAbortRegistrar) =>
new Promise<AsrResponse>((resolve, reject) => {
const task = uni.uploadFile({
url: apiUrl('/api/ai/asr'),
filePath: file.path,
name: 'file',
header: authHeaders(false),
success: (response) => {
try {
resolve(readTextPayload<AsrResponse>(response.statusCode, response.data));
} catch (error) {
reject(error);
}
},
fail: () => reject(new Error('语音上传失败')),
complete: () => registerAbort?.(null)
});
registerAbort?.(() => task.abort());
});
export const transcribeSpeechBlob = (
blob: Blob,
filename = 'practice-audio.webm',
registerAbort?: SpeechAbortRegistrar
) =>
new Promise<AsrResponse>((resolve, reject) => {
const form = new FormData();
form.append('file', blob, filename);
const xhr = new XMLHttpRequest();
xhr.open('POST', apiUrl('/api/ai/asr'));
xhr.timeout = 45000;
Object.entries(authHeaders(false)).forEach(([key, value]) => xhr.setRequestHeader(key, value));
xhr.onload = () => {
try {
resolve(readTextPayload<AsrResponse>(xhr.status, xhr.responseText || '{}'));
} catch (error) {
reject(error);
}
};
xhr.onerror = () => reject(new Error('语音上传失败'));
xhr.onabort = () => reject(new Error('语音转写已取消'));
xhr.ontimeout = () => reject(new Error('语音转写超时'));
xhr.onloadend = () => registerAbort?.(null);
registerAbort?.(() => xhr.abort());
xhr.send(form);
});
export const transcribeSpeechCapture = (
capture: SpeechCapture,
filename: string,
registerAbort?: SpeechAbortRegistrar
) => capture.kind === 'blob'
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
: transcribeSpeechFile(capture.file, registerAbort);
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
apiRequest<TtsResponse>({
url: '/api/ai/tts',
method: 'POST',
data: { text: text.trim().slice(0, 300), voiceProfile },
timeout: 60000
});
export interface SpeechPlaybackSnapshot {
key: string;
status: SpeechPlaybackStatus;
}
export const createSpeechPlaybackController = (
onState: (snapshot: SpeechPlaybackSnapshot) => void,
onError: (message: string) => void,
defaultVoiceProfile?: SpeechVoiceProfile
) => {
let audio: ReturnType<typeof uni.createInnerAudioContext> | null = null;
let browserUtterance: SpeechSynthesisUtterance | null = null;
let activeKey = '';
let generation = 0;
const sourceCache = new Map<string, string>();
const maxCachedSources = 4;
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
activeKey = key;
onState({ key, status });
};
const stop = () => {
generation += 1;
if (audio) {
audio.stop();
audio.destroy();
audio = null;
}
if (typeof window !== 'undefined' && window.speechSynthesis && browserUtterance) {
window.speechSynthesis.cancel();
browserUtterance = null;
}
publish();
};
const playWithBrowserSpeech = (
key: string,
value: string,
currentGeneration: number,
profile?: SpeechVoiceProfile
) => {
if (
typeof window === 'undefined'
|| !window.speechSynthesis
|| typeof window.SpeechSynthesisUtterance !== 'function'
) return false;
const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600));
utterance.lang = 'zh-CN';
utterance.rate = profile?.speed || 0.95;
utterance.pitch = profile?.role === 'mentor' ? 1.06 : profile?.role === 'customer' ? 1.05 : 1;
utterance.onend = () => {
if (currentGeneration === generation && activeKey === key) stop();
};
utterance.onerror = () => {
if (currentGeneration !== generation || activeKey !== key) return;
stop();
onError('当前设备语音播报不可用');
};
browserUtterance = utterance;
publish(key, 'playing');
window.speechSynthesis.speak(utterance);
return true;
};
const playAudioSource = (key: string, source: string, cacheKey: string, currentGeneration: number) => {
const context = uni.createInnerAudioContext();
audio = context;
const isCurrent = () => currentGeneration === generation && audio === context && activeKey === key;
context.onPlay(() => {
if (isCurrent()) publish(key, 'playing');
});
context.onEnded(() => {
if (isCurrent()) stop();
});
context.onStop(() => {
if (isCurrent()) publish();
});
context.onError(() => {
if (!isCurrent()) return;
sourceCache.delete(cacheKey);
stop();
onError('语音播放失败,请稍后重试');
});
context.src = source;
context.play();
};
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
if (activeKey === key) {
stop();
return;
}
stop();
const value = text.trim();
if (!value) return;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const currentGeneration = generation;
publish(key, 'loading');
try {
const cachedSource = sourceCache.get(sourceKey);
if (cachedSource) {
sourceCache.delete(sourceKey);
sourceCache.set(sourceKey, cachedSource);
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return;
}
const result = await synthesizeSpeech(value, profile);
if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频');
sourceCache.set(sourceKey, source);
while (sourceCache.size > maxCachedSources) {
const oldestKey = sourceCache.keys().next().value as string | undefined;
if (!oldestKey) break;
sourceCache.delete(oldestKey);
}
playAudioSource(key, source, sourceKey, currentGeneration);
} catch (error) {
if (currentGeneration !== generation) return;
if (playWithBrowserSpeech(key, value, currentGeneration, profile)) return;
stop();
onError(error instanceof Error ? error.message : '语音生成失败');
}
};
/**
* Synthesize (and cache) audio for `text` without playing it, so a
* WeChat-style voice bubble can show duration before the first tap.
* Returns the audio source URL, or null when remote TTS is unavailable.
*/
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
const value = text.trim();
if (!value) return null;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const cached = sourceCache.get(sourceKey);
if (cached) return cached;
try {
const result = await synthesizeSpeech(value, profile);
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null;
sourceCache.set(sourceKey, source);
while (sourceCache.size > maxCachedSources) {
const oldestKey = sourceCache.keys().next().value as string | undefined;
if (!oldestKey) break;
sourceCache.delete(oldestKey);
}
return source;
} catch {
return null;
}
};
const destroy = () => {
stop();
sourceCache.clear();
};
return { toggle, stop, destroy, preload };
};
export const measureAudioDuration = (source: string): Promise<number | null> =>
new Promise((resolve) => {
if (typeof Audio === 'undefined') {
resolve(null);
return;
}
const probe = new Audio();
let settled = false;
const finish = (value: number | null) => {
if (settled) return;
settled = true;
probe.src = '';
resolve(value);
};
const timer = setTimeout(() => finish(null), 8000);
probe.preload = 'metadata';
probe.onloadedmetadata = () => {
clearTimeout(timer);
const duration = probe.duration;
finish(Number.isFinite(duration) && duration > 0 ? Math.max(1, Math.round(duration)) : null);
};
probe.onerror = () => {
clearTimeout(timer);
finish(null);
};
probe.src = source;
});