feat(speech): add qwen tts voice preferences
This commit is contained in:
@@ -1,6 +1,8 @@
|
||||
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
||||
import { getAuth } from './auth';
|
||||
import type { SpeechCapture } from './speech-capture';
|
||||
import { applyTtsVoicePreference, getTtsDialectPreference, getTtsVoicePreference } from './tts-voice-preference';
|
||||
|
||||
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
|
||||
|
||||
@@ -103,11 +105,20 @@ export const transcribeSpeechCapture = (
|
||||
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
|
||||
: transcribeSpeechFile(capture.file, registerAbort);
|
||||
|
||||
export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => {
|
||||
const auth = getAuth();
|
||||
return applyTtsVoicePreference(
|
||||
voiceProfile,
|
||||
getTtsVoicePreference(auth.tenantId, auth.phone),
|
||||
getTtsDialectPreference(auth.tenantId, auth.phone)
|
||||
);
|
||||
};
|
||||
|
||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
||||
apiRequest<TtsResponse>({
|
||||
url: '/api/ai/tts',
|
||||
method: 'POST',
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
|
||||
data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext },
|
||||
timeout: 60000
|
||||
});
|
||||
|
||||
@@ -118,7 +129,7 @@ export interface SpeechPlaybackSnapshot {
|
||||
|
||||
const browserDialectLanguage = (dialect: SpeechVoiceProfile['dialect']) => {
|
||||
if (dialect === 'cantonese') return 'zh-HK';
|
||||
if (dialect === 'sichuanese') return '';
|
||||
if (dialect && dialect !== 'mandarin') return '';
|
||||
return 'zh-CN';
|
||||
};
|
||||
|
||||
@@ -221,8 +232,8 @@ export const createSpeechPlaybackController = (
|
||||
stop();
|
||||
const value = text.trim();
|
||||
if (!value) return;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
||||
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
|
||||
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
|
||||
const currentGeneration = generation;
|
||||
publish(key, 'loading');
|
||||
try {
|
||||
@@ -233,7 +244,7 @@ export const createSpeechPlaybackController = (
|
||||
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
||||
return;
|
||||
}
|
||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
||||
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
|
||||
if (currentGeneration !== generation || activeKey !== key) return;
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) throw new Error('语音合成未返回音频');
|
||||
@@ -246,7 +257,7 @@ export const createSpeechPlaybackController = (
|
||||
playAudioSource(key, source, sourceKey, currentGeneration);
|
||||
} catch (error) {
|
||||
if (currentGeneration !== generation) return;
|
||||
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, profile);
|
||||
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile);
|
||||
if (browserFallback.played) return;
|
||||
stop();
|
||||
onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败'));
|
||||
@@ -261,12 +272,12 @@ export const createSpeechPlaybackController = (
|
||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
|
||||
const value = text.trim();
|
||||
if (!value) return null;
|
||||
const profile = voiceProfile || defaultVoiceProfile;
|
||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
||||
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
|
||||
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
|
||||
const cached = sourceCache.get(sourceKey);
|
||||
if (cached) return cached;
|
||||
try {
|
||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
||||
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
|
||||
const source = result.inlineAudioUrl || result.audioUrl;
|
||||
if (!source) return null;
|
||||
sourceCache.set(sourceKey, source);
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
import type { SpeechVoiceProfile } from '@/types/api';
|
||||
|
||||
export const qwenTtsVoiceOptions = [
|
||||
{ value: '', label: '跟随场景' },
|
||||
{ value: 'longanhuan_v3.6', label: '龙安欢 · 中文女声' },
|
||||
{ value: 'longjielidou_v3.6', label: '龙杰力豆 · 童声' },
|
||||
{ value: 'loongeva_v3.6', label: 'Loongeva · 英文女声' },
|
||||
{ value: 'loongjohn', label: 'loongJohn · 英文男声' }
|
||||
] as const;
|
||||
|
||||
export const qwenTtsDialectOptions = [
|
||||
{ value: '', label: '跟随场景' },
|
||||
{ value: 'mandarin', label: '普通话' },
|
||||
{ value: 'cantonese', label: '粤语' },
|
||||
{ value: 'chongqing', label: '重庆话' },
|
||||
{ value: 'northeastern', label: '东北话' },
|
||||
{ value: 'gansu', label: '甘肃话' },
|
||||
{ value: 'guizhou', label: '贵州话' },
|
||||
{ value: 'zhejiang', label: '浙江话' },
|
||||
{ value: 'hebei', label: '河北话' },
|
||||
{ value: 'henan', label: '河南话' },
|
||||
{ value: 'hubei', label: '湖北话' },
|
||||
{ value: 'hunan', label: '湖南话' },
|
||||
{ value: 'jiangxi', label: '江西话' },
|
||||
{ value: 'ningbo', label: '宁波话' },
|
||||
{ value: 'ningxia', label: '宁夏话' },
|
||||
{ value: 'qingdao', label: '青岛话' },
|
||||
{ value: 'shaanxi', label: '陕西话' },
|
||||
{ value: 'shanxi', label: '山西话' },
|
||||
{ value: 'shandong', label: '山东话' },
|
||||
{ value: 'shanghai', label: '上海话' },
|
||||
{ value: 'sichuanese', label: '四川话' },
|
||||
{ value: 'yunnan', label: '云南话' }
|
||||
] as const;
|
||||
|
||||
export type TtsVoicePreference = typeof qwenTtsVoiceOptions[number]['value'];
|
||||
export type TtsDialectPreference = typeof qwenTtsDialectOptions[number]['value'];
|
||||
|
||||
const preferenceKeyPrefix = 'aihr_tts_voice';
|
||||
const dialectPreferenceKeyPrefix = 'aihr_tts_dialect';
|
||||
|
||||
export const normalizeTtsVoicePreference = (value: unknown): TtsVoicePreference =>
|
||||
qwenTtsVoiceOptions.some((option) => option.value === value) ? value as TtsVoicePreference : '';
|
||||
|
||||
export const normalizeTtsDialectPreference = (value: unknown): TtsDialectPreference =>
|
||||
qwenTtsDialectOptions.some((option) => option.value === value) ? value as TtsDialectPreference : '';
|
||||
|
||||
export const ttsVoicePreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
|
||||
const tenant = String(tenantId || '').trim();
|
||||
const account = String(phone || '').trim();
|
||||
return tenant && account ? `${preferenceKeyPrefix}:${tenant}:${account}` : '';
|
||||
};
|
||||
|
||||
export const getTtsVoicePreference = (tenantId: unknown, phone: unknown): TtsVoicePreference => {
|
||||
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
|
||||
return key ? normalizeTtsVoicePreference(uni.getStorageSync(key)) : '';
|
||||
};
|
||||
|
||||
const ttsDialectPreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
|
||||
const tenant = String(tenantId || '').trim();
|
||||
const account = String(phone || '').trim();
|
||||
return tenant && account ? `${dialectPreferenceKeyPrefix}:${tenant}:${account}` : '';
|
||||
};
|
||||
|
||||
export const getTtsDialectPreference = (tenantId: unknown, phone: unknown): TtsDialectPreference => {
|
||||
const key = ttsDialectPreferenceStorageKey(tenantId, phone);
|
||||
return key ? normalizeTtsDialectPreference(uni.getStorageSync(key)) : '';
|
||||
};
|
||||
|
||||
export const setTtsVoicePreference = (tenantId: unknown, phone: unknown, value: unknown): TtsVoicePreference => {
|
||||
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
|
||||
const normalized = normalizeTtsVoicePreference(value);
|
||||
if (key) uni.setStorageSync(key, normalized);
|
||||
return normalized;
|
||||
};
|
||||
|
||||
export const setTtsDialectPreference = (tenantId: unknown, phone: unknown, value: unknown): TtsDialectPreference => {
|
||||
const key = ttsDialectPreferenceStorageKey(tenantId, phone);
|
||||
const normalized = normalizeTtsDialectPreference(value);
|
||||
if (key) uni.setStorageSync(key, normalized);
|
||||
return normalized;
|
||||
};
|
||||
|
||||
export const applyTtsVoicePreference = (
|
||||
profile: SpeechVoiceProfile | undefined,
|
||||
preference: unknown,
|
||||
dialectPreference?: unknown
|
||||
): SpeechVoiceProfile | undefined => {
|
||||
const voice = normalizeTtsVoicePreference(preference);
|
||||
const dialect = normalizeTtsDialectPreference(dialectPreference);
|
||||
if (!voice && !dialect) return profile;
|
||||
const next: SpeechVoiceProfile = { ...(profile || { role: 'neutral' }) };
|
||||
if (voice) next.voice = voice;
|
||||
if (dialect) next.dialect = dialect;
|
||||
return next;
|
||||
};
|
||||
Reference in New Issue
Block a user