feat(speech): add qwen tts voice preferences

This commit is contained in:
2026-07-24 01:54:37 +08:00
parent faa2a77e9a
commit 474d46e728
8 changed files with 523 additions and 30 deletions
@@ -271,6 +271,25 @@
</view>
<text class="display-note">切换后同步到今日、练、问、我全部页面,并自动保存。</text>
</template>
<view v-if="loggedIn" class="voice-preference-row">
<view class="display-heading-copy">
<text class="menu-item-title">AI 播报音色</text>
<text class="menu-item-desc">{{ currentTtsVoiceLabel }} · 对练情绪和语速仍会保留</text>
</view>
<picker :range="qwenTtsVoiceOptions" range-key="label" :value="ttsVoiceOptionIndex" @change="changeTtsVoice">
<view class="voice-picker-value">{{ currentTtsVoiceLabel }} <uni-icons type="right" size="12" color="#94a3b8" /></view>
</picker>
</view>
<view v-if="loggedIn" class="voice-preference-row">
<view class="display-heading-copy">
<text class="menu-item-title">AI 播报方言</text>
<text class="menu-item-desc">{{ currentTtsDialectLabel }} · 仅影响服务端语音,失败时不退回普通话</text>
</view>
<picker :range="qwenTtsDialectOptions" range-key="label" :value="ttsDialectOptionIndex" @change="changeTtsDialect">
<view class="voice-picker-value">{{ currentTtsDialectLabel }} <uni-icons type="right" size="12" color="#94a3b8" /></view>
</picker>
</view>
</view>
</view>
</template>
@@ -297,6 +316,16 @@ import {
import { getCompetencyProfile, getPracticeHistory, getPromotionEvidence } from '@/services/practice';
import { resetPageScroll } from '@/services/navigation';
import { ensureEmployeePosition } from '@/services/position';
import {
getTtsVoicePreference,
getTtsDialectPreference,
qwenTtsDialectOptions,
qwenTtsVoiceOptions,
setTtsDialectPreference,
setTtsVoicePreference,
type TtsDialectPreference,
type TtsVoicePreference
} from '@/services/tts-voice-preference';
const loggedIn = ref(false);
const phone = ref('');
@@ -313,9 +342,19 @@ const evidence = ref<PromotionEvidence | null>(null);
const positionProfile = ref(getSelectedPositionProfile());
const supervisorLearning = ref(false);
const fontSize = ref<AppFontSize>(getAppFontSize());
const ttsVoice = ref<TtsVoicePreference>('');
const ttsDialect = ref<TtsDialectPreference>('');
const currentFontSizeLabel = computed(() =>
appFontSizeOptions.find((item) => item.value === fontSize.value)?.label || '标准'
);
const ttsVoiceOptionIndex = computed(() => Math.max(0, qwenTtsVoiceOptions.findIndex((item) => item.value === ttsVoice.value)));
const currentTtsVoiceLabel = computed(() =>
qwenTtsVoiceOptions[ttsVoiceOptionIndex.value]?.label || '跟随场景'
);
const ttsDialectOptionIndex = computed(() => Math.max(0, qwenTtsDialectOptions.findIndex((item) => item.value === ttsDialect.value)));
const currentTtsDialectLabel = computed(() =>
qwenTtsDialectOptions[ttsDialectOptionIndex.value]?.label || '跟随场景'
);
const maskedPhone = computed(() => {
const value = phone.value.trim();
if (!value) return '未登录';
@@ -333,6 +372,8 @@ const refreshAuth = () => {
positionProfile.value = getSelectedPositionProfile();
supervisorLearning.value = isSupervisorLearnerMode();
fontSize.value = getAppFontSize();
ttsVoice.value = getTtsVoicePreference(auth.tenantId, auth.phone);
ttsDialect.value = getTtsDialectPreference(auth.tenantId, auth.phone);
};
const changeFontSize = (value: AppFontSize) => {
@@ -340,6 +381,20 @@ const changeFontSize = (value: AppFontSize) => {
uni.showToast({ title: `已切换为${currentFontSizeLabel.value}`, icon: 'none' });
};
const changeTtsVoice = (event: { detail?: { value?: string | number } }) => {
const option = qwenTtsVoiceOptions[Number(event.detail?.value)];
const auth = getAuth();
ttsVoice.value = setTtsVoicePreference(auth.tenantId, auth.phone, option?.value);
uni.showToast({ title: `已切换为${currentTtsVoiceLabel.value}`, icon: 'none' });
};
const changeTtsDialect = (event: { detail?: { value?: string | number } }) => {
const option = qwenTtsDialectOptions[Number(event.detail?.value)];
const auth = getAuth();
ttsDialect.value = setTtsDialectPreference(auth.tenantId, auth.phone, option?.value);
uni.showToast({ title: `已切换为${currentTtsDialectLabel.value}`, icon: 'none' });
};
const returnToSupervisor = () => {
leaveSupervisorLearnerMode();
supervisorLearning.value = false;
@@ -650,6 +705,28 @@ onShow(() => {
box-shadow: 0 4px 12px rgba(24, 34, 48, 0.08);
}
.voice-preference-row {
display: flex;
align-items: center;
justify-content: space-between;
gap: 12px;
padding-top: 12px;
border-top: 1px solid var(--employee-line);
}
.voice-picker-value {
display: flex;
align-items: center;
gap: 3px;
max-width: 150px;
padding: 8px 10px;
border: 1px solid #e3e6ea;
border-radius: 10px;
color: var(--employee-text);
font-size: 12px;
white-space: nowrap;
}
.profile-summary {
border-color: var(--employee-line);
}
+20 -9
View File
@@ -1,6 +1,8 @@
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import { getAuth } from './auth';
import type { SpeechCapture } from './speech-capture';
import { applyTtsVoicePreference, getTtsDialectPreference, getTtsVoicePreference } from './tts-voice-preference';
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
@@ -103,11 +105,20 @@ export const transcribeSpeechCapture = (
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
: transcribeSpeechFile(capture.file, registerAbort);
export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => {
const auth = getAuth();
return applyTtsVoicePreference(
voiceProfile,
getTtsVoicePreference(auth.tenantId, auth.phone),
getTtsDialectPreference(auth.tenantId, auth.phone)
);
};
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
apiRequest<TtsResponse>({
url: '/api/ai/tts',
method: 'POST',
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext },
timeout: 60000
});
@@ -118,7 +129,7 @@ export interface SpeechPlaybackSnapshot {
const browserDialectLanguage = (dialect: SpeechVoiceProfile['dialect']) => {
if (dialect === 'cantonese') return 'zh-HK';
if (dialect === 'sichuanese') return '';
if (dialect && dialect !== 'mandarin') return '';
return 'zh-CN';
};
@@ -221,8 +232,8 @@ export const createSpeechPlaybackController = (
stop();
const value = text.trim();
if (!value) return;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile, practiceContext);
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
const currentGeneration = generation;
publish(key, 'loading');
try {
@@ -233,7 +244,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return;
}
const result = await synthesizeSpeech(value, profile, practiceContext);
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频');
@@ -246,7 +257,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, source, sourceKey, currentGeneration);
} catch (error) {
if (currentGeneration !== generation) return;
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, profile);
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile);
if (browserFallback.played) return;
stop();
onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败'));
@@ -261,12 +272,12 @@ export const createSpeechPlaybackController = (
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
const value = text.trim();
if (!value) return null;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile, practiceContext);
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
const cached = sourceCache.get(sourceKey);
if (cached) return cached;
try {
const result = await synthesizeSpeech(value, profile, practiceContext);
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null;
sourceCache.set(sourceKey, source);
@@ -0,0 +1,96 @@
import type { SpeechVoiceProfile } from '@/types/api';
export const qwenTtsVoiceOptions = [
{ value: '', label: '跟随场景' },
{ value: 'longanhuan_v3.6', label: '龙安欢 · 中文女声' },
{ value: 'longjielidou_v3.6', label: '龙杰力豆 · 童声' },
{ value: 'loongeva_v3.6', label: 'Loongeva · 英文女声' },
{ value: 'loongjohn', label: 'loongJohn · 英文男声' }
] as const;
export const qwenTtsDialectOptions = [
{ value: '', label: '跟随场景' },
{ value: 'mandarin', label: '普通话' },
{ value: 'cantonese', label: '粤语' },
{ value: 'chongqing', label: '重庆话' },
{ value: 'northeastern', label: '东北话' },
{ value: 'gansu', label: '甘肃话' },
{ value: 'guizhou', label: '贵州话' },
{ value: 'zhejiang', label: '浙江话' },
{ value: 'hebei', label: '河北话' },
{ value: 'henan', label: '河南话' },
{ value: 'hubei', label: '湖北话' },
{ value: 'hunan', label: '湖南话' },
{ value: 'jiangxi', label: '江西话' },
{ value: 'ningbo', label: '宁波话' },
{ value: 'ningxia', label: '宁夏话' },
{ value: 'qingdao', label: '青岛话' },
{ value: 'shaanxi', label: '陕西话' },
{ value: 'shanxi', label: '山西话' },
{ value: 'shandong', label: '山东话' },
{ value: 'shanghai', label: '上海话' },
{ value: 'sichuanese', label: '四川话' },
{ value: 'yunnan', label: '云南话' }
] as const;
export type TtsVoicePreference = typeof qwenTtsVoiceOptions[number]['value'];
export type TtsDialectPreference = typeof qwenTtsDialectOptions[number]['value'];
const preferenceKeyPrefix = 'aihr_tts_voice';
const dialectPreferenceKeyPrefix = 'aihr_tts_dialect';
export const normalizeTtsVoicePreference = (value: unknown): TtsVoicePreference =>
qwenTtsVoiceOptions.some((option) => option.value === value) ? value as TtsVoicePreference : '';
export const normalizeTtsDialectPreference = (value: unknown): TtsDialectPreference =>
qwenTtsDialectOptions.some((option) => option.value === value) ? value as TtsDialectPreference : '';
export const ttsVoicePreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
const tenant = String(tenantId || '').trim();
const account = String(phone || '').trim();
return tenant && account ? `${preferenceKeyPrefix}:${tenant}:${account}` : '';
};
export const getTtsVoicePreference = (tenantId: unknown, phone: unknown): TtsVoicePreference => {
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
return key ? normalizeTtsVoicePreference(uni.getStorageSync(key)) : '';
};
const ttsDialectPreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
const tenant = String(tenantId || '').trim();
const account = String(phone || '').trim();
return tenant && account ? `${dialectPreferenceKeyPrefix}:${tenant}:${account}` : '';
};
export const getTtsDialectPreference = (tenantId: unknown, phone: unknown): TtsDialectPreference => {
const key = ttsDialectPreferenceStorageKey(tenantId, phone);
return key ? normalizeTtsDialectPreference(uni.getStorageSync(key)) : '';
};
export const setTtsVoicePreference = (tenantId: unknown, phone: unknown, value: unknown): TtsVoicePreference => {
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
const normalized = normalizeTtsVoicePreference(value);
if (key) uni.setStorageSync(key, normalized);
return normalized;
};
export const setTtsDialectPreference = (tenantId: unknown, phone: unknown, value: unknown): TtsDialectPreference => {
const key = ttsDialectPreferenceStorageKey(tenantId, phone);
const normalized = normalizeTtsDialectPreference(value);
if (key) uni.setStorageSync(key, normalized);
return normalized;
};
export const applyTtsVoicePreference = (
profile: SpeechVoiceProfile | undefined,
preference: unknown,
dialectPreference?: unknown
): SpeechVoiceProfile | undefined => {
const voice = normalizeTtsVoicePreference(preference);
const dialect = normalizeTtsDialectPreference(dialectPreference);
if (!voice && !dialect) return profile;
const next: SpeechVoiceProfile = { ...(profile || { role: 'neutral' }) };
if (voice) next.voice = voice;
if (dialect) next.dialect = dialect;
return next;
};
+24 -1
View File
@@ -45,7 +45,7 @@ export interface SpeechVoiceProfile {
voice?: string;
speed?: number;
emotion?: 'warm' | 'professional' | 'serious' | 'intense' | 'calm';
dialect?: 'mandarin' | 'cantonese' | 'sichuanese';
dialect?: 'mandarin' | 'cantonese' | 'chongqing' | 'northeastern' | 'gansu' | 'guizhou' | 'zhejiang' | 'hebei' | 'henan' | 'hubei' | 'hunan' | 'jiangxi' | 'ningbo' | 'ningxia' | 'qingdao' | 'shaanxi' | 'shanxi' | 'shandong' | 'shanghai' | 'sichuanese' | 'yunnan';
}
export interface ToolItem {
@@ -582,6 +582,13 @@ export interface PracticeScenarioOption {
enabled?: boolean;
position?: string;
scenarioType?: string;
projectType?: string;
growthLevel?: string;
competencyCode?: string;
collaborationPositions?: string;
reviewStatus?: string;
curriculumVersion?: string;
difficulty?: number;
}
export interface ReviewDialogue {
@@ -601,6 +608,12 @@ export interface ReviewAnnotation {
}
export interface ReviewDetail extends PracticeRecord {
position?: string;
projectType?: string;
growthLevel?: string;
competencyCode?: string;
collaborationPositions?: string;
curriculumVersion?: string;
mentorRewrite: string;
aiComment: string;
reviewAdvice?: string;
@@ -1087,6 +1100,16 @@ export interface CompetencyProfile {
aiLevel?: string;
dimensions: CompetencyDimension[];
growthPath?: GrowthStage[];
capabilityProgress?: CapabilityProgress[];
}
export interface CapabilityProgress {
position: string;
growthLevel: string;
competencyCode: string;
completed: number;
averageScore: number;
status: string;
}
export interface PromotionEvidence {