From bf82ff969118538b5cf7675bcd001a6d4ba8d0e6 Mon Sep 17 00:00:00 2001 From: let5sne Date: Thu, 23 Jul 2026 22:12:42 +0800 Subject: [PATCH 1/3] chore: prepare local speech context for isolated tts work --- mobile-uni/src/services/speech.ts | 21 +++++++++++---------- mobile-uni/src/types/api.ts | 7 +++++++ 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/mobile-uni/src/services/speech.ts b/mobile-uni/src/services/speech.ts index bb8b533b..c2c02d81 100644 --- a/mobile-uni/src/services/speech.ts +++ b/mobile-uni/src/services/speech.ts @@ -1,4 +1,4 @@ -import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; +import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api'; import type { SpeechCapture } from './speech-capture'; @@ -103,11 +103,11 @@ export const transcribeSpeechCapture = ( ? transcribeSpeechBlob(capture.blob, filename, registerAbort) : transcribeSpeechFile(capture.file, registerAbort); -export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) => +export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => apiRequest({ url: '/api/ai/tts', method: 'POST', - data: { text: text.trim().slice(0, 300), voiceProfile }, + data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext }, timeout: 60000 }); @@ -133,7 +133,8 @@ export const createSpeechPlaybackController = ( let generation = 0; const sourceCache = new Map(); const maxCachedSources = 4; - const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]); + const cacheKey = (text: string, profile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => + JSON.stringify([text, profile || null, practiceContext || null]); const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => { activeKey = key; @@ -212,7 +213,7 @@ export const createSpeechPlaybackController = ( context.play(); }; - const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => { + const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => { if (activeKey === key) { stop(); return; @@ -221,7 +222,7 @@ export const createSpeechPlaybackController = ( const value = text.trim(); if (!value) return; const profile = voiceProfile || defaultVoiceProfile; - const sourceKey = cacheKey(value, profile); + const sourceKey = cacheKey(value, profile, practiceContext); const currentGeneration = generation; publish(key, 'loading'); try { @@ -232,7 +233,7 @@ export const createSpeechPlaybackController = ( playAudioSource(key, cachedSource, sourceKey, currentGeneration); return; } - const result = await synthesizeSpeech(value, profile); + const result = await synthesizeSpeech(value, profile, practiceContext); if (currentGeneration !== generation || activeKey !== key) return; const source = result.inlineAudioUrl || result.audioUrl; if (!source) throw new Error('语音合成未返回音频'); @@ -257,15 +258,15 @@ export const createSpeechPlaybackController = ( * WeChat-style voice bubble can show duration before the first tap. * Returns the audio source URL, or null when remote TTS is unavailable. */ - const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise => { + const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise => { const value = text.trim(); if (!value) return null; const profile = voiceProfile || defaultVoiceProfile; - const sourceKey = cacheKey(value, profile); + const sourceKey = cacheKey(value, profile, practiceContext); const cached = sourceCache.get(sourceKey); if (cached) return cached; try { - const result = await synthesizeSpeech(value, profile); + const result = await synthesizeSpeech(value, profile, practiceContext); const source = result.inlineAudioUrl || result.audioUrl; if (!source) return null; sourceCache.set(sourceKey, source); diff --git a/mobile-uni/src/types/api.ts b/mobile-uni/src/types/api.ts index cd6cdd49..2a289563 100644 --- a/mobile-uni/src/types/api.ts +++ b/mobile-uni/src/types/api.ts @@ -31,6 +31,13 @@ export interface TtsResponse { ossId?: number | string; } +/** 可选的服务端训练回合上下文;仅用于关联本次新生成的业主 TTS 录音。 */ +export interface PracticeTtsContext { + sessionId: string; + turnIndex: number; + role: 'customer'; +} + export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing'; export interface SpeechVoiceProfile { From 39ff6f6ceb554ac298d610d9369b61ade6733d90 Mon Sep 17 00:00:00 2001 From: let5sne Date: Thu, 23 Jul 2026 22:21:52 +0800 Subject: [PATCH 2/3] test: add qwen tts adapter and voice preference reproducers --- .../aihr/service/AihrSpeechServiceTest.java | 40 ++++++++++++++ .../tests/tts-voice-preference.test.mjs | 55 +++++++++++++++++++ 2 files changed, 95 insertions(+) create mode 100644 mobile-uni/tests/tts-voice-preference.test.mjs diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java b/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java index 3d7432d0..88b84971 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java @@ -60,4 +60,44 @@ class AihrSpeechServiceTest { assertThrows(IllegalArgumentException.class, () -> AihrSpeechService.buildRequestBody( mapper, runtime, "测试", null, invalid)); } + + @Test + void buildsDashScopeQwenTtsRequestAndNormalizesWorkspaceEndpoint() { + ObjectMapper mapper = new ObjectMapper(); + SpeechModel runtime = new SpeechModel( + "dashscope", + "qwen-audio-3.0-tts-flash", + "", + "" + ); + + var body = AihrSpeechService.buildDashScopeRequestBody( + mapper, runtime, "请先确认业主的诉求。", + new VoiceProfile("mentor", null, 0.88, "warm", "mandarin")); + + assertEquals("qwen-audio-3.0-tts-flash", body.path("model").asText()); + assertEquals("请先确认业主的诉求。", body.path("input").path("text").asText()); + assertEquals("longanhuan_v3.6", body.path("input").path("voice").asText()); + assertEquals("mp3", body.path("input").path("format").asText()); + assertEquals(24000, body.path("input").path("sample_rate").asInt()); + assertEquals(0.88, body.path("input").path("rate").asDouble()); + assertEquals( + "https://ws-example.cn-beijing.maas.aliyuncs.com/api/v1/services/audio/tts/SpeechSynthesizer", + AihrSpeechService.dashScopeSynthesisUri("https://ws-example.cn-beijing.maas.aliyuncs.com/compatible-mode/v1").toString() + ); + } + + @Test + void extractsDashScopeAudioUrlOnlyWhenPresent() throws Exception { + ObjectMapper mapper = new ObjectMapper(); + String response = """ + {"output":{"audio":{"url":"http://dashscope-result-bj.oss-cn-beijing.aliyuncs.com/audio.mp3?sig=ok"}}} + """; + + assertEquals( + "http://dashscope-result-bj.oss-cn-beijing.aliyuncs.com/audio.mp3?sig=ok", + AihrSpeechService.dashScopeAudioUrl(mapper, response).orElseThrow() + ); + assertTrue(AihrSpeechService.dashScopeAudioUrl(mapper, "{\"output\":{}}").isEmpty()); + } } diff --git a/mobile-uni/tests/tts-voice-preference.test.mjs b/mobile-uni/tests/tts-voice-preference.test.mjs new file mode 100644 index 00000000..03b39b7e --- /dev/null +++ b/mobile-uni/tests/tts-voice-preference.test.mjs @@ -0,0 +1,55 @@ +import assert from 'node:assert/strict'; +import { readFile } from 'node:fs/promises'; +import test from 'node:test'; +import vm from 'node:vm'; +import ts from 'typescript'; + +const source = (path) => readFile(new URL(path, import.meta.url), 'utf8'); + +const preferenceRuntime = async (storage = new Map()) => { + const service = await source('../src/services/tts-voice-preference.ts'); + const compiled = ts.transpileModule(service, { + compilerOptions: { module: ts.ModuleKind.CommonJS, target: ts.ScriptTarget.ES2020 } + }); + const runtimeModule = { exports: {} }; + vm.runInNewContext(compiled.outputText, { + exports: runtimeModule.exports, + module: runtimeModule, + uni: { + getStorageSync: (key) => storage.get(key), + setStorageSync: (key, value) => storage.set(key, value) + } + }); + return runtimeModule.exports; +}; + +test('播报音色按租户和账号隔离保存,非法值回退跟随场景', async () => { + const storage = new Map(); + const runtime = await preferenceRuntime(storage); + + assert.equal(runtime.getTtsVoicePreference('000000', '13800000000'), ''); + assert.equal(runtime.setTtsVoicePreference('000000', '13800000000', 'longanhuan_v3.6'), 'longanhuan_v3.6'); + assert.equal(runtime.getTtsVoicePreference('000000', '13800000000'), 'longanhuan_v3.6'); + assert.equal(runtime.getTtsVoicePreference('000000', '13900000000'), ''); + assert.equal(runtime.setTtsVoicePreference('000000', '13800000000', 'untrusted-voice'), ''); +}); + +test('全局选择只覆盖 voice,保留场景角色、情绪、语速和方言', async () => { + const runtime = await preferenceRuntime(); + const profile = { role: 'customer', emotion: 'intense', speed: 1.12, dialect: 'sichuanese' }; + + assert.deepEqual( + runtime.applyTtsVoicePreference(profile, 'longanhuan_v3.6'), + { ...profile, voice: 'longanhuan_v3.6' } + ); + assert.deepEqual(runtime.applyTtsVoicePreference(profile, ''), profile); +}); + +test('语音播放链路在合成前统一注入全局音色,且缓存键包含处理后的 profile', async () => { + const sourceText = await source('../src/services/speech.ts'); + + assert.match(sourceText, /applyTtsVoicePreference/); + assert.match(sourceText, /getTtsVoicePreference/); + assert.match(sourceText, /cacheKey\(value, effectiveProfile, practiceContext\)/); + assert.match(sourceText, /synthesizeSpeech\(value, effectiveProfile, practiceContext\)/); +}); From 4902830e6d24a5deb74864a38689113f756e7c84 Mon Sep 17 00:00:00 2001 From: let5sne Date: Thu, 23 Jul 2026 22:37:12 +0800 Subject: [PATCH 3/3] feat: add qwen tts voice preferences --- .../aihr/service/AihrModelSeedService.java | 18 ++- .../aihr/service/AihrSpeechService.java | 141 ++++++++++++++++++ .../aihr/service/AihrSpeechServiceTest.java | 4 + mobile-uni/src/pages/user/profile/index.vue | 51 +++++++ mobile-uni/src/services/speech.ts | 23 ++- .../src/services/tts-voice-preference.ts | 43 ++++++ .../tests/tts-voice-preference.test.mjs | 4 +- 7 files changed, 273 insertions(+), 11 deletions(-) create mode 100644 mobile-uni/src/services/tts-voice-preference.ts diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrModelSeedService.java b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrModelSeedService.java index ce719945..852b8f8e 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrModelSeedService.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrModelSeedService.java @@ -58,6 +58,12 @@ public class AihrModelSeedService { @Value("${aihr.ai-runtime.speech-enabled:${AIHR_AI_SPEECH_ENABLED:true}}") private boolean speechEnabled; + @Value("${aihr.realtime-practice.endpoint:${AIHR_QWEN_REALTIME_ENDPOINT:}}") + private String qwenRealtimeEndpoint; + + @Value("${aihr.realtime-practice.api-key:${AIHR_QWEN_REALTIME_API_KEY:}}") + private String qwenRealtimeApiKey; + public List providers() { List rows = dbProviders(); return rows.isEmpty() ? seedProviders() : mergeProviders(rows); @@ -267,7 +273,7 @@ public class AihrModelSeedService { rs.getString("resolved_api_key") ), tenantId(), category); return rows.stream() - .filter(model -> configured(model.baseUrl(), model.modelName(), model.apiKey())) + .filter(model -> configured(model.baseUrl(), model.modelName(), model.apiKey()) || qwenTtsConfigured(model)) .findFirst(); } catch (DataAccessException e) { log.debug("aihr speech model db fallback(处理错误已隐藏)"); @@ -282,6 +288,16 @@ public class AihrModelSeedService { public record SpeechModel(String providerCode, String modelName, String baseUrl, String apiKey) { } + private boolean qwenTtsConfigured(SpeechModel model) { + return isDashScopeQwenTts(model) && !isBlank(qwenRealtimeEndpoint) && !isBlank(qwenRealtimeApiKey); + } + + private static boolean isDashScopeQwenTts(SpeechModel model) { + return model != null + && ("dashscope".equalsIgnoreCase(model.providerCode()) || "qianwen".equalsIgnoreCase(model.providerCode())) + && "qwen-audio-3.0-tts-flash".equals(model.modelName()); + } + /** * 供其他模块(如三角色对练)复用的 chat 调用:模型未配置或调用失败返回 empty,由调用方决定兜底。 */ diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java index 47e24199..6a26f4e0 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java @@ -6,6 +6,7 @@ import lombok.RequiredArgsConstructor; import lombok.extern.slf4j.Slf4j; import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile; import org.dromara.aihr.service.AihrModelSeedService.SpeechModel; +import org.springframework.beans.factory.annotation.Value; import org.springframework.stereotype.Service; import java.io.ByteArrayOutputStream; @@ -33,7 +34,10 @@ public class AihrSpeechService { private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15); private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5); + private static final Duration DASHSCOPE_TTS_TIMEOUT = Duration.ofSeconds(20); private static final String DEFAULT_TTS_VOICE = "anna"; + private static final String QWEN_TTS_MODEL = "qwen-audio-3.0-tts-flash"; + private static final String QWEN_TTS_DEFAULT_VOICE = "longanhuan_v3.6"; private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}"); private static final Map ROLE_VOICES = Map.of( "mentor", "speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", @@ -52,10 +56,25 @@ public class AihrSpeechService { "cantonese", "请使用自然粤语口语表达", "sichuanese", "请使用自然四川话口语表达" ); + private static final Map QWEN_TTS_ROLE_VOICES = Map.of( + "mentor", QWEN_TTS_DEFAULT_VOICE, + "customer", QWEN_TTS_DEFAULT_VOICE, + "interviewer", QWEN_TTS_DEFAULT_VOICE, + "neutral", QWEN_TTS_DEFAULT_VOICE + ); + private static final java.util.Set QWEN_TTS_VOICES = java.util.Set.of( + "longanhuan_v3.6", "longjielidou_v3.6", "loongeva_v3.6", "loongjohn" + ); private final AihrModelSeedService modelService; private final ObjectMapper objectMapper; + @Value("${aihr.realtime-practice.endpoint:${AIHR_QWEN_REALTIME_ENDPOINT:}}") + private String qwenRealtimeEndpoint; + + @Value("${aihr.realtime-practice.api-key:${AIHR_QWEN_REALTIME_API_KEY:}}") + private String qwenRealtimeApiKey; + public boolean asrConfigured() { return modelService.speechModel("asr").isPresent(); } @@ -117,6 +136,9 @@ public class AihrSpeechService { } SpeechModel runtime = model.get(); try { + if (isDashScopeQwenTts(runtime)) { + return synthesizeDashScopeQwen(runtime, text, voice, voiceProfile); + } ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile); HttpRequest httpRequest = authorized(runtime, "/audio/speech") .header("Content-Type", "application/json") @@ -134,6 +156,38 @@ public class AihrSpeechService { } } + private Optional synthesizeDashScopeQwen(SpeechModel runtime, String text, String voice, VoiceProfile voiceProfile) { + if (isBlank(qwenRealtimeEndpoint) || isBlank(qwenRealtimeApiKey)) { + return Optional.empty(); + } + try { + ObjectNode body = buildDashScopeRequestBody(objectMapper, runtime, text, voiceProfile == null + ? new VoiceProfile("neutral", voice, null, null, null) + : withLegacyVoice(voiceProfile, voice)); + HttpRequest request = HttpRequest.newBuilder() + .uri(dashScopeSynthesisUri(qwenRealtimeEndpoint)) + .timeout(DASHSCOPE_TTS_TIMEOUT) + .header("Content-Type", "application/json") + .header("Authorization", "Bearer " + qwenRealtimeApiKey.trim()) + .POST(HttpRequest.BodyPublishers.ofString(objectMapper.writeValueAsString(body))) + .build(); + HttpResponse response = client().send(request, HttpResponse.BodyHandlers.ofString()); + if (response.statusCode() < 200 || response.statusCode() >= 300) { + log.warn("dashscope tts http {}(外部响应体已隐藏)", response.statusCode()); + return Optional.empty(); + } + Optional audioUrl = dashScopeAudioUrl(objectMapper, response.body()); + if (audioUrl.isEmpty()) { + log.warn("dashscope tts response missing audio url"); + return Optional.empty(); + } + return downloadDashScopeAudio(audioUrl.get()); + } catch (Exception e) { + log.warn("dashscope tts call failed(处理错误已隐藏)"); + return Optional.empty(); + } + } + static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text, String legacyVoice, VoiceProfile voiceProfile) { String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role()); @@ -142,6 +196,9 @@ public class AihrSpeechService { if (requestedVoice == null) { requestedVoice = safeVoice(legacyVoice); } + if (requestedVoice != null && QWEN_TTS_VOICES.contains(requestedVoice)) { + requestedVoice = null; + } if (requestedVoice == null && expressiveCosyVoice) { requestedVoice = ROLE_VOICES.get(role); } @@ -165,6 +222,80 @@ public class AihrSpeechService { return body; } + static ObjectNode buildDashScopeRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text, + VoiceProfile voiceProfile) { + String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role()); + String emotion = normalizedEmotion(voiceProfile == null ? null : voiceProfile.emotion(), role); + String dialect = normalizedDialect(voiceProfile == null ? null : voiceProfile.dialect()); + String requestedVoice = safeVoice(voiceProfile == null ? null : voiceProfile.voice()); + String voice = requestedVoice != null && QWEN_TTS_VOICES.contains(requestedVoice) + ? requestedVoice + : QWEN_TTS_ROLE_VOICES.getOrDefault(role, QWEN_TTS_DEFAULT_VOICE); + ObjectNode input = objectMapper.createObjectNode(); + input.put("text", AihrSensitiveText.forModel(text.trim())); + input.put("voice", voice); + input.put("format", "mp3"); + input.put("sample_rate", 24000); + input.put("rate", resolveSpeed(voiceProfile == null ? null : voiceProfile.speed(), emotion)); + String instruction = expressivePrompt(emotion, dialect); + if (!instruction.isBlank()) { + input.put("instruction", instruction); + } + ObjectNode body = objectMapper.createObjectNode(); + body.put("model", runtime == null || isBlank(runtime.modelName()) ? QWEN_TTS_MODEL : runtime.modelName()); + body.set("input", input); + return body; + } + + static URI dashScopeSynthesisUri(String endpoint) { + URI configured = URI.create(endpoint == null ? "" : endpoint.trim()); + String scheme = configured.getScheme(); + String authority = configured.getRawAuthority(); + if (("https".equalsIgnoreCase(scheme) || "http".equalsIgnoreCase(scheme)) && authority != null && !authority.isBlank()) { + return URI.create(scheme + "://" + authority + "/api/v1/services/audio/tts/SpeechSynthesizer"); + } + throw new IllegalArgumentException("invalid DashScope endpoint"); + } + + static Optional dashScopeAudioUrl(ObjectMapper objectMapper, String responseBody) { + try { + String value = objectMapper.readTree(responseBody == null ? "" : responseBody) + .path("output").path("audio").path("url").asText("").trim(); + URI uri = value.isBlank() ? null : URI.create(value); + if (uri == null || !("http".equalsIgnoreCase(uri.getScheme()) || "https".equalsIgnoreCase(uri.getScheme())) + || uri.getHost() == null || !uri.getHost().endsWith(".oss-cn-beijing.aliyuncs.com")) { + return Optional.empty(); + } + return Optional.of(uri.toString()); + } catch (Exception e) { + return Optional.empty(); + } + } + + private Optional downloadDashScopeAudio(String url) { + try { + HttpRequest request = HttpRequest.newBuilder() + .uri(URI.create(url)) + .timeout(DASHSCOPE_TTS_TIMEOUT) + .GET() + .build(); + HttpResponse response = client().send(request, HttpResponse.BodyHandlers.ofByteArray()); + if (response.statusCode() < 200 || response.statusCode() >= 300 || response.body().length == 0) { + return Optional.empty(); + } + return Optional.of(response.body()); + } catch (Exception e) { + return Optional.empty(); + } + } + + private static VoiceProfile withLegacyVoice(VoiceProfile profile, String legacyVoice) { + if (safeVoice(profile.voice()) != null || safeVoice(legacyVoice) == null) { + return profile; + } + return new VoiceProfile(profile.role(), legacyVoice, profile.speed(), profile.emotion(), profile.dialect()); + } + /** * 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。 */ @@ -236,6 +367,12 @@ public class AihrSpeechService { && runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice"); } + private static boolean isDashScopeQwenTts(SpeechModel runtime) { + return runtime != null + && ("dashscope".equalsIgnoreCase(runtime.providerCode()) || "qianwen".equalsIgnoreCase(runtime.providerCode())) + && QWEN_TTS_MODEL.equals(runtime.modelName()); + } + private HttpRequest.Builder authorized(SpeechModel runtime, String path) { HttpRequest.Builder builder = HttpRequest.newBuilder() .uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path)) @@ -286,6 +423,10 @@ public class AihrSpeechService { return normalized; } + private static boolean isBlank(String value) { + return value == null || value.isBlank(); + } + private static String truncate(String value) { if (value == null || value.length() <= 240) { return value; diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java b/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java index 88b84971..17ff3e56 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java @@ -32,6 +32,9 @@ class AihrSpeechServiceTest { new VoiceProfile("customer", null, 9.0, "intense", "sichuanese")); var legacy = AihrSpeechService.buildRequestBody( mapper, runtime, "普通播报", "alex", null); + var qwenVoiceBeforeSwitch = AihrSpeechService.buildRequestBody( + mapper, runtime, "普通播报", null, + new VoiceProfile("neutral", "longanhuan_v3.6", null, null, null)); assertEquals("speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", mentor.path("voice").asText()); assertEquals(0.88, mentor.path("speed").asDouble()); @@ -41,6 +44,7 @@ class AihrSpeechServiceTest { assertTrue(customer.path("input").asText().contains("情绪强烈")); assertTrue(customer.path("input").asText().contains("四川话")); assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText()); + assertEquals("FunAudioLLM/CosyVoice2-0.5B:anna", qwenVoiceBeforeSwitch.path("voice").asText()); assertFalse(legacy.has("speed")); assertFalse(legacy.path("input").asText().contains("<|endofprompt|>")); } diff --git a/mobile-uni/src/pages/user/profile/index.vue b/mobile-uni/src/pages/user/profile/index.vue index c3ae0728..b9fe9af3 100644 --- a/mobile-uni/src/pages/user/profile/index.vue +++ b/mobile-uni/src/pages/user/profile/index.vue @@ -271,6 +271,16 @@ 切换后同步到今日、练、问、我全部页面,并自动保存。 + + + + AI 播报音色 + {{ currentTtsVoiceLabel }} · 对练情绪和语速仍会保留 + + + {{ currentTtsVoiceLabel }} + + @@ -297,6 +307,12 @@ import { import { getCompetencyProfile, getPracticeHistory, getPromotionEvidence } from '@/services/practice'; import { resetPageScroll } from '@/services/navigation'; import { ensureEmployeePosition } from '@/services/position'; +import { + getTtsVoicePreference, + qwenTtsVoiceOptions, + setTtsVoicePreference, + type TtsVoicePreference +} from '@/services/tts-voice-preference'; const loggedIn = ref(false); const phone = ref(''); @@ -313,9 +329,14 @@ const evidence = ref(null); const positionProfile = ref(getSelectedPositionProfile()); const supervisorLearning = ref(false); const fontSize = ref(getAppFontSize()); +const ttsVoice = ref(''); const currentFontSizeLabel = computed(() => appFontSizeOptions.find((item) => item.value === fontSize.value)?.label || '标准' ); +const ttsVoiceOptionIndex = computed(() => Math.max(0, qwenTtsVoiceOptions.findIndex((item) => item.value === ttsVoice.value))); +const currentTtsVoiceLabel = computed(() => + qwenTtsVoiceOptions[ttsVoiceOptionIndex.value]?.label || '跟随场景' +); const maskedPhone = computed(() => { const value = phone.value.trim(); if (!value) return '未登录'; @@ -333,6 +354,7 @@ const refreshAuth = () => { positionProfile.value = getSelectedPositionProfile(); supervisorLearning.value = isSupervisorLearnerMode(); fontSize.value = getAppFontSize(); + ttsVoice.value = getTtsVoicePreference(auth.tenantId, auth.phone); }; const changeFontSize = (value: AppFontSize) => { @@ -340,6 +362,13 @@ const changeFontSize = (value: AppFontSize) => { uni.showToast({ title: `已切换为${currentFontSizeLabel.value}`, icon: 'none' }); }; +const changeTtsVoice = (event: { detail?: { value?: string | number } }) => { + const option = qwenTtsVoiceOptions[Number(event.detail?.value)]; + const auth = getAuth(); + ttsVoice.value = setTtsVoicePreference(auth.tenantId, auth.phone, option?.value); + uni.showToast({ title: `已切换为${currentTtsVoiceLabel.value}`, icon: 'none' }); +}; + const returnToSupervisor = () => { leaveSupervisorLearnerMode(); supervisorLearning.value = false; @@ -650,6 +679,28 @@ onShow(() => { box-shadow: 0 4px 12px rgba(24, 34, 48, 0.08); } +.voice-preference-row { + display: flex; + align-items: center; + justify-content: space-between; + gap: 12px; + padding-top: 12px; + border-top: 1px solid var(--employee-line); +} + +.voice-picker-value { + display: flex; + align-items: center; + gap: 3px; + max-width: 150px; + padding: 8px 10px; + border: 1px solid #e3e6ea; + border-radius: 10px; + color: var(--employee-text); + font-size: 12px; + white-space: nowrap; +} + .profile-summary { border-color: var(--employee-line); } diff --git a/mobile-uni/src/services/speech.ts b/mobile-uni/src/services/speech.ts index c2c02d81..8d43d403 100644 --- a/mobile-uni/src/services/speech.ts +++ b/mobile-uni/src/services/speech.ts @@ -1,6 +1,8 @@ import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api'; +import { getAuth } from './auth'; import type { SpeechCapture } from './speech-capture'; +import { applyTtsVoicePreference, getTtsVoicePreference } from './tts-voice-preference'; const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg']; @@ -103,11 +105,16 @@ export const transcribeSpeechCapture = ( ? transcribeSpeechBlob(capture.blob, filename, registerAbort) : transcribeSpeechFile(capture.file, registerAbort); +export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => { + const auth = getAuth(); + return applyTtsVoicePreference(voiceProfile, getTtsVoicePreference(auth.tenantId, auth.phone)); +}; + export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => apiRequest({ url: '/api/ai/tts', method: 'POST', - data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext }, + data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext }, timeout: 60000 }); @@ -221,8 +228,8 @@ export const createSpeechPlaybackController = ( stop(); const value = text.trim(); if (!value) return; - const profile = voiceProfile || defaultVoiceProfile; - const sourceKey = cacheKey(value, profile, practiceContext); + const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile); + const sourceKey = cacheKey(value, effectiveProfile, practiceContext); const currentGeneration = generation; publish(key, 'loading'); try { @@ -233,7 +240,7 @@ export const createSpeechPlaybackController = ( playAudioSource(key, cachedSource, sourceKey, currentGeneration); return; } - const result = await synthesizeSpeech(value, profile, practiceContext); + const result = await synthesizeSpeech(value, effectiveProfile, practiceContext); if (currentGeneration !== generation || activeKey !== key) return; const source = result.inlineAudioUrl || result.audioUrl; if (!source) throw new Error('语音合成未返回音频'); @@ -246,7 +253,7 @@ export const createSpeechPlaybackController = ( playAudioSource(key, source, sourceKey, currentGeneration); } catch (error) { if (currentGeneration !== generation) return; - const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, profile); + const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile); if (browserFallback.played) return; stop(); onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败')); @@ -261,12 +268,12 @@ export const createSpeechPlaybackController = ( const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise => { const value = text.trim(); if (!value) return null; - const profile = voiceProfile || defaultVoiceProfile; - const sourceKey = cacheKey(value, profile, practiceContext); + const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile); + const sourceKey = cacheKey(value, effectiveProfile, practiceContext); const cached = sourceCache.get(sourceKey); if (cached) return cached; try { - const result = await synthesizeSpeech(value, profile, practiceContext); + const result = await synthesizeSpeech(value, effectiveProfile, practiceContext); const source = result.inlineAudioUrl || result.audioUrl; if (!source) return null; sourceCache.set(sourceKey, source); diff --git a/mobile-uni/src/services/tts-voice-preference.ts b/mobile-uni/src/services/tts-voice-preference.ts new file mode 100644 index 00000000..394296a3 --- /dev/null +++ b/mobile-uni/src/services/tts-voice-preference.ts @@ -0,0 +1,43 @@ +import type { SpeechVoiceProfile } from '@/types/api'; + +export const qwenTtsVoiceOptions = [ + { value: '', label: '跟随场景' }, + { value: 'longanhuan_v3.6', label: '龙安欢 · 中文女声' }, + { value: 'longjielidou_v3.6', label: '龙杰里斗 · 童声' }, + { value: 'loongeva_v3.6', label: 'Loongeva · 英文女声' }, + { value: 'loongjohn', label: 'Loongjohn · 英文男声' } +] as const; + +export type TtsVoicePreference = typeof qwenTtsVoiceOptions[number]['value']; + +const preferenceKeyPrefix = 'aihr_tts_voice'; + +export const normalizeTtsVoicePreference = (value: unknown): TtsVoicePreference => + qwenTtsVoiceOptions.some((option) => option.value === value) ? value as TtsVoicePreference : ''; + +export const ttsVoicePreferenceStorageKey = (tenantId: unknown, phone: unknown) => { + const tenant = String(tenantId || '').trim(); + const account = String(phone || '').trim(); + return tenant && account ? `${preferenceKeyPrefix}:${tenant}:${account}` : ''; +}; + +export const getTtsVoicePreference = (tenantId: unknown, phone: unknown): TtsVoicePreference => { + const key = ttsVoicePreferenceStorageKey(tenantId, phone); + return key ? normalizeTtsVoicePreference(uni.getStorageSync(key)) : ''; +}; + +export const setTtsVoicePreference = (tenantId: unknown, phone: unknown, value: unknown): TtsVoicePreference => { + const key = ttsVoicePreferenceStorageKey(tenantId, phone); + const normalized = normalizeTtsVoicePreference(value); + if (key) uni.setStorageSync(key, normalized); + return normalized; +}; + +export const applyTtsVoicePreference = ( + profile: SpeechVoiceProfile | undefined, + preference: unknown +): SpeechVoiceProfile | undefined => { + const voice = normalizeTtsVoicePreference(preference); + if (!voice) return profile; + return { ...(profile || { role: 'neutral' }), voice }; +}; diff --git a/mobile-uni/tests/tts-voice-preference.test.mjs b/mobile-uni/tests/tts-voice-preference.test.mjs index 03b39b7e..bd33f64a 100644 --- a/mobile-uni/tests/tts-voice-preference.test.mjs +++ b/mobile-uni/tests/tts-voice-preference.test.mjs @@ -39,10 +39,10 @@ test('全局选择只覆盖 voice,保留场景角色、情绪、语速和方 const profile = { role: 'customer', emotion: 'intense', speed: 1.12, dialect: 'sichuanese' }; assert.deepEqual( - runtime.applyTtsVoicePreference(profile, 'longanhuan_v3.6'), + JSON.parse(JSON.stringify(runtime.applyTtsVoicePreference(profile, 'longanhuan_v3.6'))), { ...profile, voice: 'longanhuan_v3.6' } ); - assert.deepEqual(runtime.applyTtsVoicePreference(profile, ''), profile); + assert.deepEqual(JSON.parse(JSON.stringify(runtime.applyTtsVoicePreference(profile, ''))), profile); }); test('语音播放链路在合成前统一注入全局音色,且缓存键包含处理后的 profile', async () => {