feat(speech): add qwen tts voice preferences

This commit is contained in:
2026-07-24 01:54:37 +08:00
parent faa2a77e9a
commit 474d46e728
8 changed files with 523 additions and 30 deletions
@@ -58,6 +58,12 @@ public class AihrModelSeedService {
@Value("${aihr.ai-runtime.speech-enabled:${AIHR_AI_SPEECH_ENABLED:true}}") @Value("${aihr.ai-runtime.speech-enabled:${AIHR_AI_SPEECH_ENABLED:true}}")
private boolean speechEnabled; private boolean speechEnabled;
@Value("${aihr.realtime-practice.endpoint:${AIHR_QWEN_REALTIME_ENDPOINT:}}")
private String qwenRealtimeEndpoint;
@Value("${aihr.realtime-practice.api-key:${AIHR_QWEN_REALTIME_API_KEY:}}")
private String qwenRealtimeApiKey;
public List<ProviderResponse> providers() { public List<ProviderResponse> providers() {
List<ProviderResponse> rows = dbProviders(); List<ProviderResponse> rows = dbProviders();
return rows.isEmpty() ? seedProviders() : mergeProviders(rows); return rows.isEmpty() ? seedProviders() : mergeProviders(rows);
@@ -267,7 +273,7 @@ public class AihrModelSeedService {
rs.getString("resolved_api_key") rs.getString("resolved_api_key")
), tenantId(), category); ), tenantId(), category);
return rows.stream() return rows.stream()
.filter(model -> configured(model.baseUrl(), model.modelName(), model.apiKey())) .filter(model -> configured(model.baseUrl(), model.modelName(), model.apiKey()) || qwenTtsConfigured(model))
.findFirst(); .findFirst();
} catch (DataAccessException e) { } catch (DataAccessException e) {
log.debug("aihr speech model db fallback(处理错误已隐藏)"); log.debug("aihr speech model db fallback(处理错误已隐藏)");
@@ -282,6 +288,16 @@ public class AihrModelSeedService {
public record SpeechModel(String providerCode, String modelName, String baseUrl, String apiKey) { public record SpeechModel(String providerCode, String modelName, String baseUrl, String apiKey) {
} }
private boolean qwenTtsConfigured(SpeechModel model) {
return isDashScopeQwenTts(model) && !isBlank(qwenRealtimeEndpoint) && !isBlank(qwenRealtimeApiKey);
}
private static boolean isDashScopeQwenTts(SpeechModel model) {
return model != null
&& ("dashscope".equalsIgnoreCase(model.providerCode()) || "qianwen".equalsIgnoreCase(model.providerCode()))
&& "qwen-audio-3.0-tts-flash".equals(model.modelName());
}
/** /**
* 供其他模块(如三角色对练)复用的 chat 调用:模型未配置或调用失败返回 empty,由调用方决定兜底。 * 供其他模块(如三角色对练)复用的 chat 调用:模型未配置或调用失败返回 empty,由调用方决定兜底。
*/ */
@@ -354,7 +370,15 @@ public class AihrModelSeedService {
on p.tenant_id = c.tenant_id and p.provider_code = c.provider_code on p.tenant_id = c.tenant_id and p.provider_code = c.provider_code
where c.tenant_id = ? where c.tenant_id = ?
order by case c.category when 'chat' then 0 when 'vector' then 1 when 'rerank' then 2 else 9 end, c.id asc order by case c.category when 'chat' then 0 when 'vector' then 1 when 'rerank' then 2 else 9 end, c.id asc
""", (rs, rowNum) -> new ConfigResponse( """, (rs, rowNum) -> {
SpeechModel runtime = new SpeechModel(
rs.getString("provider_code"),
rs.getString("model_name"),
rs.getString("resolved_api_host"),
rs.getString("resolved_api_key")
);
boolean qwenTts = qwenTtsConfigured(runtime);
return new ConfigResponse(
rs.getLong("id"), rs.getLong("id"),
rs.getString("category"), rs.getString("category"),
rs.getString("model_name"), rs.getString("model_name"),
@@ -364,9 +388,10 @@ public class AihrModelSeedService {
rs.getString("model_show"), rs.getString("model_show"),
rs.getString("api_host"), rs.getString("api_host"),
rs.getInt("enabled") == 1, rs.getInt("enabled") == 1,
!isBlank(rs.getString("resolved_api_key")), qwenTts || !isBlank(runtime.apiKey()),
configured(rs.getString("resolved_api_host"), rs.getString("model_name"), rs.getString("resolved_api_key")) qwenTts || configured(runtime.baseUrl(), runtime.modelName(), runtime.apiKey())
), tenantId()); );
}, tenantId());
} catch (DataAccessException e) { } catch (DataAccessException e) {
log.debug("aihr model config db fallback(处理错误已隐藏)"); log.debug("aihr model config db fallback(处理错误已隐藏)");
return List.of(); return List.of();
@@ -6,6 +6,7 @@ import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j; import lombok.extern.slf4j.Slf4j;
import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile; import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile;
import org.dromara.aihr.service.AihrModelSeedService.SpeechModel; import org.dromara.aihr.service.AihrModelSeedService.SpeechModel;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.stereotype.Service; import org.springframework.stereotype.Service;
import java.io.ByteArrayOutputStream; import java.io.ByteArrayOutputStream;
@@ -33,7 +34,10 @@ public class AihrSpeechService {
private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15); private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15);
private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5); private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5);
private static final Duration DASHSCOPE_TTS_TIMEOUT = Duration.ofSeconds(20);
private static final String DEFAULT_TTS_VOICE = "anna"; private static final String DEFAULT_TTS_VOICE = "anna";
private static final String QWEN_TTS_MODEL = "qwen-audio-3.0-tts-flash";
private static final String QWEN_TTS_DEFAULT_VOICE = "longanhuan_v3.6";
private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}"); private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}");
private static final Map<String, String> ROLE_VOICES = Map.of( private static final Map<String, String> ROLE_VOICES = Map.of(
"mentor", "speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", "mentor", "speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr",
@@ -47,15 +51,48 @@ public class AihrSpeechService {
"intense", "请用情绪强烈、急切、有压力且明显不满的语气说", "intense", "请用情绪强烈、急切、有压力且明显不满的语气说",
"calm", "请用平静、自然的语气说" "calm", "请用平静、自然的语气说"
); );
private static final Map<String, String> DIALECT_PROMPTS = Map.of( private static final Map<String, String> DIALECT_PROMPTS = Map.ofEntries(
"mandarin", "请使用自然普通话表达", Map.entry("mandarin", "请使用自然普通话表达"),
"cantonese", "请使用自然粤语口语表达", Map.entry("cantonese", "请使用自然粤语口语表达"),
"sichuanese", "请使用自然四川话口语表达" Map.entry("chongqing", "请使用自然重庆话口语表达"),
Map.entry("northeastern", "请使用自然东北话口语表达"),
Map.entry("gansu", "请使用自然甘肃话口语表达"),
Map.entry("guizhou", "请使用自然贵州话口语表达"),
Map.entry("zhejiang", "请使用自然浙江话口语表达"),
Map.entry("hebei", "请使用自然河北话口语表达"),
Map.entry("henan", "请使用自然河南话口语表达"),
Map.entry("hubei", "请使用自然湖北话口语表达"),
Map.entry("hunan", "请使用自然湖南话口语表达"),
Map.entry("jiangxi", "请使用自然江西话口语表达"),
Map.entry("ningbo", "请使用自然宁波话口语表达"),
Map.entry("ningxia", "请使用自然宁夏话口语表达"),
Map.entry("qingdao", "请使用自然青岛话口语表达"),
Map.entry("shaanxi", "请使用自然陕西话口语表达"),
Map.entry("shanxi", "请使用自然山西话口语表达"),
Map.entry("shandong", "请使用自然山东话口语表达"),
Map.entry("shanghai", "请使用自然上海话口语表达"),
Map.entry("sichuanese", "请使用自然四川话口语表达"),
Map.entry("yunnan", "请使用自然云南话口语表达")
);
private static final Map<String, String> QWEN_TTS_ROLE_VOICES = Map.of(
"mentor", QWEN_TTS_DEFAULT_VOICE,
"customer", QWEN_TTS_DEFAULT_VOICE,
"interviewer", QWEN_TTS_DEFAULT_VOICE,
"neutral", QWEN_TTS_DEFAULT_VOICE
);
private static final java.util.Set<String> QWEN_TTS_VOICES = java.util.Set.of(
"longanhuan_v3.6", "longjielidou_v3.6", "loongeva_v3.6", "loongjohn"
); );
private final AihrModelSeedService modelService; private final AihrModelSeedService modelService;
private final ObjectMapper objectMapper; private final ObjectMapper objectMapper;
@Value("${aihr.realtime-practice.endpoint:${AIHR_QWEN_REALTIME_ENDPOINT:}}")
private String qwenRealtimeEndpoint;
@Value("${aihr.realtime-practice.api-key:${AIHR_QWEN_REALTIME_API_KEY:}}")
private String qwenRealtimeApiKey;
public boolean asrConfigured() { public boolean asrConfigured() {
return modelService.speechModel("asr").isPresent(); return modelService.speechModel("asr").isPresent();
} }
@@ -71,7 +108,7 @@ public class AihrSpeechService {
if (rawDialect.isEmpty() || DIALECT_PROMPTS.containsKey(rawDialect.toLowerCase(Locale.ROOT))) { if (rawDialect.isEmpty() || DIALECT_PROMPTS.containsKey(rawDialect.toLowerCase(Locale.ROOT))) {
return Optional.empty(); return Optional.empty();
} }
return Optional.of("dialect 仅支持 mandarin、cantonese、sichuanese"); return Optional.of("dialect 不受支持");
} }
/** /**
@@ -117,6 +154,9 @@ public class AihrSpeechService {
} }
SpeechModel runtime = model.get(); SpeechModel runtime = model.get();
try { try {
if (isDashScopeQwenTts(runtime)) {
return synthesizeDashScopeQwen(runtime, text, voice, voiceProfile);
}
ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile); ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile);
HttpRequest httpRequest = authorized(runtime, "/audio/speech") HttpRequest httpRequest = authorized(runtime, "/audio/speech")
.header("Content-Type", "application/json") .header("Content-Type", "application/json")
@@ -134,6 +174,38 @@ public class AihrSpeechService {
} }
} }
private Optional<byte[]> synthesizeDashScopeQwen(SpeechModel runtime, String text, String voice, VoiceProfile voiceProfile) {
if (isBlank(qwenRealtimeEndpoint) || isBlank(qwenRealtimeApiKey)) {
return Optional.empty();
}
try {
ObjectNode body = buildDashScopeRequestBody(objectMapper, runtime, text, voiceProfile == null
? new VoiceProfile("neutral", voice, null, null, null)
: withLegacyVoice(voiceProfile, voice));
HttpRequest request = HttpRequest.newBuilder()
.uri(dashScopeSynthesisUri(qwenRealtimeEndpoint))
.timeout(DASHSCOPE_TTS_TIMEOUT)
.header("Content-Type", "application/json")
.header("Authorization", "Bearer " + qwenRealtimeApiKey.trim())
.POST(HttpRequest.BodyPublishers.ofString(objectMapper.writeValueAsString(body)))
.build();
HttpResponse<String> response = client().send(request, HttpResponse.BodyHandlers.ofString());
if (response.statusCode() < 200 || response.statusCode() >= 300) {
log.warn("dashscope tts http {}(外部响应体已隐藏)", response.statusCode());
return Optional.empty();
}
Optional<String> audioUrl = dashScopeAudioUrl(objectMapper, response.body());
if (audioUrl.isEmpty()) {
log.warn("dashscope tts response missing audio url");
return Optional.empty();
}
return downloadDashScopeAudio(audioUrl.get());
} catch (Exception e) {
log.warn("dashscope tts call failed(处理错误已隐藏)");
return Optional.empty();
}
}
static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text, static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text,
String legacyVoice, VoiceProfile voiceProfile) { String legacyVoice, VoiceProfile voiceProfile) {
String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role()); String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role());
@@ -142,6 +214,9 @@ public class AihrSpeechService {
if (requestedVoice == null) { if (requestedVoice == null) {
requestedVoice = safeVoice(legacyVoice); requestedVoice = safeVoice(legacyVoice);
} }
if (requestedVoice != null && QWEN_TTS_VOICES.contains(requestedVoice)) {
requestedVoice = null;
}
if (requestedVoice == null && expressiveCosyVoice) { if (requestedVoice == null && expressiveCosyVoice) {
requestedVoice = ROLE_VOICES.get(role); requestedVoice = ROLE_VOICES.get(role);
} }
@@ -165,6 +240,80 @@ public class AihrSpeechService {
return body; return body;
} }
static ObjectNode buildDashScopeRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text,
VoiceProfile voiceProfile) {
String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role());
String emotion = normalizedEmotion(voiceProfile == null ? null : voiceProfile.emotion(), role);
String dialect = normalizedDialect(voiceProfile == null ? null : voiceProfile.dialect());
String requestedVoice = safeVoice(voiceProfile == null ? null : voiceProfile.voice());
String voice = requestedVoice != null && QWEN_TTS_VOICES.contains(requestedVoice)
? requestedVoice
: QWEN_TTS_ROLE_VOICES.getOrDefault(role, QWEN_TTS_DEFAULT_VOICE);
ObjectNode input = objectMapper.createObjectNode();
input.put("text", AihrSensitiveText.forModel(text.trim()));
input.put("voice", voice);
input.put("format", "mp3");
input.put("sample_rate", 24000);
input.put("rate", resolveSpeed(voiceProfile == null ? null : voiceProfile.speed(), emotion));
String instruction = expressivePrompt(emotion, dialect);
if (!instruction.isBlank()) {
input.put("instruction", instruction);
}
ObjectNode body = objectMapper.createObjectNode();
body.put("model", runtime == null || isBlank(runtime.modelName()) ? QWEN_TTS_MODEL : runtime.modelName());
body.set("input", input);
return body;
}
static URI dashScopeSynthesisUri(String endpoint) {
URI configured = URI.create(endpoint == null ? "" : endpoint.trim());
String scheme = configured.getScheme();
String authority = configured.getRawAuthority();
if (("https".equalsIgnoreCase(scheme) || "http".equalsIgnoreCase(scheme)) && authority != null && !authority.isBlank()) {
return URI.create(scheme + "://" + authority + "/api/v1/services/audio/tts/SpeechSynthesizer");
}
throw new IllegalArgumentException("invalid DashScope endpoint");
}
static Optional<String> dashScopeAudioUrl(ObjectMapper objectMapper, String responseBody) {
try {
String value = objectMapper.readTree(responseBody == null ? "" : responseBody)
.path("output").path("audio").path("url").asText("").trim();
URI uri = value.isBlank() ? null : URI.create(value);
if (uri == null || !("http".equalsIgnoreCase(uri.getScheme()) || "https".equalsIgnoreCase(uri.getScheme()))
|| uri.getHost() == null || !uri.getHost().endsWith(".oss-cn-beijing.aliyuncs.com")) {
return Optional.empty();
}
return Optional.of(uri.toString());
} catch (Exception e) {
return Optional.empty();
}
}
private Optional<byte[]> downloadDashScopeAudio(String url) {
try {
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create(url))
.timeout(DASHSCOPE_TTS_TIMEOUT)
.GET()
.build();
HttpResponse<byte[]> response = client().send(request, HttpResponse.BodyHandlers.ofByteArray());
if (response.statusCode() < 200 || response.statusCode() >= 300 || response.body().length == 0) {
return Optional.empty();
}
return Optional.of(response.body());
} catch (Exception e) {
return Optional.empty();
}
}
private static VoiceProfile withLegacyVoice(VoiceProfile profile, String legacyVoice) {
if (safeVoice(profile.voice()) != null || safeVoice(legacyVoice) == null) {
return profile;
}
return new VoiceProfile(profile.role(), legacyVoice, profile.speed(), profile.emotion(), profile.dialect());
}
/** /**
* 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。 * 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。
*/ */
@@ -196,7 +345,7 @@ public class AihrSpeechService {
if (value.isBlank() || DIALECT_PROMPTS.containsKey(value)) { if (value.isBlank() || DIALECT_PROMPTS.containsKey(value)) {
return value; return value;
} }
throw new IllegalArgumentException("dialect 仅支持 mandarin、cantonese、sichuanese"); throw new IllegalArgumentException("dialect 不受支持");
} }
private static String expressivePrompt(String emotion, String dialect) { private static String expressivePrompt(String emotion, String dialect) {
@@ -236,6 +385,12 @@ public class AihrSpeechService {
&& runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice"); && runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice");
} }
private static boolean isDashScopeQwenTts(SpeechModel runtime) {
return runtime != null
&& ("dashscope".equalsIgnoreCase(runtime.providerCode()) || "qianwen".equalsIgnoreCase(runtime.providerCode()))
&& QWEN_TTS_MODEL.equals(runtime.modelName());
}
private HttpRequest.Builder authorized(SpeechModel runtime, String path) { private HttpRequest.Builder authorized(SpeechModel runtime, String path) {
HttpRequest.Builder builder = HttpRequest.newBuilder() HttpRequest.Builder builder = HttpRequest.newBuilder()
.uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path)) .uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path))
@@ -286,6 +441,10 @@ public class AihrSpeechService {
return normalized; return normalized;
} }
private static boolean isBlank(String value) {
return value == null || value.isBlank();
}
private static String truncate(String value) { private static String truncate(String value) {
if (value == null || value.length() <= 240) { if (value == null || value.length() <= 240) {
return value; return value;
@@ -32,6 +32,9 @@ class AihrSpeechServiceTest {
new VoiceProfile("customer", null, 9.0, "intense", "sichuanese")); new VoiceProfile("customer", null, 9.0, "intense", "sichuanese"));
var legacy = AihrSpeechService.buildRequestBody( var legacy = AihrSpeechService.buildRequestBody(
mapper, runtime, "普通播报", "alex", null); mapper, runtime, "普通播报", "alex", null);
var qwenVoiceBeforeSwitch = AihrSpeechService.buildRequestBody(
mapper, runtime, "普通播报", null,
new VoiceProfile("neutral", "longanhuan_v3.6", null, null, null));
assertEquals("speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", mentor.path("voice").asText()); assertEquals("speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", mentor.path("voice").asText());
assertEquals(0.88, mentor.path("speed").asDouble()); assertEquals(0.88, mentor.path("speed").asDouble());
@@ -41,6 +44,7 @@ class AihrSpeechServiceTest {
assertTrue(customer.path("input").asText().contains("情绪强烈")); assertTrue(customer.path("input").asText().contains("情绪强烈"));
assertTrue(customer.path("input").asText().contains("四川话")); assertTrue(customer.path("input").asText().contains("四川话"));
assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText()); assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText());
assertEquals("FunAudioLLM/CosyVoice2-0.5B:anna", qwenVoiceBeforeSwitch.path("voice").asText());
assertFalse(legacy.has("speed")); assertFalse(legacy.has("speed"));
assertFalse(legacy.path("input").asText().contains("<|endofprompt|>")); assertFalse(legacy.path("input").asText().contains("<|endofprompt|>"));
} }
@@ -60,4 +64,45 @@ class AihrSpeechServiceTest {
assertThrows(IllegalArgumentException.class, () -> AihrSpeechService.buildRequestBody( assertThrows(IllegalArgumentException.class, () -> AihrSpeechService.buildRequestBody(
mapper, runtime, "测试", null, invalid)); mapper, runtime, "测试", null, invalid));
} }
@Test
void buildsDashScopeQwenTtsRequestAndNormalizesWorkspaceEndpoint() {
ObjectMapper mapper = new ObjectMapper();
SpeechModel runtime = new SpeechModel(
"dashscope",
"qwen-audio-3.0-tts-flash",
"",
""
);
var body = AihrSpeechService.buildDashScopeRequestBody(
mapper, runtime, "请先确认业主的诉求。",
new VoiceProfile("mentor", null, 0.88, "warm", "shanghai"));
assertEquals("qwen-audio-3.0-tts-flash", body.path("model").asText());
assertEquals("请先确认业主的诉求。", body.path("input").path("text").asText());
assertEquals("longanhuan_v3.6", body.path("input").path("voice").asText());
assertEquals("mp3", body.path("input").path("format").asText());
assertEquals(24000, body.path("input").path("sample_rate").asInt());
assertEquals(0.88, body.path("input").path("rate").asDouble());
assertTrue(body.path("input").path("instruction").asText().contains("上海话"));
assertEquals(
"https://ws-example.cn-beijing.maas.aliyuncs.com/api/v1/services/audio/tts/SpeechSynthesizer",
AihrSpeechService.dashScopeSynthesisUri("https://ws-example.cn-beijing.maas.aliyuncs.com/compatible-mode/v1").toString()
);
}
@Test
void extractsDashScopeAudioUrlOnlyWhenPresent() throws Exception {
ObjectMapper mapper = new ObjectMapper();
String response = """
{"output":{"audio":{"url":"http://dashscope-result-bj.oss-cn-beijing.aliyuncs.com/audio.mp3?sig=ok"}}}
""";
assertEquals(
"http://dashscope-result-bj.oss-cn-beijing.aliyuncs.com/audio.mp3?sig=ok",
AihrSpeechService.dashScopeAudioUrl(mapper, response).orElseThrow()
);
assertTrue(AihrSpeechService.dashScopeAudioUrl(mapper, "{\"output\":{}}").isEmpty());
}
} }
@@ -271,6 +271,25 @@
</view> </view>
<text class="display-note">切换后同步到今日、练、问、我全部页面,并自动保存。</text> <text class="display-note">切换后同步到今日、练、问、我全部页面,并自动保存。</text>
</template> </template>
<view v-if="loggedIn" class="voice-preference-row">
<view class="display-heading-copy">
<text class="menu-item-title">AI 播报音色</text>
<text class="menu-item-desc">{{ currentTtsVoiceLabel }} · 对练情绪和语速仍会保留</text>
</view>
<picker :range="qwenTtsVoiceOptions" range-key="label" :value="ttsVoiceOptionIndex" @change="changeTtsVoice">
<view class="voice-picker-value">{{ currentTtsVoiceLabel }} <uni-icons type="right" size="12" color="#94a3b8" /></view>
</picker>
</view>
<view v-if="loggedIn" class="voice-preference-row">
<view class="display-heading-copy">
<text class="menu-item-title">AI 播报方言</text>
<text class="menu-item-desc">{{ currentTtsDialectLabel }} · 仅影响服务端语音,失败时不退回普通话</text>
</view>
<picker :range="qwenTtsDialectOptions" range-key="label" :value="ttsDialectOptionIndex" @change="changeTtsDialect">
<view class="voice-picker-value">{{ currentTtsDialectLabel }} <uni-icons type="right" size="12" color="#94a3b8" /></view>
</picker>
</view>
</view> </view>
</view> </view>
</template> </template>
@@ -297,6 +316,16 @@ import {
import { getCompetencyProfile, getPracticeHistory, getPromotionEvidence } from '@/services/practice'; import { getCompetencyProfile, getPracticeHistory, getPromotionEvidence } from '@/services/practice';
import { resetPageScroll } from '@/services/navigation'; import { resetPageScroll } from '@/services/navigation';
import { ensureEmployeePosition } from '@/services/position'; import { ensureEmployeePosition } from '@/services/position';
import {
getTtsVoicePreference,
getTtsDialectPreference,
qwenTtsDialectOptions,
qwenTtsVoiceOptions,
setTtsDialectPreference,
setTtsVoicePreference,
type TtsDialectPreference,
type TtsVoicePreference
} from '@/services/tts-voice-preference';
const loggedIn = ref(false); const loggedIn = ref(false);
const phone = ref(''); const phone = ref('');
@@ -313,9 +342,19 @@ const evidence = ref<PromotionEvidence | null>(null);
const positionProfile = ref(getSelectedPositionProfile()); const positionProfile = ref(getSelectedPositionProfile());
const supervisorLearning = ref(false); const supervisorLearning = ref(false);
const fontSize = ref<AppFontSize>(getAppFontSize()); const fontSize = ref<AppFontSize>(getAppFontSize());
const ttsVoice = ref<TtsVoicePreference>('');
const ttsDialect = ref<TtsDialectPreference>('');
const currentFontSizeLabel = computed(() => const currentFontSizeLabel = computed(() =>
appFontSizeOptions.find((item) => item.value === fontSize.value)?.label || '标准' appFontSizeOptions.find((item) => item.value === fontSize.value)?.label || '标准'
); );
const ttsVoiceOptionIndex = computed(() => Math.max(0, qwenTtsVoiceOptions.findIndex((item) => item.value === ttsVoice.value)));
const currentTtsVoiceLabel = computed(() =>
qwenTtsVoiceOptions[ttsVoiceOptionIndex.value]?.label || '跟随场景'
);
const ttsDialectOptionIndex = computed(() => Math.max(0, qwenTtsDialectOptions.findIndex((item) => item.value === ttsDialect.value)));
const currentTtsDialectLabel = computed(() =>
qwenTtsDialectOptions[ttsDialectOptionIndex.value]?.label || '跟随场景'
);
const maskedPhone = computed(() => { const maskedPhone = computed(() => {
const value = phone.value.trim(); const value = phone.value.trim();
if (!value) return '未登录'; if (!value) return '未登录';
@@ -333,6 +372,8 @@ const refreshAuth = () => {
positionProfile.value = getSelectedPositionProfile(); positionProfile.value = getSelectedPositionProfile();
supervisorLearning.value = isSupervisorLearnerMode(); supervisorLearning.value = isSupervisorLearnerMode();
fontSize.value = getAppFontSize(); fontSize.value = getAppFontSize();
ttsVoice.value = getTtsVoicePreference(auth.tenantId, auth.phone);
ttsDialect.value = getTtsDialectPreference(auth.tenantId, auth.phone);
}; };
const changeFontSize = (value: AppFontSize) => { const changeFontSize = (value: AppFontSize) => {
@@ -340,6 +381,20 @@ const changeFontSize = (value: AppFontSize) => {
uni.showToast({ title: `已切换为${currentFontSizeLabel.value}`, icon: 'none' }); uni.showToast({ title: `已切换为${currentFontSizeLabel.value}`, icon: 'none' });
}; };
const changeTtsVoice = (event: { detail?: { value?: string | number } }) => {
const option = qwenTtsVoiceOptions[Number(event.detail?.value)];
const auth = getAuth();
ttsVoice.value = setTtsVoicePreference(auth.tenantId, auth.phone, option?.value);
uni.showToast({ title: `已切换为${currentTtsVoiceLabel.value}`, icon: 'none' });
};
const changeTtsDialect = (event: { detail?: { value?: string | number } }) => {
const option = qwenTtsDialectOptions[Number(event.detail?.value)];
const auth = getAuth();
ttsDialect.value = setTtsDialectPreference(auth.tenantId, auth.phone, option?.value);
uni.showToast({ title: `已切换为${currentTtsDialectLabel.value}`, icon: 'none' });
};
const returnToSupervisor = () => { const returnToSupervisor = () => {
leaveSupervisorLearnerMode(); leaveSupervisorLearnerMode();
supervisorLearning.value = false; supervisorLearning.value = false;
@@ -650,6 +705,28 @@ onShow(() => {
box-shadow: 0 4px 12px rgba(24, 34, 48, 0.08); box-shadow: 0 4px 12px rgba(24, 34, 48, 0.08);
} }
.voice-preference-row {
display: flex;
align-items: center;
justify-content: space-between;
gap: 12px;
padding-top: 12px;
border-top: 1px solid var(--employee-line);
}
.voice-picker-value {
display: flex;
align-items: center;
gap: 3px;
max-width: 150px;
padding: 8px 10px;
border: 1px solid #e3e6ea;
border-radius: 10px;
color: var(--employee-text);
font-size: 12px;
white-space: nowrap;
}
.profile-summary { .profile-summary {
border-color: var(--employee-line); border-color: var(--employee-line);
} }
+20 -9
View File
@@ -1,6 +1,8 @@
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api'; import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import { getAuth } from './auth';
import type { SpeechCapture } from './speech-capture'; import type { SpeechCapture } from './speech-capture';
import { applyTtsVoicePreference, getTtsDialectPreference, getTtsVoicePreference } from './tts-voice-preference';
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg']; const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
@@ -103,11 +105,20 @@ export const transcribeSpeechCapture = (
? transcribeSpeechBlob(capture.blob, filename, registerAbort) ? transcribeSpeechBlob(capture.blob, filename, registerAbort)
: transcribeSpeechFile(capture.file, registerAbort); : transcribeSpeechFile(capture.file, registerAbort);
export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => {
const auth = getAuth();
return applyTtsVoicePreference(
voiceProfile,
getTtsVoicePreference(auth.tenantId, auth.phone),
getTtsDialectPreference(auth.tenantId, auth.phone)
);
};
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) => export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
apiRequest<TtsResponse>({ apiRequest<TtsResponse>({
url: '/api/ai/tts', url: '/api/ai/tts',
method: 'POST', method: 'POST',
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext }, data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext },
timeout: 60000 timeout: 60000
}); });
@@ -118,7 +129,7 @@ export interface SpeechPlaybackSnapshot {
const browserDialectLanguage = (dialect: SpeechVoiceProfile['dialect']) => { const browserDialectLanguage = (dialect: SpeechVoiceProfile['dialect']) => {
if (dialect === 'cantonese') return 'zh-HK'; if (dialect === 'cantonese') return 'zh-HK';
if (dialect === 'sichuanese') return ''; if (dialect && dialect !== 'mandarin') return '';
return 'zh-CN'; return 'zh-CN';
}; };
@@ -221,8 +232,8 @@ export const createSpeechPlaybackController = (
stop(); stop();
const value = text.trim(); const value = text.trim();
if (!value) return; if (!value) return;
const profile = voiceProfile || defaultVoiceProfile; const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
const sourceKey = cacheKey(value, profile, practiceContext); const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
const currentGeneration = generation; const currentGeneration = generation;
publish(key, 'loading'); publish(key, 'loading');
try { try {
@@ -233,7 +244,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, cachedSource, sourceKey, currentGeneration); playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return; return;
} }
const result = await synthesizeSpeech(value, profile, practiceContext); const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
if (currentGeneration !== generation || activeKey !== key) return; if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl; const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频'); if (!source) throw new Error('语音合成未返回音频');
@@ -246,7 +257,7 @@ export const createSpeechPlaybackController = (
playAudioSource(key, source, sourceKey, currentGeneration); playAudioSource(key, source, sourceKey, currentGeneration);
} catch (error) { } catch (error) {
if (currentGeneration !== generation) return; if (currentGeneration !== generation) return;
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, profile); const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile);
if (browserFallback.played) return; if (browserFallback.played) return;
stop(); stop();
onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败')); onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败'));
@@ -261,12 +272,12 @@ export const createSpeechPlaybackController = (
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => { const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
const value = text.trim(); const value = text.trim();
if (!value) return null; if (!value) return null;
const profile = voiceProfile || defaultVoiceProfile; const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
const sourceKey = cacheKey(value, profile, practiceContext); const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
const cached = sourceCache.get(sourceKey); const cached = sourceCache.get(sourceKey);
if (cached) return cached; if (cached) return cached;
try { try {
const result = await synthesizeSpeech(value, profile, practiceContext); const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
const source = result.inlineAudioUrl || result.audioUrl; const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null; if (!source) return null;
sourceCache.set(sourceKey, source); sourceCache.set(sourceKey, source);
@@ -0,0 +1,96 @@
import type { SpeechVoiceProfile } from '@/types/api';
export const qwenTtsVoiceOptions = [
{ value: '', label: '跟随场景' },
{ value: 'longanhuan_v3.6', label: '龙安欢 · 中文女声' },
{ value: 'longjielidou_v3.6', label: '龙杰力豆 · 童声' },
{ value: 'loongeva_v3.6', label: 'Loongeva · 英文女声' },
{ value: 'loongjohn', label: 'loongJohn · 英文男声' }
] as const;
export const qwenTtsDialectOptions = [
{ value: '', label: '跟随场景' },
{ value: 'mandarin', label: '普通话' },
{ value: 'cantonese', label: '粤语' },
{ value: 'chongqing', label: '重庆话' },
{ value: 'northeastern', label: '东北话' },
{ value: 'gansu', label: '甘肃话' },
{ value: 'guizhou', label: '贵州话' },
{ value: 'zhejiang', label: '浙江话' },
{ value: 'hebei', label: '河北话' },
{ value: 'henan', label: '河南话' },
{ value: 'hubei', label: '湖北话' },
{ value: 'hunan', label: '湖南话' },
{ value: 'jiangxi', label: '江西话' },
{ value: 'ningbo', label: '宁波话' },
{ value: 'ningxia', label: '宁夏话' },
{ value: 'qingdao', label: '青岛话' },
{ value: 'shaanxi', label: '陕西话' },
{ value: 'shanxi', label: '山西话' },
{ value: 'shandong', label: '山东话' },
{ value: 'shanghai', label: '上海话' },
{ value: 'sichuanese', label: '四川话' },
{ value: 'yunnan', label: '云南话' }
] as const;
export type TtsVoicePreference = typeof qwenTtsVoiceOptions[number]['value'];
export type TtsDialectPreference = typeof qwenTtsDialectOptions[number]['value'];
const preferenceKeyPrefix = 'aihr_tts_voice';
const dialectPreferenceKeyPrefix = 'aihr_tts_dialect';
export const normalizeTtsVoicePreference = (value: unknown): TtsVoicePreference =>
qwenTtsVoiceOptions.some((option) => option.value === value) ? value as TtsVoicePreference : '';
export const normalizeTtsDialectPreference = (value: unknown): TtsDialectPreference =>
qwenTtsDialectOptions.some((option) => option.value === value) ? value as TtsDialectPreference : '';
export const ttsVoicePreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
const tenant = String(tenantId || '').trim();
const account = String(phone || '').trim();
return tenant && account ? `${preferenceKeyPrefix}:${tenant}:${account}` : '';
};
export const getTtsVoicePreference = (tenantId: unknown, phone: unknown): TtsVoicePreference => {
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
return key ? normalizeTtsVoicePreference(uni.getStorageSync(key)) : '';
};
const ttsDialectPreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
const tenant = String(tenantId || '').trim();
const account = String(phone || '').trim();
return tenant && account ? `${dialectPreferenceKeyPrefix}:${tenant}:${account}` : '';
};
export const getTtsDialectPreference = (tenantId: unknown, phone: unknown): TtsDialectPreference => {
const key = ttsDialectPreferenceStorageKey(tenantId, phone);
return key ? normalizeTtsDialectPreference(uni.getStorageSync(key)) : '';
};
export const setTtsVoicePreference = (tenantId: unknown, phone: unknown, value: unknown): TtsVoicePreference => {
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
const normalized = normalizeTtsVoicePreference(value);
if (key) uni.setStorageSync(key, normalized);
return normalized;
};
export const setTtsDialectPreference = (tenantId: unknown, phone: unknown, value: unknown): TtsDialectPreference => {
const key = ttsDialectPreferenceStorageKey(tenantId, phone);
const normalized = normalizeTtsDialectPreference(value);
if (key) uni.setStorageSync(key, normalized);
return normalized;
};
export const applyTtsVoicePreference = (
profile: SpeechVoiceProfile | undefined,
preference: unknown,
dialectPreference?: unknown
): SpeechVoiceProfile | undefined => {
const voice = normalizeTtsVoicePreference(preference);
const dialect = normalizeTtsDialectPreference(dialectPreference);
if (!voice && !dialect) return profile;
const next: SpeechVoiceProfile = { ...(profile || { role: 'neutral' }) };
if (voice) next.voice = voice;
if (dialect) next.dialect = dialect;
return next;
};
+24 -1
View File
@@ -45,7 +45,7 @@ export interface SpeechVoiceProfile {
voice?: string; voice?: string;
speed?: number; speed?: number;
emotion?: 'warm' | 'professional' | 'serious' | 'intense' | 'calm'; emotion?: 'warm' | 'professional' | 'serious' | 'intense' | 'calm';
dialect?: 'mandarin' | 'cantonese' | 'sichuanese'; dialect?: 'mandarin' | 'cantonese' | 'chongqing' | 'northeastern' | 'gansu' | 'guizhou' | 'zhejiang' | 'hebei' | 'henan' | 'hubei' | 'hunan' | 'jiangxi' | 'ningbo' | 'ningxia' | 'qingdao' | 'shaanxi' | 'shanxi' | 'shandong' | 'shanghai' | 'sichuanese' | 'yunnan';
} }
export interface ToolItem { export interface ToolItem {
@@ -582,6 +582,13 @@ export interface PracticeScenarioOption {
enabled?: boolean; enabled?: boolean;
position?: string; position?: string;
scenarioType?: string; scenarioType?: string;
projectType?: string;
growthLevel?: string;
competencyCode?: string;
collaborationPositions?: string;
reviewStatus?: string;
curriculumVersion?: string;
difficulty?: number;
} }
export interface ReviewDialogue { export interface ReviewDialogue {
@@ -601,6 +608,12 @@ export interface ReviewAnnotation {
} }
export interface ReviewDetail extends PracticeRecord { export interface ReviewDetail extends PracticeRecord {
position?: string;
projectType?: string;
growthLevel?: string;
competencyCode?: string;
collaborationPositions?: string;
curriculumVersion?: string;
mentorRewrite: string; mentorRewrite: string;
aiComment: string; aiComment: string;
reviewAdvice?: string; reviewAdvice?: string;
@@ -1087,6 +1100,16 @@ export interface CompetencyProfile {
aiLevel?: string; aiLevel?: string;
dimensions: CompetencyDimension[]; dimensions: CompetencyDimension[];
growthPath?: GrowthStage[]; growthPath?: GrowthStage[];
capabilityProgress?: CapabilityProgress[];
}
export interface CapabilityProgress {
position: string;
growthLevel: string;
competencyCode: string;
completed: number;
averageScore: number;
status: string;
} }
export interface PromotionEvidence { export interface PromotionEvidence {
@@ -0,0 +1,57 @@
import assert from 'node:assert/strict';
import { readFile } from 'node:fs/promises';
import test from 'node:test';
import vm from 'node:vm';
import ts from 'typescript';
const source = (path) => readFile(new URL(path, import.meta.url), 'utf8');
const preferenceRuntime = async (storage = new Map()) => {
const service = await source('../src/services/tts-voice-preference.ts');
const compiled = ts.transpileModule(service, {
compilerOptions: { module: ts.ModuleKind.CommonJS, target: ts.ScriptTarget.ES2020 }
});
const runtimeModule = { exports: {} };
vm.runInNewContext(compiled.outputText, {
exports: runtimeModule.exports,
module: runtimeModule,
uni: {
getStorageSync: (key) => storage.get(key),
setStorageSync: (key, value) => storage.set(key, value)
}
});
return runtimeModule.exports;
};
test('播报音色按租户和账号隔离保存,非法值回退跟随场景', async () => {
const storage = new Map();
const runtime = await preferenceRuntime(storage);
assert.equal(runtime.getTtsVoicePreference('000000', '13800000000'), '');
assert.equal(runtime.setTtsVoicePreference('000000', '13800000000', 'longanhuan_v3.6'), 'longanhuan_v3.6');
assert.equal(runtime.getTtsVoicePreference('000000', '13800000000'), 'longanhuan_v3.6');
assert.equal(runtime.getTtsVoicePreference('000000', '13900000000'), '');
assert.equal(runtime.setTtsVoicePreference('000000', '13800000000', 'untrusted-voice'), '');
assert.equal(runtime.setTtsDialectPreference('000000', '13800000000', 'shanghai'), 'shanghai');
assert.equal(runtime.getTtsDialectPreference('000000', '13800000000'), 'shanghai');
});
test('全局选择只覆盖 voice,保留场景角色、情绪、语速和方言', async () => {
const runtime = await preferenceRuntime();
const profile = { role: 'customer', emotion: 'intense', speed: 1.12, dialect: 'sichuanese' };
assert.deepEqual(
JSON.parse(JSON.stringify(runtime.applyTtsVoicePreference(profile, 'longanhuan_v3.6', 'shanghai'))),
{ ...profile, voice: 'longanhuan_v3.6', dialect: 'shanghai' }
);
assert.deepEqual(JSON.parse(JSON.stringify(runtime.applyTtsVoicePreference(profile, ''))), profile);
});
test('语音播放链路在合成前统一注入全局音色,且缓存键包含处理后的 profile', async () => {
const sourceText = await source('../src/services/speech.ts');
assert.match(sourceText, /applyTtsVoicePreference/);
assert.match(sourceText, /getTtsVoicePreference/);
assert.match(sourceText, /cacheKey\(value, effectiveProfile, practiceContext\)/);
assert.match(sourceText, /synthesizeSpeech\(value, effectiveProfile, practiceContext\)/);
});