feat: add qwen tts voice preferences
This commit is contained in:
+17
-1
@@ -58,6 +58,12 @@ public class AihrModelSeedService {
|
|||||||
@Value("${aihr.ai-runtime.speech-enabled:${AIHR_AI_SPEECH_ENABLED:true}}")
|
@Value("${aihr.ai-runtime.speech-enabled:${AIHR_AI_SPEECH_ENABLED:true}}")
|
||||||
private boolean speechEnabled;
|
private boolean speechEnabled;
|
||||||
|
|
||||||
|
@Value("${aihr.realtime-practice.endpoint:${AIHR_QWEN_REALTIME_ENDPOINT:}}")
|
||||||
|
private String qwenRealtimeEndpoint;
|
||||||
|
|
||||||
|
@Value("${aihr.realtime-practice.api-key:${AIHR_QWEN_REALTIME_API_KEY:}}")
|
||||||
|
private String qwenRealtimeApiKey;
|
||||||
|
|
||||||
public List<ProviderResponse> providers() {
|
public List<ProviderResponse> providers() {
|
||||||
List<ProviderResponse> rows = dbProviders();
|
List<ProviderResponse> rows = dbProviders();
|
||||||
return rows.isEmpty() ? seedProviders() : mergeProviders(rows);
|
return rows.isEmpty() ? seedProviders() : mergeProviders(rows);
|
||||||
@@ -267,7 +273,7 @@ public class AihrModelSeedService {
|
|||||||
rs.getString("resolved_api_key")
|
rs.getString("resolved_api_key")
|
||||||
), tenantId(), category);
|
), tenantId(), category);
|
||||||
return rows.stream()
|
return rows.stream()
|
||||||
.filter(model -> configured(model.baseUrl(), model.modelName(), model.apiKey()))
|
.filter(model -> configured(model.baseUrl(), model.modelName(), model.apiKey()) || qwenTtsConfigured(model))
|
||||||
.findFirst();
|
.findFirst();
|
||||||
} catch (DataAccessException e) {
|
} catch (DataAccessException e) {
|
||||||
log.debug("aihr speech model db fallback(处理错误已隐藏)");
|
log.debug("aihr speech model db fallback(处理错误已隐藏)");
|
||||||
@@ -282,6 +288,16 @@ public class AihrModelSeedService {
|
|||||||
public record SpeechModel(String providerCode, String modelName, String baseUrl, String apiKey) {
|
public record SpeechModel(String providerCode, String modelName, String baseUrl, String apiKey) {
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private boolean qwenTtsConfigured(SpeechModel model) {
|
||||||
|
return isDashScopeQwenTts(model) && !isBlank(qwenRealtimeEndpoint) && !isBlank(qwenRealtimeApiKey);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean isDashScopeQwenTts(SpeechModel model) {
|
||||||
|
return model != null
|
||||||
|
&& ("dashscope".equalsIgnoreCase(model.providerCode()) || "qianwen".equalsIgnoreCase(model.providerCode()))
|
||||||
|
&& "qwen-audio-3.0-tts-flash".equals(model.modelName());
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 供其他模块(如三角色对练)复用的 chat 调用:模型未配置或调用失败返回 empty,由调用方决定兜底。
|
* 供其他模块(如三角色对练)复用的 chat 调用:模型未配置或调用失败返回 empty,由调用方决定兜底。
|
||||||
*/
|
*/
|
||||||
|
|||||||
+141
@@ -6,6 +6,7 @@ import lombok.RequiredArgsConstructor;
|
|||||||
import lombok.extern.slf4j.Slf4j;
|
import lombok.extern.slf4j.Slf4j;
|
||||||
import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile;
|
import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile;
|
||||||
import org.dromara.aihr.service.AihrModelSeedService.SpeechModel;
|
import org.dromara.aihr.service.AihrModelSeedService.SpeechModel;
|
||||||
|
import org.springframework.beans.factory.annotation.Value;
|
||||||
import org.springframework.stereotype.Service;
|
import org.springframework.stereotype.Service;
|
||||||
|
|
||||||
import java.io.ByteArrayOutputStream;
|
import java.io.ByteArrayOutputStream;
|
||||||
@@ -33,7 +34,10 @@ public class AihrSpeechService {
|
|||||||
|
|
||||||
private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15);
|
private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15);
|
||||||
private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5);
|
private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5);
|
||||||
|
private static final Duration DASHSCOPE_TTS_TIMEOUT = Duration.ofSeconds(20);
|
||||||
private static final String DEFAULT_TTS_VOICE = "anna";
|
private static final String DEFAULT_TTS_VOICE = "anna";
|
||||||
|
private static final String QWEN_TTS_MODEL = "qwen-audio-3.0-tts-flash";
|
||||||
|
private static final String QWEN_TTS_DEFAULT_VOICE = "longanhuan_v3.6";
|
||||||
private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}");
|
private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}");
|
||||||
private static final Map<String, String> ROLE_VOICES = Map.of(
|
private static final Map<String, String> ROLE_VOICES = Map.of(
|
||||||
"mentor", "speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr",
|
"mentor", "speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr",
|
||||||
@@ -52,10 +56,25 @@ public class AihrSpeechService {
|
|||||||
"cantonese", "请使用自然粤语口语表达",
|
"cantonese", "请使用自然粤语口语表达",
|
||||||
"sichuanese", "请使用自然四川话口语表达"
|
"sichuanese", "请使用自然四川话口语表达"
|
||||||
);
|
);
|
||||||
|
private static final Map<String, String> QWEN_TTS_ROLE_VOICES = Map.of(
|
||||||
|
"mentor", QWEN_TTS_DEFAULT_VOICE,
|
||||||
|
"customer", QWEN_TTS_DEFAULT_VOICE,
|
||||||
|
"interviewer", QWEN_TTS_DEFAULT_VOICE,
|
||||||
|
"neutral", QWEN_TTS_DEFAULT_VOICE
|
||||||
|
);
|
||||||
|
private static final java.util.Set<String> QWEN_TTS_VOICES = java.util.Set.of(
|
||||||
|
"longanhuan_v3.6", "longjielidou_v3.6", "loongeva_v3.6", "loongjohn"
|
||||||
|
);
|
||||||
|
|
||||||
private final AihrModelSeedService modelService;
|
private final AihrModelSeedService modelService;
|
||||||
private final ObjectMapper objectMapper;
|
private final ObjectMapper objectMapper;
|
||||||
|
|
||||||
|
@Value("${aihr.realtime-practice.endpoint:${AIHR_QWEN_REALTIME_ENDPOINT:}}")
|
||||||
|
private String qwenRealtimeEndpoint;
|
||||||
|
|
||||||
|
@Value("${aihr.realtime-practice.api-key:${AIHR_QWEN_REALTIME_API_KEY:}}")
|
||||||
|
private String qwenRealtimeApiKey;
|
||||||
|
|
||||||
public boolean asrConfigured() {
|
public boolean asrConfigured() {
|
||||||
return modelService.speechModel("asr").isPresent();
|
return modelService.speechModel("asr").isPresent();
|
||||||
}
|
}
|
||||||
@@ -117,6 +136,9 @@ public class AihrSpeechService {
|
|||||||
}
|
}
|
||||||
SpeechModel runtime = model.get();
|
SpeechModel runtime = model.get();
|
||||||
try {
|
try {
|
||||||
|
if (isDashScopeQwenTts(runtime)) {
|
||||||
|
return synthesizeDashScopeQwen(runtime, text, voice, voiceProfile);
|
||||||
|
}
|
||||||
ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile);
|
ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile);
|
||||||
HttpRequest httpRequest = authorized(runtime, "/audio/speech")
|
HttpRequest httpRequest = authorized(runtime, "/audio/speech")
|
||||||
.header("Content-Type", "application/json")
|
.header("Content-Type", "application/json")
|
||||||
@@ -134,6 +156,38 @@ public class AihrSpeechService {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private Optional<byte[]> synthesizeDashScopeQwen(SpeechModel runtime, String text, String voice, VoiceProfile voiceProfile) {
|
||||||
|
if (isBlank(qwenRealtimeEndpoint) || isBlank(qwenRealtimeApiKey)) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
ObjectNode body = buildDashScopeRequestBody(objectMapper, runtime, text, voiceProfile == null
|
||||||
|
? new VoiceProfile("neutral", voice, null, null, null)
|
||||||
|
: withLegacyVoice(voiceProfile, voice));
|
||||||
|
HttpRequest request = HttpRequest.newBuilder()
|
||||||
|
.uri(dashScopeSynthesisUri(qwenRealtimeEndpoint))
|
||||||
|
.timeout(DASHSCOPE_TTS_TIMEOUT)
|
||||||
|
.header("Content-Type", "application/json")
|
||||||
|
.header("Authorization", "Bearer " + qwenRealtimeApiKey.trim())
|
||||||
|
.POST(HttpRequest.BodyPublishers.ofString(objectMapper.writeValueAsString(body)))
|
||||||
|
.build();
|
||||||
|
HttpResponse<String> response = client().send(request, HttpResponse.BodyHandlers.ofString());
|
||||||
|
if (response.statusCode() < 200 || response.statusCode() >= 300) {
|
||||||
|
log.warn("dashscope tts http {}(外部响应体已隐藏)", response.statusCode());
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
Optional<String> audioUrl = dashScopeAudioUrl(objectMapper, response.body());
|
||||||
|
if (audioUrl.isEmpty()) {
|
||||||
|
log.warn("dashscope tts response missing audio url");
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
return downloadDashScopeAudio(audioUrl.get());
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("dashscope tts call failed(处理错误已隐藏)");
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text,
|
static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text,
|
||||||
String legacyVoice, VoiceProfile voiceProfile) {
|
String legacyVoice, VoiceProfile voiceProfile) {
|
||||||
String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role());
|
String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role());
|
||||||
@@ -142,6 +196,9 @@ public class AihrSpeechService {
|
|||||||
if (requestedVoice == null) {
|
if (requestedVoice == null) {
|
||||||
requestedVoice = safeVoice(legacyVoice);
|
requestedVoice = safeVoice(legacyVoice);
|
||||||
}
|
}
|
||||||
|
if (requestedVoice != null && QWEN_TTS_VOICES.contains(requestedVoice)) {
|
||||||
|
requestedVoice = null;
|
||||||
|
}
|
||||||
if (requestedVoice == null && expressiveCosyVoice) {
|
if (requestedVoice == null && expressiveCosyVoice) {
|
||||||
requestedVoice = ROLE_VOICES.get(role);
|
requestedVoice = ROLE_VOICES.get(role);
|
||||||
}
|
}
|
||||||
@@ -165,6 +222,80 @@ public class AihrSpeechService {
|
|||||||
return body;
|
return body;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static ObjectNode buildDashScopeRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text,
|
||||||
|
VoiceProfile voiceProfile) {
|
||||||
|
String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role());
|
||||||
|
String emotion = normalizedEmotion(voiceProfile == null ? null : voiceProfile.emotion(), role);
|
||||||
|
String dialect = normalizedDialect(voiceProfile == null ? null : voiceProfile.dialect());
|
||||||
|
String requestedVoice = safeVoice(voiceProfile == null ? null : voiceProfile.voice());
|
||||||
|
String voice = requestedVoice != null && QWEN_TTS_VOICES.contains(requestedVoice)
|
||||||
|
? requestedVoice
|
||||||
|
: QWEN_TTS_ROLE_VOICES.getOrDefault(role, QWEN_TTS_DEFAULT_VOICE);
|
||||||
|
ObjectNode input = objectMapper.createObjectNode();
|
||||||
|
input.put("text", AihrSensitiveText.forModel(text.trim()));
|
||||||
|
input.put("voice", voice);
|
||||||
|
input.put("format", "mp3");
|
||||||
|
input.put("sample_rate", 24000);
|
||||||
|
input.put("rate", resolveSpeed(voiceProfile == null ? null : voiceProfile.speed(), emotion));
|
||||||
|
String instruction = expressivePrompt(emotion, dialect);
|
||||||
|
if (!instruction.isBlank()) {
|
||||||
|
input.put("instruction", instruction);
|
||||||
|
}
|
||||||
|
ObjectNode body = objectMapper.createObjectNode();
|
||||||
|
body.put("model", runtime == null || isBlank(runtime.modelName()) ? QWEN_TTS_MODEL : runtime.modelName());
|
||||||
|
body.set("input", input);
|
||||||
|
return body;
|
||||||
|
}
|
||||||
|
|
||||||
|
static URI dashScopeSynthesisUri(String endpoint) {
|
||||||
|
URI configured = URI.create(endpoint == null ? "" : endpoint.trim());
|
||||||
|
String scheme = configured.getScheme();
|
||||||
|
String authority = configured.getRawAuthority();
|
||||||
|
if (("https".equalsIgnoreCase(scheme) || "http".equalsIgnoreCase(scheme)) && authority != null && !authority.isBlank()) {
|
||||||
|
return URI.create(scheme + "://" + authority + "/api/v1/services/audio/tts/SpeechSynthesizer");
|
||||||
|
}
|
||||||
|
throw new IllegalArgumentException("invalid DashScope endpoint");
|
||||||
|
}
|
||||||
|
|
||||||
|
static Optional<String> dashScopeAudioUrl(ObjectMapper objectMapper, String responseBody) {
|
||||||
|
try {
|
||||||
|
String value = objectMapper.readTree(responseBody == null ? "" : responseBody)
|
||||||
|
.path("output").path("audio").path("url").asText("").trim();
|
||||||
|
URI uri = value.isBlank() ? null : URI.create(value);
|
||||||
|
if (uri == null || !("http".equalsIgnoreCase(uri.getScheme()) || "https".equalsIgnoreCase(uri.getScheme()))
|
||||||
|
|| uri.getHost() == null || !uri.getHost().endsWith(".oss-cn-beijing.aliyuncs.com")) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
return Optional.of(uri.toString());
|
||||||
|
} catch (Exception e) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private Optional<byte[]> downloadDashScopeAudio(String url) {
|
||||||
|
try {
|
||||||
|
HttpRequest request = HttpRequest.newBuilder()
|
||||||
|
.uri(URI.create(url))
|
||||||
|
.timeout(DASHSCOPE_TTS_TIMEOUT)
|
||||||
|
.GET()
|
||||||
|
.build();
|
||||||
|
HttpResponse<byte[]> response = client().send(request, HttpResponse.BodyHandlers.ofByteArray());
|
||||||
|
if (response.statusCode() < 200 || response.statusCode() >= 300 || response.body().length == 0) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
return Optional.of(response.body());
|
||||||
|
} catch (Exception e) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static VoiceProfile withLegacyVoice(VoiceProfile profile, String legacyVoice) {
|
||||||
|
if (safeVoice(profile.voice()) != null || safeVoice(legacyVoice) == null) {
|
||||||
|
return profile;
|
||||||
|
}
|
||||||
|
return new VoiceProfile(profile.role(), legacyVoice, profile.speed(), profile.emotion(), profile.dialect());
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。
|
* 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。
|
||||||
*/
|
*/
|
||||||
@@ -236,6 +367,12 @@ public class AihrSpeechService {
|
|||||||
&& runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice");
|
&& runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static boolean isDashScopeQwenTts(SpeechModel runtime) {
|
||||||
|
return runtime != null
|
||||||
|
&& ("dashscope".equalsIgnoreCase(runtime.providerCode()) || "qianwen".equalsIgnoreCase(runtime.providerCode()))
|
||||||
|
&& QWEN_TTS_MODEL.equals(runtime.modelName());
|
||||||
|
}
|
||||||
|
|
||||||
private HttpRequest.Builder authorized(SpeechModel runtime, String path) {
|
private HttpRequest.Builder authorized(SpeechModel runtime, String path) {
|
||||||
HttpRequest.Builder builder = HttpRequest.newBuilder()
|
HttpRequest.Builder builder = HttpRequest.newBuilder()
|
||||||
.uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path))
|
.uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path))
|
||||||
@@ -286,6 +423,10 @@ public class AihrSpeechService {
|
|||||||
return normalized;
|
return normalized;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static boolean isBlank(String value) {
|
||||||
|
return value == null || value.isBlank();
|
||||||
|
}
|
||||||
|
|
||||||
private static String truncate(String value) {
|
private static String truncate(String value) {
|
||||||
if (value == null || value.length() <= 240) {
|
if (value == null || value.length() <= 240) {
|
||||||
return value;
|
return value;
|
||||||
|
|||||||
+4
@@ -32,6 +32,9 @@ class AihrSpeechServiceTest {
|
|||||||
new VoiceProfile("customer", null, 9.0, "intense", "sichuanese"));
|
new VoiceProfile("customer", null, 9.0, "intense", "sichuanese"));
|
||||||
var legacy = AihrSpeechService.buildRequestBody(
|
var legacy = AihrSpeechService.buildRequestBody(
|
||||||
mapper, runtime, "普通播报", "alex", null);
|
mapper, runtime, "普通播报", "alex", null);
|
||||||
|
var qwenVoiceBeforeSwitch = AihrSpeechService.buildRequestBody(
|
||||||
|
mapper, runtime, "普通播报", null,
|
||||||
|
new VoiceProfile("neutral", "longanhuan_v3.6", null, null, null));
|
||||||
|
|
||||||
assertEquals("speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", mentor.path("voice").asText());
|
assertEquals("speech:shifu-warm-v1:cm3hz4wfz02jy106j6z6muix7:ysvbyzypjgmnceokxedr", mentor.path("voice").asText());
|
||||||
assertEquals(0.88, mentor.path("speed").asDouble());
|
assertEquals(0.88, mentor.path("speed").asDouble());
|
||||||
@@ -41,6 +44,7 @@ class AihrSpeechServiceTest {
|
|||||||
assertTrue(customer.path("input").asText().contains("情绪强烈"));
|
assertTrue(customer.path("input").asText().contains("情绪强烈"));
|
||||||
assertTrue(customer.path("input").asText().contains("四川话"));
|
assertTrue(customer.path("input").asText().contains("四川话"));
|
||||||
assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText());
|
assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText());
|
||||||
|
assertEquals("FunAudioLLM/CosyVoice2-0.5B:anna", qwenVoiceBeforeSwitch.path("voice").asText());
|
||||||
assertFalse(legacy.has("speed"));
|
assertFalse(legacy.has("speed"));
|
||||||
assertFalse(legacy.path("input").asText().contains("<|endofprompt|>"));
|
assertFalse(legacy.path("input").asText().contains("<|endofprompt|>"));
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -271,6 +271,16 @@
|
|||||||
</view>
|
</view>
|
||||||
<text class="display-note">切换后同步到今日、练、问、我全部页面,并自动保存。</text>
|
<text class="display-note">切换后同步到今日、练、问、我全部页面,并自动保存。</text>
|
||||||
</template>
|
</template>
|
||||||
|
|
||||||
|
<view v-if="loggedIn" class="voice-preference-row">
|
||||||
|
<view class="display-heading-copy">
|
||||||
|
<text class="menu-item-title">AI 播报音色</text>
|
||||||
|
<text class="menu-item-desc">{{ currentTtsVoiceLabel }} · 对练情绪和语速仍会保留</text>
|
||||||
|
</view>
|
||||||
|
<picker :range="qwenTtsVoiceOptions" range-key="label" :value="ttsVoiceOptionIndex" @change="changeTtsVoice">
|
||||||
|
<view class="voice-picker-value">{{ currentTtsVoiceLabel }} <uni-icons type="right" size="12" color="#94a3b8" /></view>
|
||||||
|
</picker>
|
||||||
|
</view>
|
||||||
</view>
|
</view>
|
||||||
</view>
|
</view>
|
||||||
</template>
|
</template>
|
||||||
@@ -297,6 +307,12 @@ import {
|
|||||||
import { getCompetencyProfile, getPracticeHistory, getPromotionEvidence } from '@/services/practice';
|
import { getCompetencyProfile, getPracticeHistory, getPromotionEvidence } from '@/services/practice';
|
||||||
import { resetPageScroll } from '@/services/navigation';
|
import { resetPageScroll } from '@/services/navigation';
|
||||||
import { ensureEmployeePosition } from '@/services/position';
|
import { ensureEmployeePosition } from '@/services/position';
|
||||||
|
import {
|
||||||
|
getTtsVoicePreference,
|
||||||
|
qwenTtsVoiceOptions,
|
||||||
|
setTtsVoicePreference,
|
||||||
|
type TtsVoicePreference
|
||||||
|
} from '@/services/tts-voice-preference';
|
||||||
|
|
||||||
const loggedIn = ref(false);
|
const loggedIn = ref(false);
|
||||||
const phone = ref('');
|
const phone = ref('');
|
||||||
@@ -313,9 +329,14 @@ const evidence = ref<PromotionEvidence | null>(null);
|
|||||||
const positionProfile = ref(getSelectedPositionProfile());
|
const positionProfile = ref(getSelectedPositionProfile());
|
||||||
const supervisorLearning = ref(false);
|
const supervisorLearning = ref(false);
|
||||||
const fontSize = ref<AppFontSize>(getAppFontSize());
|
const fontSize = ref<AppFontSize>(getAppFontSize());
|
||||||
|
const ttsVoice = ref<TtsVoicePreference>('');
|
||||||
const currentFontSizeLabel = computed(() =>
|
const currentFontSizeLabel = computed(() =>
|
||||||
appFontSizeOptions.find((item) => item.value === fontSize.value)?.label || '标准'
|
appFontSizeOptions.find((item) => item.value === fontSize.value)?.label || '标准'
|
||||||
);
|
);
|
||||||
|
const ttsVoiceOptionIndex = computed(() => Math.max(0, qwenTtsVoiceOptions.findIndex((item) => item.value === ttsVoice.value)));
|
||||||
|
const currentTtsVoiceLabel = computed(() =>
|
||||||
|
qwenTtsVoiceOptions[ttsVoiceOptionIndex.value]?.label || '跟随场景'
|
||||||
|
);
|
||||||
const maskedPhone = computed(() => {
|
const maskedPhone = computed(() => {
|
||||||
const value = phone.value.trim();
|
const value = phone.value.trim();
|
||||||
if (!value) return '未登录';
|
if (!value) return '未登录';
|
||||||
@@ -333,6 +354,7 @@ const refreshAuth = () => {
|
|||||||
positionProfile.value = getSelectedPositionProfile();
|
positionProfile.value = getSelectedPositionProfile();
|
||||||
supervisorLearning.value = isSupervisorLearnerMode();
|
supervisorLearning.value = isSupervisorLearnerMode();
|
||||||
fontSize.value = getAppFontSize();
|
fontSize.value = getAppFontSize();
|
||||||
|
ttsVoice.value = getTtsVoicePreference(auth.tenantId, auth.phone);
|
||||||
};
|
};
|
||||||
|
|
||||||
const changeFontSize = (value: AppFontSize) => {
|
const changeFontSize = (value: AppFontSize) => {
|
||||||
@@ -340,6 +362,13 @@ const changeFontSize = (value: AppFontSize) => {
|
|||||||
uni.showToast({ title: `已切换为${currentFontSizeLabel.value}`, icon: 'none' });
|
uni.showToast({ title: `已切换为${currentFontSizeLabel.value}`, icon: 'none' });
|
||||||
};
|
};
|
||||||
|
|
||||||
|
const changeTtsVoice = (event: { detail?: { value?: string | number } }) => {
|
||||||
|
const option = qwenTtsVoiceOptions[Number(event.detail?.value)];
|
||||||
|
const auth = getAuth();
|
||||||
|
ttsVoice.value = setTtsVoicePreference(auth.tenantId, auth.phone, option?.value);
|
||||||
|
uni.showToast({ title: `已切换为${currentTtsVoiceLabel.value}`, icon: 'none' });
|
||||||
|
};
|
||||||
|
|
||||||
const returnToSupervisor = () => {
|
const returnToSupervisor = () => {
|
||||||
leaveSupervisorLearnerMode();
|
leaveSupervisorLearnerMode();
|
||||||
supervisorLearning.value = false;
|
supervisorLearning.value = false;
|
||||||
@@ -650,6 +679,28 @@ onShow(() => {
|
|||||||
box-shadow: 0 4px 12px rgba(24, 34, 48, 0.08);
|
box-shadow: 0 4px 12px rgba(24, 34, 48, 0.08);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
.voice-preference-row {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
justify-content: space-between;
|
||||||
|
gap: 12px;
|
||||||
|
padding-top: 12px;
|
||||||
|
border-top: 1px solid var(--employee-line);
|
||||||
|
}
|
||||||
|
|
||||||
|
.voice-picker-value {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 3px;
|
||||||
|
max-width: 150px;
|
||||||
|
padding: 8px 10px;
|
||||||
|
border: 1px solid #e3e6ea;
|
||||||
|
border-radius: 10px;
|
||||||
|
color: var(--employee-text);
|
||||||
|
font-size: 12px;
|
||||||
|
white-space: nowrap;
|
||||||
|
}
|
||||||
|
|
||||||
.profile-summary {
|
.profile-summary {
|
||||||
border-color: var(--employee-line);
|
border-color: var(--employee-line);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
|
||||||
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
|
||||||
|
import { getAuth } from './auth';
|
||||||
import type { SpeechCapture } from './speech-capture';
|
import type { SpeechCapture } from './speech-capture';
|
||||||
|
import { applyTtsVoicePreference, getTtsVoicePreference } from './tts-voice-preference';
|
||||||
|
|
||||||
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
|
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
|
||||||
|
|
||||||
@@ -103,11 +105,16 @@ export const transcribeSpeechCapture = (
|
|||||||
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
|
? transcribeSpeechBlob(capture.blob, filename, registerAbort)
|
||||||
: transcribeSpeechFile(capture.file, registerAbort);
|
: transcribeSpeechFile(capture.file, registerAbort);
|
||||||
|
|
||||||
|
export const resolveTtsVoiceProfile = (voiceProfile?: SpeechVoiceProfile) => {
|
||||||
|
const auth = getAuth();
|
||||||
|
return applyTtsVoicePreference(voiceProfile, getTtsVoicePreference(auth.tenantId, auth.phone));
|
||||||
|
};
|
||||||
|
|
||||||
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext) =>
|
||||||
apiRequest<TtsResponse>({
|
apiRequest<TtsResponse>({
|
||||||
url: '/api/ai/tts',
|
url: '/api/ai/tts',
|
||||||
method: 'POST',
|
method: 'POST',
|
||||||
data: { text: text.trim().slice(0, 300), voiceProfile, practiceContext },
|
data: { text: text.trim().slice(0, 300), voiceProfile: resolveTtsVoiceProfile(voiceProfile), practiceContext },
|
||||||
timeout: 60000
|
timeout: 60000
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -221,8 +228,8 @@ export const createSpeechPlaybackController = (
|
|||||||
stop();
|
stop();
|
||||||
const value = text.trim();
|
const value = text.trim();
|
||||||
if (!value) return;
|
if (!value) return;
|
||||||
const profile = voiceProfile || defaultVoiceProfile;
|
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
|
||||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
|
||||||
const currentGeneration = generation;
|
const currentGeneration = generation;
|
||||||
publish(key, 'loading');
|
publish(key, 'loading');
|
||||||
try {
|
try {
|
||||||
@@ -233,7 +240,7 @@ export const createSpeechPlaybackController = (
|
|||||||
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
|
||||||
if (currentGeneration !== generation || activeKey !== key) return;
|
if (currentGeneration !== generation || activeKey !== key) return;
|
||||||
const source = result.inlineAudioUrl || result.audioUrl;
|
const source = result.inlineAudioUrl || result.audioUrl;
|
||||||
if (!source) throw new Error('语音合成未返回音频');
|
if (!source) throw new Error('语音合成未返回音频');
|
||||||
@@ -246,7 +253,7 @@ export const createSpeechPlaybackController = (
|
|||||||
playAudioSource(key, source, sourceKey, currentGeneration);
|
playAudioSource(key, source, sourceKey, currentGeneration);
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
if (currentGeneration !== generation) return;
|
if (currentGeneration !== generation) return;
|
||||||
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, profile);
|
const browserFallback = playWithBrowserSpeech(key, value, currentGeneration, effectiveProfile);
|
||||||
if (browserFallback.played) return;
|
if (browserFallback.played) return;
|
||||||
stop();
|
stop();
|
||||||
onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败'));
|
onError(browserFallback.message || (error instanceof Error ? error.message : '语音生成失败'));
|
||||||
@@ -261,12 +268,12 @@ export const createSpeechPlaybackController = (
|
|||||||
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
|
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile, practiceContext?: PracticeTtsContext): Promise<string | null> => {
|
||||||
const value = text.trim();
|
const value = text.trim();
|
||||||
if (!value) return null;
|
if (!value) return null;
|
||||||
const profile = voiceProfile || defaultVoiceProfile;
|
const effectiveProfile = resolveTtsVoiceProfile(voiceProfile || defaultVoiceProfile);
|
||||||
const sourceKey = cacheKey(value, profile, practiceContext);
|
const sourceKey = cacheKey(value, effectiveProfile, practiceContext);
|
||||||
const cached = sourceCache.get(sourceKey);
|
const cached = sourceCache.get(sourceKey);
|
||||||
if (cached) return cached;
|
if (cached) return cached;
|
||||||
try {
|
try {
|
||||||
const result = await synthesizeSpeech(value, profile, practiceContext);
|
const result = await synthesizeSpeech(value, effectiveProfile, practiceContext);
|
||||||
const source = result.inlineAudioUrl || result.audioUrl;
|
const source = result.inlineAudioUrl || result.audioUrl;
|
||||||
if (!source) return null;
|
if (!source) return null;
|
||||||
sourceCache.set(sourceKey, source);
|
sourceCache.set(sourceKey, source);
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
import type { SpeechVoiceProfile } from '@/types/api';
|
||||||
|
|
||||||
|
export const qwenTtsVoiceOptions = [
|
||||||
|
{ value: '', label: '跟随场景' },
|
||||||
|
{ value: 'longanhuan_v3.6', label: '龙安欢 · 中文女声' },
|
||||||
|
{ value: 'longjielidou_v3.6', label: '龙杰里斗 · 童声' },
|
||||||
|
{ value: 'loongeva_v3.6', label: 'Loongeva · 英文女声' },
|
||||||
|
{ value: 'loongjohn', label: 'Loongjohn · 英文男声' }
|
||||||
|
] as const;
|
||||||
|
|
||||||
|
export type TtsVoicePreference = typeof qwenTtsVoiceOptions[number]['value'];
|
||||||
|
|
||||||
|
const preferenceKeyPrefix = 'aihr_tts_voice';
|
||||||
|
|
||||||
|
export const normalizeTtsVoicePreference = (value: unknown): TtsVoicePreference =>
|
||||||
|
qwenTtsVoiceOptions.some((option) => option.value === value) ? value as TtsVoicePreference : '';
|
||||||
|
|
||||||
|
export const ttsVoicePreferenceStorageKey = (tenantId: unknown, phone: unknown) => {
|
||||||
|
const tenant = String(tenantId || '').trim();
|
||||||
|
const account = String(phone || '').trim();
|
||||||
|
return tenant && account ? `${preferenceKeyPrefix}:${tenant}:${account}` : '';
|
||||||
|
};
|
||||||
|
|
||||||
|
export const getTtsVoicePreference = (tenantId: unknown, phone: unknown): TtsVoicePreference => {
|
||||||
|
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
|
||||||
|
return key ? normalizeTtsVoicePreference(uni.getStorageSync(key)) : '';
|
||||||
|
};
|
||||||
|
|
||||||
|
export const setTtsVoicePreference = (tenantId: unknown, phone: unknown, value: unknown): TtsVoicePreference => {
|
||||||
|
const key = ttsVoicePreferenceStorageKey(tenantId, phone);
|
||||||
|
const normalized = normalizeTtsVoicePreference(value);
|
||||||
|
if (key) uni.setStorageSync(key, normalized);
|
||||||
|
return normalized;
|
||||||
|
};
|
||||||
|
|
||||||
|
export const applyTtsVoicePreference = (
|
||||||
|
profile: SpeechVoiceProfile | undefined,
|
||||||
|
preference: unknown
|
||||||
|
): SpeechVoiceProfile | undefined => {
|
||||||
|
const voice = normalizeTtsVoicePreference(preference);
|
||||||
|
if (!voice) return profile;
|
||||||
|
return { ...(profile || { role: 'neutral' }), voice };
|
||||||
|
};
|
||||||
@@ -39,10 +39,10 @@ test('全局选择只覆盖 voice,保留场景角色、情绪、语速和方
|
|||||||
const profile = { role: 'customer', emotion: 'intense', speed: 1.12, dialect: 'sichuanese' };
|
const profile = { role: 'customer', emotion: 'intense', speed: 1.12, dialect: 'sichuanese' };
|
||||||
|
|
||||||
assert.deepEqual(
|
assert.deepEqual(
|
||||||
runtime.applyTtsVoicePreference(profile, 'longanhuan_v3.6'),
|
JSON.parse(JSON.stringify(runtime.applyTtsVoicePreference(profile, 'longanhuan_v3.6'))),
|
||||||
{ ...profile, voice: 'longanhuan_v3.6' }
|
{ ...profile, voice: 'longanhuan_v3.6' }
|
||||||
);
|
);
|
||||||
assert.deepEqual(runtime.applyTtsVoicePreference(profile, ''), profile);
|
assert.deepEqual(JSON.parse(JSON.stringify(runtime.applyTtsVoicePreference(profile, ''))), profile);
|
||||||
});
|
});
|
||||||
|
|
||||||
test('语音播放链路在合成前统一注入全局音色,且缓存键包含处理后的 profile', async () => {
|
test('语音播放链路在合成前统一注入全局音色,且缓存键包含处理后的 profile', async () => {
|
||||||
|
|||||||
Reference in New Issue
Block a user