feat(speech): add role-aware emotional TTS profiles

This commit is contained in:
2026-07-18 18:27:14 +08:00
parent c0538ad8c3
commit 68d5600534
14 changed files with 228 additions and 45 deletions
+1 -1
View File
@@ -57,7 +57,7 @@
- 正式试点 CSV 走 `GET /api/train/practice/export?startDate=YYYY-MM-DD&endDate=YYYY-MM-DD`,起止日必填;训练、校准和 SOP 评审必须按同一窗口、唯一在职组织身份统计,完训定义为每人至少 10 次。CSV 应保留原始校准命中/SOP 可用计数,严格预检用 `AIHR_PILOT_START_DATE`、`AIHR_PILOT_END_DATE` 与 `AIHR_PILOT_STRICT=true ./scripts/demo-check.sh`;不允许用历史 seed、开发身份或四舍五入比率伪造正式试点通过。
- 移动端候选人闭环:候选人端“开始面试/面试练习”复用 `POST /api/recruit/interview/start`、`/answer`、`/finish`;“补充资料”走认证接口 `POST /api/aihr/mobile/candidate/materials`(multipart `file` + `candidateId/candidateName/materialType`)和 `GET /api/aihr/mobile/candidate/materials`,文件先写 `sys_oss`/MinIO,再写 `aihr_candidate_material` 状态 `待审核`。HR 审核复用管理端 `/recruit/interview` 页,接口为 `GET /api/aihr/hr/candidate/materials` 和 `POST /api/aihr/hr/candidate/materials/{id}/review`,状态只用 `待审核/已通过/已驳回`。上传接口不要放进 class-level `@SaIgnore` 的 `AihrMobileController`。
- 当前业务页已跑通可演示闭环;后端 API 对接先看 `docs/API_INTEGRATION.md`。三角色对练 `/turn`/`/finish` 已接真 LLM(`AihrPracticeLlmService`,评分 temperature=0 结构化输出),seed 剧本是剧情锚点与兜底,改对练逻辑时必须保留"未配置/失败回退 seed"的降级链,不要让演示依赖外部 API。
- 对练语音走 `POST /api/ai/asr`(≤5MB multipart)和 `POST /api/ai/tts`(文本≤300字,返回 base64 dataURL);模型走 `aihr_model_config` 的 `asr`/`tts` 类目(`category` 全集:chat/vector/rerank/asr/tts/vision)。移动端 `getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用隐藏 `audio/*` file input 选择/录制音频后继续走同一 ASR 接口,不要退回只有文字输入。multipart 头部的 filename/contentType 已做 CRLF 清洗,新增外发 HTTP 时同样注意。
- 对练语音走 `POST /api/ai/asr`(≤5MB multipart)和 `POST /api/ai/tts`(文本≤300字,返回 base64 dataURL);TTS 保留旧 `voice` 字符串,并支持 `voiceProfile={role,voice?,speed?,emotion?}`,当前老师傅/业主/面试官分音色,业主按对练情绪分调整语气与语速。模型走 `aihr_model_config` 的 `asr`/`tts` 类目(`category` 全集:chat/vector/rerank/asr/tts/vision)。移动端 `getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用隐藏 `audio/*` file input 选择/录制音频后继续走同一 ASR 接口,不要退回只有文字输入。multipart 头部的 filename/contentType 已做 CRLF 清洗,新增外发 HTTP 时同样注意。
- 后端业务代码不要塞进上游 `ruoyi-demo`;自有 API 放在 `backend/ruoyi-modules/ruoyi-aihr`,再接入 `ruoyi-admin`。
- AI 面试页、三角色对练页、案例沉淀页、SOP 知识库页和移动端三端首页已是 API 优先 + 本地 fallback;SOP 知识库优先查 `aihr_knowledge_fragment` 的 MySQL Fulltext,vector 模型和 Qdrant 可用时混合召回,`category=rerank` 模型启用时融合后语义重排(失败保持 RRF 顺序);支持 `.txt/.md/.markdown/.pdf/.doc/.docx/.xls/.xlsx/.ppt/.pptx`、图片 `.jpg/.jpeg/.png/.gif/.webp/.bmp` 及视频 `.mp4/.mov/.avi/.mkv/.webm/.m4v` 上传解析入库。
- 浏览器批量上传走异步队列 `POST /api/knowledge/doc/upload-async`(暂存目录 `aihr.upload.staging` 默认 `./.data/staging`,队列表 `aihr_knowledge_upload_item`,单文件重试);同步接口 `POST /api/knowledge/doc/upload` 只留给 SOP 页单文件即时预览,**不要把重加工逻辑加回同步请求线程**。视频(≤500MB/≤60分钟)只走异步队列,依赖服务器安装 ffmpeg/ffprobe。
@@ -125,7 +125,10 @@ public class AihrSpeechController {
if (text.length() > MAX_TTS_CHARS) {
text = text.substring(0, MAX_TTS_CHARS);
}
return speechService.synthesize(text, request == null ? null : request.voice())
return speechService.synthesize(
text,
request == null ? null : request.voice(),
request == null ? null : request.voiceProfile())
.map(this::storeTtsAudio)
.orElseGet(() -> R.fail("语音合成失败,已降级为文本展示"));
}
@@ -11,7 +11,10 @@ public final class AihrSpeechDto {
}
}
public record TtsRequest(String text, String voice) {
public record TtsRequest(String text, String voice, VoiceProfile voiceProfile) {
}
public record VoiceProfile(String role, String voice, Double speed, String emotion) {
}
public record TtsResponse(String audioUrl, String source, Long ossId, String inlineAudioUrl) {
@@ -4,6 +4,7 @@ import com.fasterxml.jackson.databind.ObjectMapper;
import com.fasterxml.jackson.databind.node.ObjectNode;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile;
import org.dromara.aihr.service.AihrModelSeedService.SpeechModel;
import org.springframework.stereotype.Service;
@@ -15,8 +16,11 @@ import java.net.http.HttpRequest;
import java.net.http.HttpResponse;
import java.nio.charset.StandardCharsets;
import java.time.Duration;
import java.util.Locale;
import java.util.Map;
import java.util.Optional;
import java.util.UUID;
import java.util.regex.Pattern;
/**
* ASR/TTS 网关:调用 OpenAI 兼容 audio 接口(硅基流动 SenseVoice/CosyVoice2 等)。
@@ -30,6 +34,19 @@ public class AihrSpeechService {
private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15);
private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5);
private static final String DEFAULT_TTS_VOICE = "anna";
private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}");
private static final Map<String, String> ROLE_VOICES = Map.of(
"mentor", "benjamin",
"customer", "bella",
"interviewer", "alex"
);
private static final Map<String, String> EMOTION_PROMPTS = Map.of(
"warm", "请用沉稳、耐心、有经验的老师傅语气说",
"professional", "请用沉稳、专业、清晰的语气说",
"serious", "请用严肃、克制、带明显质疑的语气说",
"intense", "请用情绪强烈、急切、有压力且明显不满的语气说",
"calm", "请用平静、自然的语气说"
);
private final AihrModelSeedService modelService;
private final ObjectMapper objectMapper;
@@ -75,17 +92,17 @@ public class AihrSpeechService {
* 语音合成:POST {base}/audio/speech,返回 mp3 字节。
*/
public Optional<byte[]> synthesize(String text, String voice) {
return synthesize(text, voice, null);
}
public Optional<byte[]> synthesize(String text, String voice, VoiceProfile voiceProfile) {
Optional<SpeechModel> model = modelService.speechModel("tts");
if (model.isEmpty() || text == null || text.isBlank()) {
return Optional.empty();
}
SpeechModel runtime = model.get();
try {
ObjectNode body = objectMapper.createObjectNode();
body.put("model", runtime.modelName());
body.put("input", AihrSensitiveText.forModel(text.trim()));
body.put("voice", resolveVoice(runtime.modelName(), voice));
body.put("response_format", "mp3");
ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile);
HttpRequest httpRequest = authorized(runtime, "/audio/speech")
.header("Content-Type", "application/json")
.POST(HttpRequest.BodyPublishers.ofString(objectMapper.writeValueAsString(body)))
@@ -102,6 +119,35 @@ public class AihrSpeechService {
}
}
static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text,
String legacyVoice, VoiceProfile voiceProfile) {
String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role());
boolean expressiveCosyVoice = isExpressiveCosyVoice(runtime);
String requestedVoice = safeVoice(voiceProfile == null ? null : voiceProfile.voice());
if (requestedVoice == null) {
requestedVoice = safeVoice(legacyVoice);
}
if (requestedVoice == null && expressiveCosyVoice) {
requestedVoice = ROLE_VOICES.get(role);
}
String emotion = normalizedEmotion(voiceProfile == null ? null : voiceProfile.emotion(), role);
String sanitizedText = AihrSensitiveText.forModel(text.trim());
String input = voiceProfile != null && expressiveCosyVoice && EMOTION_PROMPTS.containsKey(emotion)
? EMOTION_PROMPTS.get(emotion) + "。<|endofprompt|>" + sanitizedText
: sanitizedText;
ObjectNode body = objectMapper.createObjectNode();
body.put("model", runtime.modelName());
body.put("input", input);
body.put("voice", resolveVoice(runtime.modelName(), requestedVoice));
if (voiceProfile != null) {
body.put("speed", resolveSpeed(voiceProfile.speed(), emotion));
}
body.put("response_format", "mp3");
return body;
}
/**
* 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。
*/
@@ -110,6 +156,49 @@ public class AihrSpeechService {
return value.contains(":") ? value : modelName + ":" + value;
}
private static String normalizedRole(String role) {
String value = role == null ? "" : role.trim().toLowerCase(Locale.ROOT);
return ROLE_VOICES.containsKey(value) ? value : "neutral";
}
private static String normalizedEmotion(String emotion, String role) {
String value = emotion == null ? "" : emotion.trim().toLowerCase(Locale.ROOT);
if (EMOTION_PROMPTS.containsKey(value)) {
return value;
}
return switch (role) {
case "mentor" -> "warm";
case "customer" -> "serious";
case "interviewer" -> "professional";
default -> "calm";
};
}
private static double resolveSpeed(Double speed, String emotion) {
if (speed != null && Double.isFinite(speed)) {
return Math.max(0.7, Math.min(1.3, speed));
}
return switch (emotion) {
case "intense" -> 1.12;
case "serious" -> 1.04;
case "warm" -> 0.92;
case "professional" -> 0.96;
default -> 0.98;
};
}
private static String safeVoice(String voice) {
String value = voice == null ? "" : voice.trim();
return SAFE_VOICE.matcher(value).matches() ? value : null;
}
private static boolean isExpressiveCosyVoice(SpeechModel runtime) {
return runtime != null
&& "siliconflow".equalsIgnoreCase(runtime.providerCode())
&& runtime.modelName() != null
&& runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice");
}
private HttpRequest.Builder authorized(SpeechModel runtime, String path) {
HttpRequest.Builder builder = HttpRequest.newBuilder()
.uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path))
@@ -0,0 +1,45 @@
package org.dromara.aihr.service;
import com.fasterxml.jackson.databind.ObjectMapper;
import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile;
import org.dromara.aihr.service.AihrModelSeedService.SpeechModel;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
@Tag("dev")
class AihrSpeechServiceTest {
@Test
void buildsDistinctMentorAndEmotionalCustomerVoicesWithoutBreakingLegacyRequests() {
ObjectMapper mapper = new ObjectMapper();
SpeechModel runtime = new SpeechModel(
"siliconflow",
"FunAudioLLM/CosyVoice2-0.5B",
"https://api.siliconflow.cn/v1",
"test-key"
);
var mentor = AihrSpeechService.buildRequestBody(
mapper, runtime, "先承接情绪,再说明处理节点。", null,
new VoiceProfile("mentor", null, null, null));
var customer = AihrSpeechService.buildRequestBody(
mapper, runtime, "你们到底什么时候处理?", null,
new VoiceProfile("customer", null, 9.0, "intense"));
var legacy = AihrSpeechService.buildRequestBody(
mapper, runtime, "普通播报", "alex", null);
assertEquals("FunAudioLLM/CosyVoice2-0.5B:benjamin", mentor.path("voice").asText());
assertEquals(0.92, mentor.path("speed").asDouble());
assertTrue(mentor.path("input").asText().contains("老师傅语气"));
assertEquals("FunAudioLLM/CosyVoice2-0.5B:bella", customer.path("voice").asText());
assertEquals(1.3, customer.path("speed").asDouble());
assertTrue(customer.path("input").asText().contains("情绪强烈"));
assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText());
assertFalse(legacy.has("speed"));
assertFalse(legacy.path("input").asText().contains("<|endofprompt|>"));
}
}
@@ -11,7 +11,7 @@
|---|---|---|
| 员工端对练 /start /turn /finish 闭环 | ✅ | 移动端带 Authorization+clientid,mode=mobile 落 `aihr_practice_session` |
| AI 扮演业主(情绪/身份/诉求人设) | ✅ | 真 LLM 生成客户回复(seed 剧本作剧情锚点);**情绪值仍走 seed 锚点,未纳入 LLM 输出** |
| 语音输入 ASR → LLM → TTS 播报 | ✅(部分) | **原文"多语气播报"系夸大**:TTS 单一默认音色,多语气/方言/语速=BACKLOG B2 待办 |
| 语音输入 ASR → LLM → TTS 播报 | ✅(部分) | 已按老师傅/业主/面试官区分音色,并按业主情绪分切换平静/严肃/强烈语气与语速;方言、按具体人物选音色和语音克隆仍属 BACKLOG B2 |
| ASR 转写可编辑后提交 | ✅ | **原文漏计且缺口表误列为缺失**:转写先填入输入框,员工可修改再提交 |
| 每轮 AI 教练提示(coachHint) | ✅ | **原文漏计**:每轮 turn 返回教练提示(卡壳/流程偏差提醒) |
| 训练完成多维评分 | ✅ | 4 维:合规/沟通/情绪/营销 + 导师改写 + 点评;LLM temperature=0 结构化输出,评分仅 /finish 一次 |
+1 -1
View File
@@ -10,7 +10,7 @@
| 候选人入职主体关联 `/recruit/interview` | `GET/POST /api/recruit/interview/candidate-links` | HR/管理员在当前租户范围内把本地候选人 ID 关联到已同步的在职 `ext_party_id`;只保存外部主体 ID,不复制姓名、部门等组织字段,重复关联同一主体幂等,候选人更换主体或同一主体已关联其他候选人会拒绝 |
| 三角色对练 `/train/practice` | `POST /api/train/practice/start`、`/turn`、`/finish` | 已接入编排 API;数据库启用 chat 模型后,`/turn` 客户回复按人设走真 LLM 生成(seed 剧本作剧情锚点),`/finish` 走单次 temperature=0 结构化评分(4 维分+导师改写+点评);移动端 `turn/finish` 校验当前手机号与启动会话归属,未知或他人会话拒绝;模型未配置或调用失败自动回退 seed,契约不变 |
| 正式试点数据导出 | `GET /api/train/practice/export?startDate=YYYY-MM-DD&endDate=YYYY-MM-DD` | 起止日期必填且包含结束日;只统计窗口内能通过唯一手机号或外部 ID 映射到在职组织快照的正式会话,排除重复手机号和身份碰撞。完训定义为每人至少 10 次,校准必须关联同一窗口内正式会话;CSV 同时给出校准命中数、SOP 可用数、满意度响应数/平均分,以及明细级 AI 分、人工校准分、校准人、校准时间、最终采用分、满意度分和意见,避免用四舍五入后的比率反推门禁状态;汇总和明细均携带正式人员及项目口径,不混入历史 seed/开发身份 |
| 对练语音 | `POST /api/ai/asr`(multipart 字段 `file`,≤5MB)、`POST /api/ai/tts`(JSON `{text≤300字, voice}`,成功返回 `ossId`,客户端播放地址为受控的内联 `data:audio/*`,不返回原始 OSS URL)、`GET /api/aihr/mobile/oss/{ossId}` | 走 OpenAI-compatible audio 接口(如硅基流动 SenseVoice/CosyVoice2);模型管理需启用 `category=asr/tts` 配置;移动端优先用浏览器录音,`getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用 `audio/*` file input 选择/录制音频后继续调同一 ASR 接口;ASR/TTS 未配置或失败返回 fail,前端降级文本;训练/每日题录音只通过 `audioOssId` 走受保护下载,历史客户端传入的 HTTP `audioUrl` 不再回显;TTS 成功音频仍写入 `sys_oss` 留痕,同时用内联 data URL 保持旧客户端可播放;音频下载按员工本人或主管项目范围授权,系统管理端保持后台访问 |
| 对练语音 | `POST /api/ai/asr`(multipart 字段 `file`,≤5MB)、`POST /api/ai/tts`(JSON `{text≤300字, voice?, voiceProfile?:{role,voice?,speed?,emotion?}}`,旧 `voice` 兼容;成功返回 `ossId`,客户端播放地址为受控的内联 `data:audio/*`,不返回原始 OSS URL)、`GET /api/aihr/mobile/oss/{ossId}` | 走 OpenAI-compatible audio 接口;生产 SiliconFlow CosyVoice2 按角色映射老师傅/业主/面试官音色,业主对练再按已有情绪分切换平静/严肃/强烈语气和语速,设备语音降级同步调整语速与音高;模型管理需启用 `category=asr/tts` 配置;移动端优先用浏览器录音,`getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用 `audio/*` file input 选择/录制音频后继续调同一 ASR 接口;ASR/TTS 未配置或失败返回 fail,前端降级设备语音或文本;训练/每日题录音只通过 `audioOssId` 走受保护下载,历史客户端传入的 HTTP `audioUrl` 不再回显;TTS 成功音频仍写入 `sys_oss` 留痕,同时用内联 data URL 保持旧客户端可播放;音频下载按员工本人或主管项目范围授权,系统管理端保持后台访问 |
| 案例沉淀 `/knowledge/cases` | `GET /api/knowledge/case/capabilities`、`POST /api/knowledge/case/upload`、`/organize`、`/curate`、`GET /records`、`GET /records/{caseId}`、`POST /records/{caseId}/review`、`GET /records/{caseId}/media` | `/capabilities` 返回服务端判定的案例提交/查看能力,移动端不再向普通员工展示无权提交的素材表单;`/upload` 改为 multipart 真实语音上传并走 ASR,服务端只接受 MP3/WAV/M4A/WebM/OGG/AAC/FLAC,成功后原始音频写入 `sys_oss`,案例记录只保存 `mediaOssId` 供受保护媒体接口读取,不向客户端回传原始 `mediaUrl`;`/organize` 用真实转写调 chat 模型整理案例,未配置模型时按真实 transcript 本地结构化,并从背景外的真实摘要项提取学习点;APP 用户的项目范围从 `aihr_org_snapshot` 登录身份解析,上传、整理、入库、列表和详情均按项目范围校验,未完成正式组织映射时安全拒绝,不接受前端伪造项目范围;员工列表/详情只返回 `已入库` 案例,管理端系统用户保留全局运营视图;移动端和管理端案例详情通过受保护媒体接口回放原始音频,主管/项目负责人可提交脱敏点评;预渲染视频样片仍待正式媒体资产接入 |
| 案例媒体安全 | `GET /api/knowledge/case/records/{caseId}/media` | 案例详情只返回 `mediaOssId`,不返回原始 `sys_oss.url`;媒体下载会重复执行登录、后台角色或 APP 项目范围校验,再由服务端流式读取 OSS。管理端与 `mobile-uni` 通过鉴权 blob/temp 文件播放,关闭详情页时释放本地对象 URL |
| SOP知识库 `/knowledge/sop` | `POST /api/knowledge/search`、`POST /api/knowledge/answer-feedback`、`GET /api/knowledge/position-sop`、`GET /api/aihr/mobile/onboard/tasks`、`POST /api/aihr/mobile/onboard/tasks/{id}/complete`、`GET /api/aihr/mobile/qualification`、`POST /api/knowledge/doc/upload` | 已接入 MySQL Fulltext + Qdrant 混合召回、岗位学习适配摘要、OSS-first 文档上传、txt/md/PDF/Word/Excel/PPT 解析和 embedding 写入,失败回退 seed;搜索返回 `reviewId` 与 `promptVersion`,员工反馈回传并保存该评审批次,SOP 人工评审记录同时保留答案生成提示词版本,管理端可继续复核;员工学习页按当前 APP 身份读取正式岗前/入职任务,员工只能将本人处于“待完成/进行中”的任务确认完成,服务端按组织快照/手机号归属更新 `status/completed_time`,不接受前端身份参数;资格证据无正式数据时明确返回 `NOT_CONFIGURED`,不以 AI 分数代替上岗资格;`position-sop` 仍保留一期生活顾问学习导航语义 |
+2
View File
@@ -22,6 +22,8 @@
需求:AI 员工分岗位等级面试官形象;模拟多种语气声——老太说话慢、带方言、吵架、求救听不清。
> 2026-07-18 已完成首批:共享 TTS 契约支持 `voiceProfile`,老师傅/业主/面试官分音色,业主随对练情绪分调整语气与语速;剩余人物级音色、方言、克隆和信道劣化继续保留在 B2。
- 复用三角色对练人设机制,**不新建架构**:面试官 = 另一组 `practice_scenario.persona_json`(TechSpec 6.2),按岗位等级(管家/主管/项目经理)建多条,新增字段:
- `avatar_url`:岗位形象头像,由 B3 文生图生成
- `voice_profile`:`{ voice_id, speed, emotion, pitch, dialect }`
@@ -156,6 +156,7 @@ import { resetPageScroll } from '@/services/navigation';
import {
chooseSpeechAudio,
createSpeechPlaybackController,
INTERVIEWER_VOICE_PROFILE,
normalizeOssId,
transcribeSpeechBlob,
transcribeSpeechFile
@@ -186,7 +187,8 @@ const speechPlayback = createSpeechPlaybackController(
(text) => {
message.value = text;
uni.showToast({ title: text, icon: 'none' });
}
},
INTERVIEWER_VOICE_PROFILE
);
const interviewBusy = computed(() => interviewStatus.value === 'starting' || interviewStatus.value === 'submitting');
+21 -11
View File
@@ -348,6 +348,7 @@ import type {
PracticePrepCard,
PracticeAssignment,
PracticeRole,
SpeechVoiceProfile,
SpeechPlaybackStatus,
PracticeTurn,
PracticeRecord
@@ -357,7 +358,7 @@ import { getSelectedPosition, isLoggedIn, rememberLoginRedirect } from '@/servic
import { searchKnowledge } from '@/services/knowledge';
import { resetPageScroll } from '@/services/navigation';
import { ensureEmployeePosition } from '@/services/position';
import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
import { chooseSpeechAudio, createSpeechPlaybackController, customerVoiceProfile, measureAudioDuration, MENTOR_VOICE_PROFILE, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
import { createPageRequestScope, currentAccountKey, type RequestScopeTicket } from '@/services/request-scope';
import { stripInternalCodes } from '@/services/text';
@@ -433,7 +434,8 @@ const speechPlayback = createSpeechPlaybackController(
(text) => {
message.value = text;
uni.showToast({ title: text, icon: 'none' });
}
},
MENTOR_VOICE_PROFILE
);
const busy = computed(() => status.value === 'starting' || status.value === 'submitting');
const practiceView = computed<'prep' | 'active' | 'result'>(() => {
@@ -517,12 +519,19 @@ interface TurnVoice {
duration: number | null;
unread: boolean;
open: boolean;
profile: SpeechVoiceProfile;
}
const turnVoices = ref<Record<number, TurnVoice>>({});
const turnVoice = (index: number): TurnVoice =>
turnVoices.value[index] || { status: 'unavailable', duration: null, unread: false, open: false };
turnVoices.value[index] || {
status: 'unavailable',
duration: null,
unread: false,
open: false,
profile: customerVoiceProfile()
};
const patchTurnVoice = (index: number, patch: Partial<TurnVoice>) => {
turnVoices.value = { ...turnVoices.value, [index]: { ...turnVoice(index), ...patch } };
@@ -530,11 +539,12 @@ const patchTurnVoice = (index: number, patch: Partial<TurnVoice>) => {
/** Customer lines arrive as WeChat-style voice messages: pre-synthesize so the
* bubble shows a duration and an unread dot; fall back to text when TTS is off. */
const prepareTurnVoice = async (index: number, text: string) => {
const prepareTurnVoice = async (index: number, text: string, emotion = 0) => {
const spoken = (text || '').trim();
if (!spoken) return;
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false });
const source = await speechPlayback.preload(spoken);
const profile = customerVoiceProfile(emotion);
patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile });
const source = await speechPlayback.preload(spoken, profile);
if (!turnVoices.value[index]) return;
if (!source) {
patchTurnVoice(index, { status: 'unavailable', unread: false });
@@ -560,7 +570,7 @@ const playTurnVoice = (index: number, text: string) => {
return;
}
patchTurnVoice(index, { unread: false });
void speechPlayback.toggle(`turn-${index}`, (text || '').trim());
void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile);
};
const toggleTurnTranscript = (index: number) => {
@@ -748,11 +758,11 @@ const requireLogin = () => {
return false;
};
const appendTurn = (role: PracticeRole, text?: string) => {
const appendTurn = (role: PracticeRole, text?: string, emotion = 0) => {
if (!text) return;
turns.value.push({ role, text });
if (role === 'customer') {
void prepareTurnVoice(turns.value.length - 1, text);
void prepareTurnVoice(turns.value.length - 1, text, emotion);
}
scrollThreadToBottom();
};
@@ -1146,7 +1156,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => {
emotionScore.value = data.emotion ?? 0;
trustScore.value = data.trust ?? 0;
sessionId.value = data.sessionId;
appendTurn('customer', data.customerText);
appendTurn('customer', data.customerText, data.emotion);
status.value = 'active';
} catch (error) {
if (!requests.isCurrent(request)) return;
@@ -1267,7 +1277,7 @@ const submit = async () => {
await finish(true);
return;
}
appendTurn('customer', data.customerText);
appendTurn('customer', data.customerText, data.emotion);
roundIndex.value = data.roundIndex;
status.value = 'active';
} catch (error) {
+3 -2
View File
@@ -261,7 +261,7 @@ import {
} from '@/services/auth';
import { resetPageScroll } from '@/services/navigation';
import { ensureEmployeePosition } from '@/services/position';
import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, MENTOR_VOICE_PROFILE, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
import { downloadSummaryCardImage } from '@/services/summary-card-image';
import { computed } from 'vue';
@@ -337,7 +337,8 @@ const defaultToolCode = computed<'MY_PRACTICE_SUMMARY' | 'TEAM_PRACTICE_SUMMARY'
const speechPlayback = createSpeechPlaybackController(
(state) => { speechState.value = state; },
(text) => { uni.showToast({ title: text, icon: 'none' }); }
(text) => { uni.showToast({ title: text, icon: 'none' }); },
MENTOR_VOICE_PROFILE
);
const scrollToBottom = () => {
+3 -2
View File
@@ -117,7 +117,7 @@ import { onShow, onUnload } from '@dcloudio/uni-app';
import type { SpeechPlaybackStatus, WebAiCapabilities, WebAiResponse } from '@/types/api';
import ChatComposer from '@/components/chat/ChatComposer.vue';
import { isLoggedIn, rememberLoginRedirect } from '@/services/auth';
import { chooseSpeechAudio, createSpeechPlaybackController, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
import { chooseSpeechAudio, createSpeechPlaybackController, MENTOR_VOICE_PROFILE, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech';
import { getWebAiCapabilities, queryWebAi } from '@/services/web-ai';
const path = '/pages/user/web-ai/index';
@@ -164,7 +164,8 @@ let abortVoiceTranscription: (() => void) | null = null;
const speechPlayback = createSpeechPlaybackController(
(state) => { speechState.value = state; },
(text) => { uni.showToast({ title: text, icon: 'none' }); }
(text) => { uni.showToast({ title: text, icon: 'none' }); },
MENTOR_VOICE_PROFILE
);
const scrollToBottom = () => {
+39 -19
View File
@@ -1,8 +1,16 @@
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, TtsResponse } from '@/types/api';
import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api';
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.92 };
export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96 };
export const customerVoiceProfile = (emotion = 0): SpeechVoiceProfile => {
if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12 };
if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04 };
return { role: 'customer', emotion: 'calm', speed: 0.98 };
};
// OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。
export const normalizeOssId = (value: unknown): string | undefined => {
const text = String(value ?? '').trim();
@@ -83,11 +91,11 @@ export const transcribeSpeechBlob = (
xhr.send(form);
});
export const synthesizeSpeech = (text: string) =>
export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) =>
apiRequest<TtsResponse>({
url: '/api/ai/tts',
method: 'POST',
data: { text: text.trim().slice(0, 300) },
data: { text: text.trim().slice(0, 300), voiceProfile },
timeout: 60000
});
@@ -98,7 +106,8 @@ export interface SpeechPlaybackSnapshot {
export const createSpeechPlaybackController = (
onState: (snapshot: SpeechPlaybackSnapshot) => void,
onError: (message: string) => void
onError: (message: string) => void,
defaultVoiceProfile?: SpeechVoiceProfile
) => {
let audio: ReturnType<typeof uni.createInnerAudioContext> | null = null;
let browserUtterance: SpeechSynthesisUtterance | null = null;
@@ -106,6 +115,7 @@ export const createSpeechPlaybackController = (
let generation = 0;
const sourceCache = new Map<string, string>();
const maxCachedSources = 4;
const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]);
const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => {
activeKey = key;
@@ -126,7 +136,12 @@ export const createSpeechPlaybackController = (
publish();
};
const playWithBrowserSpeech = (key: string, value: string, currentGeneration: number) => {
const playWithBrowserSpeech = (
key: string,
value: string,
currentGeneration: number,
profile?: SpeechVoiceProfile
) => {
if (
typeof window === 'undefined'
|| !window.speechSynthesis
@@ -135,7 +150,8 @@ export const createSpeechPlaybackController = (
const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600));
utterance.lang = 'zh-CN';
utterance.rate = 0.95;
utterance.rate = profile?.speed || 0.95;
utterance.pitch = profile?.role === 'mentor' ? 0.85 : profile?.role === 'customer' ? 1.05 : 1;
utterance.onend = () => {
if (currentGeneration === generation && activeKey === key) stop();
};
@@ -173,7 +189,7 @@ export const createSpeechPlaybackController = (
context.play();
};
const toggle = async (key: string, text: string) => {
const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => {
if (activeKey === key) {
stop();
return;
@@ -181,30 +197,32 @@ export const createSpeechPlaybackController = (
stop();
const value = text.trim();
if (!value) return;
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const currentGeneration = generation;
publish(key, 'loading');
try {
const cachedSource = sourceCache.get(value);
const cachedSource = sourceCache.get(sourceKey);
if (cachedSource) {
sourceCache.delete(value);
sourceCache.set(value, cachedSource);
playAudioSource(key, cachedSource, value, currentGeneration);
sourceCache.delete(sourceKey);
sourceCache.set(sourceKey, cachedSource);
playAudioSource(key, cachedSource, sourceKey, currentGeneration);
return;
}
const result = await synthesizeSpeech(value);
const result = await synthesizeSpeech(value, profile);
if (currentGeneration !== generation || activeKey !== key) return;
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) throw new Error('语音合成未返回音频');
sourceCache.set(value, source);
sourceCache.set(sourceKey, source);
while (sourceCache.size > maxCachedSources) {
const oldestKey = sourceCache.keys().next().value as string | undefined;
if (!oldestKey) break;
sourceCache.delete(oldestKey);
}
playAudioSource(key, source, value, currentGeneration);
playAudioSource(key, source, sourceKey, currentGeneration);
} catch (error) {
if (currentGeneration !== generation) return;
if (playWithBrowserSpeech(key, value, currentGeneration)) return;
if (playWithBrowserSpeech(key, value, currentGeneration, profile)) return;
stop();
onError(error instanceof Error ? error.message : '语音生成失败');
}
@@ -215,16 +233,18 @@ export const createSpeechPlaybackController = (
* WeChat-style voice bubble can show duration before the first tap.
* Returns the audio source URL, or null when remote TTS is unavailable.
*/
const preload = async (text: string): Promise<string | null> => {
const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise<string | null> => {
const value = text.trim();
if (!value) return null;
const cached = sourceCache.get(value);
const profile = voiceProfile || defaultVoiceProfile;
const sourceKey = cacheKey(value, profile);
const cached = sourceCache.get(sourceKey);
if (cached) return cached;
try {
const result = await synthesizeSpeech(value);
const result = await synthesizeSpeech(value, profile);
const source = result.inlineAudioUrl || result.audioUrl;
if (!source) return null;
sourceCache.set(value, source);
sourceCache.set(sourceKey, source);
while (sourceCache.size > maxCachedSources) {
const oldestKey = sourceCache.keys().next().value as string | undefined;
if (!oldestKey) break;
+7
View File
@@ -27,6 +27,13 @@ export interface TtsResponse {
export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing';
export interface SpeechVoiceProfile {
role: 'mentor' | 'customer' | 'interviewer' | 'neutral';
voice?: string;
speed?: number;
emotion?: 'warm' | 'professional' | 'serious' | 'intense' | 'calm';
}
export interface ToolItem {
title: string;
desc: string;