diff --git a/AGENTS.md b/AGENTS.md index 4699203e..ffc471ab 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -57,7 +57,7 @@ - 正式试点 CSV 走 `GET /api/train/practice/export?startDate=YYYY-MM-DD&endDate=YYYY-MM-DD`,起止日必填;训练、校准和 SOP 评审必须按同一窗口、唯一在职组织身份统计,完训定义为每人至少 10 次。CSV 应保留原始校准命中/SOP 可用计数,严格预检用 `AIHR_PILOT_START_DATE`、`AIHR_PILOT_END_DATE` 与 `AIHR_PILOT_STRICT=true ./scripts/demo-check.sh`;不允许用历史 seed、开发身份或四舍五入比率伪造正式试点通过。 - 移动端候选人闭环:候选人端“开始面试/面试练习”复用 `POST /api/recruit/interview/start`、`/answer`、`/finish`;“补充资料”走认证接口 `POST /api/aihr/mobile/candidate/materials`(multipart `file` + `candidateId/candidateName/materialType`)和 `GET /api/aihr/mobile/candidate/materials`,文件先写 `sys_oss`/MinIO,再写 `aihr_candidate_material` 状态 `待审核`。HR 审核复用管理端 `/recruit/interview` 页,接口为 `GET /api/aihr/hr/candidate/materials` 和 `POST /api/aihr/hr/candidate/materials/{id}/review`,状态只用 `待审核/已通过/已驳回`。上传接口不要放进 class-level `@SaIgnore` 的 `AihrMobileController`。 - 当前业务页已跑通可演示闭环;后端 API 对接先看 `docs/API_INTEGRATION.md`。三角色对练 `/turn`/`/finish` 已接真 LLM(`AihrPracticeLlmService`,评分 temperature=0 结构化输出),seed 剧本是剧情锚点与兜底,改对练逻辑时必须保留"未配置/失败回退 seed"的降级链,不要让演示依赖外部 API。 -- 对练语音走 `POST /api/ai/asr`(≤5MB multipart)和 `POST /api/ai/tts`(文本≤300字,返回 base64 dataURL);模型走 `aihr_model_config` 的 `asr`/`tts` 类目(`category` 全集:chat/vector/rerank/asr/tts/vision)。移动端 `getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用隐藏 `audio/*` file input 选择/录制音频后继续走同一 ASR 接口,不要退回只有文字输入。multipart 头部的 filename/contentType 已做 CRLF 清洗,新增外发 HTTP 时同样注意。 +- 对练语音走 `POST /api/ai/asr`(≤5MB multipart)和 `POST /api/ai/tts`(文本≤300字,返回 base64 dataURL);TTS 保留旧 `voice` 字符串,并支持 `voiceProfile={role,voice?,speed?,emotion?}`,当前老师傅/业主/面试官分音色,业主按对练情绪分调整语气与语速。模型走 `aihr_model_config` 的 `asr`/`tts` 类目(`category` 全集:chat/vector/rerank/asr/tts/vision)。移动端 `getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用隐藏 `audio/*` file input 选择/录制音频后继续走同一 ASR 接口,不要退回只有文字输入。multipart 头部的 filename/contentType 已做 CRLF 清洗,新增外发 HTTP 时同样注意。 - 后端业务代码不要塞进上游 `ruoyi-demo`;自有 API 放在 `backend/ruoyi-modules/ruoyi-aihr`,再接入 `ruoyi-admin`。 - AI 面试页、三角色对练页、案例沉淀页、SOP 知识库页和移动端三端首页已是 API 优先 + 本地 fallback;SOP 知识库优先查 `aihr_knowledge_fragment` 的 MySQL Fulltext,vector 模型和 Qdrant 可用时混合召回,`category=rerank` 模型启用时融合后语义重排(失败保持 RRF 顺序);支持 `.txt/.md/.markdown/.pdf/.doc/.docx/.xls/.xlsx/.ppt/.pptx`、图片 `.jpg/.jpeg/.png/.gif/.webp/.bmp` 及视频 `.mp4/.mov/.avi/.mkv/.webm/.m4v` 上传解析入库。 - 浏览器批量上传走异步队列 `POST /api/knowledge/doc/upload-async`(暂存目录 `aihr.upload.staging` 默认 `./.data/staging`,队列表 `aihr_knowledge_upload_item`,单文件重试);同步接口 `POST /api/knowledge/doc/upload` 只留给 SOP 页单文件即时预览,**不要把重加工逻辑加回同步请求线程**。视频(≤500MB/≤60分钟)只走异步队列,依赖服务器安装 ffmpeg/ffprobe。 diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/controller/AihrSpeechController.java b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/controller/AihrSpeechController.java index 85c389c8..3a97eb89 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/controller/AihrSpeechController.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/controller/AihrSpeechController.java @@ -125,7 +125,10 @@ public class AihrSpeechController { if (text.length() > MAX_TTS_CHARS) { text = text.substring(0, MAX_TTS_CHARS); } - return speechService.synthesize(text, request == null ? null : request.voice()) + return speechService.synthesize( + text, + request == null ? null : request.voice(), + request == null ? null : request.voiceProfile()) .map(this::storeTtsAudio) .orElseGet(() -> R.fail("语音合成失败,已降级为文本展示")); } diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/domain/AihrSpeechDto.java b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/domain/AihrSpeechDto.java index 87654e5f..b7443ec8 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/domain/AihrSpeechDto.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/domain/AihrSpeechDto.java @@ -11,7 +11,10 @@ public final class AihrSpeechDto { } } - public record TtsRequest(String text, String voice) { + public record TtsRequest(String text, String voice, VoiceProfile voiceProfile) { + } + + public record VoiceProfile(String role, String voice, Double speed, String emotion) { } public record TtsResponse(String audioUrl, String source, Long ossId, String inlineAudioUrl) { diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java index 9d458541..7e8fab4f 100644 --- a/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java +++ b/backend/ruoyi-modules/ruoyi-aihr/src/main/java/org/dromara/aihr/service/AihrSpeechService.java @@ -4,6 +4,7 @@ import com.fasterxml.jackson.databind.ObjectMapper; import com.fasterxml.jackson.databind.node.ObjectNode; import lombok.RequiredArgsConstructor; import lombok.extern.slf4j.Slf4j; +import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile; import org.dromara.aihr.service.AihrModelSeedService.SpeechModel; import org.springframework.stereotype.Service; @@ -15,8 +16,11 @@ import java.net.http.HttpRequest; import java.net.http.HttpResponse; import java.nio.charset.StandardCharsets; import java.time.Duration; +import java.util.Locale; +import java.util.Map; import java.util.Optional; import java.util.UUID; +import java.util.regex.Pattern; /** * ASR/TTS 网关:调用 OpenAI 兼容 audio 接口(硅基流动 SenseVoice/CosyVoice2 等)。 @@ -30,6 +34,19 @@ public class AihrSpeechService { private static final Duration CONNECT_TIMEOUT = Duration.ofSeconds(15); private static final Duration REQUEST_TIMEOUT = Duration.ofSeconds(5); private static final String DEFAULT_TTS_VOICE = "anna"; + private static final Pattern SAFE_VOICE = Pattern.compile("[A-Za-z0-9_./:-]{1,200}"); + private static final Map ROLE_VOICES = Map.of( + "mentor", "benjamin", + "customer", "bella", + "interviewer", "alex" + ); + private static final Map EMOTION_PROMPTS = Map.of( + "warm", "请用沉稳、耐心、有经验的老师傅语气说", + "professional", "请用沉稳、专业、清晰的语气说", + "serious", "请用严肃、克制、带明显质疑的语气说", + "intense", "请用情绪强烈、急切、有压力且明显不满的语气说", + "calm", "请用平静、自然的语气说" + ); private final AihrModelSeedService modelService; private final ObjectMapper objectMapper; @@ -75,17 +92,17 @@ public class AihrSpeechService { * 语音合成:POST {base}/audio/speech,返回 mp3 字节。 */ public Optional synthesize(String text, String voice) { + return synthesize(text, voice, null); + } + + public Optional synthesize(String text, String voice, VoiceProfile voiceProfile) { Optional model = modelService.speechModel("tts"); if (model.isEmpty() || text == null || text.isBlank()) { return Optional.empty(); } SpeechModel runtime = model.get(); try { - ObjectNode body = objectMapper.createObjectNode(); - body.put("model", runtime.modelName()); - body.put("input", AihrSensitiveText.forModel(text.trim())); - body.put("voice", resolveVoice(runtime.modelName(), voice)); - body.put("response_format", "mp3"); + ObjectNode body = buildRequestBody(objectMapper, runtime, text, voice, voiceProfile); HttpRequest httpRequest = authorized(runtime, "/audio/speech") .header("Content-Type", "application/json") .POST(HttpRequest.BodyPublishers.ofString(objectMapper.writeValueAsString(body))) @@ -102,6 +119,35 @@ public class AihrSpeechService { } } + static ObjectNode buildRequestBody(ObjectMapper objectMapper, SpeechModel runtime, String text, + String legacyVoice, VoiceProfile voiceProfile) { + String role = normalizedRole(voiceProfile == null ? null : voiceProfile.role()); + boolean expressiveCosyVoice = isExpressiveCosyVoice(runtime); + String requestedVoice = safeVoice(voiceProfile == null ? null : voiceProfile.voice()); + if (requestedVoice == null) { + requestedVoice = safeVoice(legacyVoice); + } + if (requestedVoice == null && expressiveCosyVoice) { + requestedVoice = ROLE_VOICES.get(role); + } + + String emotion = normalizedEmotion(voiceProfile == null ? null : voiceProfile.emotion(), role); + String sanitizedText = AihrSensitiveText.forModel(text.trim()); + String input = voiceProfile != null && expressiveCosyVoice && EMOTION_PROMPTS.containsKey(emotion) + ? EMOTION_PROMPTS.get(emotion) + "。<|endofprompt|>" + sanitizedText + : sanitizedText; + + ObjectNode body = objectMapper.createObjectNode(); + body.put("model", runtime.modelName()); + body.put("input", input); + body.put("voice", resolveVoice(runtime.modelName(), requestedVoice)); + if (voiceProfile != null) { + body.put("speed", resolveSpeed(voiceProfile.speed(), emotion)); + } + body.put("response_format", "mp3"); + return body; + } + /** * 硅基流动 voice 格式为 "{model}:{voice}";调用方只传短名(如 anna/粤语音色名)时自动补模型前缀。 */ @@ -110,6 +156,49 @@ public class AihrSpeechService { return value.contains(":") ? value : modelName + ":" + value; } + private static String normalizedRole(String role) { + String value = role == null ? "" : role.trim().toLowerCase(Locale.ROOT); + return ROLE_VOICES.containsKey(value) ? value : "neutral"; + } + + private static String normalizedEmotion(String emotion, String role) { + String value = emotion == null ? "" : emotion.trim().toLowerCase(Locale.ROOT); + if (EMOTION_PROMPTS.containsKey(value)) { + return value; + } + return switch (role) { + case "mentor" -> "warm"; + case "customer" -> "serious"; + case "interviewer" -> "professional"; + default -> "calm"; + }; + } + + private static double resolveSpeed(Double speed, String emotion) { + if (speed != null && Double.isFinite(speed)) { + return Math.max(0.7, Math.min(1.3, speed)); + } + return switch (emotion) { + case "intense" -> 1.12; + case "serious" -> 1.04; + case "warm" -> 0.92; + case "professional" -> 0.96; + default -> 0.98; + }; + } + + private static String safeVoice(String voice) { + String value = voice == null ? "" : voice.trim(); + return SAFE_VOICE.matcher(value).matches() ? value : null; + } + + private static boolean isExpressiveCosyVoice(SpeechModel runtime) { + return runtime != null + && "siliconflow".equalsIgnoreCase(runtime.providerCode()) + && runtime.modelName() != null + && runtime.modelName().toLowerCase(Locale.ROOT).contains("cosyvoice"); + } + private HttpRequest.Builder authorized(SpeechModel runtime, String path) { HttpRequest.Builder builder = HttpRequest.newBuilder() .uri(URI.create(normalizeBaseUrl(runtime.baseUrl()) + path)) diff --git a/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java b/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java new file mode 100644 index 00000000..ffd84682 --- /dev/null +++ b/backend/ruoyi-modules/ruoyi-aihr/src/test/java/org/dromara/aihr/service/AihrSpeechServiceTest.java @@ -0,0 +1,45 @@ +package org.dromara.aihr.service; + +import com.fasterxml.jackson.databind.ObjectMapper; +import org.dromara.aihr.domain.AihrSpeechDto.VoiceProfile; +import org.dromara.aihr.service.AihrModelSeedService.SpeechModel; +import org.junit.jupiter.api.Tag; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; + +@Tag("dev") +class AihrSpeechServiceTest { + + @Test + void buildsDistinctMentorAndEmotionalCustomerVoicesWithoutBreakingLegacyRequests() { + ObjectMapper mapper = new ObjectMapper(); + SpeechModel runtime = new SpeechModel( + "siliconflow", + "FunAudioLLM/CosyVoice2-0.5B", + "https://api.siliconflow.cn/v1", + "test-key" + ); + + var mentor = AihrSpeechService.buildRequestBody( + mapper, runtime, "先承接情绪,再说明处理节点。", null, + new VoiceProfile("mentor", null, null, null)); + var customer = AihrSpeechService.buildRequestBody( + mapper, runtime, "你们到底什么时候处理?", null, + new VoiceProfile("customer", null, 9.0, "intense")); + var legacy = AihrSpeechService.buildRequestBody( + mapper, runtime, "普通播报", "alex", null); + + assertEquals("FunAudioLLM/CosyVoice2-0.5B:benjamin", mentor.path("voice").asText()); + assertEquals(0.92, mentor.path("speed").asDouble()); + assertTrue(mentor.path("input").asText().contains("老师傅语气")); + assertEquals("FunAudioLLM/CosyVoice2-0.5B:bella", customer.path("voice").asText()); + assertEquals(1.3, customer.path("speed").asDouble()); + assertTrue(customer.path("input").asText().contains("情绪强烈")); + assertEquals("FunAudioLLM/CosyVoice2-0.5B:alex", legacy.path("voice").asText()); + assertFalse(legacy.has("speed")); + assertFalse(legacy.path("input").asText().contains("<|endofprompt|>")); + } +} diff --git a/docs/AI陪练功能优化与缺口分析(修订).md b/docs/AI陪练功能优化与缺口分析(修订).md index 7bdd9bed..78ec7857 100644 --- a/docs/AI陪练功能优化与缺口分析(修订).md +++ b/docs/AI陪练功能优化与缺口分析(修订).md @@ -11,7 +11,7 @@ |---|---|---| | 员工端对练 /start /turn /finish 闭环 | ✅ | 移动端带 Authorization+clientid,mode=mobile 落 `aihr_practice_session` | | AI 扮演业主(情绪/身份/诉求人设) | ✅ | 真 LLM 生成客户回复(seed 剧本作剧情锚点);**情绪值仍走 seed 锚点,未纳入 LLM 输出** | -| 语音输入 ASR → LLM → TTS 播报 | ✅(部分) | **原文"多语气播报"系夸大**:TTS 单一默认音色,多语气/方言/语速=BACKLOG B2 待办 | +| 语音输入 ASR → LLM → TTS 播报 | ✅(部分) | 已按老师傅/业主/面试官区分音色,并按业主情绪分切换平静/严肃/强烈语气与语速;方言、按具体人物选音色和语音克隆仍属 BACKLOG B2 | | ASR 转写可编辑后提交 | ✅ | **原文漏计且缺口表误列为缺失**:转写先填入输入框,员工可修改再提交 | | 每轮 AI 教练提示(coachHint) | ✅ | **原文漏计**:每轮 turn 返回教练提示(卡壳/流程偏差提醒) | | 训练完成多维评分 | ✅ | 4 维:合规/沟通/情绪/营销 + 导师改写 + 点评;LLM temperature=0 结构化输出,评分仅 /finish 一次 | diff --git a/docs/API_INTEGRATION.md b/docs/API_INTEGRATION.md index 602698fc..55897e25 100644 --- a/docs/API_INTEGRATION.md +++ b/docs/API_INTEGRATION.md @@ -10,7 +10,7 @@ | 候选人入职主体关联 `/recruit/interview` | `GET/POST /api/recruit/interview/candidate-links` | HR/管理员在当前租户范围内把本地候选人 ID 关联到已同步的在职 `ext_party_id`;只保存外部主体 ID,不复制姓名、部门等组织字段,重复关联同一主体幂等,候选人更换主体或同一主体已关联其他候选人会拒绝 | | 三角色对练 `/train/practice` | `POST /api/train/practice/start`、`/turn`、`/finish` | 已接入编排 API;数据库启用 chat 模型后,`/turn` 客户回复按人设走真 LLM 生成(seed 剧本作剧情锚点),`/finish` 走单次 temperature=0 结构化评分(4 维分+导师改写+点评);移动端 `turn/finish` 校验当前手机号与启动会话归属,未知或他人会话拒绝;模型未配置或调用失败自动回退 seed,契约不变 | | 正式试点数据导出 | `GET /api/train/practice/export?startDate=YYYY-MM-DD&endDate=YYYY-MM-DD` | 起止日期必填且包含结束日;只统计窗口内能通过唯一手机号或外部 ID 映射到在职组织快照的正式会话,排除重复手机号和身份碰撞。完训定义为每人至少 10 次,校准必须关联同一窗口内正式会话;CSV 同时给出校准命中数、SOP 可用数、满意度响应数/平均分,以及明细级 AI 分、人工校准分、校准人、校准时间、最终采用分、满意度分和意见,避免用四舍五入后的比率反推门禁状态;汇总和明细均携带正式人员及项目口径,不混入历史 seed/开发身份 | -| 对练语音 | `POST /api/ai/asr`(multipart 字段 `file`,≤5MB)、`POST /api/ai/tts`(JSON `{text≤300字, voice}`,成功返回 `ossId`,客户端播放地址为受控的内联 `data:audio/*`,不返回原始 OSS URL)、`GET /api/aihr/mobile/oss/{ossId}` | 走 OpenAI-compatible audio 接口(如硅基流动 SenseVoice/CosyVoice2);模型管理需启用 `category=asr/tts` 配置;移动端优先用浏览器录音,`getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用 `audio/*` file input 选择/录制音频后继续调同一 ASR 接口;ASR/TTS 未配置或失败返回 fail,前端降级文本;训练/每日题录音只通过 `audioOssId` 走受保护下载,历史客户端传入的 HTTP `audioUrl` 不再回显;TTS 成功音频仍写入 `sys_oss` 留痕,同时用内联 data URL 保持旧客户端可播放;音频下载按员工本人或主管项目范围授权,系统管理端保持后台访问 | +| 对练语音 | `POST /api/ai/asr`(multipart 字段 `file`,≤5MB)、`POST /api/ai/tts`(JSON `{text≤300字, voice?, voiceProfile?:{role,voice?,speed?,emotion?}}`,旧 `voice` 兼容;成功返回 `ossId`,客户端播放地址为受控的内联 `data:audio/*`,不返回原始 OSS URL)、`GET /api/aihr/mobile/oss/{ossId}` | 走 OpenAI-compatible audio 接口;生产 SiliconFlow CosyVoice2 按角色映射老师傅/业主/面试官音色,业主对练再按已有情绪分切换平静/严肃/强烈语气和语速,设备语音降级同步调整语速与音高;模型管理需启用 `category=asr/tts` 配置;移动端优先用浏览器录音,`getUserMedia/MediaRecorder` 不可用或麦克风权限失败时,用 `audio/*` file input 选择/录制音频后继续调同一 ASR 接口;ASR/TTS 未配置或失败返回 fail,前端降级设备语音或文本;训练/每日题录音只通过 `audioOssId` 走受保护下载,历史客户端传入的 HTTP `audioUrl` 不再回显;TTS 成功音频仍写入 `sys_oss` 留痕,同时用内联 data URL 保持旧客户端可播放;音频下载按员工本人或主管项目范围授权,系统管理端保持后台访问 | | 案例沉淀 `/knowledge/cases` | `GET /api/knowledge/case/capabilities`、`POST /api/knowledge/case/upload`、`/organize`、`/curate`、`GET /records`、`GET /records/{caseId}`、`POST /records/{caseId}/review`、`GET /records/{caseId}/media` | `/capabilities` 返回服务端判定的案例提交/查看能力,移动端不再向普通员工展示无权提交的素材表单;`/upload` 改为 multipart 真实语音上传并走 ASR,服务端只接受 MP3/WAV/M4A/WebM/OGG/AAC/FLAC,成功后原始音频写入 `sys_oss`,案例记录只保存 `mediaOssId` 供受保护媒体接口读取,不向客户端回传原始 `mediaUrl`;`/organize` 用真实转写调 chat 模型整理案例,未配置模型时按真实 transcript 本地结构化,并从背景外的真实摘要项提取学习点;APP 用户的项目范围从 `aihr_org_snapshot` 登录身份解析,上传、整理、入库、列表和详情均按项目范围校验,未完成正式组织映射时安全拒绝,不接受前端伪造项目范围;员工列表/详情只返回 `已入库` 案例,管理端系统用户保留全局运营视图;移动端和管理端案例详情通过受保护媒体接口回放原始音频,主管/项目负责人可提交脱敏点评;预渲染视频样片仍待正式媒体资产接入 | | 案例媒体安全 | `GET /api/knowledge/case/records/{caseId}/media` | 案例详情只返回 `mediaOssId`,不返回原始 `sys_oss.url`;媒体下载会重复执行登录、后台角色或 APP 项目范围校验,再由服务端流式读取 OSS。管理端与 `mobile-uni` 通过鉴权 blob/temp 文件播放,关闭详情页时释放本地对象 URL | | SOP知识库 `/knowledge/sop` | `POST /api/knowledge/search`、`POST /api/knowledge/answer-feedback`、`GET /api/knowledge/position-sop`、`GET /api/aihr/mobile/onboard/tasks`、`POST /api/aihr/mobile/onboard/tasks/{id}/complete`、`GET /api/aihr/mobile/qualification`、`POST /api/knowledge/doc/upload` | 已接入 MySQL Fulltext + Qdrant 混合召回、岗位学习适配摘要、OSS-first 文档上传、txt/md/PDF/Word/Excel/PPT 解析和 embedding 写入,失败回退 seed;搜索返回 `reviewId` 与 `promptVersion`,员工反馈回传并保存该评审批次,SOP 人工评审记录同时保留答案生成提示词版本,管理端可继续复核;员工学习页按当前 APP 身份读取正式岗前/入职任务,员工只能将本人处于“待完成/进行中”的任务确认完成,服务端按组织快照/手机号归属更新 `status/completed_time`,不接受前端身份参数;资格证据无正式数据时明确返回 `NOT_CONFIGURED`,不以 AI 分数代替上岗资格;`position-sop` 仍保留一期生活顾问学习导航语义 | diff --git a/docs/BACKLOG.md b/docs/BACKLOG.md index 2911f50a..eb58e988 100644 --- a/docs/BACKLOG.md +++ b/docs/BACKLOG.md @@ -22,6 +22,8 @@ 需求:AI 员工分岗位等级面试官形象;模拟多种语气声——老太说话慢、带方言、吵架、求救听不清。 +> 2026-07-18 已完成首批:共享 TTS 契约支持 `voiceProfile`,老师傅/业主/面试官分音色,业主随对练情绪分调整语气与语速;剩余人物级音色、方言、克隆和信道劣化继续保留在 B2。 + - 复用三角色对练人设机制,**不新建架构**:面试官 = 另一组 `practice_scenario.persona_json`(TechSpec 6.2),按岗位等级(管家/主管/项目经理)建多条,新增字段: - `avatar_url`:岗位形象头像,由 B3 文生图生成 - `voice_profile`:`{ voice_id, speed, emotion, pitch, dialect }` diff --git a/mobile-uni/src/pages/candidate/interview/index.vue b/mobile-uni/src/pages/candidate/interview/index.vue index 5e4da634..9325e8e3 100644 --- a/mobile-uni/src/pages/candidate/interview/index.vue +++ b/mobile-uni/src/pages/candidate/interview/index.vue @@ -156,6 +156,7 @@ import { resetPageScroll } from '@/services/navigation'; import { chooseSpeechAudio, createSpeechPlaybackController, + INTERVIEWER_VOICE_PROFILE, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile @@ -186,7 +187,8 @@ const speechPlayback = createSpeechPlaybackController( (text) => { message.value = text; uni.showToast({ title: text, icon: 'none' }); - } + }, + INTERVIEWER_VOICE_PROFILE ); const interviewBusy = computed(() => interviewStatus.value === 'starting' || interviewStatus.value === 'submitting'); diff --git a/mobile-uni/src/pages/user/practice/index.vue b/mobile-uni/src/pages/user/practice/index.vue index 5fe6935d..79a5588c 100644 --- a/mobile-uni/src/pages/user/practice/index.vue +++ b/mobile-uni/src/pages/user/practice/index.vue @@ -348,6 +348,7 @@ import type { PracticePrepCard, PracticeAssignment, PracticeRole, + SpeechVoiceProfile, SpeechPlaybackStatus, PracticeTurn, PracticeRecord @@ -357,7 +358,7 @@ import { getSelectedPosition, isLoggedIn, rememberLoginRedirect } from '@/servic import { searchKnowledge } from '@/services/knowledge'; import { resetPageScroll } from '@/services/navigation'; import { ensureEmployeePosition } from '@/services/position'; -import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech'; +import { chooseSpeechAudio, createSpeechPlaybackController, customerVoiceProfile, measureAudioDuration, MENTOR_VOICE_PROFILE, normalizeOssId, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech'; import { createPageRequestScope, currentAccountKey, type RequestScopeTicket } from '@/services/request-scope'; import { stripInternalCodes } from '@/services/text'; @@ -433,7 +434,8 @@ const speechPlayback = createSpeechPlaybackController( (text) => { message.value = text; uni.showToast({ title: text, icon: 'none' }); - } + }, + MENTOR_VOICE_PROFILE ); const busy = computed(() => status.value === 'starting' || status.value === 'submitting'); const practiceView = computed<'prep' | 'active' | 'result'>(() => { @@ -517,12 +519,19 @@ interface TurnVoice { duration: number | null; unread: boolean; open: boolean; + profile: SpeechVoiceProfile; } const turnVoices = ref>({}); const turnVoice = (index: number): TurnVoice => - turnVoices.value[index] || { status: 'unavailable', duration: null, unread: false, open: false }; + turnVoices.value[index] || { + status: 'unavailable', + duration: null, + unread: false, + open: false, + profile: customerVoiceProfile() + }; const patchTurnVoice = (index: number, patch: Partial) => { turnVoices.value = { ...turnVoices.value, [index]: { ...turnVoice(index), ...patch } }; @@ -530,11 +539,12 @@ const patchTurnVoice = (index: number, patch: Partial) => { /** Customer lines arrive as WeChat-style voice messages: pre-synthesize so the * bubble shows a duration and an unread dot; fall back to text when TTS is off. */ -const prepareTurnVoice = async (index: number, text: string) => { +const prepareTurnVoice = async (index: number, text: string, emotion = 0) => { const spoken = (text || '').trim(); if (!spoken) return; - patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false }); - const source = await speechPlayback.preload(spoken); + const profile = customerVoiceProfile(emotion); + patchTurnVoice(index, { status: 'preparing', duration: null, unread: true, open: false, profile }); + const source = await speechPlayback.preload(spoken, profile); if (!turnVoices.value[index]) return; if (!source) { patchTurnVoice(index, { status: 'unavailable', unread: false }); @@ -560,7 +570,7 @@ const playTurnVoice = (index: number, text: string) => { return; } patchTurnVoice(index, { unread: false }); - void speechPlayback.toggle(`turn-${index}`, (text || '').trim()); + void speechPlayback.toggle(`turn-${index}`, (text || '').trim(), turnVoice(index).profile); }; const toggleTurnTranscript = (index: number) => { @@ -748,11 +758,11 @@ const requireLogin = () => { return false; }; -const appendTurn = (role: PracticeRole, text?: string) => { +const appendTurn = (role: PracticeRole, text?: string, emotion = 0) => { if (!text) return; turns.value.push({ role, text }); if (role === 'customer') { - void prepareTurnVoice(turns.value.length - 1, text); + void prepareTurnVoice(turns.value.length - 1, text, emotion); } scrollThreadToBottom(); }; @@ -1146,7 +1156,7 @@ const start = async (scenarioId = defaultScenarioId, assignmentId?: number) => { emotionScore.value = data.emotion ?? 0; trustScore.value = data.trust ?? 0; sessionId.value = data.sessionId; - appendTurn('customer', data.customerText); + appendTurn('customer', data.customerText, data.emotion); status.value = 'active'; } catch (error) { if (!requests.isCurrent(request)) return; @@ -1267,7 +1277,7 @@ const submit = async () => { await finish(true); return; } - appendTurn('customer', data.customerText); + appendTurn('customer', data.customerText, data.emotion); roundIndex.value = data.roundIndex; status.value = 'active'; } catch (error) { diff --git a/mobile-uni/src/pages/user/sop/index.vue b/mobile-uni/src/pages/user/sop/index.vue index 57e9b6d2..6b0185f3 100644 --- a/mobile-uni/src/pages/user/sop/index.vue +++ b/mobile-uni/src/pages/user/sop/index.vue @@ -261,7 +261,7 @@ import { } from '@/services/auth'; import { resetPageScroll } from '@/services/navigation'; import { ensureEmployeePosition } from '@/services/position'; -import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech'; +import { chooseSpeechAudio, createSpeechPlaybackController, measureAudioDuration, MENTOR_VOICE_PROFILE, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech'; import { downloadSummaryCardImage } from '@/services/summary-card-image'; import { computed } from 'vue'; @@ -337,7 +337,8 @@ const defaultToolCode = computed<'MY_PRACTICE_SUMMARY' | 'TEAM_PRACTICE_SUMMARY' const speechPlayback = createSpeechPlaybackController( (state) => { speechState.value = state; }, - (text) => { uni.showToast({ title: text, icon: 'none' }); } + (text) => { uni.showToast({ title: text, icon: 'none' }); }, + MENTOR_VOICE_PROFILE ); const scrollToBottom = () => { diff --git a/mobile-uni/src/pages/user/web-ai/index.vue b/mobile-uni/src/pages/user/web-ai/index.vue index cd0b8bea..8fe16bd0 100644 --- a/mobile-uni/src/pages/user/web-ai/index.vue +++ b/mobile-uni/src/pages/user/web-ai/index.vue @@ -117,7 +117,7 @@ import { onShow, onUnload } from '@dcloudio/uni-app'; import type { SpeechPlaybackStatus, WebAiCapabilities, WebAiResponse } from '@/types/api'; import ChatComposer from '@/components/chat/ChatComposer.vue'; import { isLoggedIn, rememberLoginRedirect } from '@/services/auth'; -import { chooseSpeechAudio, createSpeechPlaybackController, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech'; +import { chooseSpeechAudio, createSpeechPlaybackController, MENTOR_VOICE_PROFILE, transcribeSpeechBlob, transcribeSpeechFile } from '@/services/speech'; import { getWebAiCapabilities, queryWebAi } from '@/services/web-ai'; const path = '/pages/user/web-ai/index'; @@ -164,7 +164,8 @@ let abortVoiceTranscription: (() => void) | null = null; const speechPlayback = createSpeechPlaybackController( (state) => { speechState.value = state; }, - (text) => { uni.showToast({ title: text, icon: 'none' }); } + (text) => { uni.showToast({ title: text, icon: 'none' }); }, + MENTOR_VOICE_PROFILE ); const scrollToBottom = () => { diff --git a/mobile-uni/src/services/speech.ts b/mobile-uni/src/services/speech.ts index db6cc2c9..acce534e 100644 --- a/mobile-uni/src/services/speech.ts +++ b/mobile-uni/src/services/speech.ts @@ -1,8 +1,16 @@ -import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, TtsResponse } from '@/types/api'; +import type { AsrResponse, SpeechPlaybackStatus, SpeechSelectedFile, SpeechVoiceProfile, TtsResponse } from '@/types/api'; import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api'; const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg']; +export const MENTOR_VOICE_PROFILE: SpeechVoiceProfile = { role: 'mentor', emotion: 'warm', speed: 0.92 }; +export const INTERVIEWER_VOICE_PROFILE: SpeechVoiceProfile = { role: 'interviewer', emotion: 'professional', speed: 0.96 }; +export const customerVoiceProfile = (emotion = 0): SpeechVoiceProfile => { + if (emotion >= 80) return { role: 'customer', emotion: 'intense', speed: 1.12 }; + if (emotion >= 70) return { role: 'customer', emotion: 'serious', speed: 1.04 }; + return { role: 'customer', emotion: 'calm', speed: 0.98 }; +}; + // OSS 编号是雪花 ID,超出 JS 安全整数范围,必须按字符串透传,禁止 Number() 强转。 export const normalizeOssId = (value: unknown): string | undefined => { const text = String(value ?? '').trim(); @@ -83,11 +91,11 @@ export const transcribeSpeechBlob = ( xhr.send(form); }); -export const synthesizeSpeech = (text: string) => +export const synthesizeSpeech = (text: string, voiceProfile?: SpeechVoiceProfile) => apiRequest({ url: '/api/ai/tts', method: 'POST', - data: { text: text.trim().slice(0, 300) }, + data: { text: text.trim().slice(0, 300), voiceProfile }, timeout: 60000 }); @@ -98,7 +106,8 @@ export interface SpeechPlaybackSnapshot { export const createSpeechPlaybackController = ( onState: (snapshot: SpeechPlaybackSnapshot) => void, - onError: (message: string) => void + onError: (message: string) => void, + defaultVoiceProfile?: SpeechVoiceProfile ) => { let audio: ReturnType | null = null; let browserUtterance: SpeechSynthesisUtterance | null = null; @@ -106,6 +115,7 @@ export const createSpeechPlaybackController = ( let generation = 0; const sourceCache = new Map(); const maxCachedSources = 4; + const cacheKey = (text: string, profile?: SpeechVoiceProfile) => JSON.stringify([text, profile || null]); const publish = (key = '', status: SpeechPlaybackStatus = 'idle') => { activeKey = key; @@ -126,7 +136,12 @@ export const createSpeechPlaybackController = ( publish(); }; - const playWithBrowserSpeech = (key: string, value: string, currentGeneration: number) => { + const playWithBrowserSpeech = ( + key: string, + value: string, + currentGeneration: number, + profile?: SpeechVoiceProfile + ) => { if ( typeof window === 'undefined' || !window.speechSynthesis @@ -135,7 +150,8 @@ export const createSpeechPlaybackController = ( const utterance = new window.SpeechSynthesisUtterance(value.slice(0, 600)); utterance.lang = 'zh-CN'; - utterance.rate = 0.95; + utterance.rate = profile?.speed || 0.95; + utterance.pitch = profile?.role === 'mentor' ? 0.85 : profile?.role === 'customer' ? 1.05 : 1; utterance.onend = () => { if (currentGeneration === generation && activeKey === key) stop(); }; @@ -173,7 +189,7 @@ export const createSpeechPlaybackController = ( context.play(); }; - const toggle = async (key: string, text: string) => { + const toggle = async (key: string, text: string, voiceProfile?: SpeechVoiceProfile) => { if (activeKey === key) { stop(); return; @@ -181,30 +197,32 @@ export const createSpeechPlaybackController = ( stop(); const value = text.trim(); if (!value) return; + const profile = voiceProfile || defaultVoiceProfile; + const sourceKey = cacheKey(value, profile); const currentGeneration = generation; publish(key, 'loading'); try { - const cachedSource = sourceCache.get(value); + const cachedSource = sourceCache.get(sourceKey); if (cachedSource) { - sourceCache.delete(value); - sourceCache.set(value, cachedSource); - playAudioSource(key, cachedSource, value, currentGeneration); + sourceCache.delete(sourceKey); + sourceCache.set(sourceKey, cachedSource); + playAudioSource(key, cachedSource, sourceKey, currentGeneration); return; } - const result = await synthesizeSpeech(value); + const result = await synthesizeSpeech(value, profile); if (currentGeneration !== generation || activeKey !== key) return; const source = result.inlineAudioUrl || result.audioUrl; if (!source) throw new Error('语音合成未返回音频'); - sourceCache.set(value, source); + sourceCache.set(sourceKey, source); while (sourceCache.size > maxCachedSources) { const oldestKey = sourceCache.keys().next().value as string | undefined; if (!oldestKey) break; sourceCache.delete(oldestKey); } - playAudioSource(key, source, value, currentGeneration); + playAudioSource(key, source, sourceKey, currentGeneration); } catch (error) { if (currentGeneration !== generation) return; - if (playWithBrowserSpeech(key, value, currentGeneration)) return; + if (playWithBrowserSpeech(key, value, currentGeneration, profile)) return; stop(); onError(error instanceof Error ? error.message : '语音生成失败'); } @@ -215,16 +233,18 @@ export const createSpeechPlaybackController = ( * WeChat-style voice bubble can show duration before the first tap. * Returns the audio source URL, or null when remote TTS is unavailable. */ - const preload = async (text: string): Promise => { + const preload = async (text: string, voiceProfile?: SpeechVoiceProfile): Promise => { const value = text.trim(); if (!value) return null; - const cached = sourceCache.get(value); + const profile = voiceProfile || defaultVoiceProfile; + const sourceKey = cacheKey(value, profile); + const cached = sourceCache.get(sourceKey); if (cached) return cached; try { - const result = await synthesizeSpeech(value); + const result = await synthesizeSpeech(value, profile); const source = result.inlineAudioUrl || result.audioUrl; if (!source) return null; - sourceCache.set(value, source); + sourceCache.set(sourceKey, source); while (sourceCache.size > maxCachedSources) { const oldestKey = sourceCache.keys().next().value as string | undefined; if (!oldestKey) break; diff --git a/mobile-uni/src/types/api.ts b/mobile-uni/src/types/api.ts index 0247d5e8..dc854ce1 100644 --- a/mobile-uni/src/types/api.ts +++ b/mobile-uni/src/types/api.ts @@ -27,6 +27,13 @@ export interface TtsResponse { export type SpeechPlaybackStatus = 'idle' | 'loading' | 'playing'; +export interface SpeechVoiceProfile { + role: 'mentor' | 'customer' | 'interviewer' | 'neutral'; + voice?: string; + speed?: number; + emotion?: 'warm' | 'professional' | 'serious' | 'intense' | 'calm'; +} + export interface ToolItem { title: string; desc: string;