feat(mobile): expand TTS voices and align App SDK

This commit is contained in:
2026-07-26 17:05:43 +08:00
parent 94ee29f59c
commit f57e0a53c6
18 changed files with 1171 additions and 419 deletions
+6 -6
View File
@@ -34,37 +34,37 @@ const positionProfiles: Record<UserPosition, UserPositionProfile> = {
生活顾问: {
value: '生活顾问',
heroTitle: '客服管家',
quickQuestions: ['催费话术', '投诉应对', '收费标准'],
quickQuestions: ['催费话术怎么说', '投诉怎么应对', '收费标准是什么'],
tip: '接待客户三步法:听清诉求 → 共情理解 → 解决闭环'
},
保安: {
value: '保安',
heroTitle: '秩序岗',
quickQuestions: ['门岗核验', '巡逻异常', '车辆指引'],
quickQuestions: ['门岗怎么核验', '巡逻发现异常怎么办', '车辆怎么指引'],
tip: '今日小贴士:异常先确认人车物,再登记,再同步主管。'
},
保洁: {
value: '保洁',
heroTitle: '环境岗',
quickQuestions: ['公区清洁', '垃圾分类', '消杀标准'],
quickQuestions: ['公区清洁怎么做', '垃圾怎么分类', '消杀标准是什么'],
tip: '今日小贴士:高频区域先看地面、扶手、电梯按钮。'
},
保修: {
value: '保修',
heroTitle: '工程维修',
quickQuestions: ['报修接单', '上门话术', '安全检查'],
quickQuestions: ['报修怎么接单', '上门话术怎么说', '安全检查怎么做'],
tip: '今日小贴士:上门前确认故障、工具、预约时间。'
},
客服: {
value: '客服',
heroTitle: '服务前台',
quickQuestions: ['来访接待', '投诉登记', '工单跟进'],
quickQuestions: ['来访怎么接待', '投诉怎么登记', '工单怎么跟进'],
tip: '今日小贴士:复杂诉求先复述确认,再记录责任人与时限。'
},
主管: {
value: '主管',
heroTitle: '项目管理',
quickQuestions: ['团队沟通', '现场协调', '投诉复盘'],
quickQuestions: ['团队怎么沟通', '现场怎么协调', '投诉怎么复盘'],
tip: '今日小贴士:先确认事实与责任人,再明确动作、时限和复盘节点。'
}
};
@@ -0,0 +1,94 @@
/**
* 从 MP3 字节流数出时长,不依赖任何宿主音频 API。
*
* 起因:`measureAudioDuration` 用浏览器的 `Audio` 构造器探测元数据,而
* App-Plus 的逻辑层是 JsCore 而非 webview,`typeof Audio === 'undefined'`
* 直接成立,于是 App 上语音条永远拿不到秒数,退化成「语音」二字。
*
* 逐帧累加而不是用 `字节数 / 比特率` 估算:后者只在 CBR 下准确,VBR 会偏。
*/
/** kbps。索引 [是否 MPEG1][bitrateIndex],Layer III 专用。 */
const LAYER3_BITRATES = {
mpeg1: [0, 32, 40, 48, 56, 64, 80, 96, 112, 128, 160, 192, 224, 256, 320, 0],
mpeg2: [0, 8, 16, 24, 32, 40, 48, 56, 64, 80, 96, 112, 128, 144, 160, 0]
} as const;
/** Hz。索引 [versionBits][sampleRateIndex]。 */
const SAMPLE_RATES: Record<number, readonly number[]> = {
3: [44100, 48000, 32000], // MPEG1
2: [22050, 24000, 16000], // MPEG2
0: [11025, 12000, 8000] // MPEG2.5
};
/** Layer III 每帧采样数:MPEG1 为 1152,MPEG2/2.5 减半。 */
const samplesPerFrame = (isMpeg1: boolean) => (isMpeg1 ? 1152 : 576);
/** ID3v2 头长 10 字节,尺寸是 4 个 synchsafe 字节(每字节仅低 7 位有效)。 */
const id3v2Length = (bytes: Uint8Array): number => {
if (bytes.length < 10) return 0;
if (bytes[0] !== 0x49 || bytes[1] !== 0x44 || bytes[2] !== 0x33) return 0;
const size = ((bytes[6] & 0x7f) << 21) | ((bytes[7] & 0x7f) << 14) | ((bytes[8] & 0x7f) << 7) | (bytes[9] & 0x7f);
const footer = (bytes[5] & 0x10) !== 0 ? 10 : 0;
return 10 + size + footer;
};
interface Frame {
length: number;
seconds: number;
}
/** 解析一个帧头;不是合法 Layer III 帧头时返回 null。 */
const readFrame = (bytes: Uint8Array, at: number): Frame | null => {
if (at + 4 > bytes.length) return null;
// 11 位帧同步
if (bytes[at] !== 0xff || (bytes[at + 1] & 0xe0) !== 0xe0) return null;
const versionBits = (bytes[at + 1] >> 3) & 0x03;
if (versionBits === 1) return null; // 保留值
const layerBits = (bytes[at + 1] >> 1) & 0x03;
if (layerBits !== 1) return null; // 只处理 Layer III
const bitrateIndex = (bytes[at + 2] >> 4) & 0x0f;
const sampleRateIndex = (bytes[at + 2] >> 2) & 0x03;
if (sampleRateIndex === 3) return null;
const isMpeg1 = versionBits === 3;
const bitrate = (isMpeg1 ? LAYER3_BITRATES.mpeg1 : LAYER3_BITRATES.mpeg2)[bitrateIndex];
const sampleRate = SAMPLE_RATES[versionBits]?.[sampleRateIndex];
if (!bitrate || !sampleRate) return null; // free-format / 保留值
const samples = samplesPerFrame(isMpeg1);
const padding = (bytes[at + 2] >> 1) & 0x01;
const length = Math.floor((samples / 8) * bitrate * 1000 / sampleRate) + padding;
if (length <= 4) return null;
return { length, seconds: samples / sampleRate };
};
/**
* 累加所有帧得到秒数;一个合法帧都找不到时返回 null。
*
* 首帧通常是 Xing/LAME 信息帧,它本身也是合法音频帧,计入只多约 24ms,
* 不足以影响向上取整后的展示值,因此不做特殊剔除。
*/
export const estimateMpegAudioDuration = (bytes: Uint8Array): number | null => {
let at = id3v2Length(bytes);
let seconds = 0;
let frames = 0;
while (at < bytes.length) {
const frame = readFrame(bytes, at);
if (frame) {
seconds += frame.seconds;
frames += 1;
at += frame.length;
continue;
}
// 遇到 ID3v1 尾标或其它非帧数据就停,不再向后重新找同步字,
// 免得把任意 0xff 字节误判成帧头而虚增时长。
if (frames > 0) break;
at += 1;
}
return frames > 0 && seconds > 0 ? seconds : null;
};
+22 -1
View File
@@ -2,7 +2,8 @@ import type { AsrResponse, PracticeTtsContext, SpeechPlaybackStatus, SpeechSelec
import { apiRequest, apiUrl, authHeaders, readTextPayload } from './api';
import { getAuth } from './auth';
import type { SpeechCapture } from './speech-capture';
import { dataUriToBlobUrl, hasNativeAudioFileSupport, isDataAudioUri, removeNativeTempAudio, writeNativeTempAudio } from './speech-audio-source';
import { estimateMpegAudioDuration } from './mpeg-audio-duration';
import { dataUriToBlobUrl, decodeDataAudioUri, hasNativeAudioFileSupport, isDataAudioUri, removeNativeTempAudio, writeNativeTempAudio } from './speech-audio-source';
import { applyTtsVoicePreference, getTtsDialectPreference, getTtsVoicePreference } from './tts-voice-preference';
const audioExtensions = ['.mp3', '.wav', '.m4a', '.webm', '.aac', '.ogg'];
@@ -344,8 +345,28 @@ export const createSpeechPlaybackController = (
return { toggle, stop, destroy, preload };
};
/**
* data: URI 直接数帧,不经宿主音频 API。
*
* App-Plus 的逻辑层是 JsCore,没有 `Audio`,原先在这里直接返回 null,
* 导致语音条只显示「语音」而没有秒数。字节解析在 H5 上同样可用且更快,
* 所以两端都优先走它,`Audio` 只作为远端 URL 的兜底。
*/
const measureFromBytes = (source: string): number | null => {
if (!isDataAudioUri(source)) return null;
const decoded = decodeDataAudioUri(source);
if (!decoded) return null;
const seconds = estimateMpegAudioDuration(decoded.bytes);
return seconds ? Math.max(1, Math.round(seconds)) : null;
};
export const measureAudioDuration = (source: string): Promise<number | null> =>
new Promise((resolve) => {
const fromBytes = measureFromBytes(source);
if (fromBytes !== null) {
resolve(fromBytes);
return;
}
if (typeof Audio === 'undefined') {
resolve(null);
return;
@@ -5,7 +5,15 @@ export const qwenTtsVoiceOptions = [
{ value: 'longanhuan_v3.6', label: '龙安欢 · 中文女声' },
{ value: 'longjielidou_v3.6', label: '龙杰力豆 · 童声' },
{ value: 'loongeva_v3.6', label: 'Loongeva · 英文女声' },
{ value: 'loongjohn', label: 'loongJohn · 英文男声' }
{ value: 'loongjohn', label: 'loongJohn · 英文男声' },
{ value: 'qwen-audio-3.0-tts-flash-longyingmuyu', label: '龙应沐语 · 温柔客服女声' },
{ value: 'qwen-audio-3.0-tts-flash-longyingshiyu', label: '龙应时渝 · 干练客服女声' },
{ value: 'qwen-audio-3.0-tts-flash-longchezhuyu', label: '龙澈竹雨 · 自然助手女声' },
{ value: 'qwen-audio-3.0-tts-flash-longyingyueyue', label: '龙应越越 · 耐心讲解女声' },
{ value: 'qwen-audio-3.0-tts-flash-longluxiaohui', label: '龙露潇晖 · 亲切客服男声' },
{ value: 'qwen-audio-3.0-tts-flash-longlanqinluan', label: '龙兰琴鸾 · 沉稳知性女声' },
{ value: 'qwen-audio-3.0-tts-flash-longluliuche', label: '龙露柳澈 · 新闻播报男声' },
{ value: 'qwen-audio-3.0-tts-flash-longliuxulan', label: '龙柳旭澜 · 新闻播报女声' }
] as const;
export const qwenTtsDialectOptions = [