fix(sop): restore voice-first replies on App without withholding text

73712994 disabled voice-first on native by gating the TTS preload behind
!isNativeAppRuntime(), which left App users with the degraded text bubble
plus a 播报 button. The voice-line, 转文字 toggle and 收起全文 markup were
still present but unreachable.

Re-enable the preload on every platform and open the transcript when the
message is created, so the answer is readable while synthesis runs.

Collapse the transcript when playback starts rather than when synthesis
finishes: on-device synthesis settles in 2-6s, so collapsing on ready
pulled the answer away mid-sentence. transcriptPinned records a manual
toggle so the automatic collapse never overrides the reader.

Bound the preparing bubble at 20s (the upstream DashScope ceiling) so a
stalled synthesis degrades to the plain text bubble instead of showing an
endless placeholder duration.

Replace the tautological assertion that pinned the previous behaviour by
matching source strings with assertions on the behaviour itself.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-07-26 02:36:14 +08:00
co-authored by Claude Opus 5
parent 7096545e82
commit d84ad87cb6
2 changed files with 62 additions and 9 deletions
+39 -5
View File
@@ -108,7 +108,7 @@
<text class="voice-duration">{{ voiceLabel(msg) }}</text> <text class="voice-duration">{{ voiceLabel(msg) }}</text>
</view> </view>
<view v-if="msg.unread" class="voice-dot" /> <view v-if="msg.unread" class="voice-dot" />
<button class="transcript-toggle" @click="msg.transcriptOpen = !msg.transcriptOpen"> <button class="transcript-toggle" @click="toggleTranscript(msg)">
{{ msg.transcriptOpen ? '收起' : '转文字' }} {{ msg.transcriptOpen ? '收起' : '转文字' }}
</button> </button>
</view> </view>
@@ -231,7 +231,7 @@
</view> </view>
</view> </view>
<view v-if="msg.voiceStatus !== 'unavailable'" class="transcript-collapse" @click="msg.transcriptOpen = false"> <view v-if="msg.voiceStatus !== 'unavailable'" class="transcript-collapse" @click="collapseTranscript(msg)">
<text class="transcript-collapse-text">收起全文 ▲</text> <text class="transcript-collapse-text">收起全文 ▲</text>
</view> </view>
</view> </view>
@@ -497,6 +497,9 @@ interface MasterMsg {
voiceDuration: number | null; voiceDuration: number | null;
unread: boolean; unread: boolean;
transcriptOpen: boolean; transcriptOpen: boolean;
/** Set once the reader toggles the transcript, so preparing voice never
* collapses text the reader chose to keep open. */
transcriptPinned: boolean;
} }
interface SystemMsg { interface SystemMsg {
@@ -993,7 +996,7 @@ const sendQuestion = async (
} }
retireEarlierMemoryCandidate(nextResult.memoryCandidate?.id); retireEarlierMemoryCandidate(nextResult.memoryCandidate?.id);
const spoken = (nextResult.answer || '').trim(); const spoken = (nextResult.answer || '').trim();
const shouldPreloadVoice = Boolean(spoken) && !isNativeAppRuntime(); const shouldPreloadVoice = Boolean(spoken);
replaceMsg(pending.id, { replaceMsg(pending.id, {
id: pending.id, id: pending.id,
role: 'master', role: 'master',
@@ -1008,7 +1011,10 @@ const sendQuestion = async (
voiceStatus: shouldPreloadVoice ? 'preparing' : 'unavailable', voiceStatus: shouldPreloadVoice ? 'preparing' : 'unavailable',
voiceDuration: null, voiceDuration: null,
unread: Boolean(spoken), unread: Boolean(spoken),
transcriptOpen: false // Keep the answer readable while the voice is still synthesizing; it
// collapses on its own once the voice bubble can show a duration.
transcriptOpen: true,
transcriptPinned: false
}); });
if (shouldPreloadVoice) void prepareVoice(pending.id, spoken); if (shouldPreloadVoice) void prepareVoice(pending.id, spoken);
if (!nextResult.answer && !nextResult.snippets?.length && !nextResult.memoryCandidate) { if (!nextResult.answer && !nextResult.snippets?.length && !nextResult.memoryCandidate) {
@@ -1155,22 +1161,47 @@ const findMasterMsg = (id: number): MasterMsg | null => {
return found && found.role === 'master' ? found : null; return found && found.role === 'master' ? found : null;
}; };
/** Upper bound on the `preparing` voice bubble. Remote TTS has been seen to
* hang well past its own timeout; after this the reply degrades to the plain
* text bubble instead of showing an endless placeholder duration. */
const VOICE_PREPARE_TIMEOUT_MS = 20000;
const spokenTextOf = (msg: MasterMsg) => (msg.result.answer || '').trim(); const spokenTextOf = (msg: MasterMsg) => (msg.result.answer || '').trim();
/** Pre-synthesize the reply audio so the voice bubble can show a duration /** Pre-synthesize the reply audio so the voice bubble can show a duration
* before the first tap, WeChat-style. Falls back to a plain text bubble * before the first tap, WeChat-style. Falls back to a plain text bubble
* when remote TTS is unavailable. */ * when remote TTS is unavailable. */
const prepareVoice = async (id: number, spoken: string) => { const prepareVoice = async (id: number, spoken: string) => {
const source = await speechPlayback.preload(spoken); const source = await Promise.race([
speechPlayback.preload(spoken),
new Promise<null>((resolve) => {
setTimeout(() => resolve(null), VOICE_PREPARE_TIMEOUT_MS);
})
]);
const msg = findMasterMsg(id); const msg = findMasterMsg(id);
if (!msg) return; if (!msg) return;
if (!source) { if (!source) {
// No voice to offer: leave the text bubble and its play button in place.
msg.voiceStatus = 'unavailable'; msg.voiceStatus = 'unavailable';
msg.unread = false; msg.unread = false;
return; return;
} }
msg.voiceDuration = await measureAudioDuration(source); msg.voiceDuration = await measureAudioDuration(source);
msg.voiceStatus = 'ready'; msg.voiceStatus = 'ready';
// Deliberately not collapsing here: synthesis often finishes in 2-4s, which
// would pull the answer away mid-sentence. The transcript collapses when the
// reader starts the voice instead.
};
/** Pin on any manual toggle so a later `ready` voice keeps the reader's choice. */
const toggleTranscript = (msg: MasterMsg) => {
msg.transcriptOpen = !msg.transcriptOpen;
msg.transcriptPinned = true;
};
const collapseTranscript = (msg: MasterMsg) => {
msg.transcriptOpen = false;
msg.transcriptPinned = true;
}; };
const isVoicePlaying = (msg: MasterMsg) => const isVoicePlaying = (msg: MasterMsg) =>
@@ -1188,6 +1219,9 @@ const playVoice = (msg: MasterMsg) => {
return; return;
} }
msg.unread = false; msg.unread = false;
// Listening replaces reading, so fold the transcript away unless the reader
// explicitly asked to keep it open.
if (!msg.transcriptPinned) msg.transcriptOpen = false;
void speechPlayback.toggle(`answer-${msg.id}`, spokenTextOf(msg)); void speechPlayback.toggle(`answer-${msg.id}`, spokenTextOf(msg));
}; };
+23 -4
View File
@@ -38,12 +38,31 @@ test('Agent 响应按结构化状态渲染来源、澄清和确认动作', async
assert.doesNotMatch(page, /result\.answer\.(?:includes|match|startsWith)/); assert.doesNotMatch(page, /result\.answer\.(?:includes|match|startsWith)/);
}); });
test('原生 App 立即显示 Agent 文字回答,语音仅按需生成', async () => { test('回复语音优先,但文字在合成期间可读且不覆盖读者选择', async () => {
const page = await source('../src/pages/user/sop/index.vue'); const page = await source('../src/pages/user/sop/index.vue');
assert.match(page, /const shouldPreloadVoice = Boolean\(spoken\) && !isNativeAppRuntime\(\)/); // Voice-first applies on every platform, not just H5.
assert.match(page, /voiceStatus: shouldPreloadVoice \? 'preparing' : 'unavailable'/); assert.match(page, /const shouldPreloadVoice = Boolean\(spoken\);/);
assert.match(page, /if \(shouldPreloadVoice\) void prepareVoice\(pending\.id, spoken\)/); assert.doesNotMatch(page, /shouldPreloadVoice = [^;]*isNativeAppRuntime/);
// Text is readable while the voice is still preparing.
assert.match(page, /transcriptOpen: true/);
// Finishing synthesis must not yank the text away; only starting playback folds it.
assert.match(
page,
/msg\.unread = false;\s*(?:\/\/[^\n]*\n\s*)*if \(!msg\.transcriptPinned\) msg\.transcriptOpen = false;\s*void speechPlayback\.toggle/
);
assert.doesNotMatch(
page,
/msg\.voiceStatus = 'ready';\s*if \(!msg\.transcriptPinned\)/
);
assert.match(page, /const toggleTranscript = [^}]*transcriptPinned = true;/s);
assert.match(page, /const collapseTranscript = [^}]*transcriptPinned = true;/s);
// A hanging synthesis degrades to the plain text bubble instead of an endless placeholder.
assert.match(page, /VOICE_PREPARE_TIMEOUT_MS/);
assert.match(page, /Promise\.race\(\[\s*speechPlayback\.preload\(spoken\)/);
}); });
test('Agent 确认卡通过不透明 draftId 调用动作端点', async () => { test('Agent 确认卡通过不透明 draftId 调用动作端点', async () => {