From aab8f8dfe5bf94089b92a0d0d65680f07cf970f9 Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 15:05:49 +0800 Subject: [PATCH 1/9] =?UTF-8?q?fix(assistant):=20=E6=8C=89=E4=BD=8F?= =?UTF-8?q?=E8=AF=B4=E8=AF=9D=E9=95=BF=E5=9B=9E=E5=A4=8D=E4=B8=8D=E5=86=8D?= =?UTF-8?q?=E5=9C=A8=E8=AF=AD=E9=9F=B3=E6=92=AD=E6=94=BE=E4=B8=AD=E8=A2=AB?= =?UTF-8?q?=E6=B8=85=E7=A9=BA?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _startTurn() 开头无条件清 replyText:上一轮回复还在流式、语音还在播时 用户又按住说话,已经显示的气泡会立刻消失,而 TTS 不受影响照常播—— 现象就是"长回复气泡完全不出现"。回复越长,用户在播放中再按一次的 概率越大,所以只有长回复才容易踩到。 叠加了 #364 加的 request_id 轮次门控:新一轮换了 request_id 后, 上一轮剩余的流式增量全部被当成"别人的"丢掉,即使气泡真的显示出来, 后续增量也进不来,replyText 永远停在清空前那一刻。 现在 _startTurn 不再清 replyText,气泡留到被新一轮自己的回复覆盖, 或用户点掉/取消为止;门控放宽成"当前轮,或者气泡本来就在显示的 那一轮"都放行,让上一轮的剩余增量能继续更新已经显示的气泡。 dismissReply()/cancelTurn() 里一并清掉这个"气泡属于哪一轮"的记录, 保住 #364 本来要防的场景——气泡点掉后不会被晚到的旧回复重新弹出来。 --- .../AssistantConversationService.ts | 29 +++- .../AssistantConversationService.test.ts | 126 ++++++++++++++++++ 2 files changed, 149 insertions(+), 6 deletions(-) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index 69505f9a..c2df3090 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -48,6 +48,11 @@ export class AssistantConversationService implements AssistantApplicationPort { private replyText: string | null = null; /** 当前 turn 的 request_id;voice.dialogue.reply 拿它认轮次,对不上的是上一轮迟到的。 */ private turnRequestId: string | null = null; + /** 当前气泡里显示的是哪个 request_id 的回复。新一轮开始不再清 replyText, + * 所以上一轮剩余的流式增量(同一个 request_id)应该继续更新它,不能被 + * isCurrentTurnReply 当成"别人的"丢掉;点掉/取消气泡时清空,避免真正过期的 + * 回复把已经点掉的气泡重新弹出来。 */ + private displayedReplyRequestId: string | null = null; /** 当前这一帧麦克风音量(dBFS),给波形展示;不在录音时是 null。 */ private soundLevel: number | null = null; // 展示层的 onPressIn/onPressOut 不等待彼此:快速按放会让 endTurn() 在 @@ -125,7 +130,10 @@ export class AssistantConversationService implements AssistantApplicationPort { } private async _startTurn(turnId: number): Promise { - this.replyText = null; + // 不清 replyText:上一轮的回复可能还在流式(用户在它说完前又按住说话), + // 这里清掉会让已经在显示的气泡瞬间消失,即使语音还在正常播(#392 同类问题 + // 在按住说话这条路径上的版本)。旧气泡留到被新一轮自己的回复覆盖,或者 + // 用户点掉/取消为止。 this.soundLevel = null; const previousCaptureCleanup = this.captureCleanup; if (this.connection === null) { @@ -223,6 +231,7 @@ export class AssistantConversationService implements AssistantApplicationPort { this.pendingStartTurn = null; this.soundLevel = null; this.replyText = null; + this.displayedReplyRequestId = null; this.currentAudioId = null; const connection = this.connection; @@ -244,6 +253,7 @@ export class AssistantConversationService implements AssistantApplicationPort { async dismissReply(): Promise { this.replyText = null; + this.displayedReplyRequestId = null; this.currentAudioId = null; this.notifyListeners(); await this.deps.playback.stop().catch(() => {}); @@ -339,16 +349,23 @@ export class AssistantConversationService implements AssistantApplicationPort { speechText: message.payload.speech_text, }); return; - case 'voice.dialogue.reply': - // 上一轮被打断时它的 reply 可能晚到,把已经点掉的旧气泡重新弹出来——气泡 - // 是全屏点击层,会挡住语音条,用户得再点一次才能说话。只认当前这一轮的 - // request_id;服务端把 voice.stream.start 上带的 id 回显在这条消息上。 - if (!this.isCurrentTurnReply(message.request_id)) { + case 'voice.dialogue.reply': { + // 只认当前这一轮,或者当前气泡本来就在显示的那一轮——后者是为了让 + // "上一轮回复还在流式时用户又按住说话"这种情况下,上一轮剩余的增量 + // 还能继续更新它已经显示出来的气泡,不被当成别的轮次的内容丢掉。 + // 除此之外一律当成上一轮被打断后晚到的、已经点掉的旧气泡,不能让它 + // 重新弹出来——气泡是全屏点击层,会挡住语音条,用户得再点一次才能说话。 + const requestId = message.request_id; + if (!this.isCurrentTurnReply(requestId) && requestId !== this.displayedReplyRequestId) { return; } this.replyText = message.payload.speech_text; + if (typeof requestId === 'string') { + this.displayedReplyRequestId = requestId; + } this.notifyListeners(); return; + } case 'voice.tts.start': this.currentAudioId = message.audio_id; this.setState({ conversationId: message.conversation_id, phase: 'speaking' }); diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index af183bcb..3c5cc920 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -742,6 +742,132 @@ describe('AssistantConversationService', () => { await completeStreamStart(fake, nextTurn); }); + it('keeps the bubble visible while a reply streams, and shows its full text', async () => { + // 长回复 = 后端分多条 voice.dialogue.reply 流式下发(composed/realtime agent + // 每次文本增量一条,text 是累计的),最后一条 done=true。气泡应随增量逐步 + // 更新并保持可见,最终显示完整文本。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + await completeStreamStart(fake, service.startTurn()); + const turnId = fake.sent.filter((message) => message.type === 'voice.stream.start')[0] + .request_id; + + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turnId, + payload: { done: false, reply_id: 'reply_1', speech_text: '好的,明天下午' }, + type: 'voice.dialogue.reply', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getReplyText()).toBe('好的,明天下午'); + + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turnId, + payload: { + done: false, + reply_id: 'reply_1', + speech_text: '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', + }, + type: 'voice.dialogue.reply', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getReplyText()).toBe( + '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', + ); + + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turnId, + payload: { + done: true, + reply_id: 'reply_1', + speech_text: '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', + }, + type: 'voice.dialogue.reply', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getReplyText()).toBe( + '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', + ); + }); + + it('keeps a streaming reply visible when a new press starts mid-reply, instead of dropping it while TTS keeps playing', async () => { + // 用户报告:PTT 长回复时气泡完全不出现,但语音照常播放。序列是—— + // turn 1 的回复还在流式(文本增量 + TTS 并行下发),语音播放中用户又按住 + // 说话(PushToTalkBar 在 speaking/awaiting_result 期间不禁用自己): + // 1. 新一轮 _startTurn 先清空 replyText(旧气泡消失); + // 2. turnRequestId 换新,turn 1 剩余的流式 reply 增量按 request_id 被 + // #364 gate 全部丢弃; + // 3. 结果 replyText 永远为空——气泡不出现;而 voice.tts.start 不经过 + // gate,TTS 照常播放。 + // + // 连续模式不受影响:它的 reply 写进 turns,不按 request_id 门控。这也是 + // 用户说"只有 PTT 模式有这个问题"的原因。 + // + // 期望行为(本测试锁定):reply 一旦到达就应该留在气泡里——新一轮开始 + // 不应该把上一轮已经收到的回复文本清掉,流式增量也不应该被 request_id + // gate 吞掉。修复前本用例失败(replyText 被清空且增量被丢弃)。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + // turn 1:回复开始流式(第一条增量已显示)。 + await completeStreamStart(fake, service.startTurn()); + const turn1Id = fake.sent.filter((message) => message.type === 'voice.stream.start')[0] + .request_id; + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turn1Id, + payload: { done: false, reply_id: 'reply_1', speech_text: '好的,明天下午' }, + type: 'voice.dialogue.reply', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getReplyText()).toBe('好的,明天下午'); + + // TTS 开始播(不受 request_id gate 影响)。 + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + payload: { + format: 'pcm', + purpose: 'command_result', + sample_rate_hz: 24000, + speech_text: '好的,明天下午', + }, + type: 'voice.tts.start', + } as AssistantServerMessage); + await flushAsync(); + expect(deps.playback.startStream).toHaveBeenCalledTimes(1); + + // 语音还在播,用户又按住说话 → 新一轮开始。已收到的回复文本应保留在 + // 气泡里(BUG:_startTurn 开头把 replyText 清成 null)。 + const nextTurn = service.startTurn(); + await flushAsync(); + expect(service.getReplyText()).toBe('好的,明天下午'); + + // turn 1 剩余的流式 reply 增量(同一个 request_id)晚到 → 应继续更新气泡 + // (BUG:request_id 已换新,增量被 #364 gate 丢弃)。 + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turn1Id, + payload: { + done: true, + reply_id: 'reply_1', + speech_text: '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', + }, + type: 'voice.dialogue.reply', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getReplyText()).toBe( + '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', + ); + + await completeStreamStart(fake, nextTurn); + }); + it('still shows a reply whose request_id is null', async () => { // 后端 model_dump() 把缺省的 request_id 序列化成 null(不是省掉字段),所以 // null 必须当作"不知道是哪一轮"放行,否则回复永远显示不出来。 From 440d6e531fb53d04151d287810bb564810e695b4 Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 15:06:42 +0800 Subject: [PATCH 2/9] =?UTF-8?q?fix(assistant):=20=E6=8C=89=E4=BD=8F?= =?UTF-8?q?=E8=AF=B4=E8=AF=9D=E5=86=8D=E6=8C=89=E4=B8=80=E6=AC=A1=E4=B8=8D?= =?UTF-8?q?=E5=86=8D=E6=89=93=E6=96=AD=E4=B8=8A=E4=B8=80=E8=BD=AE=E7=9A=84?= =?UTF-8?q?=E8=AF=AD=E9=9F=B3=E7=8A=B6=E6=80=81=E5=92=8C=E8=BF=BD=E9=97=AE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 同一类问题的另外两处:voice.dialogue.question 完全没有轮次门控, 上一轮被打断后晚到的追问会把已经推进到新一轮的 phase 强行掰回 asking;voice.tts.start/voice.tts.end 同样没有门控(协议本身也不带 request_id,没法照搬 dialogue.reply 那套判断),上一轮语音还没播完 就开始新一轮时两轮音频会同时播,且上一轮迟到的 tts.end 会把新一轮 的 phase 强行掰回 idle。 voice.dialogue.question 补上跟 reply 一样的 request_id 门控。 新增 abandonedAudioId:新一轮开始(或点掉气泡)时如果上一轮语音还 没播完,主动停止播放并记住这个 audio_id;它迟到的 tts.end 到达时 只清记录,不再碰 phase。isCurrentTurnReply 改名 isCurrentTurn,因为 现在不止 reply 一处在用它。 已知局限:voice.tts.start/voice.tts.end 协议本身不带任何轮次标识, 新一轮开始的瞬间如果恰好卡在服务端取消生效前那个协程调度间隙内 (毫秒级窗口),上一轮迟到的 tts.start 仍会被当成合法消息接受。这 道窄缝要彻底堵上需要后端配合,这次范围只做前端,先记录着。 --- .../AssistantConversationService.ts | 30 +++++++- .../AssistantConversationService.test.ts | 74 +++++++++++++++++++ 2 files changed, 101 insertions(+), 3 deletions(-) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index c2df3090..f765fb1b 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -50,9 +50,14 @@ export class AssistantConversationService implements AssistantApplicationPort { private turnRequestId: string | null = null; /** 当前气泡里显示的是哪个 request_id 的回复。新一轮开始不再清 replyText, * 所以上一轮剩余的流式增量(同一个 request_id)应该继续更新它,不能被 - * isCurrentTurnReply 当成"别人的"丢掉;点掉/取消气泡时清空,避免真正过期的 + * isCurrentTurn 当成"别人的"丢掉;点掉/取消气泡时清空,避免真正过期的 * 回复把已经点掉的气泡重新弹出来。 */ private displayedReplyRequestId: string | null = null; + /** 上一轮还没播完就被新一轮/dismissReply() 主动放弃的 audio_id。voice.tts.start/ + * voice.tts.end 协议里不带 request_id,没法像 dialogue.reply 那样按轮次门控, + * 只能靠这个记住"这条迟到的收尾消息属于已经不关心的那条音频",不让它把 + * phase 强行掰回 idle、打断已经在推进的新一轮状态。 */ + private abandonedAudioId: string | null = null; /** 当前这一帧麦克风音量(dBFS),给波形展示;不在录音时是 null。 */ private soundLevel: number | null = null; // 展示层的 onPressIn/onPressOut 不等待彼此:快速按放会让 endTurn() 在 @@ -135,6 +140,13 @@ export class AssistantConversationService implements AssistantApplicationPort { // 在按住说话这条路径上的版本)。旧气泡留到被新一轮自己的回复覆盖,或者 // 用户点掉/取消为止。 this.soundLevel = null; + if (this.currentAudioId !== null) { + // 上一轮的语音还没播完,新一轮开始时主动掐掉:不能让两轮音频同时播, + // 也不能放着让它的 tts.end 晚到后把 phase 强行掰回 idle。 + this.abandonedAudioId = this.currentAudioId; + this.currentAudioId = null; + void this.deps.playback.stop().catch(() => {}); + } const previousCaptureCleanup = this.captureCleanup; if (this.connection === null) { this.setState({ phase: 'connecting' }); @@ -254,6 +266,7 @@ export class AssistantConversationService implements AssistantApplicationPort { async dismissReply(): Promise { this.replyText = null; this.displayedReplyRequestId = null; + this.abandonedAudioId = this.currentAudioId; this.currentAudioId = null; this.notifyListeners(); await this.deps.playback.stop().catch(() => {}); @@ -343,6 +356,11 @@ export class AssistantConversationService implements AssistantApplicationPort { void this.applyCategoryUpdate(message.payload.schedule_id, message.payload.category); return; case 'voice.dialogue.question': + // 跟 voice.dialogue.reply 同理:上一轮被新一轮打断后晚到的追问,不能 + // 把已经推进到新一轮的 phase 强行掰回 asking。 + if (!this.isCurrentTurn(message.request_id)) { + return; + } this.setState({ conversationId: message.conversation_id, phase: 'asking', @@ -356,7 +374,7 @@ export class AssistantConversationService implements AssistantApplicationPort { // 除此之外一律当成上一轮被打断后晚到的、已经点掉的旧气泡,不能让它 // 重新弹出来——气泡是全屏点击层,会挡住语音条,用户得再点一次才能说话。 const requestId = message.request_id; - if (!this.isCurrentTurnReply(requestId) && requestId !== this.displayedReplyRequestId) { + if (!this.isCurrentTurn(requestId) && requestId !== this.displayedReplyRequestId) { return; } this.replyText = message.payload.speech_text; @@ -377,6 +395,12 @@ export class AssistantConversationService implements AssistantApplicationPort { .catch(() => {}); return; case 'voice.tts.end': + if (message.audio_id === this.abandonedAudioId) { + // 已经主动放弃的那条音频,收尾消息迟到了:清掉记录就行,不能再碰 + // phase——这时它早就是新一轮的状态,不能被这条迟到消息掰回 idle。 + this.abandonedAudioId = null; + return; + } this.currentAudioId = null; this.deps.playback.endStream().catch(() => {}); this.setState({ phase: 'idle' }); @@ -388,7 +412,7 @@ export class AssistantConversationService implements AssistantApplicationPort { /** 拿不到判断依据时一律放行:后端把缺省的 request_id 序列化成 null 而不是省略字段, * 所以 null 和 undefined 都要当作"不知道是哪一轮",不能误伤。 */ - private isCurrentTurnReply(requestId: string | null | undefined): boolean { + private isCurrentTurn(requestId: string | null | undefined): boolean { if (this.turnRequestId === null || requestId === null || requestId === undefined) { return true; } diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 3c5cc920..271d14b2 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -868,6 +868,80 @@ describe('AssistantConversationService', () => { await completeStreamStart(fake, nextTurn); }); + it('does not let a stale tts.end from an abandoned turn clobber the new turn\'s phase', async () => { + // 同类问题:voice.tts.start/voice.tts.end 协议里不带 request_id,没法像 + // voice.dialogue.reply 那样按轮次门控。turn 1 的语音还没播完,用户又按住 + // 说话开始 turn 2——turn 1 的音频应该被主动掐掉;turn 1 迟到的 tts.end + // 不能把已经推进到 turn 2 的 phase 强行掰回 idle。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + // turn 1:语音开始播。 + await completeStreamStart(fake, service.startTurn()); + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + payload: { + format: 'pcm', + purpose: 'command_result', + sample_rate_hz: 24000, + speech_text: '', + }, + type: 'voice.tts.start', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toMatchObject({ phase: 'speaking' }); + + // 语音还没播完,用户又按住说话 → turn 2 开始:应该主动掐掉 turn 1 的播放。 + const nextTurn = service.startTurn(); + await flushAsync(); + expect(deps.playback.stop).toHaveBeenCalledTimes(1); + + await completeStreamStart(fake, nextTurn); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + + // turn 1 迟到的 tts.end 到达:不能把 phase 掰回 idle。 + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + type: 'voice.tts.end', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + }); + + it('drops a stale dialogue.question from a previous turn so it cannot hijack the new turn\'s phase', async () => { + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + await completeStreamStart(fake, service.startTurn()); + const turn1Id = fake.sent.filter((message) => message.type === 'voice.stream.start')[0] + .request_id; + + const nextTurn = service.startTurn(); + await flushAsync(); + const stateBeforeStaleQuestion = service.getState(); + + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turn1Id, + payload: { + candidates: [], + question_id: 'q_1', + question_kind: 'missing_field', + speech_text: '你是想订哪一天的会议室?', + }, + type: 'voice.dialogue.question', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toEqual(stateBeforeStaleQuestion); + + await completeStreamStart(fake, nextTurn); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + }); + it('still shows a reply whose request_id is null', async () => { // 后端 model_dump() 把缺省的 request_id 序列化成 null(不是省掉字段),所以 // null 必须当作"不知道是哪一轮"放行,否则回复永远显示不出来。 From fb937703cd0ca9548393e4e50341d7771e3555ad Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 15:17:45 +0800 Subject: [PATCH 3/9] =?UTF-8?q?fix(assistant):=20=E6=94=BE=E5=BC=83?= =?UTF-8?q?=E7=9A=84=E9=9F=B3=E9=A2=91=20id=20=E6=94=B9=E7=94=A8=E9=9B=86?= =?UTF-8?q?=E5=90=88=EF=BC=8C=E9=98=B2=E6=AD=A2=E8=BF=9E=E7=BB=AD=E6=89=93?= =?UTF-8?q?=E6=96=AD=E6=97=B6=E4=BA=92=E7=9B=B8=E8=A6=86=E7=9B=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit abandonedAudioId 之前是单个值:turn 1 的音频还没等到自己的 tts.end 就被 turn 2 自己的音频顶替、又被 turn 3 打断覆盖掉记录,turn 1 迟到 的 tts.end 落进正常分支,把已经推进到 turn 3 的 phase 错误地掰回 idle(PR #401 review 指出)。 改用 Set 记录所有还没等到 tts.end 的放弃 id,命中就从集合 里摘掉;cancelTurn() 里连接整个关闭时顺手清空,避免长会话攒垃圾。 补了对应的多轮打断回归测试。 --- .../AssistantConversationService.ts | 26 ++++--- .../AssistantConversationService.test.ts | 68 ++++++++++++++++++- 2 files changed, 82 insertions(+), 12 deletions(-) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index f765fb1b..d87d87b4 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -53,11 +53,13 @@ export class AssistantConversationService implements AssistantApplicationPort { * isCurrentTurn 当成"别人的"丢掉;点掉/取消气泡时清空,避免真正过期的 * 回复把已经点掉的气泡重新弹出来。 */ private displayedReplyRequestId: string | null = null; - /** 上一轮还没播完就被新一轮/dismissReply() 主动放弃的 audio_id。voice.tts.start/ - * voice.tts.end 协议里不带 request_id,没法像 dialogue.reply 那样按轮次门控, - * 只能靠这个记住"这条迟到的收尾消息属于已经不关心的那条音频",不让它把 - * phase 强行掰回 idle、打断已经在推进的新一轮状态。 */ - private abandonedAudioId: string | null = null; + /** 被新一轮/dismissReply() 主动放弃、但还没等到 tts.end 的 audio_id 集合。 + * voice.tts.start/voice.tts.end 协议里不带 request_id,没法像 dialogue.reply + * 那样按轮次门控,只能靠这个记住"这条迟到的收尾消息属于已经不关心的哪条 + * 音频",不让它把 phase 强行掰回 idle、打断已经在推进的新一轮状态。用 + * 集合而不是单个值:连续按两次,第一条还没等到 tts.end 就被第二条覆盖掉的话, + * 第一条迟到的 tts.end 会落进正常分支,把 phase 错误地掰回 idle。 */ + private readonly abandonedAudioIds = new Set(); /** 当前这一帧麦克风音量(dBFS),给波形展示;不在录音时是 null。 */ private soundLevel: number | null = null; // 展示层的 onPressIn/onPressOut 不等待彼此:快速按放会让 endTurn() 在 @@ -143,7 +145,7 @@ export class AssistantConversationService implements AssistantApplicationPort { if (this.currentAudioId !== null) { // 上一轮的语音还没播完,新一轮开始时主动掐掉:不能让两轮音频同时播, // 也不能放着让它的 tts.end 晚到后把 phase 强行掰回 idle。 - this.abandonedAudioId = this.currentAudioId; + this.abandonedAudioIds.add(this.currentAudioId); this.currentAudioId = null; void this.deps.playback.stop().catch(() => {}); } @@ -245,6 +247,9 @@ export class AssistantConversationService implements AssistantApplicationPort { this.replyText = null; this.displayedReplyRequestId = null; this.currentAudioId = null; + // 连接马上就要整个关掉,不会再有消息进来;清空避免长会话里攒一堆再也 + // 用不上的 id。 + this.abandonedAudioIds.clear(); const connection = this.connection; this.unsubscribeConnection?.(); @@ -266,8 +271,10 @@ export class AssistantConversationService implements AssistantApplicationPort { async dismissReply(): Promise { this.replyText = null; this.displayedReplyRequestId = null; - this.abandonedAudioId = this.currentAudioId; - this.currentAudioId = null; + if (this.currentAudioId !== null) { + this.abandonedAudioIds.add(this.currentAudioId); + this.currentAudioId = null; + } this.notifyListeners(); await this.deps.playback.stop().catch(() => {}); } @@ -395,10 +402,9 @@ export class AssistantConversationService implements AssistantApplicationPort { .catch(() => {}); return; case 'voice.tts.end': - if (message.audio_id === this.abandonedAudioId) { + if (this.abandonedAudioIds.delete(message.audio_id)) { // 已经主动放弃的那条音频,收尾消息迟到了:清掉记录就行,不能再碰 // phase——这时它早就是新一轮的状态,不能被这条迟到消息掰回 idle。 - this.abandonedAudioId = null; return; } this.currentAudioId = null; diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 271d14b2..0f91b019 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -868,7 +868,7 @@ describe('AssistantConversationService', () => { await completeStreamStart(fake, nextTurn); }); - it('does not let a stale tts.end from an abandoned turn clobber the new turn\'s phase', async () => { + it("does not let a stale tts.end from an abandoned turn clobber the new turn's phase", async () => { // 同类问题:voice.tts.start/voice.tts.end 协议里不带 request_id,没法像 // voice.dialogue.reply 那样按轮次门控。turn 1 的语音还没播完,用户又按住 // 说话开始 turn 2——turn 1 的音频应该被主动掐掉;turn 1 迟到的 tts.end @@ -911,7 +911,71 @@ describe('AssistantConversationService', () => { expect(service.getState()).toMatchObject({ phase: 'recording' }); }); - it('drops a stale dialogue.question from a previous turn so it cannot hijack the new turn\'s phase', async () => { + it('tracks every abandoned audio id, not just the most recently abandoned one', async () => { + // 连续按两次:turn 1 的音频 A 还没等到自己的 tts.end,就被 turn 2 自己的 + // 音频 B 顶替成"当前正在播的",然后 turn 2 也被 turn 3 打断——B 被放弃时 + // 如果放弃记录只存一个值,会把 A 那条覆盖掉,A 迟到的 tts.end 就会落进 + // 正常分支,把已经推进到 turn 3 的 phase 错误地掰回 idle。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + // turn 1:音频 A 开始播。 + await completeStreamStart(fake, service.startTurn()); + fake.emitMessage({ + audio_id: 'audio_a', + conversation_id: 'conv_001', + payload: { + format: 'pcm', + purpose: 'command_result', + sample_rate_hz: 24000, + speech_text: '', + }, + type: 'voice.tts.start', + } as AssistantServerMessage); + await flushAsync(); + + // turn 2 开始:A 被放弃。 + await completeStreamStart(fake, service.startTurn()); + + // turn 2 自己的音频 B 开始播。 + fake.emitMessage({ + audio_id: 'audio_b', + conversation_id: 'conv_001', + payload: { + format: 'pcm', + purpose: 'command_result', + sample_rate_hz: 24000, + speech_text: '', + }, + type: 'voice.tts.start', + } as AssistantServerMessage); + await flushAsync(); + + // turn 3 开始:B 也被放弃,此时 A 还没等到自己的 tts.end。 + await completeStreamStart(fake, service.startTurn()); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + + // A 迟到的 tts.end 到达:不能把已经推进到 turn 3 的 phase 掰回 idle。 + fake.emitMessage({ + audio_id: 'audio_a', + conversation_id: 'conv_001', + type: 'voice.tts.end', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + + // B 迟到的 tts.end 到达:同样不能把 phase 掰回 idle。 + fake.emitMessage({ + audio_id: 'audio_b', + conversation_id: 'conv_001', + type: 'voice.tts.end', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + }); + + it("drops a stale dialogue.question from a previous turn so it cannot hijack the new turn's phase", async () => { const fake = createFakeConnection(); const deps = createDeps({ connection: fake.connection }); const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); From 74d0aaec7cf5c7a0df4eb4d9a949b0f9a7fb93b4 Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 15:32:04 +0800 Subject: [PATCH 4/9] =?UTF-8?q?test(assistant):=20=E8=A1=A5=E4=B8=8A?= =?UTF-8?q?=E7=82=B9=E6=8E=89=E6=B0=94=E6=B3=A1=E6=97=B6=E8=AF=AD=E9=9F=B3?= =?UTF-8?q?=E6=AD=A3=E5=9C=A8=E6=92=AD=E6=94=BE=E7=9A=84=E8=A6=86=E7=9B=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit dismissReply() 里新增的放弃-音频逻辑(abandonedAudioIds.add / 清空 currentAudioId)之前没有测试覆盖——所有已有用例点掉气泡时都没有语音 正在播。补一个用例:语音播放中点掉气泡,确认立即停播且这条 audio_id 被记进放弃集合,迟到的 tts.end 不会把随后开始的新一轮状态掰回 idle。 --- .../AssistantConversationService.test.ts | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 0f91b019..0d11c1af 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -975,6 +975,46 @@ describe('AssistantConversationService', () => { expect(service.getState()).toMatchObject({ phase: 'recording' }); }); + it('abandons the currently-playing audio when dismissReply() is called mid-speech', async () => { + // dismissReply() 点掉气泡时如果语音还在播,也要走跟新一轮开始时一样的 + // 放弃流程:立即停播、把这条 audio_id 记进 abandonedAudioIds,不然它 + // 迟到的 tts.end 会落进正常分支,把点掉气泡之后的状态又碰一遍。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + await completeStreamStart(fake, service.startTurn()); + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + payload: { + format: 'pcm', + purpose: 'command_result', + sample_rate_hz: 24000, + speech_text: '', + }, + type: 'voice.tts.start', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toMatchObject({ phase: 'speaking' }); + + await service.dismissReply(); + expect(deps.playback.stop).toHaveBeenCalledTimes(1); + + // 点掉之后开始新一轮。 + await completeStreamStart(fake, service.startTurn()); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + + // 被放弃的那条音频迟到的 tts.end 到达:不能把 phase 掰回 idle。 + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + type: 'voice.tts.end', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getState()).toMatchObject({ phase: 'recording' }); + }); + it("drops a stale dialogue.question from a previous turn so it cannot hijack the new turn's phase", async () => { const fake = createFakeConnection(); const deps = createDeps({ connection: fake.connection }); From fd745dcb264e9be21235a4514460368f97136bcf Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 15:54:43 +0800 Subject: [PATCH 5/9] =?UTF-8?q?fix(assistant):=20=E5=86=8D=E6=AC=A1?= =?UTF-8?q?=E6=8C=89=E4=BD=8F=E8=AF=B4=E8=AF=9D=E6=94=B9=E5=9B=9E=E6=B8=85?= =?UTF-8?q?=E7=A9=BA=E6=B0=94=E6=B3=A1=E5=B9=B6=E5=81=9C=E6=92=AD=EF=BC=8C?= =?UTF-8?q?=E8=B7=9F=E7=82=B9=E6=8E=89=E6=B0=94=E6=B3=A1=E4=B8=80=E8=87=B4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 上一版把 _startTurn() 改成不清 replyText,让气泡跨轮次持续显示。产品 预期不是这样:再次按住说话应该等同于主动点掉气泡(dismissReply)—— 文字立即清空、语音立即停播,然后干净地开始听新一段,不应该让上一轮 内容跟新一轮混在一起。 _startTurn() 里重新加回 this.replyText = null(放弃音频那部分不动)。 删掉 displayedReplyRequestId 字段和相关逻辑——它是上一版"气泡不清空" 设计专用的(让已显示的气泡继续吃旧轮次的流式增量),现在用不上了。 voice.dialogue.reply 门控简化回单纯的 isCurrentTurn(requestId)。 --- .../AssistantConversationService.ts | 32 ++++------- .../AssistantConversationService.test.ts | 53 +++++++++---------- 2 files changed, 33 insertions(+), 52 deletions(-) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index d87d87b4..504144b0 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -48,11 +48,6 @@ export class AssistantConversationService implements AssistantApplicationPort { private replyText: string | null = null; /** 当前 turn 的 request_id;voice.dialogue.reply 拿它认轮次,对不上的是上一轮迟到的。 */ private turnRequestId: string | null = null; - /** 当前气泡里显示的是哪个 request_id 的回复。新一轮开始不再清 replyText, - * 所以上一轮剩余的流式增量(同一个 request_id)应该继续更新它,不能被 - * isCurrentTurn 当成"别人的"丢掉;点掉/取消气泡时清空,避免真正过期的 - * 回复把已经点掉的气泡重新弹出来。 */ - private displayedReplyRequestId: string | null = null; /** 被新一轮/dismissReply() 主动放弃、但还没等到 tts.end 的 audio_id 集合。 * voice.tts.start/voice.tts.end 协议里不带 request_id,没法像 dialogue.reply * 那样按轮次门控,只能靠这个记住"这条迟到的收尾消息属于已经不关心的哪条 @@ -137,10 +132,10 @@ export class AssistantConversationService implements AssistantApplicationPort { } private async _startTurn(turnId: number): Promise { - // 不清 replyText:上一轮的回复可能还在流式(用户在它说完前又按住说话), - // 这里清掉会让已经在显示的气泡瞬间消失,即使语音还在正常播(#392 同类问题 - // 在按住说话这条路径上的版本)。旧气泡留到被新一轮自己的回复覆盖,或者 - // 用户点掉/取消为止。 + // 新一轮开始等于用户主动放弃上一轮,跟点掉气泡(dismissReply)是同一个 + // 动作:文字气泡立即清空、语音立即停播,然后干净地开始听新一段——不能 + // 留着上一轮的气泡/声音跟新一轮混在一起。 + this.replyText = null; this.soundLevel = null; if (this.currentAudioId !== null) { // 上一轮的语音还没播完,新一轮开始时主动掐掉:不能让两轮音频同时播, @@ -245,7 +240,6 @@ export class AssistantConversationService implements AssistantApplicationPort { this.pendingStartTurn = null; this.soundLevel = null; this.replyText = null; - this.displayedReplyRequestId = null; this.currentAudioId = null; // 连接马上就要整个关掉,不会再有消息进来;清空避免长会话里攒一堆再也 // 用不上的 id。 @@ -270,7 +264,6 @@ export class AssistantConversationService implements AssistantApplicationPort { async dismissReply(): Promise { this.replyText = null; - this.displayedReplyRequestId = null; if (this.currentAudioId !== null) { this.abandonedAudioIds.add(this.currentAudioId); this.currentAudioId = null; @@ -374,23 +367,16 @@ export class AssistantConversationService implements AssistantApplicationPort { speechText: message.payload.speech_text, }); return; - case 'voice.dialogue.reply': { - // 只认当前这一轮,或者当前气泡本来就在显示的那一轮——后者是为了让 - // "上一轮回复还在流式时用户又按住说话"这种情况下,上一轮剩余的增量 - // 还能继续更新它已经显示出来的气泡,不被当成别的轮次的内容丢掉。 - // 除此之外一律当成上一轮被打断后晚到的、已经点掉的旧气泡,不能让它 - // 重新弹出来——气泡是全屏点击层,会挡住语音条,用户得再点一次才能说话。 - const requestId = message.request_id; - if (!this.isCurrentTurn(requestId) && requestId !== this.displayedReplyRequestId) { + case 'voice.dialogue.reply': + // 上一轮被新一轮打断(或被点掉)时它的 reply 可能晚到,把已经清空的 + // 气泡重新弹出来——气泡是全屏点击层,会挡住语音条,用户得再点一次 + // 才能说话。只认当前这一轮的 request_id。 + if (!this.isCurrentTurn(message.request_id)) { return; } this.replyText = message.payload.speech_text; - if (typeof requestId === 'string') { - this.displayedReplyRequestId = requestId; - } this.notifyListeners(); return; - } case 'voice.tts.start': this.currentAudioId = message.audio_id; this.setState({ conversationId: message.conversation_id, phase: 'speaking' }); diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 0d11c1af..6aac4ae3 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -794,27 +794,15 @@ describe('AssistantConversationService', () => { ); }); - it('keeps a streaming reply visible when a new press starts mid-reply, instead of dropping it while TTS keeps playing', async () => { - // 用户报告:PTT 长回复时气泡完全不出现,但语音照常播放。序列是—— - // turn 1 的回复还在流式(文本增量 + TTS 并行下发),语音播放中用户又按住 - // 说话(PushToTalkBar 在 speaking/awaiting_result 期间不禁用自己): - // 1. 新一轮 _startTurn 先清空 replyText(旧气泡消失); - // 2. turnRequestId 换新,turn 1 剩余的流式 reply 增量按 request_id 被 - // #364 gate 全部丢弃; - // 3. 结果 replyText 永远为空——气泡不出现;而 voice.tts.start 不经过 - // gate,TTS 照常播放。 - // - // 连续模式不受影响:它的 reply 写进 turns,不按 request_id 门控。这也是 - // 用户说"只有 PTT 模式有这个问题"的原因。 - // - // 期望行为(本测试锁定):reply 一旦到达就应该留在气泡里——新一轮开始 - // 不应该把上一轮已经收到的回复文本清掉,流式增量也不应该被 request_id - // gate 吞掉。修复前本用例失败(replyText 被清空且增量被丢弃)。 + it('clears the bubble and stops the audio immediately when a new press starts mid-reply', async () => { + // 产品预期:再次按住说话等于主动放弃上一轮,跟点掉气泡(dismissReply) + // 是同一个动作——文字气泡立即清空、语音立即停播,然后干净地开始听 + // 新一段,不应该把上一轮的内容留着跟新一轮混在一起。 const fake = createFakeConnection(); const deps = createDeps({ connection: fake.connection }); const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); - // turn 1:回复开始流式(第一条增量已显示)。 + // turn 1:回复开始流式,TTS 也开始播。 await completeStreamStart(fake, service.startTurn()); const turn1Id = fake.sent.filter((message) => message.type === 'voice.stream.start')[0] .request_id; @@ -824,10 +812,6 @@ describe('AssistantConversationService', () => { payload: { done: false, reply_id: 'reply_1', speech_text: '好的,明天下午' }, type: 'voice.dialogue.reply', } as AssistantServerMessage); - await flushAsync(); - expect(service.getReplyText()).toBe('好的,明天下午'); - - // TTS 开始播(不受 request_id gate 影响)。 fake.emitMessage({ audio_id: 'audio_1', conversation_id: 'conv_001', @@ -840,16 +824,17 @@ describe('AssistantConversationService', () => { type: 'voice.tts.start', } as AssistantServerMessage); await flushAsync(); + expect(service.getReplyText()).toBe('好的,明天下午'); expect(deps.playback.startStream).toHaveBeenCalledTimes(1); - // 语音还在播,用户又按住说话 → 新一轮开始。已收到的回复文本应保留在 - // 气泡里(BUG:_startTurn 开头把 replyText 清成 null)。 + // 语音还在播,用户又按住说话 → 新一轮开始:气泡立即清空,语音立即停播。 const nextTurn = service.startTurn(); await flushAsync(); - expect(service.getReplyText()).toBe('好的,明天下午'); + expect(service.getReplyText()).toBeNull(); + expect(deps.playback.stop).toHaveBeenCalledTimes(1); - // turn 1 剩余的流式 reply 增量(同一个 request_id)晚到 → 应继续更新气泡 - // (BUG:request_id 已换新,增量被 #364 gate 丢弃)。 + // turn 1 剩余的流式增量(还是旧 request_id)晚到:已经被放弃的这一轮, + // 不能让它重新把气泡填回去。 fake.emitMessage({ conversation_id: 'conv_001', request_id: turn1Id, @@ -861,11 +846,21 @@ describe('AssistantConversationService', () => { type: 'voice.dialogue.reply', } as AssistantServerMessage); await flushAsync(); - expect(service.getReplyText()).toBe( - '好的,明天下午三点在203会议室开会,我会提前十五分钟提醒你', - ); + expect(service.getReplyText()).toBeNull(); await completeStreamStart(fake, nextTurn); + + // turn 2 自己的回复到了才显示。 + const turn2Id = fake.sent.filter((message) => message.type === 'voice.stream.start')[1] + .request_id; + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turn2Id, + payload: { done: true, reply_id: 'reply_2', speech_text: '好的,帮你查一下' }, + type: 'voice.dialogue.reply', + } as AssistantServerMessage); + await flushAsync(); + expect(service.getReplyText()).toBe('好的,帮你查一下'); }); it("does not let a stale tts.end from an abandoned turn clobber the new turn's phase", async () => { From 892c130d5dc80893f356f8a20d1088b7bb181be8 Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 16:18:41 +0800 Subject: [PATCH 6/9] =?UTF-8?q?fix(assistant):=20=E5=86=8D=E6=8C=89?= =?UTF-8?q?=E4=BD=8F=E8=AF=B4=E8=AF=9D=E6=97=B6=20stop()=20=E6=94=B9?= =?UTF-8?q?=E6=88=90=E6=97=A0=E6=9D=A1=E4=BB=B6=E8=B0=83=E7=94=A8=EF=BC=8C?= =?UTF-8?q?=E4=B8=8D=E5=86=8D=E8=A2=AB=E6=BC=8F=E6=8E=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 之前 stop() 放在 if (currentAudioId !== null) 里面,只有当前正追踪着 一个 audio_id 时才会调用。长回复的 TTS 可能分好几段下发(一句一个 tts.start/tts.end),如果按下的瞬间恰好卡在上一段 tts.end 和下一段 tts.start 之间,currentAudioId 这时候是 null,stop() 直接被跳过—— 原生播放器完全没收到停止指令,会正常播完当前缓冲区里的音频、甚至 接着播下一段。dismissReply() 里 stop() 本来就是无条件调用的,这处 不一致正是"点气泡能停、按住说话不能停"的根因。 现在 stop() 挪到 if 外面,跟 dismissReply() 保持一致;abandonedAudioIds 的记录仍然只在 currentAudioId 非空时才添加(没有 id 可记就不记)。 --- .../AssistantConversationService.ts | 6 ++- .../AssistantConversationService.test.ts | 43 +++++++++++++++++++ 2 files changed, 48 insertions(+), 1 deletion(-) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index 504144b0..d63f6f01 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -142,8 +142,12 @@ export class AssistantConversationService implements AssistantApplicationPort { // 也不能放着让它的 tts.end 晚到后把 phase 强行掰回 idle。 this.abandonedAudioIds.add(this.currentAudioId); this.currentAudioId = null; - void this.deps.playback.stop().catch(() => {}); } + // stop() 放在 if 外面、无条件调用(跟 dismissReply() 一致):长回复的 TTS + // 可能分好几段下发,如果按下的瞬间恰好卡在上一段 tts.end 和下一段 + // tts.start 之间,currentAudioId 这时候是 null,放在 if 里面的话 stop() + // 会被跳过,原生播放器收不到停止指令,只会眼睁睁看着它继续播下一段。 + void this.deps.playback.stop().catch(() => {}); const previousCaptureCleanup = this.captureCleanup; if (this.connection === null) { this.setState({ phase: 'connecting' }); diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 6aac4ae3..88af8204 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -826,6 +826,7 @@ describe('AssistantConversationService', () => { await flushAsync(); expect(service.getReplyText()).toBe('好的,明天下午'); expect(deps.playback.startStream).toHaveBeenCalledTimes(1); + jest.mocked(deps.playback.stop).mockClear(); // 语音还在播,用户又按住说话 → 新一轮开始:气泡立即清空,语音立即停播。 const nextTurn = service.startTurn(); @@ -887,6 +888,7 @@ describe('AssistantConversationService', () => { } as AssistantServerMessage); await flushAsync(); expect(service.getState()).toMatchObject({ phase: 'speaking' }); + jest.mocked(deps.playback.stop).mockClear(); // 语音还没播完,用户又按住说话 → turn 2 开始:应该主动掐掉 turn 1 的播放。 const nextTurn = service.startTurn(); @@ -992,6 +994,7 @@ describe('AssistantConversationService', () => { } as AssistantServerMessage); await flushAsync(); expect(service.getState()).toMatchObject({ phase: 'speaking' }); + jest.mocked(deps.playback.stop).mockClear(); await service.dismissReply(); expect(deps.playback.stop).toHaveBeenCalledTimes(1); @@ -1010,6 +1013,46 @@ describe('AssistantConversationService', () => { expect(service.getState()).toMatchObject({ phase: 'recording' }); }); + it('still stops playback when a new press lands in the gap between two TTS segments', async () => { + // 长回复的 TTS 可能分好几段下发(一句一个 tts.start/tts.end)。如果按下 + // 的瞬间恰好卡在上一段 tts.end 和下一段 tts.start 之间,currentAudioId + // 这时候是 null——stop() 之前是放在 `if (currentAudioId !== null)` 里面 + // 调用的,这种情况下会被跳过,原生播放器收不到停止指令,眼睁睁看着它 + // 继续播下一段。stop() 现在挪到 if 外面、无条件调用,不依赖这个判断。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + await completeStreamStart(fake, service.startTurn()); + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + payload: { + format: 'pcm', + purpose: 'command_result', + sample_rate_hz: 24000, + speech_text: '', + }, + type: 'voice.tts.start', + } as AssistantServerMessage); + await flushAsync(); + + // 第一段说完,收尾——currentAudioId 回到 null,但原生播放器可能还没真正 + // 播完硬件缓冲区里的音频,下一段的 tts.start 也还没到。 + fake.emitMessage({ + audio_id: 'audio_1', + conversation_id: 'conv_001', + type: 'voice.tts.end', + } as AssistantServerMessage); + await flushAsync(); + jest.mocked(deps.playback.stop).mockClear(); + + // 恰好在这个间隙按住说话:即使没有正在追踪的 audio_id,也要把 stop() + // 发出去。 + await completeStreamStart(fake, service.startTurn()); + expect(deps.playback.stop).toHaveBeenCalledTimes(1); + }); + it("drops a stale dialogue.question from a previous turn so it cannot hijack the new turn's phase", async () => { const fake = createFakeConnection(); const deps = createDeps({ connection: fake.connection }); From 873ba78bdd86542844bf98519957357a900c2357 Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 16:19:00 +0800 Subject: [PATCH 7/9] =?UTF-8?q?fix(patch):=20AudioPlaybackManager.stopPlay?= =?UTF-8?q?back()=20=E7=A9=BA=E9=98=9F=E5=88=97=E4=B9=9F=E8=A6=81=E7=9C=9F?= =?UTF-8?q?=E7=9A=84=E5=81=9C=E6=92=AD?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit playbackChannel 是还没解码写入 AudioTrack 的排队分片,流式播放时它 经常是空的(解码速度快于播放速度),空不代表 AudioTrack 硬件缓冲区 里没有还在播的音频。原实现把它也当"没在播"的判据,导致 stop() 在 这种(很常见的)时刻直接跳过 audioTrack.stop()/flush(),调用方以为 已经停了,实际上已经写进硬件缓冲区的音频还会继续播完。 判据改成只看 isPlaying。这是通过 patch-package 打的补丁,会在 npm install 时自动重新应用。 --- .../@irvingouj+expo-audio-stream+3.1.0.patch | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/frontend/patches/@irvingouj+expo-audio-stream+3.1.0.patch b/frontend/patches/@irvingouj+expo-audio-stream+3.1.0.patch index 6936bf65..2664de01 100644 --- a/frontend/patches/@irvingouj+expo-audio-stream+3.1.0.patch +++ b/frontend/patches/@irvingouj+expo-audio-stream+3.1.0.patch @@ -21,6 +21,24 @@ index c3038e4..dde8fe9 100644 } namespace "expo.modules.audiostream" +diff --git a/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioPlaybackManager.kt b/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioPlaybackManager.kt +index 6783bba..3cffedc 100644 +--- a/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioPlaybackManager.kt ++++ b/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioPlaybackManager.kt +@@ -289,7 +289,12 @@ class AudioPlaybackManager(private val eventSender: EventSender? = null) { + + fun stopPlayback(promise: Promise? = null) { + Log.d("ExpoPlayStreamModule", "Stopping playback") +- if (!isPlaying || playbackChannel.isEmpty ) { ++ // playbackChannel 只是还没解码写入 AudioTrack 的排队分片;它在正常 ++ // 流式播放中经常是空的(解码速度快于播放速度),空不代表 AudioTrack ++ // 硬件缓冲区里没有还在播的音频。原来把它也当"没在播"的判据,会让 ++ // stop() 在这种(很常见的)时刻直接跳过 audioTrack.stop()/flush(), ++ // 调用方以为已经停了,实际上已经写进硬件缓冲区的音频还会继续播完。 ++ if (!isPlaying) { + promise?.resolve(null) + Log.d("ExpoPlayStreamModule", "Nothing is played return") + return diff --git a/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioRecorderManager.kt b/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioRecorderManager.kt index 9ee227e..0373656 100644 --- a/node_modules/@irvingouj/expo-audio-stream/android/src/main/java/expo/modules/audiostream/AudioRecorderManager.kt From 883069905e666509091c4881f58127eb81873718 Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 16:42:46 +0800 Subject: [PATCH 8/9] =?UTF-8?q?fix(assistant):=20=E7=A1=AE=E8=AE=A4?= =?UTF-8?q?=E7=B1=BB=E8=BF=BD=E9=97=AE=E4=B9=9F=E8=A6=81=E5=86=99=E8=BF=9B?= =?UTF-8?q?=20replyText=EF=BC=8C=E4=B8=8D=E7=84=B6=E6=B0=94=E6=B3=A1?= =?UTF-8?q?=E6=B0=B8=E8=BF=9C=E4=B8=8D=E5=87=BA=E7=8E=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit voice.dialogue.question(比如"删除某一个"触发的"确认要删除吗?")之前 只写 state.speechText,气泡 UI(AssistantVoiceOverlay)只读 replyText—— 两个完全独立的字段,追问从来没被接进气泡显示逻辑。语音照常播(走 voice.tts.start,跟这个无关),但气泡永远不会出现,也没法点掉。这跟 打断/连续按没有关系,是这条消息类型本身漏了这一步。 现在 voice.dialogue.question 也把 speech_text 写进 replyText。 --- .../AssistantConversationService.ts | 4 +++ .../AssistantConversationService.test.ts | 30 +++++++++++++++++++ 2 files changed, 34 insertions(+) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index d63f6f01..d7adfbeb 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -365,6 +365,10 @@ export class AssistantConversationService implements AssistantApplicationPort { if (!this.isCurrentTurn(message.request_id)) { return; } + // 追问也要写进 replyText:气泡 UI 只读这个字段,state.speechText 只是 + // phase 内部携带的文案,不接气泡——之前只写 speechText,语音照常播 + // (走 voice.tts.start,跟这个无关),但追问永远不会显示成气泡。 + this.replyText = message.payload.speech_text; this.setState({ conversationId: message.conversation_id, phase: 'asking', diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 88af8204..6ee2ebb7 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -1053,6 +1053,36 @@ describe('AssistantConversationService', () => { expect(deps.playback.stop).toHaveBeenCalledTimes(1); }); + it('shows a clarifying question as the reply bubble, even though TTS plays it independently', async () => { + // 用户报告:问"有什么"能看到气泡,问"删除某一个"(触发确认类追问) + // 听得见语音但看不见气泡,也点不了。voice.dialogue.question 之前只写 + // state.speechText,气泡 UI(AssistantVoiceOverlay)只读 replyText—— + // 两个完全独立的字段,追问从来没被接进气泡显示逻辑。这跟打断/连续按 + // 没有关系,是这条消息类型本身漏了这一步。 + const fake = createFakeConnection(); + const deps = createDeps({ connection: fake.connection }); + const service = new AssistantConversationService({ accountId: 'acc_001' }, deps); + + await completeStreamStart(fake, service.startTurn()); + const turnId = fake.sent.filter((message) => message.type === 'voice.stream.start')[0] + .request_id; + fake.emitMessage({ + conversation_id: 'conv_001', + request_id: turnId, + payload: { + candidates: [], + question_id: 'q_1', + question_kind: 'confirmation', + speech_text: '确认要删除这条日程吗?', + }, + type: 'voice.dialogue.question', + } as AssistantServerMessage); + await flushAsync(); + + expect(service.getReplyText()).toBe('确认要删除这条日程吗?'); + expect(service.getState()).toMatchObject({ phase: 'asking' }); + }); + it("drops a stale dialogue.question from a previous turn so it cannot hijack the new turn's phase", async () => { const fake = createFakeConnection(); const deps = createDeps({ connection: fake.connection }); From 12b891645c10da9f8e2c29dd2afa93cf8cf5011f Mon Sep 17 00:00:00 2001 From: LUPENGHAN Date: Thu, 27 Aug 2026 17:14:07 +0800 Subject: [PATCH 9/9] =?UTF-8?q?fix(assistant):=20=E6=96=B0=E4=B8=80?= =?UTF-8?q?=E8=BD=AE=E5=BC=80=E5=A7=8B=E7=9A=84=20stop=20=E6=8E=A5?= =?UTF-8?q?=E5=85=A5=20playbackChain/generation=20=E5=BA=8F=E5=88=97?= =?UTF-8?q?=E5=8C=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit main 上合并进来的 #399 给播放操作加了 playbackChain/playbackGeneration 序列化(避免 pushChunk 分片交错、stop 和在途写入竞争)。这条分支里 _startTurn() 开始新一轮时的 stop 调用还是合并前的旧写法,直接调 deps.playback.stop(),绕过了这套序列化——会跟 chainPlayback 里在途的 pushChunk 竞争,且不会让已经排队但还没执行的旧流分片失效。 改成调 stopPlaybackImmediately(),跟 dismissReply()/cancelTurn() 保持 一致。同时修一条现有测试的断言:这次改动让 stop 在每次 _startTurn() 都会被调用一次(即使没有正在播的音频,对应原生层已经修过的空操作 分支),测试原来只预期 dismiss 触发的那一次。 --- .../application/AssistantConversationService.ts | 11 +++++++---- .../application/AssistantConversationService.test.ts | 1 + 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/frontend/src/features/assistant/application/AssistantConversationService.ts b/frontend/src/features/assistant/application/AssistantConversationService.ts index 74f93655..6e955cec 100644 --- a/frontend/src/features/assistant/application/AssistantConversationService.ts +++ b/frontend/src/features/assistant/application/AssistantConversationService.ts @@ -150,11 +150,14 @@ export class AssistantConversationService implements AssistantApplicationPort { this.abandonedAudioIds.add(this.currentAudioId); this.currentAudioId = null; } - // stop() 放在 if 外面、无条件调用(跟 dismissReply() 一致):长回复的 TTS + // stop 放在 if 外面、无条件调用(跟 dismissReply() 一致):长回复的 TTS // 可能分好几段下发,如果按下的瞬间恰好卡在上一段 tts.end 和下一段 - // tts.start 之间,currentAudioId 这时候是 null,放在 if 里面的话 stop() - // 会被跳过,原生播放器收不到停止指令,只会眼睁睁看着它继续播下一段。 - void this.deps.playback.stop().catch(() => {}); + // tts.start 之间,currentAudioId 这时候是 null,放在 if 里面的话 stop + // 会被跳过,原生播放器收不到停止指令,只会眼睁睁看着它继续播下一段。走 + // stopPlaybackImmediately() 而不是直接调 deps.playback.stop():后者绕过了 + // playbackChain/generation,会跟 chainPlayback 里在途的 pushChunk 竞争, + // 已经排队但还没执行的旧流分片也不会被这次 stop 作废。 + void this.stopPlaybackImmediately(); const previousCaptureCleanup = this.captureCleanup; if (this.connection === null) { this.setState({ phase: 'connecting' }); diff --git a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts index 6b0734cb..321020c3 100644 --- a/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts +++ b/frontend/tests/unit/features/assistant/application/AssistantConversationService.test.ts @@ -594,6 +594,7 @@ describe('AssistantConversationService', () => { fake.emitAudioFrame(new ArrayBuffer(4)); await flushAsync(); expect(pushOrder).toEqual(['push-1-start']); + jest.mocked(deps.playback.stop).mockClear(); const dismiss = service.dismissReply(); resolveFirst();