From 66186e70253cde54f41a82d747aeabd6d1029d35 Mon Sep 17 00:00:00 2001 From: conrad mugabe Date: Sun, 26 Jul 2026 23:30:55 +0200 Subject: [PATCH 01/14] style: apply prettier formatting to CHANGELOG Pre-existing formatting drift surfaced by the pre-commit hook, which runs prettier across the whole codebase. Isolated here so it does not pollute the functional commits that follow. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d2fb7a22..966cd4ef 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,13 +4,13 @@ ### Tests -* **e2e:** deflake journey 62 csd-02 against backend auto-titling ([dce701c](https://github.com/iblai/os/commit/dce701c881163a586c076c873dfb4ff97f1b1ae6)) +- **e2e:** deflake journey 62 csd-02 against backend auto-titling ([dce701c](https://github.com/iblai/os/commit/dce701c881163a586c076c873dfb4ff97f1b1ae6)) ## [0.102.4](https://github.com/iblai/os/compare/v0.102.3...v0.102.4) (2026-07-24) ### Bug Fixes -* **mentor:** adding fix for the RBAC permissions ([5948b7b](https://github.com/iblai/os/commit/5948b7b86189dc54bce06d4468238f6d0ffbec4a)) +- **mentor:** adding fix for the RBAC permissions ([5948b7b](https://github.com/iblai/os/commit/5948b7b86189dc54bce06d4468238f6d0ffbec4a)) ## [0.102.3](https://github.com/iblai/os/compare/v0.102.2...v0.102.3) (2026-07-24) From 7e2c8851b717aa77588187daa64892bf39ab54b9 Mon Sep 17 00:00:00 2001 From: conrad mugabe Date: Sun, 26 Jul 2026 23:34:54 +0200 Subject: [PATCH 02/14] chore(i18n): add voice call status and transcript strings Adds the message keys used by the voice call modal's status caption and live transcript band: call status states, speaker labels, the empty state and the jump-to-latest affordance. Translated across en, es, fr and zh. Co-Authored-By: Claude Opus 5 (1M context) --- messages/en.json | 17 ++++++++++++++++- messages/es.json | 17 ++++++++++++++++- messages/fr.json | 17 ++++++++++++++++- messages/zh.json | 17 ++++++++++++++++- 4 files changed, 64 insertions(+), 4 deletions(-) diff --git a/messages/en.json b/messages/en.json index 7b312ae0..d4e8411a 100644 --- a/messages/en.json +++ b/messages/en.json @@ -1355,7 +1355,22 @@ "unmute": "Unmute", "mute": "Mute", "closeVoiceChat": "Close voice chat", - "endCall": "End Call" + "endCall": "End Call", + "muteAgentAudio": "Mute agent audio", + "unmuteAgentAudio": "Unmute agent audio", + "muteAgent": "Mute agent", + "unmuteAgent": "Unmute agent", + "agentSpeaking": "Agent speaking", + "agentMuted": "Agent muted", + "callStatus": "Call status", + "micMuted": "Mic muted", + "listening": "Listening…", + "callTranscript": "Call transcript", + "transcriptEmpty": "The live transcript will appear here as the conversation starts.", + "transcriptSpeakerYou": "You", + "transcriptSpeakerAgent": "Agent", + "transcriptLive": "Live", + "jumpToLatest": "Jump to latest" }, "modelDownloadModelDownloadStatus": { "downloading": "Downloading...", diff --git a/messages/es.json b/messages/es.json index 786b033a..a9b1108e 100644 --- a/messages/es.json +++ b/messages/es.json @@ -1355,7 +1355,22 @@ "unmute": "Activar", "mute": "Silenciar", "closeVoiceChat": "Cerrar chat de voz", - "endCall": "Finalizar llamada" + "endCall": "Finalizar llamada", + "muteAgentAudio": "Silenciar audio del agente", + "unmuteAgentAudio": "Activar audio del agente", + "muteAgent": "Silenciar agente", + "unmuteAgent": "Activar agente", + "agentSpeaking": "Agente hablando", + "agentMuted": "Agente silenciado", + "callStatus": "Estado de la llamada", + "micMuted": "Micrófono silenciado", + "listening": "Escuchando…", + "callTranscript": "Transcripción de la llamada", + "transcriptEmpty": "La transcripción en directo aparecerá aquí cuando comience la conversación.", + "transcriptSpeakerYou": "Tú", + "transcriptSpeakerAgent": "Agente", + "transcriptLive": "En directo", + "jumpToLatest": "Ir a lo más reciente" }, "modelDownloadModelDownloadStatus": { "downloading": "Descargando...", diff --git a/messages/fr.json b/messages/fr.json index 043bbcf3..970c793c 100644 --- a/messages/fr.json +++ b/messages/fr.json @@ -1355,7 +1355,22 @@ "unmute": "Réactiver", "mute": "Couper", "closeVoiceChat": "Fermer le chat vocal", - "endCall": "Terminer l'appel" + "endCall": "Terminer l'appel", + "muteAgentAudio": "Couper l'audio de l'agent", + "unmuteAgentAudio": "Réactiver l'audio de l'agent", + "muteAgent": "Couper l'agent", + "unmuteAgent": "Réactiver l'agent", + "agentSpeaking": "L'agent parle", + "agentMuted": "Agent coupé", + "callStatus": "État de l'appel", + "micMuted": "Microphone coupé", + "listening": "À l'écoute…", + "callTranscript": "Transcription de l'appel", + "transcriptEmpty": "La transcription en direct s'affichera ici dès le début de la conversation.", + "transcriptSpeakerYou": "Vous", + "transcriptSpeakerAgent": "Agent", + "transcriptLive": "En direct", + "jumpToLatest": "Aller au plus récent" }, "modelDownloadModelDownloadStatus": { "downloading": "Téléchargement...", diff --git a/messages/zh.json b/messages/zh.json index f3934886..3ce58d88 100644 --- a/messages/zh.json +++ b/messages/zh.json @@ -1355,7 +1355,22 @@ "unmute": "取消静音", "mute": "静音", "closeVoiceChat": "关闭语音聊天", - "endCall": "结束通话" + "endCall": "结束通话", + "muteAgentAudio": "静音智能体音频", + "unmuteAgentAudio": "取消静音智能体音频", + "muteAgent": "静音智能体", + "unmuteAgent": "取消静音智能体", + "agentSpeaking": "智能体正在讲话", + "agentMuted": "智能体已静音", + "callStatus": "通话状态", + "micMuted": "麦克风已静音", + "listening": "正在聆听…", + "callTranscript": "通话文字记录", + "transcriptEmpty": "对话开始后,实时文字记录将显示在这里。", + "transcriptSpeakerYou": "您", + "transcriptSpeakerAgent": "智能体", + "transcriptLive": "实时", + "jumpToLatest": "跳到最新" }, "modelDownloadModelDownloadStatus": { "downloading": "下载中...", From b6827c1a6c759bdd91cd334ece13ab1fd1b5ebfc Mon Sep 17 00:00:00 2001 From: conrad mugabe Date: Sun, 26 Jul 2026 23:39:51 +0200 Subject: [PATCH 03/14] feat(voice): add useLiveKitTranscription hook Subscribes to RoomEvent.TranscriptionReceived and accumulates an ordered transcript for the lifetime of a call. Unlike the screen-sharing relay, which holds a single utterance and clears it five seconds after it finalises, this keeps the full session so the user can read back what was said. - segment ids update their entry in place as speech grows; unknown ids append - isFinal is sticky, so a late partial cannot reopen a finalised line - timestamps are first-seen, so entries never reorder while updating - speaker attribution compares against the local participant identity rather than falling back to the mentor name first, which would label every remote speaker as the mentor - segment joining tolerates both cumulative and disjoint server behaviour - every event is logged so an empty transcript can be diagnosed as agent silence rather than a UI fault Not yet wired into the modal; the screen-sharing relay is left untouched and marked as a convergence opportunity. Co-Authored-By: Claude Opus 5 (1M context) --- .../use-livekit-transcription.test.ts | 542 ++++++++++++++++++ hooks/use-livekit-transcription.ts | 244 ++++++++ 2 files changed, 786 insertions(+) create mode 100644 hooks/__tests__/use-livekit-transcription.test.ts create mode 100644 hooks/use-livekit-transcription.ts diff --git a/hooks/__tests__/use-livekit-transcription.test.ts b/hooks/__tests__/use-livekit-transcription.test.ts new file mode 100644 index 00000000..f7172a4f --- /dev/null +++ b/hooks/__tests__/use-livekit-transcription.test.ts @@ -0,0 +1,542 @@ +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { act, renderHook } from '@testing-library/react'; + +// The hook only needs the RoomEvent enum from livekit-client; everything else +// it touches is a plain object we hand it. +vi.mock('livekit-client', () => ({ + RoomEvent: { + TranscriptionReceived: 'transcriptionReceived', + }, +})); + +import { + DEFAULT_MAX_TRANSCRIPT_ENTRIES, + joinTranscriptionSegments, + useLiveKitTranscription, +} from '../use-livekit-transcription'; + +const TRANSCRIPTION_EVENT = 'transcriptionReceived'; + +type Handler = (...args: unknown[]) => void; + +function createRoom(localIdentity: string | undefined = 'local-user') { + const handlers: Record = {}; + + return { + state: 'connected', + localParticipant: localIdentity ? { identity: localIdentity } : undefined, + on: vi.fn((event: string, handler: Handler) => { + handlers[event] = [...(handlers[event] ?? []), handler]; + }), + off: vi.fn((event: string, handler: Handler) => { + handlers[event] = (handlers[event] ?? []).filter((h) => h !== handler); + }), + listenerCount: (event: string) => (handlers[event] ?? []).length, + emit: (event: string, ...args: unknown[]) => + (handlers[event] ?? []).forEach((handler) => handler(...args)), + }; +} + +type FakeRoom = ReturnType; + +function segment( + id: string, + text: string, + final = false, +): Record { + return { id, text, final, startTime: 0, endTime: 0, language: 'en' }; +} + +const localParticipant = { identity: 'local-user', name: 'Me', isLocal: true }; +const agentParticipant = { + identity: 'agent-7', + name: 'Agent Seven', + isLocal: false, +}; + +const render = (room: FakeRoom | null, options: Record = {}) => + renderHook( + (props: Record) => useLiveKitTranscription(props as any), + { initialProps: { room, ...options } }, + ); + +function emit( + room: FakeRoom, + segments: unknown, + participant?: Record, +) { + act(() => { + room.emit(TRANSCRIPTION_EVENT, segments, participant); + }); +} + +describe('useLiveKitTranscription', () => { + let logSpy: ReturnType; + let warnSpy: ReturnType; + + beforeEach(() => { + logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}); + warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {}); + }); + + afterEach(() => { + logSpy.mockRestore(); + warnSpy.mockRestore(); + vi.clearAllMocks(); + }); + + describe('subscription lifecycle', () => { + it('subscribes to TranscriptionReceived on mount and logs it', () => { + const room = createRoom(); + render(room); + + expect(room.on).toHaveBeenCalledWith( + TRANSCRIPTION_EVENT, + expect.any(Function), + ); + expect(logSpy).toHaveBeenCalledWith( + '[VoiceChat:Transcription]', + 'Subscribed to RoomEvent.TranscriptionReceived', + expect.objectContaining({ + roomState: 'connected', + localParticipant: 'local-user', + }), + ); + }); + + it('removes the listener on unmount', () => { + const room = createRoom(); + const { unmount } = render(room); + + expect(room.listenerCount(TRANSCRIPTION_EVENT)).toBe(1); + + unmount(); + + expect(room.off).toHaveBeenCalledWith( + TRANSCRIPTION_EVENT, + expect.any(Function), + ); + expect(room.listenerCount(TRANSCRIPTION_EVENT)).toBe(0); + expect(logSpy).toHaveBeenCalledWith( + '[VoiceChat:Transcription]', + 'Unsubscribed from RoomEvent.TranscriptionReceived', + ); + }); + + it('does not update state after unmount', () => { + const room = createRoom(); + const { unmount } = render(room); + unmount(); + + // Nothing is listening anymore, so this is a no-op rather than a + // "setState on an unmounted component" warning. + expect(() => + room.emit(TRANSCRIPTION_EVENT, [segment('s1', 'ghost', true)]), + ).not.toThrow(); + }); + + it('warns and stays inert when there is no room', () => { + const { result } = render(null); + + expect(result.current.entries).toEqual([]); + expect(warnSpy).toHaveBeenCalledWith( + '[VoiceChat:Transcription]', + 'No room available; not subscribing', + ); + }); + + it('re-subscribes when the room instance changes', () => { + const first = createRoom(); + const second = createRoom(); + const { rerender } = render(first); + + rerender({ room: second }); + + expect(first.listenerCount(TRANSCRIPTION_EVENT)).toBe(0); + expect(second.listenerCount(TRANSCRIPTION_EVENT)).toBe(1); + }); + }); + + describe('accumulation', () => { + it('appends a new segment id as a new entry', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Hello', true)], agentParticipant); + emit(room, [segment('s2', 'How are you', true)], localParticipant); + + expect(result.current.entries.map((e) => e.text)).toEqual([ + 'Hello', + 'How are you', + ]); + }); + + it('keeps every utterance for the life of the call', () => { + const room = createRoom(); + const { result } = render(room); + + for (let i = 0; i < 12; i += 1) { + emit(room, [segment(`s${i}`, `line ${i}`, true)], agentParticipant); + } + + expect(result.current.entries).toHaveLength(12); + expect(result.current.entries[0].text).toBe('line 0'); + }); + + it('updates a known segment id in place instead of appending', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Hel')], agentParticipant); + emit(room, [segment('s1', 'Hello the')], agentParticipant); + emit(room, [segment('s1', 'Hello there', true)], agentParticipant); + + expect(result.current.entries).toHaveLength(1); + expect(result.current.entries[0]).toMatchObject({ + id: 's1', + text: 'Hello there', + isFinal: true, + }); + }); + + it('preserves the original timestamp across partial updates', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Hel')], agentParticipant); + const firstTimestamp = result.current.entries[0].timestamp; + + emit(room, [segment('s1', 'Hello', true)], agentParticipant); + + expect(result.current.entries[0].timestamp).toBe(firstTimestamp); + }); + + it('keeps ordering stable when two speakers interleave partials', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('u1', 'What is')], localParticipant); + emit(room, [segment('a1', 'Let me')], agentParticipant); + emit(room, [segment('u1', 'What is the time', true)], localParticipant); + emit(room, [segment('a1', 'Let me check', true)], agentParticipant); + + expect( + result.current.entries.map((e) => [e.id, e.speaker, e.text]), + ).toEqual([ + ['u1', 'user', 'What is the time'], + ['a1', 'agent', 'Let me check'], + ]); + }); + + it('does not reopen a finalised entry when a late partial arrives', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Done', true)], agentParticipant); + emit(room, [segment('s1', 'Done later')], agentParticipant); + + expect(result.current.entries).toHaveLength(1); + expect(result.current.entries[0].isFinal).toBe(true); + expect(result.current.entries[0].text).toBe('Done later'); + }); + + it('treats a duplicate final event for the same id as an update', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Same', true)], agentParticipant); + emit(room, [segment('s1', 'Same', true)], agentParticipant); + + expect(result.current.entries).toHaveLength(1); + }); + + it('updates an older entry in place when ids arrive out of order', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'first')], agentParticipant); + emit(room, [segment('s2', 'second')], localParticipant); + emit(room, [segment('s1', 'first, corrected', true)], agentParticipant); + + expect(result.current.entries.map((e) => e.text)).toEqual([ + 'first, corrected', + 'second', + ]); + }); + + it('trims the oldest entries past the retention ceiling', () => { + const room = createRoom(); + const { result } = render(room, { maxEntries: 3 }); + + ['a', 'b', 'c', 'd'].forEach((id) => + emit(room, [segment(id, id, true)], agentParticipant), + ); + + expect(result.current.entries.map((e) => e.id)).toEqual(['b', 'c', 'd']); + }); + + it('defaults the retention ceiling to 500 entries', () => { + expect(DEFAULT_MAX_TRANSCRIPT_ENTRIES).toBe(500); + }); + + it('resets the accumulated transcript', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Hello', true)], agentParticipant); + expect(result.current.hasReceivedTranscription).toBe(true); + + act(() => result.current.reset()); + + expect(result.current.entries).toEqual([]); + expect(result.current.hasReceivedTranscription).toBe(false); + }); + }); + + describe('degenerate events', () => { + it('records that an event arrived even when it carries no segments', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [], agentParticipant); + + expect(result.current.entries).toEqual([]); + expect(result.current.hasReceivedTranscription).toBe(true); + }); + + it('survives an undefined segment list', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, undefined, agentParticipant); + + expect(result.current.entries).toEqual([]); + expect(result.current.hasReceivedTranscription).toBe(true); + }); + + it('treats a segment with no final flag as still in progress', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [{ id: 's1', text: 'Half a thought' }], agentParticipant); + + expect(result.current.entries[0].isFinal).toBe(false); + expect(result.current.isTranscribing).toBe(true); + }); + + it('drops a segment with no id and says so', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [{ text: 'orphan', final: true }], agentParticipant); + + expect(result.current.entries).toEqual([]); + expect(warnSpy).toHaveBeenCalledWith( + '[VoiceChat:Transcription]', + 'Dropping transcription without a segment id', + { text: 'orphan' }, + ); + }); + + it('tolerates a segment with no text', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [{ id: 's1', final: false }], agentParticipant); + + expect(result.current.entries[0].text).toBe(''); + }); + + it('logs every event so a silent agent is distinguishable', () => { + const room = createRoom(); + render(room); + + emit(room, [segment('s1', 'Hello', true)], agentParticipant); + + expect(logSpy).toHaveBeenCalledWith( + '[VoiceChat:Transcription]', + 'TranscriptionReceived', + expect.objectContaining({ + segmentCount: 1, + participantIdentity: 'agent-7', + participantName: 'Agent Seven', + isLocal: false, + finals: [true], + }), + ); + }); + }); + + describe('speaker attribution', () => { + it('labels the local participant as the user by identity', () => { + const room = createRoom('local-user'); + const { result } = render(room); + + emit(room, [segment('s1', 'mine', true)], { + identity: 'local-user', + name: 'Me', + }); + + expect(result.current.entries[0].speaker).toBe('user'); + }); + + it('labels the local participant as the user by isLocal', () => { + // Identity is not known yet (room still connecting) but LiveKit still + // flags the participant as local. + const room = createRoom(undefined); + const { result } = render(room); + + emit(room, [segment('s1', 'mine', true)], { + identity: 'someone-else', + isLocal: true, + }); + + expect(result.current.entries[0].speaker).toBe('user'); + }); + + it('labels any other participant as the agent', () => { + const room = createRoom('local-user'); + const { result } = render(room); + + emit(room, [segment('s1', 'theirs', true)], agentParticipant); + + expect(result.current.entries[0]).toMatchObject({ + speaker: 'agent', + participantName: 'Agent Seven', + participantIdentity: 'agent-7', + }); + }); + + it('labels an unattributed transcription as the agent', () => { + const room = createRoom('local-user'); + const { result } = render(room); + + emit(room, [segment('s1', 'from nowhere', true)]); + + expect(result.current.entries[0].speaker).toBe('agent'); + expect(result.current.entries[0].participantName).toBeUndefined(); + }); + + it('prefers an explicitly supplied local identity over the room one', () => { + const room = createRoom('stale-identity'); + const { result } = render(room, { + localParticipantIdentity: 'real-local', + }); + + emit(room, [segment('s1', 'mine', true)], { identity: 'real-local' }); + emit(room, [segment('s2', 'theirs', true)], { + identity: 'stale-identity', + }); + + expect(result.current.entries.map((e) => e.speaker)).toEqual([ + 'user', + 'agent', + ]); + }); + + it('keeps the known participant details when a later event omits them', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'part')], agentParticipant); + emit(room, [segment('s1', 'part two', true)]); + + expect(result.current.entries[0]).toMatchObject({ + participantName: 'Agent Seven', + participantIdentity: 'agent-7', + }); + }); + + it('does not label a remote speaker as local when identity is unknown', () => { + const room = createRoom(undefined); + const { result } = render(room); + + emit(room, [segment('s1', 'theirs', true)], { identity: 'agent-7' }); + + expect(result.current.entries[0].speaker).toBe('agent'); + }); + }); + + describe('isTranscribing', () => { + it('is false before anything arrives', () => { + const room = createRoom(); + const { result } = render(room); + + expect(result.current.isTranscribing).toBe(false); + expect(result.current.hasReceivedTranscription).toBe(false); + }); + + it('is true while the newest entry is still partial', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Hel')], agentParticipant); + + expect(result.current.isTranscribing).toBe(true); + }); + + it('settles to false once the newest entry finalises', () => { + const room = createRoom(); + const { result } = render(room); + + emit(room, [segment('s1', 'Hel')], agentParticipant); + emit(room, [segment('s1', 'Hello', true)], agentParticipant); + + expect(result.current.isTranscribing).toBe(false); + }); + }); + + describe('joinTranscriptionSegments', () => { + it('joins disjoint fragments with a space', () => { + expect(joinTranscriptionSegments(['Hello', 'there'])).toBe('Hello there'); + }); + + it('collapses cumulative snapshots instead of duplicating them', () => { + expect(joinTranscriptionSegments(['Hello', 'Hello there'])).toBe( + 'Hello there', + ); + }); + + it('ignores blank and whitespace-only segments', () => { + expect(joinTranscriptionSegments(['', ' ', 'Hello', ' '])).toBe('Hello'); + }); + + it('trims each fragment', () => { + expect(joinTranscriptionSegments([' Hello ', ' there '])).toBe( + 'Hello there', + ); + }); + + it('returns an empty string for no segments', () => { + expect(joinTranscriptionSegments([])).toBe(''); + }); + + it('survives a null or undefined fragment', () => { + expect( + joinTranscriptionSegments([ + undefined as unknown as string, + 'Hello', + null as unknown as string, + ]), + ).toBe('Hello'); + }); + + it('is applied to multi-segment events', () => { + const room = createRoom(); + const { result } = render(room); + + emit( + room, + [segment('s1', 'Hello'), segment('s2', 'Hello there', true)], + agentParticipant, + ); + + // The last segment supplies the id, the fold supplies the text. + expect(result.current.entries).toHaveLength(1); + expect(result.current.entries[0]).toMatchObject({ + id: 's2', + text: 'Hello there', + }); + }); + }); +}); diff --git a/hooks/use-livekit-transcription.ts b/hooks/use-livekit-transcription.ts new file mode 100644 index 00000000..010a5674 --- /dev/null +++ b/hooks/use-livekit-transcription.ts @@ -0,0 +1,244 @@ +'use client'; + +import React from 'react'; + +import { + RoomEvent, + type Participant, + type Room, + type TranscriptionSegment, +} from 'livekit-client'; + +/** + * Live transcription for a LiveKit room. + * + * The screen-sharing flow already consumes `RoomEvent.TranscriptionReceived` + * (see `TranscriptionRelay` in `components/live-kit-screen-sharing.tsx`), but it + * forwards each event over `postMessage` to a Picture-in-Picture window that + * renders a single, self-clearing line. The voice call modal lives in the main + * document and wants the opposite: an accumulated, ordered transcript for the + * whole call. + * + * CONVERGENCE OPPORTUNITY: `TranscriptionRelay` could be reimplemented on top + * of this hook (relaying `entries[entries.length - 1]` instead of hand-rolling + * the same segment handling). That refactor is deliberately out of scope here + * to keep the blast radius of the voice-call change small. + */ + +const TRANSCRIPTION_DEBUG_PREFIX = '[VoiceChat:Transcription]'; + +// Matches the `voiceLog`/`voiceWarn` convention in +// `components/live-kit-voice-chat.tsx` so a single console filter shows the +// whole voice-call story. +function transcriptionLog(...args: unknown[]) { + console.log(TRANSCRIPTION_DEBUG_PREFIX, ...args); +} + +function transcriptionWarn(...args: unknown[]) { + console.warn(TRANSCRIPTION_DEBUG_PREFIX, ...args); +} + +/** + * Who produced an utterance. Anything that is not the local participant is the + * mentor agent — a voice call is always a two-party room. + */ +export type TranscriptSpeaker = 'user' | 'agent'; + +export type TranscriptEntry = { + /** LiveKit segment id. Stable across the partial -> final updates. */ + id: string; + text: string; + speaker: TranscriptSpeaker; + /** Display name LiveKit reported for the speaker, when it sent one. */ + participantName?: string; + participantIdentity?: string; + /** `false` while the utterance is still being spoken. */ + isFinal: boolean; + /** When the entry first appeared. Kept stable so ordering never jumps. */ + timestamp: number; +}; + +export type UseLiveKitTranscriptionOptions = { + room: Room | null | undefined; + /** + * Identity of the local participant. Optional: the room usually has not + * finished connecting when the hook mounts, so the identity is re-read from + * `room.localParticipant` on every event when this is not supplied. + */ + localParticipantIdentity?: string; + /** + * Hard ceiling on retained entries. An hour-long call is thousands of + * utterances, and every one of them is a live DOM node in the transcript + * band; past this many the oldest are dropped. + */ + maxEntries?: number; +}; + +export type UseLiveKitTranscriptionResult = { + /** Oldest first. New ids append; known ids update in place. */ + entries: TranscriptEntry[]; + /** True while the newest entry is still a partial. */ + isTranscribing: boolean; + /** True once at least one transcription event has been observed. */ + hasReceivedTranscription: boolean; + /** Clears the accumulated transcript. */ + reset: () => void; +}; + +export const DEFAULT_MAX_TRANSCRIPT_ENTRIES = 500; + +/** + * Combines the segments carried by a single transcription event. + * + * KNOWN UNKNOWN: LiveKit does not guarantee whether the segments in one event + * are disjoint fragments of an utterance ("Hello", "there") or successive + * CUMULATIVE snapshots of it ("Hello", "Hello there"). The screen-sharing relay + * assumes disjoint and does a naive `join(' ')`, which duplicates text under + * the cumulative shape. We have not verified which shape our voice agent emits. + * + * The fold below is safe under both: a segment that starts with everything we + * have so far supersedes it (cumulative), anything else is appended (disjoint). + * The one accepted cost is that two identical fragments inside a single event + * ("no", "no") collapse to one — vastly less damaging than echoing an entire + * utterance twice, and it cannot happen at all under the disjoint shape unless + * the speaker literally repeated a word within the same event window. + */ +export function joinTranscriptionSegments(texts: readonly string[]): string { + return texts.reduce((accumulated, raw) => { + const text = (raw ?? '').trim(); + if (!text) return accumulated; + if (!accumulated) return text; + if (text.startsWith(accumulated)) return text; + return `${accumulated} ${text}`; + }, ''); +} + +export function useLiveKitTranscription({ + room, + localParticipantIdentity, + maxEntries = DEFAULT_MAX_TRANSCRIPT_ENTRIES, +}: UseLiveKitTranscriptionOptions): UseLiveKitTranscriptionResult { + const [entries, setEntries] = React.useState([]); + const [hasReceivedTranscription, setHasReceivedTranscription] = + React.useState(false); + + // Read inside the handler so a late-arriving identity is still honoured + // without re-subscribing (which would drop events during the swap). + const localIdentityRef = React.useRef(localParticipantIdentity); + localIdentityRef.current = localParticipantIdentity; + const maxEntriesRef = React.useRef(maxEntries); + maxEntriesRef.current = maxEntries; + + const reset = React.useCallback(() => { + setEntries([]); + setHasReceivedTranscription(false); + }, []); + + React.useEffect(() => { + if (!room) { + transcriptionWarn('No room available; not subscribing'); + return; + } + + const handleTranscription = ( + segments: TranscriptionSegment[], + participant?: Participant, + ) => { + const list = segments ?? []; + // Logged on every event on purpose: when a user reports "no captions" we + // need to tell "the agent never emitted anything" apart from "the UI + // dropped it". Absence of this line is itself the diagnosis. + transcriptionLog('TranscriptionReceived', { + segmentCount: list.length, + participantIdentity: participant?.identity, + participantName: participant?.name, + isLocal: participant?.isLocal, + finals: list.map((segment) => segment?.final), + }); + + setHasReceivedTranscription(true); + + if (list.length === 0) return; + + const text = joinTranscriptionSegments(list.map((s) => s?.text ?? '')); + const lastSegment = list[list.length - 1]; + const id = lastSegment?.id; + if (!id) { + transcriptionWarn('Dropping transcription without a segment id', { + text, + }); + return; + } + + // Speaker attribution. `pip-chat.tsx` checks the mentor name FIRST in its + // fallback chain, so every non-local speaker is labelled with the mentor's + // name regardless of who actually spoke. Here identity decides, and the + // name is only ever a display detail. + const resolvedLocalIdentity = + localIdentityRef.current ?? room.localParticipant?.identity; + const speaker: TranscriptSpeaker = + participant?.isLocal === true || + (!!participant?.identity && + !!resolvedLocalIdentity && + participant.identity === resolvedLocalIdentity) + ? 'user' + : 'agent'; + + const isFinal = lastSegment?.final ?? false; + + setEntries((previous) => { + const index = previous.findIndex((entry) => entry.id === id); + + if (index === -1) { + const appended = [ + ...previous, + { + id, + text, + speaker, + participantName: participant?.name, + participantIdentity: participant?.identity, + isFinal, + timestamp: Date.now(), + }, + ]; + return appended.length > maxEntriesRef.current + ? appended.slice(appended.length - maxEntriesRef.current) + : appended; + } + + // Known id: the utterance grew. Update in place — appending would + // print every prefix of the sentence as its own line. + const existing = previous[index]; + const next = previous.slice(); + next[index] = { + ...existing, + text, + // `final` is sticky: a late partial for an already-finalised segment + // must not reopen it and restart the in-progress affordance. + isFinal: existing.isFinal || isFinal, + participantName: participant?.name ?? existing.participantName, + participantIdentity: + participant?.identity ?? existing.participantIdentity, + }; + return next; + }); + }; + + room.on(RoomEvent.TranscriptionReceived, handleTranscription); + transcriptionLog('Subscribed to RoomEvent.TranscriptionReceived', { + roomState: room.state, + localParticipant: room.localParticipant?.identity, + }); + + return () => { + room.off(RoomEvent.TranscriptionReceived, handleTranscription); + transcriptionLog('Unsubscribed from RoomEvent.TranscriptionReceived'); + }; + }, [room]); + + const isTranscribing = + entries.length > 0 && !entries[entries.length - 1].isFinal; + + return { entries, isTranscribing, hasReceivedTranscription, reset }; +} From 7430a56f17f5b508325982d3583b55d3d077ac09 Mon Sep 17 00:00:00 2001 From: conrad mugabe Date: Sun, 26 Jul 2026 23:46:23 +0200 Subject: [PATCH 04/14] fix(voice): separate mic and mentor audio, show speaker state and transcript A single isMuted flag drove both setMicrophoneEnabled and the muted prop on RoomAudioRenderer, so muting your own microphone also silenced the mentor, with nothing in the UI to say so. Splits it into isMicMuted and isMentorAudioMuted with independent controls. RoomAudioRenderer's muted prop is used alone rather than also calling setVolume: it maps over subscribed remote tracks via useTracks, which is reactive, so late joiners are covered without a re-apply. The animated blob had the same conflation in visual form. It keyed off !isMicMuted, so muting yourself made a live call look dead, and it never reflected the mentor at all. It now animates whenever connected: the wave bars track the user's voice, a violet ring tracks the mentor's, and both can show at once. Adds the live transcript band beneath the blob, backed by useLiveKitTranscription. Newest line is prominent, older lines recede, in-progress lines carry a caret until they finalise. Auto-scroll pins to the latest line but yields when the user scrolls up to read back. The two stacked status rows from the first pass are gone; they duplicated both the blob and the button icons. One role=status caption replaces them, and the transcript is a polite role=log so new lines are announced without re-reading the history. Regression tests assert each control leaves the other untouched, and that the animation survives muting the microphone. Co-Authored-By: Claude Opus 5 (1M context) --- .../__tests__/live-kit-voice-chat.test.tsx | 389 +++++++- components/live-kit-voice-chat.tsx | 68 +- .../__tests__/voice-chat-modal.test.tsx | 832 +++++++++++++++++- components/modals/voice-chat-modal.tsx | 377 +++++++- 4 files changed, 1582 insertions(+), 84 deletions(-) diff --git a/components/__tests__/live-kit-voice-chat.test.tsx b/components/__tests__/live-kit-voice-chat.test.tsx index 449ce58c..d13effd2 100644 --- a/components/__tests__/live-kit-voice-chat.test.tsx +++ b/components/__tests__/live-kit-voice-chat.test.tsx @@ -36,6 +36,7 @@ const mockAudioTrackPublications = new Map([ ]); const mockLocalParticipant = { + identity: 'testuser', setMicrophoneEnabled: mockSetMicrophoneEnabled, audioTrackPublications: mockAudioTrackPublications, on: vi.fn((event: string, handler: (...args: any[]) => void) => { @@ -44,12 +45,41 @@ const mockLocalParticipant = { off: vi.fn(), }; +// A remote (mentor) participant so the diagnostic dumps that iterate +// `room.remoteParticipants` are exercised. +const makeRemoteParticipants = () => + new Map([ + [ + 'p1', + { + identity: 'mentor-agent', + sid: 'p1', + isSpeaking: false, + audioLevel: 0, + isLocal: false, + trackPublications: new Map([ + [ + 't1', + { + trackSid: 't1', + source: 'microphone', + kind: 'audio', + isMuted: false, + isSubscribed: true, + isEnabled: true, + }, + ], + ]), + }, + ], + ]); + vi.mock('livekit-client', () => ({ Room: vi.fn(() => ({ connect: mockRoomConnect, disconnect: mockRoomDisconnect, localParticipant: mockLocalParticipant, - remoteParticipants: new Map(), + remoteParticipants: makeRemoteParticipants(), name: 'test-room', state: 'disconnected', on: vi.fn((event: string, handler: (...args: any[]) => void) => { @@ -72,6 +102,7 @@ vi.mock('livekit-client', () => ({ ActiveSpeakersChanged: 'activeSpeakersChanged', MediaDevicesError: 'mediaDevicesError', SignalConnected: 'signalConnected', + TranscriptionReceived: 'transcriptionReceived', }, ConnectionState: { Connected: 'connected', @@ -173,13 +204,24 @@ describe('LiveKitChat', () => { expect.objectContaining({ isOpen: true, onClose: defaultProps.onClose, - toggleMute: expect.any(Function), - isMuted: expect.any(Boolean), + toggleMicMute: expect.any(Function), + isMicMuted: expect.any(Boolean), + toggleMentorAudio: expect.any(Function), + isMentorAudioMuted: expect.any(Boolean), connectionState: expect.any(String), isSpeaking: false, + isMentorSpeaking: false, }), ); }); + + it('should start with mentor audio unmuted', async () => { + const { getByTestId } = render(); + expect(getByTestId('room-audio-renderer')).toHaveAttribute( + 'data-muted', + 'false', + ); + }); }); describe('successful connection flow', () => { @@ -224,14 +266,14 @@ describe('LiveKitChat', () => { }); }); - it('should pass isMuted=false to modal after successful connection', async () => { + it('should pass isMicMuted=false to modal after successful connection', async () => { render(); await vi.waitFor(() => { const lastCall = mockVoiceChatModal.mock.calls[ mockVoiceChatModal.mock.calls.length - 1 ][0]; - expect(lastCall.isMuted).toBe(false); + expect(lastCall.isMicMuted).toBe(false); }); }); @@ -452,36 +494,133 @@ describe('LiveKitChat', () => { }); }); - describe('mute toggle', () => { - it('should toggle mute state when toggleMute is called', async () => { - render(); + describe('mute toggles', () => { + const latestModalProps = () => + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + const renderConnected = async () => { + const utils = render(); await vi.waitFor(() => { - const lastCall = - mockVoiceChatModal.mock.calls[ - mockVoiceChatModal.mock.calls.length - 1 - ][0]; - expect(lastCall.isMuted).toBe(false); + expect(latestModalProps().isMicMuted).toBe(false); }); + return utils; + }; + + it('should toggle mic mute state when toggleMicMute is called', async () => { + await renderConnected(); - // Get the toggleMute function and call it - const lastCall = - mockVoiceChatModal.mock.calls[ - mockVoiceChatModal.mock.calls.length - 1 - ][0]; act(() => { - (lastCall.toggleMute as () => void)(); + (latestModalProps().toggleMicMute as () => void)(); }); await vi.waitFor(() => { - const updatedCall = - mockVoiceChatModal.mock.calls[ - mockVoiceChatModal.mock.calls.length - 1 - ][0]; - expect(updatedCall.isMuted).toBe(true); + expect(latestModalProps().isMicMuted).toBe(true); }); expect(mockSetMicrophoneEnabled).toHaveBeenCalledWith(false); }); + + it('should re-enable the microphone when toggled back on', async () => { + await renderConnected(); + + act(() => { + (latestModalProps().toggleMicMute as () => void)(); + }); + await vi.waitFor(() => { + expect(latestModalProps().isMicMuted).toBe(true); + }); + + act(() => { + (latestModalProps().toggleMicMute as () => void)(); + }); + await vi.waitFor(() => { + expect(latestModalProps().isMicMuted).toBe(false); + }); + expect(mockSetMicrophoneEnabled).toHaveBeenLastCalledWith(true); + }); + + // ── Regression test for the bug this change fixes ── + // A single `isMuted` flag used to drive BOTH `setMicrophoneEnabled` and + // ``, so muting your own mic silently silenced the + // mentor too. Muting the mic must leave mentor audio playing. + it('should NOT mute mentor audio when the microphone is muted', async () => { + const { getByTestId } = await renderConnected(); + + expect(getByTestId('room-audio-renderer')).toHaveAttribute( + 'data-muted', + 'false', + ); + + act(() => { + (latestModalProps().toggleMicMute as () => void)(); + }); + + await vi.waitFor(() => { + expect(latestModalProps().isMicMuted).toBe(true); + }); + + // The whole point: mentor audio playback is untouched. + expect(getByTestId('room-audio-renderer')).toHaveAttribute( + 'data-muted', + 'false', + ); + expect(latestModalProps().isMentorAudioMuted).toBe(false); + }); + + it('should mute mentor audio playback when toggleMentorAudio is called', async () => { + const { getByTestId } = await renderConnected(); + + act(() => { + (latestModalProps().toggleMentorAudio as () => void)(); + }); + + await vi.waitFor(() => { + expect(latestModalProps().isMentorAudioMuted).toBe(true); + }); + expect(getByTestId('room-audio-renderer')).toHaveAttribute( + 'data-muted', + 'true', + ); + }); + + it('should NOT touch the microphone when mentor audio is toggled', async () => { + await renderConnected(); + + mockSetMicrophoneEnabled.mockClear(); + + act(() => { + (latestModalProps().toggleMentorAudio as () => void)(); + }); + + await vi.waitFor(() => { + expect(latestModalProps().isMentorAudioMuted).toBe(true); + }); + expect(mockSetMicrophoneEnabled).not.toHaveBeenCalled(); + expect(latestModalProps().isMicMuted).toBe(false); + }); + + it('should unmute mentor audio when toggled back on', async () => { + const { getByTestId } = await renderConnected(); + + act(() => { + (latestModalProps().toggleMentorAudio as () => void)(); + }); + await vi.waitFor(() => { + expect(latestModalProps().isMentorAudioMuted).toBe(true); + }); + + act(() => { + (latestModalProps().toggleMentorAudio as () => void)(); + }); + await vi.waitFor(() => { + expect(latestModalProps().isMentorAudioMuted).toBe(false); + }); + expect(getByTestId('room-audio-renderer')).toHaveAttribute( + 'data-muted', + 'false', + ); + }); }); describe('cleanup on unmount', () => { @@ -717,6 +856,116 @@ describe('LiveKitChat', () => { }); }); + it('should keep isSpeaking false while the mic is muted', async () => { + render(); + + await vi.waitFor(() => { + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.connectionState).toBe('connected'); + }); + + const beforeMute = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + act(() => { + (beforeMute.toggleMicMute as () => void)(); + }); + + await vi.waitFor(() => { + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.isMicMuted).toBe(true); + }); + + act(() => { + participantEventHandlers['isSpeakingChanged']?.(true); + }); + + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.isSpeaking).toBe(false); + }); + + it('should set isMentorSpeaking when a remote participant speaks', async () => { + render(); + await vi.waitFor(() => { + expect(mockRoomConnect).toHaveBeenCalled(); + }); + + act(() => { + roomEventHandlers['activeSpeakersChanged']?.([ + { identity: 'mentor-agent', sid: 's1', isLocal: false }, + ]); + }); + + await vi.waitFor(() => { + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.isMentorSpeaking).toBe(true); + }); + }); + + it('should not set isMentorSpeaking when only the local participant speaks', async () => { + render(); + await vi.waitFor(() => { + expect(mockRoomConnect).toHaveBeenCalled(); + }); + + act(() => { + roomEventHandlers['activeSpeakersChanged']?.([ + { identity: 'testuser', sid: 's0', isLocal: true }, + ]); + }); + + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.isMentorSpeaking).toBe(false); + }); + + it('should clear isMentorSpeaking when the active speaker list empties', async () => { + render(); + await vi.waitFor(() => { + expect(mockRoomConnect).toHaveBeenCalled(); + }); + + act(() => { + roomEventHandlers['activeSpeakersChanged']?.([ + { identity: 'mentor-agent', sid: 's1', isLocal: false }, + ]); + }); + await vi.waitFor(() => { + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.isMentorSpeaking).toBe(true); + }); + + act(() => { + roomEventHandlers['activeSpeakersChanged']?.([]); + }); + + await vi.waitFor(() => { + const lastCall = + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + expect(lastCall.isMentorSpeaking).toBe(false); + }); + }); + it('should set isSpeaking to false on trackMuted event', async () => { render(); @@ -1002,4 +1251,96 @@ describe('LiveKitChat', () => { }); }); }); + + describe('live transcription', () => { + const lastProps = (): any => + mockVoiceChatModal.mock.calls[ + mockVoiceChatModal.mock.calls.length - 1 + ][0]; + + const emitTranscription = ( + segments: Record[], + participant?: Record, + ) => { + act(() => { + roomEventHandlers['transcriptionReceived']?.(segments, participant); + }); + }; + + let consoleLogSpy: ReturnType; + + beforeEach(() => { + consoleLogSpy = vi.spyOn(console, 'log').mockImplementation(() => {}); + }); + + afterEach(() => { + consoleLogSpy.mockRestore(); + }); + + it('subscribes to transcription events for the call', async () => { + render(); + + await vi.waitFor(() => { + expect(roomEventHandlers['transcriptionReceived']).toBeTypeOf( + 'function', + ); + }); + }); + + it('starts with an empty transcript', () => { + render(); + + expect(lastProps().transcript).toEqual([]); + expect(lastProps().isTranscriptLive).toBe(false); + }); + + it('forwards the mentor name for transcript labelling', () => { + render(); + + expect(lastProps().mentorName).toBe('Ada'); + }); + + it('accumulates transcript entries and attributes the speaker', async () => { + render(); + await vi.waitFor(() => { + expect(roomEventHandlers['transcriptionReceived']).toBeDefined(); + }); + + emitTranscription([{ id: 's1', text: 'Hello', final: true }], { + identity: 'mentor-agent', + name: 'Ada', + }); + emitTranscription([{ id: 's2', text: 'Hi back', final: true }], { + identity: 'testuser', + }); + + expect( + lastProps().transcript.map((e: { speaker: string; text: string }) => [ + e.speaker, + e.text, + ]), + ).toEqual([ + ['agent', 'Hello'], + ['user', 'Hi back'], + ]); + }); + + it('reports a partial line as live and settles once it finalises', async () => { + render(); + await vi.waitFor(() => { + expect(roomEventHandlers['transcriptionReceived']).toBeDefined(); + }); + + emitTranscription([{ id: 's1', text: 'Hel', final: false }], { + identity: 'mentor-agent', + }); + expect(lastProps().isTranscriptLive).toBe(true); + + emitTranscription([{ id: 's1', text: 'Hello', final: true }], { + identity: 'mentor-agent', + }); + expect(lastProps().isTranscriptLive).toBe(false); + expect(lastProps().transcript).toHaveLength(1); + }); + }); }); diff --git a/components/live-kit-voice-chat.tsx b/components/live-kit-voice-chat.tsx index e80de0d0..7c339b56 100644 --- a/components/live-kit-voice-chat.tsx +++ b/components/live-kit-voice-chat.tsx @@ -9,6 +9,8 @@ import { import { useCreateCallCredentialsMutation } from '@iblai/iblai-js/data-layer'; import { RoomAudioRenderer, RoomContext } from '@livekit/components-react'; +import { useLiveKitTranscription } from '@/hooks/use-livekit-transcription'; + import { VoiceChatModal } from './modals/voice-chat-modal'; const VOICE_DEBUG_PREFIX = '[VoiceChat:LiveKit]'; @@ -32,6 +34,8 @@ type Props = { username: string; onClose: () => void; isOpen: boolean; + /** Display name of the mentor, used to label its transcript lines. */ + mentorName?: string; }; type ConnectionState = @@ -69,6 +73,7 @@ export function LiveKitChat({ username, onClose, isOpen, + mentorName, }: Props) { const [initiateCall] = useCreateCallCredentialsMutation(); @@ -76,13 +81,26 @@ export function LiveKitChat({ voiceLog('Creating new Room instance'); return new Room({}); }); - const [isMuted, setIsMuted] = React.useState(true); + // Outbound audio: whether the user's own microphone is muted. + const [isMicMuted, setIsMicMuted] = React.useState(true); + // Inbound audio: whether the mentor's voice playback is muted. These are two + // unrelated concerns and must never share a single flag — muting your mic + // used to silence the mentor as a side effect. + const [isMentorAudioMuted, setIsMentorAudioMuted] = React.useState(false); const [connectionState, setConnectionState] = React.useState( 'requesting-permission', ); const [isSpeaking, setIsSpeaking] = React.useState(false); + const [isMentorSpeaking, setIsMentorSpeaking] = React.useState(false); const permissionStreamRef = React.useRef(null); + // Live captions for the call. A voice-only call is unusable for deaf and + // hard-of-hearing users without this, so it is mounted for the whole call + // rather than gated behind a toggle. + const { entries: transcript, isTranscribing } = useLiveKitTranscription({ + room, + }); + function stopPermissionStream() { voiceLog('Stopping permission stream', { hasTracks: !!permissionStreamRef.current, @@ -226,7 +244,7 @@ export function LiveKitChat({ kind: pub.kind, })), }); - setIsMuted(false); // Auto-unmute when successfully connected + setIsMicMuted(false); // Auto-unmute the mic when successfully connected setConnectionState('connected'); postRoomStatusToOpener('connected', 'voice-call'); } catch (error) { @@ -262,10 +280,10 @@ export function LiveKitChat({ const handleIsSpeakingChanged = (speaking: boolean) => { voiceLog('Local participant speaking changed', { speaking, - isMuted, - effectiveSpeaking: speaking && !isMuted, + isMicMuted, + effectiveSpeaking: speaking && !isMicMuted, }); - setIsSpeaking(speaking && !isMuted); + setIsSpeaking(speaking && !isMicMuted); }; const handleTrackMuted = () => { @@ -282,7 +300,7 @@ export function LiveKitChat({ room.localParticipant.off('isSpeakingChanged', handleIsSpeakingChanged); room.localParticipant.off('trackMuted', handleTrackMuted); }; - }, [connectionState, room, isMuted]); + }, [connectionState, room, isMicMuted]); // Listen to room connection state changes from LiveKit React.useEffect(() => { @@ -406,6 +424,8 @@ export function LiveKitChat({ count: speakers.length, speakers: speakers.map((s) => ({ identity: s.identity, sid: s.sid })), }); + // Anything speaking that is not us is the mentor agent. + setIsMentorSpeaking(speakers.some((s) => !s.isLocal)); }; const handleMediaDevicesError = (error: any) => { @@ -483,17 +503,31 @@ export function LiveKitChat({ }; }, []); - const handleToggleMute = () => { - const newMutedState = !isMuted; - voiceLog('Toggle mute', { - currentMuted: isMuted, + // Outbound only: stops publishing the user's microphone. Must not touch + // mentor audio playback. + const handleToggleMicMute = () => { + const newMutedState = !isMicMuted; + voiceLog('Toggle microphone mute', { + currentMuted: isMicMuted, newMuted: newMutedState, roomState: room.state, }); - setIsMuted(newMutedState); + setIsMicMuted(newMutedState); room.localParticipant.setMicrophoneEnabled(!newMutedState); }; + // Inbound only: gates playback of the mentor's voice via RoomAudioRenderer. + // Must not touch the microphone. + const handleToggleMentorAudio = () => { + const newMutedState = !isMentorAudioMuted; + voiceLog('Toggle mentor audio mute', { + currentMuted: isMentorAudioMuted, + newMuted: newMutedState, + roomState: room.state, + }); + setIsMentorAudioMuted(newMutedState); + }; + // Periodic room state dump (every 5 seconds while connected) React.useEffect(() => { if (connectionState !== 'connected') return; @@ -547,14 +581,20 @@ export function LiveKitChat({ return ( - + ); diff --git a/components/modals/__tests__/voice-chat-modal.test.tsx b/components/modals/__tests__/voice-chat-modal.test.tsx index 912cd7f0..cf57a073 100644 --- a/components/modals/__tests__/voice-chat-modal.test.tsx +++ b/components/modals/__tests__/voice-chat-modal.test.tsx @@ -1,15 +1,53 @@ import { describe, it, expect, vi, beforeEach } from 'vitest'; import { render, screen, fireEvent } from '@testing-library/react'; import { VoiceChatModal } from '../voice-chat-modal'; +import type { TranscriptEntry } from '@/hooks/use-livekit-transcription'; + +function entry( + id: string, + text: string, + speaker: TranscriptEntry['speaker'], + isFinal = true, + extra: Partial = {}, +): TranscriptEntry { + return { id, text, speaker, isFinal, timestamp: 0, ...extra }; +} + +/** + * jsdom has no layout, so the scroll container reports 0 for every metric and + * ignores `scrollTop` writes. Give the element real own-properties so the + * auto-scroll logic can be exercised. + */ +function makeScrollable( + element: HTMLElement, + { scrollHeight = 500, clientHeight = 100, scrollTop = 400 } = {}, +) { + Object.defineProperty(element, 'scrollHeight', { + value: scrollHeight, + configurable: true, + }); + Object.defineProperty(element, 'clientHeight', { + value: clientHeight, + configurable: true, + }); + Object.defineProperty(element, 'scrollTop', { + value: scrollTop, + writable: true, + configurable: true, + }); +} describe('VoiceChatModal', () => { const defaultProps = { isOpen: true, onClose: vi.fn(), - toggleMute: vi.fn(), - isMuted: false, + toggleMicMute: vi.fn(), + isMicMuted: false, + toggleMentorAudio: vi.fn(), + isMentorAudioMuted: false, connectionState: 'connected' as const, isSpeaking: false, + isMentorSpeaking: false, }; beforeEach(() => { @@ -67,6 +105,48 @@ describe('VoiceChatModal', () => { expect(screen.getByLabelText('Mute microphone')).toBeDisabled(); }); + + it('disables the agent audio button while connecting', () => { + render(); + + expect(screen.getByLabelText('Mute agent audio')).toBeDisabled(); + }); + + it('hides the call status caption while connecting', () => { + render(); + + expect(screen.queryByLabelText('Call status')).toBeNull(); + }); + + it('hides the call status caption while requesting permission', () => { + render( + , + ); + + expect(screen.queryByLabelText('Call status')).toBeNull(); + }); + + it('shows the loading spinner while connecting', () => { + render(); + + expect(document.querySelector('.animate-spin')).toBeInTheDocument(); + }); + + it('does not tint the controls red while connecting', () => { + render(); + + // The loading state renders MicOff/VolumeX icons, but that is not a + // muted state and must not read as one. + expect(screen.getByLabelText('Mute microphone')).not.toHaveClass( + 'border-red-500', + ); + expect(screen.getByLabelText('Mute agent audio')).not.toHaveClass( + 'border-red-500', + ); + }); }); describe('connected state', () => { @@ -75,7 +155,7 @@ describe('VoiceChatModal', () => { , ); @@ -92,22 +172,31 @@ describe('VoiceChatModal', () => { expect(screen.getByLabelText('Mute microphone')).toBeEnabled(); }); + + it('enables the agent audio button when connected', () => { + render(); + + expect(screen.getByLabelText('Mute agent audio')).toBeEnabled(); + }); }); - describe('speaking state', () => { - it('uses faster pulse animation when speaking', () => { + describe('blob animation', () => { + const pulsingBg = () => document.querySelector('.bg-blue-100'); + const waveBars = () => + Array.from(document.querySelectorAll('.transform-gpu')); + + it('uses faster pulse animation when the user is speaking', () => { render( , ); // Speaking uses 1.5s pulse (faster than non-speaking 2s) - const pulsingBg = document.querySelector('.bg-blue-100'); - expect(pulsingBg).toHaveStyle({ + expect(pulsingBg()).toHaveStyle({ animation: 'randomPulse1 1.5s ease-in-out infinite', }); }); @@ -117,16 +206,195 @@ describe('VoiceChatModal', () => { , ); // Not speaking uses 2s pulse (slower) - const pulsingBg = document.querySelector('.bg-blue-100'); - expect(pulsingBg).toHaveStyle({ + expect(pulsingBg()).toHaveStyle({ + animation: 'randomPulse1 2s ease-in-out infinite', + }); + }); + + // Regression: the animation used to be gated on `!isMicMuted`, so muting + // your own microphone froze the entire indicator and the call looked dead + // even while the agent was talking. + it('keeps the blob animating while the mic is muted', () => { + render( + , + ); + + expect(pulsingBg()).toHaveStyle({ animation: 'randomPulse1 2s ease-in-out infinite', }); + expect(document.querySelector('.from-blue-200')).toHaveStyle({ + animation: 'randomPulse2 2.5s ease-in-out infinite', + }); + }); + + it('keeps the particles animating while the mic is muted', () => { + render( + , + ); + + const particles = document.querySelectorAll( + 'div[style*="particlePulse"]', + ); + expect(particles).toHaveLength(10); + }); + + it('falls back to the idle pulse when not connected', () => { + render( + , + ); + + expect(pulsingBg()).toHaveStyle({ + animation: 'pulse 2s cubic-bezier(0.4, 0, 0.6, 1) infinite', + }); + expect(document.querySelector('.from-blue-200')).toHaveStyle({ + animation: 'none', + opacity: '0.8', + }); + }); + + it('animates the sound wave bars when the user is speaking', () => { + render(); + + const bars = waveBars(); + expect(bars).toHaveLength(5); + expect(bars[0]).toHaveStyle({ height: '30px', opacity: '1' }); + expect(bars[0]).toHaveStyle({ + animation: 'soundWave1 0.8s ease-in-out infinite', + }); + }); + + it('flattens and dims the sound wave bars when the mic is muted', () => { + render( + , + ); + + const bars = waveBars(); + expect(bars).toHaveLength(5); + bars.forEach((bar) => { + expect(bar).toHaveStyle({ + height: '12px', + opacity: '0.35', + animation: 'none', + }); + }); + }); + + it('keeps the bars idle-but-live when the mic is on and the user is silent', () => { + render(); + + const bars = waveBars(); + expect(bars[0]).toHaveStyle({ height: '30px', opacity: '0.7' }); + expect(bars[0]).toHaveStyle({ + animation: 'soundWave1 1.2s ease-in-out infinite', + }); + }); + + it('shows the violet mentor ring while the agent is speaking', () => { + render(); + + const ring = screen.getByTestId('mentor-speaking-ring'); + expect(ring).toBeInTheDocument(); + expect(ring.querySelector('.border-violet-500')).toHaveStyle({ + animation: 'mentorRingPulse 1.4s ease-in-out infinite', + }); + expect(ring.querySelector('.border-indigo-400')).toHaveStyle({ + animation: 'mentorRingRipple 1.4s ease-out infinite', + }); + }); + + it('hides the mentor ring when the agent is silent', () => { + render(); + + expect(screen.queryByTestId('mentor-speaking-ring')).toBeNull(); + }); + + it('hides the mentor ring when agent audio is muted', () => { + render( + , + ); + + expect(screen.queryByTestId('mentor-speaking-ring')).toBeNull(); + }); + + it('hides the mentor ring when not connected', () => { + render( + , + ); + + expect(screen.queryByTestId('mentor-speaking-ring')).toBeNull(); + }); + + it('shows the mentor ring and the user waves at the same time', () => { + render( + , + ); + + // The two parties are separate layers, never mutually exclusive. + expect(screen.getByTestId('mentor-speaking-ring')).toBeInTheDocument(); + expect(waveBars()[0]).toHaveStyle({ opacity: '1' }); + }); + + it('shows the mentor ring even while the user mic is muted', () => { + render( + , + ); + + expect(screen.getByTestId('mentor-speaking-ring')).toBeInTheDocument(); + }); + + it('dims the whole blob when agent audio is muted', () => { + render(); + + expect(screen.getByTestId('voice-blob')).toHaveStyle({ + opacity: '0.45', + filter: 'saturate(0.35)', + }); + }); + + it('leaves the blob at full strength when agent audio is on', () => { + render(); + + expect(screen.getByTestId('voice-blob')).toHaveStyle({ opacity: '1' }); + }); + + it('does not dim the blob when only the mic is muted', () => { + render(); + + expect(screen.getByTestId('voice-blob')).toHaveStyle({ opacity: '1' }); }); }); @@ -136,7 +404,7 @@ describe('VoiceChatModal', () => { , ); @@ -148,7 +416,7 @@ describe('VoiceChatModal', () => { , ); @@ -160,7 +428,7 @@ describe('VoiceChatModal', () => { , ); @@ -168,6 +436,166 @@ describe('VoiceChatModal', () => { }); }); + describe('agent audio control', () => { + it('shows the mute label when agent audio is playing', () => { + render(); + + expect(screen.getByLabelText('Mute agent audio')).toBeInTheDocument(); + }); + + it('shows the unmute label when agent audio is muted', () => { + render(); + + expect(screen.getByLabelText('Unmute agent audio')).toBeInTheDocument(); + }); + + it('calls toggleMentorAudio when the agent audio button is clicked', () => { + render(); + + fireEvent.click(screen.getByLabelText('Mute agent audio')); + + expect(defaultProps.toggleMentorAudio).toHaveBeenCalledTimes(1); + expect(defaultProps.toggleMicMute).not.toHaveBeenCalled(); + }); + + it('leaves the microphone control untouched when agent audio is muted', () => { + render(); + + // Mic is still live even though the agent is silenced. + expect(screen.getByLabelText('Mute microphone')).toBeInTheDocument(); + expect(screen.getByLabelText('Mute microphone')).not.toHaveClass( + 'border-red-500', + ); + }); + }); + + describe('call status caption', () => { + const caption = () => screen.getByLabelText('Call status'); + + it('renders a single live region', () => { + render(); + + const regions = screen.getAllByRole('status'); + expect(regions).toHaveLength(1); + expect(regions[0]).toHaveAttribute('aria-label', 'Call status'); + }); + + it('shows "Listening…" when connected and nobody is muted or speaking', () => { + render(); + + expect(caption()).toHaveTextContent('Listening…'); + }); + + it('shows "Mic muted" when only the mic is muted', () => { + render(); + + expect(caption()).toHaveTextContent('Mic muted'); + }); + + it('shows "Agent speaking" when the agent is speaking', () => { + render(); + + expect(caption()).toHaveTextContent('Agent speaking'); + }); + + it('shows "Agent muted" when agent audio is muted', () => { + render(); + + expect(caption()).toHaveTextContent('Agent muted'); + }); + + it('prefers "Agent muted" over the agent speaking', () => { + render( + , + ); + + expect(caption()).toHaveTextContent('Agent muted'); + }); + + it('prefers "Agent muted" over the mic being muted', () => { + render( + , + ); + + expect(caption()).toHaveTextContent('Agent muted'); + }); + + it('prefers "Agent speaking" over the mic being muted', () => { + render( + , + ); + + expect(caption()).toHaveTextContent('Agent speaking'); + }); + + it('does not report the user speaking - the blob shows that', () => { + render(); + + expect(caption()).toHaveTextContent('Listening…'); + }); + + it('no longer renders the removed per-party status rows', () => { + render(); + + expect(screen.queryByLabelText('Agent audio status')).toBeNull(); + expect(screen.queryByLabelText('Microphone status')).toBeNull(); + }); + }); + + describe('control button muted treatment', () => { + it('tints the mic button red when the mic is muted', () => { + render(); + + expect(screen.getByLabelText('Unmute microphone')).toHaveClass( + 'border-red-500', + ); + }); + + it('leaves the mic button blue when the mic is live', () => { + render(); + + expect(screen.getByLabelText('Mute microphone')).toHaveClass( + 'border-blue-500', + ); + }); + + it('tints the agent audio button red when agent audio is muted', () => { + render(); + + expect(screen.getByLabelText('Unmute agent audio')).toHaveClass( + 'border-red-500', + ); + }); + + it('leaves the agent audio button blue when agent audio is on', () => { + render(); + + expect(screen.getByLabelText('Mute agent audio')).toHaveClass( + 'border-blue-500', + ); + }); + + it('does not tint the mic button when only agent audio is muted', () => { + render(); + + expect(screen.getByLabelText('Mute microphone')).toHaveClass( + 'border-blue-500', + ); + }); + }); + describe('disconnected state', () => { it('renders disconnected state without sound waves', () => { render( @@ -179,12 +607,13 @@ describe('VoiceChatModal', () => { }); describe('button interactions', () => { - it('calls toggleMute when mute button is clicked', () => { + it('calls toggleMicMute when mute button is clicked', () => { render(); fireEvent.click(screen.getByLabelText('Mute microphone')); - expect(defaultProps.toggleMute).toHaveBeenCalledTimes(1); + expect(defaultProps.toggleMicMute).toHaveBeenCalledTimes(1); + expect(defaultProps.toggleMentorAudio).not.toHaveBeenCalled(); }); it('calls onClose when close button is clicked', () => { @@ -196,6 +625,377 @@ describe('VoiceChatModal', () => { }); }); + describe('transcript band', () => { + const logRegion = () => screen.getByRole('log'); + const lines = () => screen.queryAllByTestId('voice-transcript-line'); + + it('exposes a polite log region with a stable label', () => { + render(); + + const region = logRegion(); + expect(region).toHaveAttribute('aria-label', 'Call transcript'); + expect(region).toHaveAttribute('aria-live', 'polite'); + // Only new/changed lines are announced - the accumulated history is + // never re-read when the user scrolls. + expect(region).toHaveAttribute('aria-relevant', 'additions text'); + // The scroll region must be reachable without a mouse. + expect(region).toHaveAttribute('tabindex', '0'); + }); + + it('does not turn the transcript into a second status region', () => { + render( + , + ); + + const statusRegions = screen.getAllByRole('status'); + expect(statusRegions).toHaveLength(1); + expect(statusRegions[0]).toHaveAttribute('aria-label', 'Call status'); + }); + + it('shows a neutral empty state before the first utterance', () => { + render(); + + expect(screen.getByTestId('voice-transcript-empty')).toHaveTextContent( + 'The live transcript will appear here as the conversation starts.', + ); + expect(lines()).toHaveLength(0); + }); + + it('replaces the empty state once lines arrive', () => { + render( + , + ); + + expect(screen.queryByTestId('voice-transcript-empty')).toBeNull(); + expect(lines()).toHaveLength(1); + }); + + it('renders lines oldest first', () => { + render( + , + ); + + expect(lines().map((l) => l.textContent)).toEqual([ + 'AgentFirst', + 'YouSecond', + ]); + }); + + it('labels the user with "You" in blue', () => { + render( + , + ); + + const label = lines()[0].querySelector('span'); + expect(label).toHaveTextContent('You'); + expect(label).toHaveClass('text-blue-600'); + }); + + it('labels the agent with the mentor name in the ring violet', () => { + render( + , + ); + + const label = lines()[0].querySelector('span'); + expect(label).toHaveTextContent('Ada'); + // Same family as the mentor-speaking ring, on purpose. + expect(label).toHaveClass('text-violet-600'); + }); + + it('falls back to the LiveKit participant name when no mentor name is given', () => { + render( + , + ); + + expect(lines()[0].querySelector('span')).toHaveTextContent('Agent Seven'); + }); + + it('falls back to a generic agent label as a last resort', () => { + render( + , + ); + + expect(lines()[0].querySelector('span')).toHaveTextContent('Agent'); + }); + + it('never labels a user line with the mentor name', () => { + render( + , + ); + + expect(lines()[0].querySelector('span')).toHaveTextContent('You'); + }); + + it('gives the newest line full weight and dims the older ones', () => { + render( + , + ); + + const [older, newest] = lines(); + expect(older).toHaveAttribute('data-newest', 'false'); + expect(older).toHaveClass('opacity-60'); + expect(older).toHaveClass('text-xs'); + expect(newest).toHaveAttribute('data-newest', 'true'); + expect(newest).toHaveClass('opacity-100'); + expect(newest).toHaveClass('text-sm'); + }); + + it('marks an in-progress line with a caret', () => { + render( + , + ); + + const caret = screen.getByTestId('voice-transcript-caret'); + expect(caret).toHaveAttribute('aria-hidden', 'true'); + expect(caret).toHaveStyle({ + animation: 'transcriptCaret 1s ease-in-out infinite', + }); + expect(lines()[0]).toHaveAttribute('data-final', 'false'); + }); + + it('settles the caret when the line finalises', () => { + const { rerender } = render( + , + ); + + expect(screen.getByTestId('voice-transcript-caret')).toBeInTheDocument(); + + rerender( + , + ); + + expect(screen.queryByTestId('voice-transcript-caret')).toBeNull(); + expect(lines()[0]).toHaveAttribute('data-final', 'true'); + }); + + it('shows a live indicator while words are still arriving', () => { + render(); + + const indicator = screen.getByTestId('voice-transcript-live'); + expect(indicator).toHaveTextContent('Live'); + // The log already announces the words; this is decoration only. + expect(indicator).toHaveAttribute('aria-hidden', 'true'); + }); + + it('hides the live indicator when nothing is in progress', () => { + render(); + + expect(screen.queryByTestId('voice-transcript-live')).toBeNull(); + }); + + it('tags each line with its speaker for styling and selection', () => { + render( + , + ); + + expect(lines().map((l) => l.getAttribute('data-speaker'))).toEqual([ + 'user', + 'agent', + ]); + }); + }); + + describe('transcript auto-scroll', () => { + const scrollBox = () => screen.getByTestId('voice-transcript-scroll'); + + it('scrolls to the newest line when a line arrives', () => { + const { rerender } = render(); + makeScrollable(scrollBox(), { scrollTop: 0 }); + + rerender( + , + ); + + expect(scrollBox().scrollTop).toBe(500); + }); + + it('does not offer "jump to latest" while pinned to the bottom', () => { + render( + , + ); + + expect(screen.queryByTestId('voice-transcript-jump')).toBeNull(); + }); + + it('pauses auto-scroll and offers a jump affordance when scrolled up', () => { + const { rerender } = render( + , + ); + + const box = scrollBox(); + makeScrollable(box, { scrollTop: 0 }); + fireEvent.scroll(box); + + expect(screen.getByTestId('voice-transcript-jump')).toHaveTextContent( + 'Jump to latest', + ); + + // A new line must not yank the reader back down. + rerender( + , + ); + + expect(scrollBox().scrollTop).toBe(0); + }); + + it('resumes auto-scroll when the user scrolls back to the bottom', () => { + const { rerender } = render( + , + ); + + const box = scrollBox(); + makeScrollable(box, { scrollTop: 0 }); + fireEvent.scroll(box); + expect(screen.getByTestId('voice-transcript-jump')).toBeInTheDocument(); + + // Back within the pin threshold of the bottom (500 - 100 = 400). + box.scrollTop = 390; + fireEvent.scroll(box); + + expect(screen.queryByTestId('voice-transcript-jump')).toBeNull(); + + rerender( + , + ); + + expect(scrollBox().scrollTop).toBe(500); + }); + + it('jumps back to the newest line when the affordance is clicked', () => { + render( + , + ); + + const box = scrollBox(); + makeScrollable(box, { scrollTop: 0 }); + fireEvent.scroll(box); + + fireEvent.click(screen.getByTestId('voice-transcript-jump')); + + expect(scrollBox().scrollTop).toBe(500); + expect(screen.queryByTestId('voice-transcript-jump')).toBeNull(); + }); + + it('does not offer the jump affordance with an empty transcript', () => { + render(); + + const box = scrollBox(); + makeScrollable(box, { scrollTop: 0 }); + fireEvent.scroll(box); + + expect(screen.queryByTestId('voice-transcript-jump')).toBeNull(); + }); + }); + + describe('layout', () => { + it('keeps the blob, the caption and all three controls alongside the band', () => { + render( + , + ); + + expect(screen.getByTestId('voice-blob')).toBeInTheDocument(); + expect(screen.getByLabelText('Call status')).toBeInTheDocument(); + expect(screen.getByLabelText('Mute microphone')).toBeInTheDocument(); + expect(screen.getByLabelText('Mute agent audio')).toBeInTheDocument(); + expect(screen.getByLabelText('Close voice chat')).toBeInTheDocument(); + expect(screen.getByRole('log')).toBeInTheDocument(); + }); + + it('lets the band flex instead of pushing the controls off-screen', () => { + render(); + + const band = screen.getByTestId('voice-transcript'); + expect(band).toHaveClass('flex-1'); + expect(band).toHaveClass('min-h-0'); + expect(screen.getByTestId('voice-blob')).toHaveClass('shrink-0'); + }); + }); + describe('reconnecting/error states', () => { it('does not show loading message for reconnecting state', () => { render( diff --git a/components/modals/voice-chat-modal.tsx b/components/modals/voice-chat-modal.tsx index 8bf1da39..aeec19d2 100644 --- a/components/modals/voice-chat-modal.tsx +++ b/components/modals/voice-chat-modal.tsx @@ -1,5 +1,7 @@ 'use client'; +import { useCallback, useEffect, useRef, useState } from 'react'; + import { useTranslations } from 'next-intl'; import { Dialog, @@ -7,7 +9,16 @@ import { DialogDescription, DialogTitle, } from '@/components/ui/dialog'; -import { Mic, MicOff, X, Loader2 } from 'lucide-react'; +import type { TranscriptEntry } from '@/hooks/use-livekit-transcription'; +import { + Mic, + MicOff, + Volume2, + VolumeX, + X, + Loader2, + ArrowDown, +} from 'lucide-react'; import { Tooltip, TooltipTrigger, @@ -86,31 +97,82 @@ const pulseAnimations = ` 0%, 100% { transform: scale(0.9); opacity: 0.75; } 50% { transform: scale(1.6); opacity: 0.9; } } + + @keyframes mentorRingPulse { + 0%, 100% { transform: scale(1); opacity: 0.55; } + 50% { transform: scale(1.04); opacity: 0.95; } + } + + @keyframes mentorRingRipple { + 0% { transform: scale(0.96); opacity: 0.7; } + 70% { opacity: 0.12; } + 100% { transform: scale(1.22); opacity: 0; } + } + + @keyframes transcriptCaret { + 0%, 45% { opacity: 1; } + 50%, 95% { opacity: 0.15; } + 100% { opacity: 1; } + } `; +/** + * How close to the bottom (px) still counts as "pinned to the newest line". + * Anything further up is treated as the user reading back, which suspends + * auto-scroll until they return. + */ +const AUTO_SCROLL_PIN_THRESHOLD_PX = 24; + interface VoiceChatModalProps { isOpen: boolean; onClose: () => void; - toggleMute: () => void; - isMuted: boolean; + /** Toggles the user's own microphone (outbound audio). */ + toggleMicMute: () => void; + isMicMuted: boolean; + /** Toggles playback of the mentor's voice (inbound audio). */ + toggleMentorAudio: () => void; + isMentorAudioMuted: boolean; connectionState: ConnectionState; + /** Whether the user is currently speaking. */ isSpeaking: boolean; + /** Whether the mentor agent is currently speaking. */ + isMentorSpeaking: boolean; + /** Accumulated call transcript, oldest first. */ + transcript?: TranscriptEntry[]; + /** Whether the newest transcript line is still in progress. */ + isTranscriptLive?: boolean; + /** Display name of the mentor, used to label its transcript lines. */ + mentorName?: string; } export function VoiceChatModal({ isOpen, onClose, - toggleMute, - isMuted, + toggleMicMute, + isMicMuted, + toggleMentorAudio, + isMentorAudioMuted, connectionState, isSpeaking, + isMentorSpeaking, + transcript = [], + isTranscriptLive = false, + mentorName, }: VoiceChatModalProps) { const t = useTranslations('modalsVoiceChatModal'); const isLoading = connectionState === 'requesting-permission' || connectionState === 'connecting'; const isConnected = connectionState === 'connected'; - const shouldAnimate = isConnected && !isMuted; + // The blob is the single indicator for the whole conversation, so it stays + // alive for as long as the call is up. Muting your own microphone must never + // make the call look dead — that only silences the sound-wave bars below. + const shouldAnimate = isConnected; + // The sound-wave bars belong to the user: they react to the local mic only. + const isUserVoiceActive = isSpeaking && !isMicMuted; + // The violet ring belongs to the mentor: it only shows while the agent is + // actually audible. Both can be on at once — real conversations overlap. + const isMentorVoiceActive = isMentorSpeaking && !isMentorAudioMuted; const loadingMessage = isLoading ? connectionState === 'requesting-permission' @@ -118,6 +180,49 @@ export function VoiceChatModal({ : t('connectingToVoiceChat') : null; + // One caption replaces the old pair of status rows. Highest-precedence fact + // wins: the agent being silenced is the most surprising state, then who is + // talking, then the user's own mic. + // --- Transcript auto-scroll ------------------------------------------- + // The band follows the newest line, but only while the reader is already at + // the bottom. Scrolling up to re-read must never be yanked back down. + const transcriptScrollRef = useRef(null); + const [isPinnedToLatest, setIsPinnedToLatest] = useState(true); + + const scrollToLatest = useCallback(() => { + const element = transcriptScrollRef.current; + if (!element) return; + element.scrollTop = element.scrollHeight; + }, []); + + useEffect(() => { + if (!isPinnedToLatest) return; + scrollToLatest(); + }, [transcript, isPinnedToLatest, scrollToLatest]); + + const handleTranscriptScroll = useCallback(() => { + const element = transcriptScrollRef.current; + if (!element) return; + const distanceFromBottom = + element.scrollHeight - element.scrollTop - element.clientHeight; + setIsPinnedToLatest(distanceFromBottom <= AUTO_SCROLL_PIN_THRESHOLD_PX); + }, []); + + const handleJumpToLatest = useCallback(() => { + setIsPinnedToLatest(true); + scrollToLatest(); + }, [scrollToLatest]); + + const showJumpToLatest = !isPinnedToLatest && transcript.length > 0; + + const callStatusLabel = isMentorAudioMuted + ? t('agentMuted') + : isMentorSpeaking + ? t('agentSpeaking') + : isMicMuted + ? t('micMuted') + : t('listening'); + return ( <>