diff --git a/apps/docs/docs.json b/apps/docs/docs.json index a859ab3baf..71d62c34d7 100644 --- a/apps/docs/docs.json +++ b/apps/docs/docs.json @@ -58,7 +58,8 @@ "goal-mode", "fast-sessions", "memory", - "file-attachments" + "file-attachments", + "voice" ] }, { diff --git a/apps/docs/environment-variables.mdx b/apps/docs/environment-variables.mdx index 7fff9905ef..57c655b072 100644 --- a/apps/docs/environment-variables.mdx +++ b/apps/docs/environment-variables.mdx @@ -372,6 +372,7 @@ as per-task auth tokens or workspace paths. | `R_ALLOWED_EMAILS` | Optional | Comma-separated email allowlist for deployments that restrict sign-in by email. | | `R_ELEVENLABS_API_KEY` | Optional | ElevenLabs API key for narrated feature-demo videos. The key stays on the control plane; sandboxes reach text-to-speech only through an authenticated Roomote endpoint. A key scoped to text-to-speech only is sufficient and recommended. | | `R_ELEVENLABS_VOICE_ID` | Optional | ElevenLabs voice ID used for feature-demo narration. Required alongside the API key for narration to be available. | +| `R_VOICE_OPENAI_API_KEY` | Optional | OpenAI key dedicated to GPT-Live voice conversations. Falls back to `OPENAI_API_KEY` when unset and requires project access to `gpt-live-1`. The key stays on the control plane; browsers receive only a server-negotiated WebRTC session answer. | During Microsoft Teams setup, Roomote uses the Microsoft Entra app values for the Teams bot by default. Use **Show advanced config** after the Directory diff --git a/apps/docs/voice.mdx b/apps/docs/voice.mdx new file mode 100644 index 0000000000..3fc78c3f7b --- /dev/null +++ b/apps/docs/voice.mdx @@ -0,0 +1,68 @@ +--- +title: Voice +icon: audio-lines +description: Talk naturally with a Fast Session using GPT-Live. +--- + +Voice turns a Fast Session into a natural, full-duplex spoken conversation. +GPT-Live handles listening, speaking, and interruptions while the actual Fast +Session handles questions and work with its selected model, tools, context, and +durable transcript. You can move between voice and text without switching to a +separate voice-only agent. + +## Enabling voice + +Voice uses OpenAI GPT-Live-1 and is available on deployments that set +`R_VOICE_OPENAI_API_KEY` to a key from an OpenAI project with GPT-Live access. +Voice is opt-in: the deployment's general `OPENAI_API_KEY` is not used, so +enabling OpenAI for task inference does not turn voice on. + +When the voice key is not configured the voice button does not appear. The key +stays on the control plane. The browser sends its WebRTC connection offer to Roomote +and receives only the negotiated session answer; it never receives the API key. + +## Using voice + +1. Select the voice button in a composer: in an open Session, or on the home + page and the **New Session** dialog. From the home page or dialog a new + Session is created and the call starts inside it. +2. Grant microphone access when the browser asks. A short rising tone + confirms the call is open; a falling tone marks the end. A **Call started** + marker appears in the Session. +3. Talk to Roomote the way you would on a phone call. It acknowledges each + request in a few words, hands the work to the Fast Session, and reports the + result out loud when it lands. Greetings, thanks, and small talk are + answered directly without starting Fast work. +4. Speak at any time to interrupt. Roomote keeps listening while it speaks, + and follow-ups go back through the same Fast Session. You can also type in + the composer during the call. +5. Use the in-call controls to mute your microphone, silence Roomote's audio + without muting yourself, or end the call. The button stays highlighted + while the call is active, and a **Call ended** marker records its length. + +Voice input requires a browser with microphone and WebRTC support, which +includes current Chrome, Edge, Safari, and Firefox. + +## How the transcript works + +A voice call is transcribed into the Session as the record of what was said. +Your speech appears as your messages, and what Roomote said out loud appears +as its replies. The Fast Session's work, such as tool calls, launched tasks, +and reports, appears between those turns exactly as it does in a typed +Session, so the timeline shows both the conversation and the work behind it. + +During a call the Fast Session returns its results to the voice rather than +writing them as chat replies; Roomote then reports them in its own words, +keeping numbers, names, paths, and link labels exact. If a result cannot be +spoken, for example because the call drops, it is written to the transcript +as a normal reply so nothing is lost. Typed messages sent during a call are +answered in writing as usual. + +Each spoken request is cleaned up (filler words, false starts, and misheard +terms) by the deployment's helper model before it reaches the Fast Session. +GPT-Live is told which repositories, environments, and integrations the Fast +Session can reach, so it recognises their names, and the same names guide the +cleanup so a misheard repository name is corrected to the real one. + +Ending the call stops the microphone; Fast work already started remains +visible in the Session and follows the normal Session lifecycle. diff --git a/apps/web/src/app/(authenticated)/home/Home.client.test.tsx b/apps/web/src/app/(authenticated)/home/Home.client.test.tsx index e3ee5a02d0..b42297ac4b 100644 --- a/apps/web/src/app/(authenticated)/home/Home.client.test.tsx +++ b/apps/web/src/app/(authenticated)/home/Home.client.test.tsx @@ -25,6 +25,7 @@ let capturedDefaultReasoningEffort: string | null | undefined; let submittedPromptText = 'Test prompt'; const { + voiceState, mockPush, mockToast, mockToastError, @@ -34,6 +35,13 @@ const { mockPreparePromptAttachments, mockStartFastSession, } = vi.hoisted(() => ({ + voiceState: { + enabled: false, + active: false, + start: vi.fn(), + stop: vi.fn(), + onUtterance: undefined as ((text: string) => void) | undefined, + }, mockPush: vi.fn(), mockToast: vi.fn(), mockToastError: vi.fn(), @@ -96,6 +104,24 @@ vi.mock('@/hooks/task-runs', () => ({ }), })); +vi.mock('@/hooks/useVoiceEnabled', () => ({ + useVoiceEnabled: () => voiceState.enabled, +})); + +vi.mock('@/hooks/useLiveVoice', () => ({ + useLiveVoice: ({ onUtterance }: { onUtterance: (text: string) => void }) => { + voiceState.onUtterance = onUtterance; + return { + active: voiceState.active, + status: voiceState.active ? 'listening' : 'idle', + start: voiceState.start, + stop: voiceState.stop, + speak: vi.fn(), + addContext: vi.fn(), + }; + }, +})); + vi.mock('@/lib/prompt-attachments', async () => { const actual = await vi.importActual< typeof import('@/lib/prompt-attachments') @@ -147,6 +173,7 @@ vi.mock('@/components/tasks', async () => { submitDisabledReason, submitWithMetaKey, tools, + voice, }: { onSubmit: (message: PromptInputMessage) => Promise | void; onPromptTextChange?: (value: string) => void; @@ -155,6 +182,7 @@ vi.mock('@/components/tasks', async () => { submitDisabledReason?: string; submitWithMetaKey?: boolean; tools?: import('react').ReactNode; + voice?: { active: boolean; onToggle: () => void }; }) => { capturedSubmitWithMetaKey = submitWithMetaKey; @@ -177,6 +205,15 @@ vi.mock('@/components/tasks', async () => { + {tools} + {voice ? ( + + ) : null}
{placeholder}