/** * @fileoverview Voice input with three providers: Claude (this server's Claude Code * login), Deepgram Nova-3, and the Web Speech API. * * Defines three singleton objects: * * - ClaudeVoiceProvider — Dictation through Codeman's own `/ws/voice/stream`, which * relays to the speech-to-text service Claude Code's `/voice` mode uses. No API key: * the server holds the OAuth token, the browser only sends PCM16 @16 kHz (AudioWorklet, * since MediaRecorder cannot emit raw PCM) and receives text. See docs/claude-voice-plan.md. * * - DeepgramProvider — Direct browser-to-Deepgram WebSocket connection for speech-to-text. * Captures audio via MediaRecorder, streams chunks every 250ms, handles KeepAlive pings, * auto-detects MIME type (opus/webm/mp4), and supports custom key terms for dev vocabulary. * * - VoiceInput — High-level voice input controller. Toggle mode: tap mic to start, tap * again to stop. Auto-stops after 3s silence. Shows floating preview overlay with recording * indicator, level meter (AnalyserNode), and elapsed timer. Two insert modes: "direct" * (inject into local echo overlay or PTY) and "compose" (editable textarea overlay). * Includes a temporary green Send button that replaces the settings gear icon after voice input. * Web Speech API has auto-retry (up to 2x) for premature onend and iOS Safari stability check. * * @globals {object} ClaudeVoiceProvider * @globals {object} DeepgramProvider * @globals {object} VoiceInput * * @dependency mobile-handlers.js (MobileDetection for device checks) * @dependency app.js (uses global `app` for sendInput, showToast, terminal focus) * @loadorder 3 of 15 — loaded after mobile-handlers.js, before notification-manager.js */ // Codeman — Voice input with Claude, Deepgram Nova-3 and Web Speech API // Loaded after mobile-handlers.js, before app.js /** Dev vocabulary sent to the recognizer as a hint. Shared by every provider and the settings form. */ const DEFAULT_VOICE_KEYTERMS = 'refactor, endpoint, middleware, callback, async, regex, TypeScript, npm, API, deploy, config, linter, env, webhook, schema, CLI, JSON, CSS, DOM, SSE, backend, frontend, localhost, dependencies, repository, merge, rebase, diff, commit, com'; // ═══════════════════════════════════════════════════════════════ // Voice Input (Deepgram Nova-3 + Web Speech API fallback) // ═══════════════════════════════════════════════════════════════ /** * DeepgramProvider - Speech-to-text via Deepgram Nova-3 WebSocket API. * Direct browser-to-Deepgram connection (no server proxy). * Uses MediaRecorder to capture audio and streams via WebSocket. */ const DeepgramProvider = { _ws: null, _mediaRecorder: null, _stream: null, _silenceTimeout: null, _keepAliveInterval: null, _onResult: null, _onError: null, _onEnd: null, /** * Start streaming audio to Deepgram. * @param {object} opts - { apiKey, language, keyterms[], onResult(text, isFinal), onError(msg), onEnd(), onStream(stream) } */ async start(opts) { this._onResult = opts.onResult; this._onError = opts.onError; this._onEnd = opts.onEnd; // 1. Get microphone access if (!navigator.mediaDevices?.getUserMedia) { this._onError?.('Microphone requires a secure context (HTTPS). Use --https flag or access via localhost.'); this._cleanup(); return; } try { this._stream = await navigator.mediaDevices.getUserMedia({ audio: { noiseSuppression: true, echoCancellation: true, autoGainControl: true } }); } catch (err) { const msg = err.name === 'NotAllowedError' ? 'Microphone access denied. Check browser settings.' : 'Microphone error: ' + err.message; this._onError?.(msg); this._cleanup(); return; } // Notify caller so it can set up audio level meter opts.onStream?.(this._stream); // 2. Detect best supported MIME type for MediaRecorder const mimeTypes = ['audio/webm;codecs=opus', 'audio/webm', 'audio/mp4']; this._selectedMime = null; for (const mt of mimeTypes) { if (typeof MediaRecorder !== 'undefined' && MediaRecorder.isTypeSupported(mt)) { this._selectedMime = mt; break; } } // 3. Build WebSocket URL (no encoding param — Deepgram auto-detects from container format) const params = new URLSearchParams({ model: 'nova-3', smart_format: 'false', punctuate: 'false', interim_results: 'true', utterance_end_ms: '1500', vad_events: 'true', }); if (opts.language && opts.language !== 'multi') { params.set('language', opts.language); } else if (opts.language === 'multi') { params.set('detect_language', 'true'); } if (opts.keyterms?.length) { for (const term of opts.keyterms) { const trimmed = term.trim(); if (trimmed) params.append('keyterm', trimmed + ':2'); } } // 4. Connect WebSocket (trim API key to avoid whitespace auth failures) const apiKey = (opts.apiKey || '').trim(); if (!apiKey) { this._onError?.('No Deepgram API key configured. Add one in Settings > Voice.'); this._cleanup(); return; } const wsUrl = `wss://api.deepgram.com/v1/listen?${params}`; try { this._ws = new WebSocket(wsUrl, ['token', apiKey]); } catch (err) { this._onError?.('Failed to connect to Deepgram: ' + err.message); this._cleanup(); return; } this._ws.onopen = () => { // 5. Send KeepAlive every 8s to prevent Deepgram from closing idle connections // (covers the gap before MediaRecorder produces its first chunk) this._keepAliveInterval = setInterval(() => { if (this._ws?.readyState === WebSocket.OPEN) { try { this._ws.send(JSON.stringify({ type: 'KeepAlive' })); } catch (_e) { /* ignore */ } } }, 8000); // 6. Start MediaRecorder once connected this._startRecording(); }; this._ws.onmessage = (event) => { try { const data = JSON.parse(event.data); if (data.type === 'Results' && data.channel?.alternatives?.[0]) { const alt = data.channel.alternatives[0]; const transcript = alt.transcript || ''; if (transcript) { const isFinal = data.is_final === true; this._onResult?.(transcript, isFinal); this._resetSilenceTimeout(); } } } catch (_e) { // Ignore parse errors for non-JSON messages } }; this._ws.onerror = () => { // WebSocket onerror doesn't carry useful info — onclose handles it }; this._ws.onclose = (event) => { clearInterval(this._keepAliveInterval); this._keepAliveInterval = null; if (event.code === 1008) { this._onError?.('Authentication failed. Check your Deepgram API key in Settings > Voice.'); } else if (event.code === 1006) { // 1006 = abnormal closure (no close frame). Usually auth failure, expired key, or no credits. this._onError?.('Deepgram connection failed (1006). Check your API key is valid and has credits in Settings > Voice.'); } else if (event.code !== 1000) { this._onError?.('Deepgram connection closed: ' + (event.reason || `code ${event.code}`)); } this._stopRecording(); this._onEnd?.(); }; }, _startRecording() { if (!this._stream || !this._ws || this._ws.readyState !== WebSocket.OPEN) return; const recorderOpts = this._selectedMime ? { mimeType: this._selectedMime } : {}; try { this._mediaRecorder = new MediaRecorder(this._stream, recorderOpts); } catch (err) { this._onError?.('MediaRecorder failed: ' + err.message); this._cleanup(); return; } this._mediaRecorder.ondataavailable = (event) => { if (event.data.size > 0 && this._ws?.readyState === WebSocket.OPEN) { this._ws.send(event.data); } }; this._mediaRecorder.start(250); // Send chunks every 250ms this._resetSilenceTimeout(); }, _stopRecording() { if (this._mediaRecorder && this._mediaRecorder.state !== 'inactive') { try { this._mediaRecorder.stop(); } catch (_e) { /* already stopped */ } } // Stop all mic tracks if (this._stream) { this._stream.getTracks().forEach(t => t.stop()); } }, _resetSilenceTimeout() { clearTimeout(this._silenceTimeout); this._silenceTimeout = setTimeout(() => { this.stop(); }, 3000); }, stop() { clearTimeout(this._silenceTimeout); this._silenceTimeout = null; clearInterval(this._keepAliveInterval); this._keepAliveInterval = null; this._stopRecording(); // Detach WS handlers before closing to prevent stale onclose from // killing a subsequent recording that starts before the close completes if (this._ws) { this._ws.onclose = null; this._ws.onmessage = null; this._ws.onerror = null; if (this._ws.readyState === WebSocket.OPEN) { try { this._ws.close(1000); } catch (_e) { /* ignore */ } } this._ws = null; } // Save onEnd before nulling — must notify VoiceInput when silence timeout // triggers stop internally (VoiceInput.onEnd guards with isRecording check) const onEnd = this._onEnd; this._onResult = null; this._onError = null; this._onEnd = null; onEnd?.(); }, _cleanup() { this.stop(); this._mediaRecorder = null; this._stream = null; this._selectedMime = null; } }; /** * ClaudeVoiceProvider - Speech-to-text through this Codeman server's Claude Code * login, i.e. the same service the CLI's own `/voice` mode uses. No API key. * * Audio goes browser -> Codeman -> Anthropic: the OAuth token never leaves the * server, so the browser only ever sends PCM and receives text * (docs/claude-voice-plan.md). * * ⚠️ The upstream endpoint is opened as linear16 / 16 kHz / mono, so capture MUST * be raw PCM at that rate. MediaRecorder cannot emit raw PCM (container formats * only), which is why this path uses an AudioWorklet rather than reusing * DeepgramProvider's recorder. The AudioContext is constructed at 16000 Hz so the * browser does the resampling. * * ⚠️ Transcript frames carry the WHOLE running transcript, not deltas. Callers * must replace, never concatenate. */ const ClaudeVoiceProvider = { _ws: null, _stream: null, _audioContext: null, _workletNode: null, _sourceNode: null, _scriptNode: null, _silenceTimeout: null, _onResult: null, _onError: null, _onEnd: null, _finalized: false, /** How long without any transcript before the recording gives up on its own. */ SILENCE_MS: 6000, /** * Start streaming. * @param {object} opts - { language, keyterms[], onResult(text, isFinal), onError(msg), onEnd(), onStream(stream) } */ async start(opts) { this._onResult = opts.onResult; this._onError = opts.onError; this._onEnd = opts.onEnd; this._finalized = false; if (!navigator.mediaDevices?.getUserMedia) { this._onError?.('Microphone requires a secure context (HTTPS). Use --https flag or access via localhost.'); this._cleanup(); return; } try { this._stream = await navigator.mediaDevices.getUserMedia({ audio: { noiseSuppression: true, echoCancellation: true, autoGainControl: true } }); } catch (err) { const msg = err.name === 'NotAllowedError' ? 'Microphone access denied. Check browser settings.' : 'Microphone error: ' + err.message; this._onError?.(msg); this._cleanup(); return; } opts.onStream?.(this._stream); const params = new URLSearchParams(); if (opts.language) params.set('language', opts.language); if (opts.keyterms?.length) params.set('keyterms', opts.keyterms.join(',')); const proto = location.protocol === 'https:' ? 'wss:' : 'ws:'; try { this._ws = new WebSocket(`${proto}//${location.host}/ws/voice/stream?${params}`); } catch (err) { this._onError?.('Failed to open voice stream: ' + err.message); this._cleanup(); return; } this._ws.binaryType = 'arraybuffer'; this._ws.onopen = () => { // Capture starts only once the socket is up: PCM buffered before that would // be the oldest audio, and dropping it keeps the transcript aligned with what // the user hears themselves saying. this._startCapture().catch((err) => { this._onError?.('Microphone capture failed: ' + err.message); this.stop(); }); this._resetSilenceTimeout(); }; this._ws.onmessage = (event) => { let msg; try { msg = JSON.parse(event.data); } catch (_e) { return; } if (msg.t === 'transcript' && msg.text) { this._resetSilenceTimeout(); this._onResult?.(msg.text, msg.final === true); } else if (msg.t === 'error') { this._onError?.(msg.message || 'Voice transcription failed'); } }; this._ws.onerror = () => { // onclose carries the actionable detail (close code); nothing useful here. }; this._ws.onclose = (event) => { if (event.code === 4004) { this._onError?.(this._unavailableMessage(event.reason)); } else if (event.code === 4008) { this._onError?.('Too many voice streams are already running on this server.'); } else if (event.code === 4003) { this._onError?.('Voice stream refused (origin not allowed).'); } else if (event.code !== 1000 && !this._finalized) { this._onError?.('Voice stream closed: ' + (event.reason || `code ${event.code}`)); } this._stopCapture(); const onEnd = this._onEnd; this._onEnd = null; onEnd?.(); }; }, /** Map the server's close reason onto something a user can act on. */ _unavailableMessage(reason) { if (reason === 'expired') return 'Claude login expired. Run a Claude session to refresh it, then try again.'; if (reason === 'disabled') return 'Claude voice is off. Enable it in Settings > Voice.'; return 'No Claude Code login found on the server. Sign in with `claude` there, or use Deepgram.'; }, /** Wire mic -> 16 kHz PCM16 frames -> WebSocket. */ async _startCapture() { const Ctx = window.AudioContext || window.webkitAudioContext; // Ask for 16 kHz directly so the browser resamples; Safari may hand back its // own rate, which _pcmFromFloat32 then downsamples to match. this._audioContext = new Ctx({ sampleRate: 16000 }); if (this._audioContext.state === 'suspended') await this._audioContext.resume(); this._sourceNode = this._audioContext.createMediaStreamSource(this._stream); if (this._audioContext.audioWorklet) { await this._audioContext.audioWorklet.addModule(this._workletUrl()); this._workletNode = new AudioWorkletNode(this._audioContext, 'pcm-frame-processor'); this._workletNode.port.onmessage = (event) => this._sendAudio(event.data); this._sourceNode.connect(this._workletNode); // A worklet with no destination is not pulled in some engines; a zero-gain // sink keeps the graph running without echoing the mic to the speakers. const sink = this._audioContext.createGain(); sink.gain.value = 0; this._workletNode.connect(sink).connect(this._audioContext.destination); return; } // Fallback for engines without AudioWorklet (older Safari): deprecated, but // it is this or no dictation at all there. this._scriptNode = this._audioContext.createScriptProcessor(4096, 1, 1); this._scriptNode.onaudioprocess = (event) => { this._sendAudio(this._pcmFromFloat32(event.inputBuffer.getChannelData(0), this._audioContext.sampleRate)); }; this._sourceNode.connect(this._scriptNode); this._scriptNode.connect(this._audioContext.destination); }, /** * Worklet URL carrying this page's cache-bust token. * * ⚠️ Static assets are served `immutable` for a year, and `cacheBustAssets` * only rewrites `.js` refs in `