feat(voice): dictate through the server's Claude Code login, no API key

The mic button previously needed a Deepgram API key, or fell back to the
browser's Web Speech engine. It can now transcribe through the same
speech-to-text service Claude Code's own /voice mode uses, so anyone signed
in to Claude Code on the server gets dictation with no third-party account.

Claude Code's voice mode cannot be driven directly: it opens the HOST's
microphone (sox/arecord), and the CLI runs in a headless tmux pane while the
human is in a browser somewhere else. So capture stays in the browser and only
the transcription backend is borrowed.

Audio goes browser -> Codeman -> Anthropic. The OAuth token never reaches the
page: the browser sends PCM16 (16 kHz mono, produced by an AudioWorklet since
MediaRecorder cannot emit raw PCM) and receives text.

- GET /api/voice/status reports readiness and never the token
- GET /ws/voice/stream relays one dictation, with the same Host/Origin upgrade
  guard as the terminal socket, plus caps on concurrency, stream length and
  frame size
- credentials are read-only: Codeman never refreshes them, since a refresh
  rotates the refresh token and could sign the user out of their own CLI
- claudeVoiceEnabled (synced, default OFF) gates the whole server side
- voiceSettings.provider picks auto/claude/deepgram/webspeech; auto prefers
  Claude, then a configured Deepgram key, then the browser

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Codeman maintainer
2026-08-10 11:19:51 +02:00
parent 752374abc7
commit 4b51ba306e
19 changed files with 1904 additions and 22 deletions
+126
View File
@@ -0,0 +1,126 @@
/**
* @fileoverview Read-only access to the Claude Code OAuth credentials.
*
* Claude Code stores its subscription OAuth tokens in
* `$CLAUDE_CONFIG_DIR/.credentials.json` (default `~/.claude/.credentials.json`,
* mode 0600) on Linux/Windows, and in the login keychain on macOS. Codeman reads
* the access token to authenticate the voice-dictation relay
* (`src/web/voice-stream.ts`) against the same speech-to-text service the CLI's
* own `/voice` mode uses.
*
* ⚠️ READ-ONLY, deliberately. Codeman never writes this file and never performs
* an OAuth refresh: a refresh ROTATES the refresh token, so racing Claude Code's
* own refresh could invalidate the user's CLI login. An expired access token is
* reported as `expired` and the caller tells the user to run a Claude session
* (which refreshes it) instead.
*
* ⚠️ The token is a bearer secret: it is never logged, never persisted, never
* included in any API response, and never sent to the browser.
*/
import { readFile } from 'fs/promises';
import { execFile } from 'child_process';
import { homedir, userInfo } from 'os';
import { join } from 'path';
/** Result of inspecting the credential store. The token is present only on 'ok'. */
export type ClaudeCredentialStatus = 'ok' | 'expired' | 'missing' | 'malformed';
export interface ClaudeOAuthCredentials {
status: ClaudeCredentialStatus;
/** Bearer token. Present only when status is 'ok'. Never log or serialize this. */
accessToken?: string;
/** Epoch ms the access token expires at, when the store reports one. */
expiresAt?: number;
/** e.g. 'max', 'pro'. Display-only, safe to surface. */
subscriptionType?: string;
}
/** Skew applied to the stored expiry so a token that dies mid-stream is refused up front. */
const EXPIRY_SKEW_MS = 60_000;
/** macOS keychain service holding the same JSON blob as `.credentials.json`. */
const KEYCHAIN_SERVICE = 'Claude Code-credentials';
/** Keychain lookups shell out; keep them short so a locked keychain cannot hang a request. */
const KEYCHAIN_TIMEOUT_MS = 3000;
/**
* Parse a `.credentials.json` payload. Pure: no IO, no clock read (pass `now`),
* so the expiry and shape handling are unit-testable.
*
* Returns 'malformed' for anything that is not the expected `claudeAiOauth`
* shape rather than throwing — a hand-edited or half-written file must degrade
* to "voice unavailable", never to a 500.
*/
export function parseClaudeCredentials(raw: string, now: number): ClaudeOAuthCredentials {
let parsed: unknown;
try {
parsed = JSON.parse(raw);
} catch {
return { status: 'malformed' };
}
if (!parsed || typeof parsed !== 'object') return { status: 'malformed' };
const oauth = (parsed as { claudeAiOauth?: unknown }).claudeAiOauth;
if (!oauth || typeof oauth !== 'object') return { status: 'malformed' };
const record = oauth as Record<string, unknown>;
const accessToken = typeof record.accessToken === 'string' ? record.accessToken.trim() : '';
if (!accessToken) return { status: 'malformed' };
const expiresAt = typeof record.expiresAt === 'number' ? record.expiresAt : undefined;
const subscriptionType = typeof record.subscriptionType === 'string' ? record.subscriptionType : undefined;
// An expired token is a real state (the CLI refreshes on its next run), not a
// malformed store: report it separately so the UI can say something useful.
if (expiresAt !== undefined && expiresAt - EXPIRY_SKEW_MS <= now) {
return { status: 'expired', expiresAt, subscriptionType };
}
return { status: 'ok', accessToken, expiresAt, subscriptionType };
}
/** Path of the credentials file, honoring CLAUDE_CONFIG_DIR like the CLI does. */
export function claudeCredentialsPath(env: NodeJS.ProcessEnv = process.env): string {
const configDir = typeof env.CLAUDE_CONFIG_DIR === 'string' && env.CLAUDE_CONFIG_DIR.trim();
return join(configDir || join(homedir(), '.claude'), '.credentials.json');
}
/** Read the macOS keychain entry. Resolves to null on any failure (locked, absent, non-mac). */
function readKeychainCredentials(): Promise<string | null> {
return new Promise((resolve) => {
execFile(
'security',
['find-generic-password', '-a', userInfo().username, '-w', '-s', KEYCHAIN_SERVICE],
{ encoding: 'utf-8', timeout: KEYCHAIN_TIMEOUT_MS },
(err, stdout) => resolve(err ? null : stdout.trim() || null)
);
});
}
/**
* Locate and parse the Claude Code OAuth credentials.
*
* File first (present on every platform once the CLI has run there), keychain
* second on macOS. Never caches: Claude Code rewrites the store roughly every
* 8 hours, and a cached token would go stale inside a long-lived server.
*/
export async function readClaudeOAuthCredentials(now: number = Date.now()): Promise<ClaudeOAuthCredentials> {
let fileResult: ClaudeOAuthCredentials | null = null;
try {
fileResult = parseClaudeCredentials(await readFile(claudeCredentialsPath(), 'utf-8'), now);
} catch {
fileResult = null;
}
if (fileResult && fileResult.status !== 'malformed') return fileResult;
if (process.platform === 'darwin') {
const raw = await readKeychainCredentials();
if (raw) {
const keychainResult = parseClaudeCredentials(raw, now);
if (keychainResult.status !== 'malformed') return keychainResult;
}
}
return fileResult ?? { status: 'missing' };
}
+56
View File
@@ -0,0 +1,56 @@
/**
* @fileoverview Bounds and endpoint config for Claude voice dictation.
*
* Backs the browser → Codeman → Anthropic dictation relay (`src/web/voice-stream.ts`,
* `src/web/routes/voice-routes.ts`; design in `docs/claude-voice-plan.md`).
*
* Why everything here is bounded: an open microphone is an open pipe. Each live
* stream holds a browser socket, an upstream socket and a keepalive timer, and
* every second of audio is billed against the server owner's Claude subscription.
* A tab left recording (phone in a pocket, forgotten laptop) must cost a bounded
* amount, so streams die on their own at `MAX_STREAM_MS` and the server refuses
* more than `MAX_CONCURRENT_STREAMS` at once.
*
* The audio frame cap is a memory guard on a socket that carries attacker-shaped
* binary data: PCM16 at 16 kHz mono is 32 KB/s, so a 256 ms frame is ~8 KB and
* anything near 64 KB is either a broken client or an attempt to make the relay
* buffer for someone else.
*/
/** Upstream speech-to-text service (the one Claude Code's own `/voice` mode uses). */
export const VOICE_STREAM_HOST = 'wss://api.anthropic.com';
/** Path of the streaming speech-to-text endpoint. */
export const VOICE_STREAM_PATH = '/api/ws/speech_to_text/voice_stream';
/**
* Base override, for tests (point the relay at a local mock) and for users on an
* Anthropic-compatible gateway. Must be a ws:// or wss:// origin.
*/
export function voiceStreamBase(env: NodeJS.ProcessEnv = process.env): string {
const override = typeof env.CODEMAN_VOICE_STREAM_BASE === 'string' ? env.CODEMAN_VOICE_STREAM_BASE.trim() : '';
if (override && /^wss?:\/\//.test(override)) return override.replace(/\/+$/, '');
return VOICE_STREAM_HOST;
}
/** Upstream drops an idle socket; the CLI pings at 8s and so do we. */
export const KEEPALIVE_INTERVAL_MS = 8000;
/** Hard ceiling on one dictation. Long enough for any real utterance, short enough to bound a forgotten mic. */
export const MAX_STREAM_MS = 5 * 60_000;
/** Concurrent relays server-wide. Dictation is a human-paced, one-at-a-time act. */
export const MAX_CONCURRENT_STREAMS = 4;
/** Largest single audio frame accepted from the browser (~2s of PCM16 @16 kHz mono). */
export const MAX_AUDIO_FRAME_BYTES = 64 * 1024;
/** How long to wait for the final transcript after the client asks to finalize. */
export const FINALIZE_TIMEOUT_MS = 3000;
/** Upstream caps the keyterms header; mirrors the CLI's own limit. */
export const MAX_KEYTERMS_HEADER_CHARS = 1024;
/** Audio format the endpoint is opened with. The browser worklet must match exactly. */
export const AUDIO_SAMPLE_RATE = 16000;
export const AUDIO_CHANNELS = 1;
+2
View File
@@ -19,6 +19,8 @@ export interface ConfigPort {
getTerminalHistoryConfig(): Promise<TerminalHistoryConfig>;
/** Synced `agentSkillEnabled` app setting (default OFF); gates per-case agent-skill injection. */
getAgentSkillEnabled(): Promise<boolean>;
/** Synced `claudeVoiceEnabled` app setting (default OFF); gates the Claude voice dictation relay. */
getClaudeVoiceEnabled(): Promise<boolean>;
getDefaultClaudeMdPath(): Promise<string | undefined>;
getLightState(identity?: { username: string; role: 'admin' | 'user' }): unknown;
getLightSessionsState(): unknown[];
+29
View File
@@ -1966,6 +1966,18 @@
<div class="set-row-text"><span class="set-row-label">Active provider</span></div>
<span class="voice-provider-status" id="voiceProviderStatus">&mdash;</span>
</div>
<div class="set-row has-field" data-search="voice provider claude deepgram web speech engine">
<div class="set-row-text">
<span class="set-row-label">Speech engine</span>
<span class="set-row-desc">Auto prefers Claude when this server can transcribe, then Deepgram, then the browser.</span>
</div>
<select id="voiceProvider" class="set-select">
<option value="auto">Auto</option>
<option value="claude">Claude</option>
<option value="deepgram">Deepgram</option>
<option value="webspeech">Browser (Web Speech)</option>
</select>
</div>
<div class="set-row has-field" data-search="voice insert mode compose">
<div class="set-row-text">
<span class="set-row-label">Insert mode</span>
@@ -1979,6 +1991,23 @@
</div>
</div>
<div class="set-group">
<div class="set-group-head"><h4>Claude</h4><span class="set-scope">synced</span></div>
<div class="set-group-body">
<div class="set-row" data-search="claude voice dictation subscription no api key">
<div class="set-row-text">
<span class="set-row-label">Transcribe with this server's Claude login</span>
<span class="set-row-desc">Dictation with no API key, through the same service Claude Code's own /voice mode uses. Microphone audio goes to Anthropic and is billed to this machine's Claude subscription, for everyone who can reach this UI.</span>
</div>
<label class="switch switch-sm"><input type="checkbox" id="appSettingsClaudeVoice"><span class="slider"></span></label>
</div>
<div class="set-row" data-search="claude voice status credentials">
<div class="set-row-text"><span class="set-row-label">Server status</span></div>
<span class="voice-provider-status" id="voiceClaudeStatus" data-i18n-skip>&mdash;</span>
</div>
</div>
</div>
<div class="set-group">
<div class="set-group-head"><h4>Deepgram Nova-3</h4><span class="set-scope">device</span></div>
<div class="set-group-body">
+44 -6
View File
@@ -482,17 +482,18 @@ Object.assign(CodemanApp.prototype, {
const voiceCfg = VoiceInput._getDeepgramConfig();
document.getElementById('voiceDeepgramKey').value = voiceCfg.apiKey || '';
document.getElementById('voiceLanguage').value = voiceCfg.language || 'en-US';
document.getElementById('voiceKeyterms').value = voiceCfg.keyterms || 'refactor, endpoint, middleware, callback, async, regex, TypeScript, npm, API, deploy, config, linter, env, webhook, schema, CLI, JSON, CSS, DOM, SSE, backend, frontend, localhost, dependencies, repository, merge, rebase, diff, commit, com';
document.getElementById('voiceKeyterms').value = voiceCfg.keyterms || DEFAULT_VOICE_KEYTERMS;
document.getElementById('voiceInsertMode').value = voiceCfg.insertMode || 'direct';
document.getElementById('voiceProvider').value = voiceCfg.provider || 'auto';
document.getElementById('appSettingsClaudeVoice').checked = settings.claudeVoiceEnabled ?? false;
// Reset key visibility to hidden
const keyInput = document.getElementById('voiceDeepgramKey');
keyInput.type = 'password';
document.getElementById('voiceKeyToggleBtn').textContent = 'Show';
// Update provider status
const providerName = VoiceInput.getActiveProviderName();
const providerEl = document.getElementById('voiceProviderStatus');
providerEl.textContent = providerName;
providerEl.className = 'voice-provider-status' + (providerName.startsWith('Deepgram') ? ' active' : '');
// Update provider status. The Claude row needs a fresh server probe: the
// setting is synced, so another device may have flipped it since page load.
this._renderVoiceProviderStatus();
VoiceInput.refreshClaudeStatus().then(() => this._renderVoiceProviderStatus());
// Updates section — show current version, reset transient result/progress UI.
this._initUpdatesSection();
@@ -1830,6 +1831,37 @@ Object.assign(CodemanApp.prototype, {
}
},
/**
* Paint both Voice status rows: which provider a mic press would use, and what
* the server reports about its Claude login. Called on open and again once the
* /api/voice/status probe resolves.
*/
_renderVoiceProviderStatus() {
const providerEl = document.getElementById('voiceProviderStatus');
if (providerEl) {
const providerName = VoiceInput.getActiveProviderName();
providerEl.textContent = providerName;
const live = providerName.startsWith('Deepgram Nova') || providerName.startsWith('Claude (this');
providerEl.className = 'voice-provider-status' + (live ? ' active' : '');
}
const claudeEl = document.getElementById('voiceClaudeStatus');
if (!claudeEl) return;
const status = VoiceInput._claudeStatus;
const text = !status
? 'Checking...'
: status.available
? `Ready${status.subscriptionType ? ` (${status.subscriptionType})` : ''}`
: status.reason === 'expired'
? 'Login expired - run a Claude session to refresh'
: status.reason === 'no-credentials'
? 'No Claude Code login on the server'
: status.reason === 'malformed'
? 'Claude credentials unreadable'
: 'Off - enable it above';
claudeEl.textContent = text;
claudeEl.className = 'voice-provider-status' + (status?.available ? ' active' : '');
},
async saveAppSettings() {
// Gesture overlay is injected at page render (server-side), so a change to it
// only takes effect on reload — remember the prior value to decide below.
@@ -1892,6 +1924,7 @@ Object.assign(CodemanApp.prototype, {
// Claude Permissions settings
agentTeamsEnabled: document.getElementById('appSettingsAgentTeams').checked,
agentSkillEnabled: document.getElementById('appSettingsAgentSkill').checked,
claudeVoiceEnabled: document.getElementById('appSettingsClaudeVoice').checked,
claudeModel: document.getElementById('appSettingsClaudeModel').value,
opusContext1mEnabled: document.getElementById('appSettingsOpusContext1m').checked,
remoteAutoReconnect: document.getElementById('appSettingsRemoteAutoReconnect').checked,
@@ -1931,6 +1964,7 @@ Object.assign(CodemanApp.prototype, {
// Save voice settings to localStorage + include in server payload for cross-device sync
const voiceSettings = {
provider: document.getElementById('voiceProvider').value,
apiKey: document.getElementById('voiceDeepgramKey').value.trim(),
language: document.getElementById('voiceLanguage').value,
keyterms: document.getElementById('voiceKeyterms').value.trim(),
@@ -2110,6 +2144,10 @@ Object.assign(CodemanApp.prototype, {
this.closeAppSettings();
// Voice availability is a server-side answer, so re-probe after a save:
// otherwise the mic keeps using the pre-save provider until the next reload.
VoiceInput.refreshClaudeStatus();
// The gesture overlay is injected at page render (server reads
// gestureControlEnabled from settings.json), so a change only takes effect on
// reload. Reload when it actually changed — the server PUT above already
+421 -13
View File
@@ -1,7 +1,13 @@
/**
* @fileoverview Voice input with Deepgram Nova-3 (primary) and Web Speech API (fallback).
* @fileoverview Voice input with three providers: Claude (this server's Claude Code
* login), Deepgram Nova-3, and the Web Speech API.
*
* Defines two singleton objects:
* Defines three singleton objects:
*
* - ClaudeVoiceProvider — Dictation through Codeman's own `/ws/voice/stream`, which
* relays to the speech-to-text service Claude Code's `/voice` mode uses. No API key:
* the server holds the OAuth token, the browser only sends PCM16 @16 kHz (AudioWorklet,
* since MediaRecorder cannot emit raw PCM) and receives text. See docs/claude-voice-plan.md.
*
* - DeepgramProvider — Direct browser-to-Deepgram WebSocket connection for speech-to-text.
* Captures audio via MediaRecorder, streams chunks every 250ms, handles KeepAlive pings,
@@ -14,6 +20,7 @@
* Includes a temporary green Send button that replaces the settings gear icon after voice input.
* Web Speech API has auto-retry (up to 2x) for premature onend and iOS Safari stability check.
*
* @globals {object} ClaudeVoiceProvider
* @globals {object} DeepgramProvider
* @globals {object} VoiceInput
*
@@ -22,9 +29,13 @@
* @loadorder 3 of 15 — loaded after mobile-handlers.js, before notification-manager.js
*/
// Codeman — Voice input with Deepgram Nova-3 and Web Speech API fallback
// Codeman — Voice input with Claude, Deepgram Nova-3 and Web Speech API
// Loaded after mobile-handlers.js, before app.js
/** Dev vocabulary sent to the recognizer as a hint. Shared by every provider and the settings form. */
const DEFAULT_VOICE_KEYTERMS =
'refactor, endpoint, middleware, callback, async, regex, TypeScript, npm, API, deploy, config, linter, env, webhook, schema, CLI, JSON, CSS, DOM, SSE, backend, frontend, localhost, dependencies, repository, merge, rebase, diff, commit, com';
// ═══════════════════════════════════════════════════════════════
// Voice Input (Deepgram Nova-3 + Web Speech API fallback)
// ═══════════════════════════════════════════════════════════════
@@ -245,7 +256,282 @@ const DeepgramProvider = {
};
/**
* VoiceInput - Speech-to-text with Deepgram Nova-3 (primary) and Web Speech API (fallback).
* ClaudeVoiceProvider - Speech-to-text through this Codeman server's Claude Code
* login, i.e. the same service the CLI's own `/voice` mode uses. No API key.
*
* Audio goes browser -> Codeman -> Anthropic: the OAuth token never leaves the
* server, so the browser only ever sends PCM and receives text
* (docs/claude-voice-plan.md).
*
* ⚠️ The upstream endpoint is opened as linear16 / 16 kHz / mono, so capture MUST
* be raw PCM at that rate. MediaRecorder cannot emit raw PCM (container formats
* only), which is why this path uses an AudioWorklet rather than reusing
* DeepgramProvider's recorder. The AudioContext is constructed at 16000 Hz so the
* browser does the resampling.
*
* ⚠️ Transcript frames carry the WHOLE running transcript, not deltas. Callers
* must replace, never concatenate.
*/
const ClaudeVoiceProvider = {
_ws: null,
_stream: null,
_audioContext: null,
_workletNode: null,
_sourceNode: null,
_scriptNode: null,
_silenceTimeout: null,
_onResult: null,
_onError: null,
_onEnd: null,
_finalized: false,
/** How long without any transcript before the recording gives up on its own. */
SILENCE_MS: 6000,
/**
* Start streaming.
* @param {object} opts - { language, keyterms[], onResult(text, isFinal), onError(msg), onEnd(), onStream(stream) }
*/
async start(opts) {
this._onResult = opts.onResult;
this._onError = opts.onError;
this._onEnd = opts.onEnd;
this._finalized = false;
if (!navigator.mediaDevices?.getUserMedia) {
this._onError?.('Microphone requires a secure context (HTTPS). Use --https flag or access via localhost.');
this._cleanup();
return;
}
try {
this._stream = await navigator.mediaDevices.getUserMedia({
audio: { noiseSuppression: true, echoCancellation: true, autoGainControl: true }
});
} catch (err) {
const msg = err.name === 'NotAllowedError'
? 'Microphone access denied. Check browser settings.'
: 'Microphone error: ' + err.message;
this._onError?.(msg);
this._cleanup();
return;
}
opts.onStream?.(this._stream);
const params = new URLSearchParams();
if (opts.language) params.set('language', opts.language);
if (opts.keyterms?.length) params.set('keyterms', opts.keyterms.join(','));
const proto = location.protocol === 'https:' ? 'wss:' : 'ws:';
try {
this._ws = new WebSocket(`${proto}//${location.host}/ws/voice/stream?${params}`);
} catch (err) {
this._onError?.('Failed to open voice stream: ' + err.message);
this._cleanup();
return;
}
this._ws.binaryType = 'arraybuffer';
this._ws.onopen = () => {
// Capture starts only once the socket is up: PCM buffered before that would
// be the oldest audio, and dropping it keeps the transcript aligned with what
// the user hears themselves saying.
this._startCapture().catch((err) => {
this._onError?.('Microphone capture failed: ' + err.message);
this.stop();
});
this._resetSilenceTimeout();
};
this._ws.onmessage = (event) => {
let msg;
try {
msg = JSON.parse(event.data);
} catch (_e) {
return;
}
if (msg.t === 'transcript' && msg.text) {
this._resetSilenceTimeout();
this._onResult?.(msg.text, msg.final === true);
} else if (msg.t === 'error') {
this._onError?.(msg.message || 'Voice transcription failed');
}
};
this._ws.onerror = () => {
// onclose carries the actionable detail (close code); nothing useful here.
};
this._ws.onclose = (event) => {
if (event.code === 4004) {
this._onError?.(this._unavailableMessage(event.reason));
} else if (event.code === 4008) {
this._onError?.('Too many voice streams are already running on this server.');
} else if (event.code === 4003) {
this._onError?.('Voice stream refused (origin not allowed).');
} else if (event.code !== 1000 && !this._finalized) {
this._onError?.('Voice stream closed: ' + (event.reason || `code ${event.code}`));
}
this._stopCapture();
const onEnd = this._onEnd;
this._onEnd = null;
onEnd?.();
};
},
/** Map the server's close reason onto something a user can act on. */
_unavailableMessage(reason) {
if (reason === 'expired') return 'Claude login expired. Run a Claude session to refresh it, then try again.';
if (reason === 'disabled') return 'Claude voice is off. Enable it in Settings > Voice.';
return 'No Claude Code login found on the server. Sign in with `claude` there, or use Deepgram.';
},
/** Wire mic -> 16 kHz PCM16 frames -> WebSocket. */
async _startCapture() {
const Ctx = window.AudioContext || window.webkitAudioContext;
// Ask for 16 kHz directly so the browser resamples; Safari may hand back its
// own rate, which _pcmFromFloat32 then downsamples to match.
this._audioContext = new Ctx({ sampleRate: 16000 });
if (this._audioContext.state === 'suspended') await this._audioContext.resume();
this._sourceNode = this._audioContext.createMediaStreamSource(this._stream);
if (this._audioContext.audioWorklet) {
await this._audioContext.audioWorklet.addModule(this._workletUrl());
this._workletNode = new AudioWorkletNode(this._audioContext, 'pcm-frame-processor');
this._workletNode.port.onmessage = (event) => this._sendAudio(event.data);
this._sourceNode.connect(this._workletNode);
// A worklet with no destination is not pulled in some engines; a zero-gain
// sink keeps the graph running without echoing the mic to the speakers.
const sink = this._audioContext.createGain();
sink.gain.value = 0;
this._workletNode.connect(sink).connect(this._audioContext.destination);
return;
}
// Fallback for engines without AudioWorklet (older Safari): deprecated, but
// it is this or no dictation at all there.
this._scriptNode = this._audioContext.createScriptProcessor(4096, 1, 1);
this._scriptNode.onaudioprocess = (event) => {
this._sendAudio(this._pcmFromFloat32(event.inputBuffer.getChannelData(0), this._audioContext.sampleRate));
};
this._sourceNode.connect(this._scriptNode);
this._scriptNode.connect(this._audioContext.destination);
},
/**
* Worklet URL carrying this page's cache-bust token.
*
* ⚠️ Static assets are served `immutable` for a year, and `cacheBustAssets`
* only rewrites `.js` refs in `<script>`/`<link>` tags — a URL built here in JS
* is invisible to it. So the token is borrowed from voice-input.js's own script
* tag, which the server DID rewrite. Consequence: **edit the worklet and this
* file together**, or the browser keeps serving the old worklet.
*/
_workletUrl() {
const src = document.querySelector('script[src*="voice-input.js"]')?.getAttribute('src') || '';
const q = src.indexOf('?');
return 'voice-pcm-worklet.js' + (q === -1 ? '' : src.slice(q));
},
/** Float32 [-1,1] at any rate -> Int16 PCM at 16 kHz (nearest-neighbour decimation). */
_pcmFromFloat32(input, sampleRate) {
const ratio = sampleRate / 16000;
const outLength = Math.floor(input.length / ratio);
const out = new Int16Array(outLength);
for (let i = 0; i < outLength; i++) {
const sample = Math.max(-1, Math.min(1, input[Math.floor(i * ratio)]));
out[i] = sample < 0 ? sample * 0x8000 : sample * 0x7fff;
}
return out.buffer;
},
_sendAudio(arrayBuffer) {
if (this._finalized) return;
if (this._ws?.readyState !== WebSocket.OPEN) return;
try {
this._ws.send(arrayBuffer);
} catch (_e) {
/* socket died mid-frame */
}
},
_resetSilenceTimeout() {
clearTimeout(this._silenceTimeout);
this._silenceTimeout = setTimeout(() => this.stop(), this.SILENCE_MS);
},
/**
* Ask for the final transcript and let the server close the socket. Capture stops
* immediately, but the WebSocket stays open: the last (and usually best) transcript
* arrives AFTER the audio does, so closing here would throw away the utterance.
*/
stop() {
clearTimeout(this._silenceTimeout);
this._silenceTimeout = null;
if (this._finalized) return;
this._finalized = true;
this._stopCapture();
if (this._ws?.readyState === WebSocket.OPEN) {
try {
this._ws.send(JSON.stringify({ t: 'finalize' }));
} catch (_e) {
/* ignore */
}
} else {
const onEnd = this._onEnd;
this._onEnd = null;
onEnd?.();
}
},
/** Tear down the audio graph and release the mic. Idempotent. */
_stopCapture() {
if (this._workletNode) {
this._workletNode.port.onmessage = null;
try { this._workletNode.disconnect(); } catch (_e) { /* ignore */ }
this._workletNode = null;
}
if (this._scriptNode) {
this._scriptNode.onaudioprocess = null;
try { this._scriptNode.disconnect(); } catch (_e) { /* ignore */ }
this._scriptNode = null;
}
if (this._sourceNode) {
try { this._sourceNode.disconnect(); } catch (_e) { /* ignore */ }
this._sourceNode = null;
}
if (this._audioContext) {
try { this._audioContext.close(); } catch (_e) { /* ignore */ }
this._audioContext = null;
}
if (this._stream) {
this._stream.getTracks().forEach(t => t.stop());
this._stream = null;
}
},
/** Hard stop: drop the socket without waiting for a final transcript. */
_cleanup() {
this._finalized = true;
clearTimeout(this._silenceTimeout);
this._silenceTimeout = null;
this._stopCapture();
if (this._ws) {
this._ws.onclose = null;
this._ws.onmessage = null;
this._ws.onerror = null;
if (this._ws.readyState === WebSocket.OPEN) {
try { this._ws.close(1000); } catch (_e) { /* ignore */ }
}
this._ws = null;
}
this._onResult = null;
this._onError = null;
this._onEnd = null;
}
};
/**
* VoiceInput - Speech-to-text with Claude (this server's Claude Code login),
* Deepgram Nova-3, or the Web Speech API.
* Toggle mode: tap mic to start, tap again to stop. Auto-stops after silence.
* Shows interim transcription in a floating preview overlay.
* Inserts final text into the active session (user presses Enter to submit).
@@ -273,6 +559,29 @@ const VoiceInput = {
this._initRecognition();
// Always show buttons — if unsupported, toggle() shows a toast
this._showButtons();
// Probe the server's Claude voice availability in the background. `auto`
// resolution reads the cached answer, so the first mic press does not wait
// on a round trip; a miss just falls through to the next provider.
this.refreshClaudeStatus();
},
/** Last /api/voice/status answer, or null before the first probe resolves. */
_claudeStatus: null,
/**
* Re-probe whether this server can transcribe with its Claude Code login.
* Called at init and whenever App Settings opens (the setting is server-side,
* so another device could have flipped it).
*/
async refreshClaudeStatus() {
try {
const res = await fetch('/api/voice/status');
const json = await res.json();
this._claudeStatus = json?.success ? json.data : { available: false, reason: 'disabled' };
} catch (_e) {
this._claudeStatus = { available: false, reason: 'disabled' };
}
return this._claudeStatus;
},
// --- Deepgram config (localStorage only, never sent to server) ---
@@ -294,11 +603,37 @@ const VoiceInput = {
return !!(cfg.apiKey && cfg.apiKey.trim());
},
_claudeAvailable() {
return this._claudeStatus?.available === true;
},
/**
* Which provider a press of the mic would use.
*
* An explicit pick always wins, even when it cannot run — the resulting error
* ("Claude voice is off", "no Deepgram key") is more useful than silently
* transcribing somewhere the user did not choose. `auto` prefers Claude because
* it needs no key and no per-word billing, then the configured Deepgram key,
* then the browser's own engine.
*/
_resolveProvider() {
const pinned = this._getDeepgramConfig().provider;
if (pinned === 'claude' || pinned === 'deepgram' || pinned === 'webspeech') return pinned;
if (this._claudeAvailable()) return 'claude';
if (this._shouldUseDeepgram()) return 'deepgram';
return 'webspeech';
},
/** Get the active provider name for display */
getActiveProviderName() {
if (this._shouldUseDeepgram()) return 'Deepgram Nova-3';
if (this.supported) return 'Web Speech API';
return 'None';
switch (this._resolveProvider()) {
case 'claude':
return this._claudeAvailable() ? 'Claude (this server’s login)' : 'Claude (unavailable)';
case 'deepgram':
return this._shouldUseDeepgram() ? 'Deepgram Nova-3' : 'Deepgram (no API key)';
default:
return this.supported ? 'Web Speech API' : 'None';
}
},
/** Try to create a SpeechRecognition instance */
@@ -334,13 +669,81 @@ const VoiceInput = {
}
this._retryCount = 0;
if (this._shouldUseDeepgram()) {
const provider = this._resolveProvider();
if (provider === 'claude') {
this._startClaude();
} else if (provider === 'deepgram') {
this._startDeepgram();
} else {
this._startWebSpeech();
}
},
_startClaude() {
if (!this._claudeAvailable()) {
const reason = this._claudeStatus?.reason;
app.showToast(
reason === 'expired'
? 'Claude login expired on the server. Run a Claude session to refresh it.'
: reason === 'no-credentials'
? 'No Claude Code login found on the server. Sign in there with `claude`.'
: 'Claude voice is off. Enable it in Settings > Voice.',
'warning'
);
// Re-probe so a setting flipped on another device is picked up by the next press.
this.refreshClaudeStatus();
return;
}
const cfg = this._getDeepgramConfig();
this.isRecording = true;
this._activeProvider = 'claude';
this._accumulatedFinal = '';
this._lastTranscript = '';
this._hasReceivedResult = false;
this._recordingStartedAt = Date.now();
this._updateButtons('recording');
this._showPreview('Listening...', 'claude');
this._startDurationTimer();
const keyterms = (cfg.keyterms || DEFAULT_VOICE_KEYTERMS)
.split(',').map(t => t.trim()).filter(Boolean);
ClaudeVoiceProvider.start({
// The upstream endpoint wants a bare language tag; the Deepgram picker's
// 'en-US' style narrows to its base, and 'multi' means auto-detect.
language: (cfg.language || 'en-US').split('-')[0],
keyterms,
onStream: (stream) => this._startLevelMeter(stream),
onResult: (text, isFinal) => {
if (!this.isRecording) return;
this._hasReceivedResult = true;
// Each frame is the WHOLE running transcript, so replace rather than append.
this._accumulatedFinal = text;
if (isFinal) {
this._hidePreview();
this._insertText(text);
this.stop();
} else {
this._showPreview(text, 'claude');
}
},
onError: (msg) => {
const wasRecording = this.isRecording;
this.stop();
if (wasRecording) app.showToast(msg, 'error');
},
onEnd: () => {
if (this.isRecording) {
if (this._accumulatedFinal) this._insertText(this._accumulatedFinal);
this.stop();
}
}
});
if (navigator.vibrate) navigator.vibrate(50);
},
_startDeepgram() {
const cfg = this._getDeepgramConfig();
this.isRecording = true;
@@ -353,7 +756,7 @@ const VoiceInput = {
this._showPreview('Listening...', 'deepgram');
this._startDurationTimer();
const keyterms = (cfg.keyterms || 'refactor, endpoint, middleware, callback, async, regex, TypeScript, npm, API, deploy, config, linter, env, webhook, schema, CLI, JSON, CSS, DOM, SSE, backend, frontend, localhost, dependencies, repository, merge, rebase, diff, commit, com')
const keyterms = (cfg.keyterms || DEFAULT_VOICE_KEYTERMS)
.split(',').map(t => t.trim()).filter(Boolean);
DeepgramProvider.start({
@@ -452,7 +855,10 @@ const VoiceInput = {
this._updateButtons('idle');
this._hidePreview();
if (this._activeProvider === 'deepgram') {
if (this._activeProvider === 'claude') {
// Finalize, don't hang up: the last transcript arrives after the audio does.
ClaudeVoiceProvider.stop();
} else if (this._activeProvider === 'deepgram') {
DeepgramProvider.stop();
} else if (this._activeProvider === 'webspeech') {
try {
@@ -803,11 +1209,12 @@ const VoiceInput = {
timerEl.textContent = '0:00';
indicator.appendChild(timerEl);
this.previewEl.appendChild(indicator);
// Provider badge for Deepgram
if (provider === 'deepgram') {
// Provider badge (Web Speech gets none — it is the fallback, not a choice)
const badgeText = provider === 'deepgram' ? 'DG' : provider === 'claude' ? 'CLAUDE' : '';
if (badgeText) {
const badge = document.createElement('span');
badge.className = 'voice-preview-badge';
badge.textContent = 'DG';
badge.textContent = badgeText;
this.previewEl.appendChild(badge);
this.previewEl.appendChild(document.createTextNode(' '));
}
@@ -861,6 +1268,7 @@ const VoiceInput = {
if (this.isRecording) this.stop();
this._hideVoiceSendBtn();
DeepgramProvider._cleanup();
ClaudeVoiceProvider._cleanup();
this.recognition = null;
this._activeProvider = null;
this._stopDurationTimer();
+57
View File
@@ -0,0 +1,57 @@
/**
* @fileoverview AudioWorklet that turns microphone audio into the PCM frames the
* Claude voice endpoint expects.
*
* The endpoint is opened as `encoding=linear16, sample_rate=16000, channels=1`,
* i.e. raw signed 16-bit little-endian mono. MediaRecorder cannot produce that
* (it only emits container formats — webm/opus, mp4), which is why the Deepgram
* path's capture code cannot be reused here: Deepgram sniffs the container,
* Anthropic's endpoint does not.
*
* Sample rate is handled by the AudioContext, constructed at 16000 Hz so the
* browser resamples the mic for us. This processor only converts Float32 [-1,1]
* to Int16 and batches, because a raw 128-sample render quantum is a ~4 ms
* WebSocket frame — 250 frames a second of pure overhead.
*
* Loaded via `audioWorklet.addModule()` from voice-input.js. Runs on the audio
* thread: no DOM, no globals from the page.
*
* ⚠️ Edit this file and voice-input.js together. Static assets are served
* `immutable` for a year and this one is fetched from JS, so it inherits its
* cache-bust token from voice-input.js's script tag (see `_workletUrl()`); a
* change here alone would keep serving the old copy to every returning browser.
*/
/** ~256 ms at 16 kHz. Big enough to keep frame overhead down, small enough that interim transcripts stay live. */
const FRAME_SAMPLES = 4096;
class PcmFrameProcessor extends AudioWorkletProcessor {
constructor() {
super();
this._buffer = new Int16Array(FRAME_SAMPLES);
this._offset = 0;
}
process(inputs) {
const channel = inputs[0]?.[0];
// No input yet (mic still warming) — keep the processor alive.
if (!channel) return true;
for (let i = 0; i < channel.length; i++) {
// Clamp before scaling: values slightly outside [-1,1] are legal in Web Audio
// and would wrap around to the opposite sign as Int16, which sounds like a click.
const sample = Math.max(-1, Math.min(1, channel[i]));
this._buffer[this._offset++] = sample < 0 ? sample * 0x8000 : sample * 0x7fff;
if (this._offset === FRAME_SAMPLES) {
// Transfer a copy: the worklet keeps reusing its own buffer.
const frame = new Int16Array(this._buffer);
this.port.postMessage(frame.buffer, [frame.buffer]);
this._offset = 0;
}
}
return true;
}
}
registerProcessor('pcm-frame-processor', PcmFrameProcessor);
+1
View File
@@ -24,4 +24,5 @@ export { registerSearchRoutes } from './search-routes.js';
export { registerMeRoutes } from './me-routes.js';
export { registerAdminRoutes } from './admin-routes.js';
export { registerWsRoutes } from './ws-routes.js';
export { registerVoiceRoutes } from './voice-routes.js';
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
+194
View File
@@ -0,0 +1,194 @@
/**
* @fileoverview Claude voice dictation routes.
*
* - `GET /api/voice/status` — can this server transcribe? (settings gate + credential state)
* - `GET /ws/voice/stream` — one dictation: PCM16 audio up, transcripts down
*
* Design and the upstream protocol: `docs/claude-voice-plan.md`. The relay itself
* lives in `../voice-stream.ts`; this file is the auth, gating and lifetime shell
* around it.
*
* ⚠️ `/api/voice/status` reports STATE, never the token: `{ available, reason,
* subscriptionType?, expiresAt? }`. The Claude OAuth access token stays inside the
* server process — the browser sends audio and receives text, nothing else.
*
* ⚠️ The WebSocket carries the same upgrade guard as `/ws/sessions/:id/terminal`
* (allowed Host + same-site Origin, on top of the global auth hook that already ran
* on the handshake). Without it a cross-site page could open a dictation stream on
* the user's credentials and bill their subscription.
*
* ⚠️ The feature is OFF unless `claudeVoiceEnabled` is set: turning it on spends the
* server owner's Claude subscription on transcription for anyone who can reach the
* UI, which is a decision for the operator rather than a default.
*/
import { createRequire } from 'module';
import { FastifyInstance } from 'fastify';
import type { WebSocket } from 'ws';
import { ApiErrorCode, createErrorResponse } from '../../types.js';
import { isAllowedRequestHost, isAllowedRequestOrigin, type HostPolicy } from '../network-auth-policy.js';
import { readClaudeOAuthCredentials } from '../../claude-credentials.js';
import { VoiceStreamRelay } from '../voice-stream.js';
import { MAX_AUDIO_FRAME_BYTES, MAX_CONCURRENT_STREAMS } from '../../config/voice.js';
import type { ConfigPort } from '../ports/index.js';
const require = createRequire(import.meta.url);
const { version: APP_VERSION } = require('../../../package.json') as { version: string };
/** Why voice is unavailable, in a form the frontend can branch on. */
export type VoiceUnavailableReason = 'disabled' | 'no-credentials' | 'expired' | 'malformed';
export interface VoiceStatus {
available: boolean;
reason?: VoiceUnavailableReason;
/** Display-only ('max', 'pro'); present when the credential store reported one. */
subscriptionType?: string;
expiresAt?: number;
}
/**
* Resolve the server's dictation readiness. Split out and exported so the status
* endpoint and the WebSocket upgrade cannot drift apart: the socket must never
* accept a stream the status endpoint calls unavailable.
*/
export async function resolveVoiceStatus(enabled: boolean): Promise<VoiceStatus> {
if (!enabled) return { available: false, reason: 'disabled' };
const creds = await readClaudeOAuthCredentials();
switch (creds.status) {
case 'ok':
return { available: true, subscriptionType: creds.subscriptionType, expiresAt: creds.expiresAt };
case 'expired':
return { available: false, reason: 'expired', expiresAt: creds.expiresAt };
case 'malformed':
return { available: false, reason: 'malformed' };
default:
return { available: false, reason: 'no-credentials' };
}
}
/** Live relays, server-wide. Dictation is human-paced, so the cap is small. */
let activeStreams = 0;
/** Test seam: the cap is process-wide state, so suites must be able to reset it. */
export function _resetVoiceStreamCountForTesting(): void {
activeStreams = 0;
}
/** Split a comma-separated keyterms query value into terms. */
function parseKeyterms(raw: unknown): string[] {
if (typeof raw !== 'string' || !raw) return [];
return raw
.split(',')
.map((t) => t.trim())
.filter(Boolean)
.slice(0, 100);
}
export function registerVoiceRoutes(app: FastifyInstance, ctx: ConfigPort, getHostPolicy: () => HostPolicy): void {
app.get('/api/voice/status', async (_req, reply) => {
try {
return { success: true, data: await resolveVoiceStatus(await ctx.getClaudeVoiceEnabled()) };
} catch {
reply.code(500);
return createErrorResponse(ApiErrorCode.INTERNAL_ERROR, 'Failed to read voice status');
}
});
app.get<{ Querystring: { language?: string; keyterms?: string } }>(
'/ws/voice/stream',
{ websocket: true },
async (socket: WebSocket, req) => {
// Cross-site upgrade guard first: this socket spends the operator's Claude
// subscription, so it must be reachable only from Codeman's own origin.
const policy = getHostPolicy();
if (!isAllowedRequestHost(req.headers.host, policy) || !isAllowedRequestOrigin(req.headers.origin, policy)) {
socket.close(4003, 'Forbidden');
return;
}
const status = await resolveVoiceStatus(await ctx.getClaudeVoiceEnabled());
if (!status.available) {
socket.close(4004, status.reason ?? 'unavailable');
return;
}
// Re-read rather than trusting resolveVoiceStatus's discarded token: the
// status helper deliberately never returns it.
const creds = await readClaudeOAuthCredentials();
if (creds.status !== 'ok' || !creds.accessToken) {
socket.close(4004, 'no-credentials');
return;
}
if (activeStreams >= MAX_CONCURRENT_STREAMS) {
socket.close(4008, 'Too many voice streams');
return;
}
activeStreams++;
let released = false;
const release = () => {
if (released) return;
released = true;
activeStreams--;
};
const send = (payload: Record<string, unknown>) => {
if (socket.readyState !== 1) return;
try {
socket.send(JSON.stringify(payload));
} catch {
/* client vanished mid-write */
}
};
const relay = new VoiceStreamRelay({
accessToken: creds.accessToken,
appVersion: APP_VERSION,
language: req.query.language,
keyterms: parseKeyterms(req.query.keyterms),
onReady: () => send({ t: 'ready' }),
onTranscript: (text, final) => send({ t: 'transcript', text, final }),
onError: (message) => send({ t: 'error', message }),
onClose: () => {
release();
send({ t: 'closed' });
if (socket.readyState === 1) {
try {
socket.close(1000, 'Voice stream ended');
} catch {
/* already closing */
}
}
},
});
// Handlers are attached synchronously before any further await
// (@fastify/websocket drops messages that arrive before they exist).
socket.on('message', (raw: Buffer, isBinary: boolean) => {
if (isBinary) {
if (raw.length === 0 || raw.length > MAX_AUDIO_FRAME_BYTES) return;
relay.sendAudio(raw);
return;
}
try {
const msg = JSON.parse(String(raw)) as { t?: string };
if (msg.t === 'finalize') relay.finalize();
else if (msg.t === 'stop') relay.close();
} catch {
/* non-JSON control frame — ignore */
}
});
socket.on('close', () => {
relay.close();
release();
});
socket.on('error', () => {
relay.close();
release();
});
relay.connect();
}
);
}
+12
View File
@@ -857,6 +857,16 @@ export const SettingsUpdateSchema = z
* add-only at create; a marker keeps user-authored copies untouched.
*/
agentSkillEnabled: z.boolean().optional(),
/**
* Let browser dictation transcribe through this machine's Claude Code login,
* the same speech-to-text service the CLI's own `/voice` mode uses
* (docs/claude-voice-plan.md). SYNCED, default OFF: enabling it spends the
* operator's Claude subscription on transcription for anyone who can reach
* the UI, and routes microphone audio to Anthropic rather than to whichever
* provider was configured before. The Deepgram and Web Speech paths are
* untouched by this flag.
*/
claudeVoiceEnabled: z.boolean().optional(),
/**
* Approvals Inbox (header bell + drawer, phone overview answer buttons,
* push Approve/Deny action buttons). SYNCED, default OFF (opt-in): even
@@ -970,6 +980,8 @@ export const SettingsUpdateSchema = z
// Voice settings (cross-device sync)
voiceSettings: z
.object({
/** 'auto' | 'claude' | 'deepgram' | 'webspeech'. Unknown values fall back to auto client-side. */
provider: z.string().max(20).optional(),
apiKey: z.string().max(200).optional(),
language: z.string().max(20).optional(),
keyterms: z.string().max(500).optional(),
+12
View File
@@ -166,6 +166,7 @@ import {
registerMeRoutes,
registerAdminRoutes,
registerWsRoutes,
registerVoiceRoutes,
registerWebviewRoutes,
tryWebviewRefererFallback,
} from './routes/index.js';
@@ -635,6 +636,7 @@ export class WebServer extends EventEmitter {
getClaudeModeConfig: this.getClaudeModeConfig.bind(this),
getTerminalHistoryConfig: this.getTerminalHistoryConfig.bind(this),
getAgentSkillEnabled: this.getAgentSkillEnabled.bind(this),
getClaudeVoiceEnabled: this.getClaudeVoiceEnabled.bind(this),
getDefaultClaudeMdPath: this.getDefaultClaudeMdPath.bind(this),
getLightState: this.getLightState.bind(this),
getLightSessionsState: this.getLightSessionsState.bind(this),
@@ -982,6 +984,7 @@ export class WebServer extends EventEmitter {
registerCronRoutes(this.app, { ...ctx, cron: this.cronService });
registerWsRoutes(this.app, ctx, () => this.getHostPolicy());
registerVoiceRoutes(this.app, ctx, () => this.getHostPolicy());
}
/**
@@ -1704,6 +1707,15 @@ export class WebServer extends EventEmitter {
return settings.agentSkillEnabled === true;
}
// Whether browser dictation may use this machine's Claude Code credentials
// (synced `claudeVoiceEnabled` setting, default OFF; docs/claude-voice-plan.md).
// OFF by default because turning it on spends the operator's Claude subscription
// on transcription for anyone who can reach the UI.
private async getClaudeVoiceEnabled(): Promise<boolean> {
const settings = await this.readSettings();
return settings.claudeVoiceEnabled === true;
}
/**
* Read My Mind predictor model (docs/readmymind-plan.md): `readMyMindModel`
* setting, defaulting to the AI-checker opus model. Prediction quality is
+300
View File
@@ -0,0 +1,300 @@
/**
* @fileoverview Upstream half of Claude voice dictation: one browser recording
* relayed to the speech-to-text service Claude Code's own `/voice` mode uses.
*
* The browser cannot talk to that service directly — it would need the Claude
* OAuth bearer token in page JavaScript, and the endpoint is not CORS-open — so
* Codeman sits in the middle and is the only thing that ever holds the token.
* See `docs/claude-voice-plan.md` for the protocol table this implements.
*
* Wire contract (mirrors the CLI's `connectVoiceStream`):
* - Query pins the audio format: linear16 PCM, 16 kHz, mono. The browser worklet
* produces exactly that; a mismatch transcribes as silence or noise, never an error.
* - `{"type":"KeepAlive"}` on open and every 8s, or upstream drops the socket
* between utterances.
* - Audio frames go up as raw binary.
* - Downstream, `TranscriptText`/`TranscriptInterim` carry the RUNNING transcript
* (each frame supersedes the previous one — they are not deltas to concatenate),
* and `TranscriptEndpoint` promotes the pending interim to final.
* - `{"type":"CloseStream"}` finalizes; the endpoint frame that follows is the
* last transcript, so `finalize()` waits briefly for it rather than closing.
*
* The pure builders at the top are unit-tested; `VoiceStreamRelay` owns the socket,
* the keepalive timer and the lifetime cap.
*/
import WebSocket from 'ws';
import {
AUDIO_CHANNELS,
AUDIO_SAMPLE_RATE,
FINALIZE_TIMEOUT_MS,
KEEPALIVE_INTERVAL_MS,
MAX_KEYTERMS_HEADER_CHARS,
MAX_STREAM_MS,
VOICE_STREAM_PATH,
voiceStreamBase,
} from '../config/voice.js';
const KEEPALIVE_FRAME = '{"type":"KeepAlive"}';
const CLOSE_STREAM_FRAME = '{"type":"CloseStream"}';
export interface VoiceStreamParams {
/** BCP-47-ish language hint. Anything unusable falls back to 'en'. */
language?: string;
/** Domain vocabulary sent as a recognition hint. */
keyterms?: string[];
}
/**
* Collapse keyterms into the single ASCII header value upstream accepts.
*
* Commas separate terms, so a comma INSIDE a term would silently split it; it is
* replaced with a space rather than dropped. Non-ASCII is stripped because the
* value travels as an HTTP header, where anything outside the visible ASCII range
* is not portable. Deduped and truncated on a term boundary so a long list degrades
* to a shorter list instead of a mangled final term.
*/
export function sanitizeKeyterms(terms: string[]): string {
const seen = new Set<string>();
const out: string[] = [];
let length = 0;
for (const term of terms) {
const cleaned = term
.replace(/,/g, ' ')
.replace(/[^\x20-\x7E]/g, '')
.replace(/\s+/g, ' ')
.trim();
if (!cleaned || seen.has(cleaned)) continue;
const cost = cleaned.length + (out.length > 0 ? 1 : 0);
if (length + cost > MAX_KEYTERMS_HEADER_CHARS) break;
seen.add(cleaned);
out.push(cleaned);
length += cost;
}
return out.join(',');
}
/** Normalize a language hint to what the endpoint expects, defaulting to English. */
export function normalizeVoiceLanguage(language: string | undefined): string {
const trimmed = (language ?? '').trim();
if (!trimmed || !/^[a-zA-Z]{2,3}(-[a-zA-Z0-9]{2,8})?$|^multi$/.test(trimmed)) return 'en';
return trimmed;
}
/** Full upstream URL with the audio format pinned. */
export function buildVoiceStreamUrl(params: VoiceStreamParams = {}, env: NodeJS.ProcessEnv = process.env): string {
const query = new URLSearchParams({
encoding: 'linear16',
sample_rate: String(AUDIO_SAMPLE_RATE),
channels: String(AUDIO_CHANNELS),
endpointing_ms: '300',
utterance_end_ms: '1000',
language: normalizeVoiceLanguage(params.language),
use_conversation_engine: 'true',
stt_provider: 'deepgram-nova3',
});
return `${voiceStreamBase(env)}${VOICE_STREAM_PATH}?${query.toString()}`;
}
/**
* Upstream headers. Codeman identifies itself honestly (it is not the CLI), which
* the endpoint accepts; the bearer token is the only thing that authenticates.
*/
export function buildVoiceStreamHeaders(
accessToken: string,
appVersion: string,
keyterms: string[] = []
): Record<string, string> {
const headers: Record<string, string> = {
Authorization: `Bearer ${accessToken}`,
'User-Agent': `codeman/${appVersion} (voice-bridge)`,
'x-app': 'codeman',
'anthropic-client-platform': 'codeman_web',
};
const sanitized = sanitizeKeyterms(keyterms);
if (sanitized) headers['x-config-keyterms'] = sanitized;
return headers;
}
export interface VoiceStreamRelayOptions extends VoiceStreamParams {
accessToken: string;
appVersion: string;
/** Called once the upstream socket is open and audio may flow. */
onReady: () => void;
/** Running transcript. `final` marks the utterance as complete. */
onTranscript: (text: string, final: boolean) => void;
/** Human-readable failure. The relay is dead (or dying) by the time this fires. */
onError: (message: string) => void;
/** Terminal: the relay released its socket and timers. Fires exactly once. */
onClose: () => void;
}
/**
* One dictation, upstream. Owns exactly one WebSocket and dies with it: every
* exit path (error, upstream close, lifetime cap, caller close) funnels through
* `_teardown()`, which fires `onClose` once and clears both timers.
*/
export class VoiceStreamRelay {
private ws: WebSocket | null = null;
private keepAlive: ReturnType<typeof setInterval> | null = null;
private lifetimeTimer: ReturnType<typeof setTimeout> | null = null;
private finalizeTimer: ReturnType<typeof setTimeout> | null = null;
private closed = false;
private finalizing = false;
/** Latest interim, held so a close/finalize can promote it to final. */
private pendingTranscript = '';
constructor(private readonly opts: VoiceStreamRelayOptions) {}
/** Open the upstream socket. Safe to call once; a second call is a no-op. */
connect(): void {
if (this.ws || this.closed) return;
const url = buildVoiceStreamUrl({ language: this.opts.language, keyterms: this.opts.keyterms });
const ws = new WebSocket(url, {
headers: buildVoiceStreamHeaders(this.opts.accessToken, this.opts.appVersion, this.opts.keyterms ?? []),
});
this.ws = ws;
ws.on('open', () => {
// Ping immediately: the gap between upgrade and the browser's first audio
// frame is long enough (mic permission, worklet boot) for upstream to drop us.
this.safeSend(KEEPALIVE_FRAME);
this.keepAlive = setInterval(() => this.safeSend(KEEPALIVE_FRAME), KEEPALIVE_INTERVAL_MS);
this.lifetimeTimer = setTimeout(() => {
this.opts.onError('Voice stream reached its maximum length');
this.close();
}, MAX_STREAM_MS);
this.opts.onReady();
});
ws.on('message', (raw) => this.handleMessage(String(raw)));
// An upgrade rejection never reaches 'open', so its status is the only signal
// that the token was refused rather than the network being down.
ws.on('unexpected-response', (_req, res) => {
const status = res.statusCode ?? 0;
res.resume();
this.opts.onError(
status === 401 || status === 403
? 'Claude rejected the voice credentials. Run a Claude session to refresh your login.'
: `Voice service refused the connection (HTTP ${status})`
);
this.teardown();
});
ws.on('error', (err: Error) => {
if (this.closed) return;
this.opts.onError(`Voice stream error: ${err.message}`);
});
ws.on('close', () => {
this.promotePending();
this.teardown();
});
}
/** Relay one raw PCM16 frame upstream. Dropped after finalize, as upstream ignores it. */
sendAudio(chunk: Buffer): void {
if (this.finalizing || this.closed) return;
if (this.ws?.readyState !== WebSocket.OPEN) return;
this.ws.send(chunk);
}
/**
* Ask upstream for the final transcript. The endpoint frame usually follows
* within a few hundred ms; the timer is the backstop so a silent upstream still
* yields whatever interim we already have instead of hanging the caller.
*/
finalize(): void {
if (this.finalizing || this.closed) return;
this.finalizing = true;
if (this.ws?.readyState !== WebSocket.OPEN) {
this.promotePending();
this.close();
return;
}
this.safeSend(CLOSE_STREAM_FRAME);
this.finalizeTimer = setTimeout(() => {
this.promotePending();
this.close();
}, FINALIZE_TIMEOUT_MS);
}
/** Terminal shutdown. Idempotent. */
close(): void {
if (this.closed) return;
const ws = this.ws;
this.teardown();
if (ws && (ws.readyState === WebSocket.OPEN || ws.readyState === WebSocket.CONNECTING)) {
try {
ws.close();
} catch {
/* already closing */
}
}
}
private handleMessage(raw: string): void {
let msg: { type?: string; data?: string; description?: string; error_code?: string; message?: string };
try {
msg = JSON.parse(raw);
} catch {
return;
}
switch (msg.type) {
case 'TranscriptText':
case 'TranscriptInterim': {
// Each frame is the whole running transcript, not a delta.
if (typeof msg.data === 'string' && msg.data) {
this.pendingTranscript = msg.data;
this.opts.onTranscript(msg.data, false);
}
break;
}
case 'TranscriptEndpoint': {
this.promotePending();
if (this.finalizing) this.close();
break;
}
case 'TranscriptError': {
this.opts.onError(msg.description || msg.error_code || 'Transcription failed');
break;
}
case 'error': {
this.opts.onError(msg.message || 'Voice service error');
break;
}
default:
break;
}
}
/** Emit the held interim as final, exactly once per utterance. */
private promotePending(): void {
if (!this.pendingTranscript) return;
const text = this.pendingTranscript;
this.pendingTranscript = '';
this.opts.onTranscript(text, true);
}
private safeSend(frame: string): void {
if (this.ws?.readyState !== WebSocket.OPEN) return;
try {
this.ws.send(frame);
} catch {
/* socket died between the check and the send */
}
}
private teardown(): void {
if (this.closed) return;
this.closed = true;
if (this.keepAlive) clearInterval(this.keepAlive);
if (this.lifetimeTimer) clearTimeout(this.lifetimeTimer);
if (this.finalizeTimer) clearTimeout(this.finalizeTimer);
this.keepAlive = null;
this.lifetimeTimer = null;
this.finalizeTimer = null;
this.opts.onClose();
}
}