diff --git a/src/web/public/image-input.js b/src/web/public/image-input.js index d2518fc4..92ee87ed 100644 --- a/src/web/public/image-input.js +++ b/src/web/public/image-input.js @@ -185,8 +185,12 @@ Object.assign(CodemanApp.prototype, { const paths = results.filter(Boolean); if (paths.length > 0 && options.insert !== false) { - // Insert all paths in one shot, space-separated, in selection order. - await this.sendInput(paths.join(' ')); + // Insert all paths in one shot, space-separated, in selection order, into + // the session the batch was uploaded TO. Not sendInput(): it re-reads + // activeSessionId, and after the awaits above that is whatever tab the + // user switched to mid-upload, so the paths landed in the wrong session. + // Same delivery sendInput() uses (durable queue, useMux for the POST path). + this._sendInputAsync(sessionId, paths.join(' '), { useMux: true }); } // Final status: successes, plus any failures / cap so nothing is silent. diff --git a/src/web/public/voice-input.js b/src/web/public/voice-input.js index 1c6355e9..f792242d 100644 --- a/src/web/public/voice-input.js +++ b/src/web/public/voice-input.js @@ -554,6 +554,11 @@ const VoiceInput = { _analyserSource: null, // MediaStreamSource for level meter _audioContext: null, // AudioContext for level meter _levelAnimFrame: null, // rAF handle for level meter + // The session dictation was started FOR, captured in start(). Transcripts + // arrive seconds later and the green send button / compose overlay can be + // used later still; reading app.activeSessionId at that point sent the text + // to whatever tab the user had switched to in the meantime. + _targetSessionId: null, init() { this._initRecognition(); @@ -663,10 +668,12 @@ const VoiceInput = { start() { if (this.isRecording) return; - if (!app.activeSessionId) { + const target = app._focusedPane?.()?.sessionId || app.activeSessionId; + if (!target) { app.showToast('No active session', 'warning'); return; } + this._targetSessionId = target; this._retryCount = 0; const provider = this._resolveProvider(); @@ -959,8 +966,31 @@ const VoiceInput = { this.stop(); }, + /** The session this dictation belongs to (see _targetSessionId). */ + _targetSession() { + return this._targetSessionId || app.activeSessionId; + }, + + /** + * Send text to the dictation's own session. The active session keeps going + * through app.sendInput() exactly as before; any other session goes straight + * to the durable queue with the same useMux flag sendInput() passes. + */ + _sendToTarget(target, text) { + if (target === app.activeSessionId) return app.sendInput(text); + // Closed while dictating: say so instead of queueing text for a session + // that will only answer 404 (and never typing it into some other tab). + if (app.sessions && !app.sessions.has(target)) { + app.showToast?.('That session has closed; dictation not sent', 'warning'); + return Promise.resolve(); + } + app._sendInputAsync(target, text, { useMux: true }); + return Promise.resolve(); + }, + _insertText(text) { - if (!app.activeSessionId || !text.trim()) return; + const target = this._targetSession(); + if (!target || !text.trim()) return; const trimmed = text.trim(); const mode = this._getDeepgramConfig().insertMode || 'direct'; @@ -975,14 +1005,17 @@ const VoiceInput = { this._showComposeOverlay(trimmed); } } else { - // Direct mode: inject into local echo overlay if available, else send to PTY - if (app._localEchoEnabled && app._localEchoOverlay) { + // Direct mode: inject into local echo overlay if available, else send to PTY. + // The overlay belongs to the ACTIVE session's terminal, so text dictated + // for any other session must not be typed into it. + const isActive = target === app.activeSessionId; + if (isActive && app._localEchoEnabled && app._localEchoOverlay) { app._localEchoOverlay.appendText(trimmed); } else { - app.sendInput(trimmed).catch(() => {}); + this._sendToTarget(target, trimmed).catch(() => {}); } this._showVoiceSendBtn(); - setTimeout(() => { if (app.terminal) app.terminal.focus(); }, 150); + setTimeout(() => { if (isActive && app.terminal) app.terminal.focus(); }, 150); } }, @@ -1008,10 +1041,15 @@ const VoiceInput = { // Click handler this._voiceSendHandler = () => { - if (!app.activeSessionId) return; + const target = this._targetSession(); + if (!target) return; // Simulate Enter key: if local echo is active, flush its buffer + send \r; - // otherwise just send \r directly to the PTY - if (app._localEchoEnabled && app._localEchoOverlay) { + // otherwise just send \r directly to the PTY. Both the overlay and the + // predictions belong to the ACTIVE session's terminal, so a dictation + // for another session just sends its Enter there. + if (target !== app.activeSessionId) { + this._sendToTarget(target, '\r').catch(() => {}); + } else if (app._localEchoEnabled && app._localEchoOverlay) { const text = app._localEchoOverlay.pendingText || ''; app._localEchoOverlay.clear(); app._localEchoOverlay.suppressBufferDetection(); @@ -1065,7 +1103,7 @@ const VoiceInput = { const send = () => { const val = textarea.value.trim(); overlay.remove(); - if (val) app.sendInput(val + '\r').catch(() => {}); + if (val) this._sendToTarget(this._targetSession(), val + '\r').catch(() => {}); }; const cancel = () => overlay.remove(); const newInput = () => { diff --git a/test/image-paste-trap.test.ts b/test/image-paste-trap.test.ts index 230993f2..f0df2e21 100644 --- a/test/image-paste-trap.test.ts +++ b/test/image-paste-trap.test.ts @@ -143,6 +143,7 @@ function loadImageInputApp() { app.activeSessionId = 'session-1'; app.showToast = vi.fn(); app.sendInput = vi.fn(async () => {}); + app._sendInputAsync = vi.fn(); app._normalizeImageForUpload = vi.fn(async (file) => file); app._uploadPasteImage = vi.fn(async (_sessionId, file: { path: string }) => file.path); return app as Record; @@ -199,6 +200,7 @@ describe('image upload insertion policy', () => { expect(Array.from(paths)).toEqual(['/tmp/first.png', '/tmp/second.png']); expect(app.sendInput).not.toHaveBeenCalled(); + expect(app._sendInputAsync).not.toHaveBeenCalled(); }); it('preserves terminal insertion by default', async () => { @@ -207,6 +209,33 @@ describe('image upload insertion policy', () => { const paths = await app._uploadAndInsertImages([{ path: '/tmp/legacy.png' }]); expect(Array.from(paths)).toEqual(['/tmp/legacy.png']); - expect(app.sendInput).toHaveBeenCalledWith('/tmp/legacy.png'); + // The same delivery sendInput() uses (durable queue, useMux for the POST + // fallback), but addressed to the session the batch was uploaded to. + expect(app._sendInputAsync).toHaveBeenCalledWith('session-1', '/tmp/legacy.png', { useMux: true }); + }); + + it('inserts into the session the upload started in, even after a tab switch mid-upload', async () => { + const app = loadImageInputApp(); + // The user switches tabs while the upload is in flight. sendInput() re-read + // activeSessionId after the awaits, so the paths used to land in session-2. + app._uploadPasteImage = vi.fn(async (_sessionId, file: { path: string }) => { + app.activeSessionId = 'session-2'; + return file.path; + }); + + await app._uploadAndInsertImages([{ path: '/tmp/shot.png' }]); + + expect(app._uploadPasteImage).toHaveBeenCalledWith('session-1', { path: '/tmp/shot.png' }); + expect(app._sendInputAsync).toHaveBeenCalledWith('session-1', '/tmp/shot.png', { useMux: true }); + expect(app.sendInput).not.toHaveBeenCalled(); + }); + + it('uploads to and inserts into an explicitly named session', async () => { + const app = loadImageInputApp(); + + await app._uploadAndInsertImages([{ path: '/tmp/pane-b.png' }], { sessionId: 'session-b' }); + + expect(app._uploadPasteImage).toHaveBeenCalledWith('session-b', { path: '/tmp/pane-b.png' }); + expect(app._sendInputAsync).toHaveBeenCalledWith('session-b', '/tmp/pane-b.png', { useMux: true }); }); }); diff --git a/test/voice-input-target.test.ts b/test/voice-input-target.test.ts new file mode 100644 index 00000000..624431b2 --- /dev/null +++ b/test/voice-input-target.test.ts @@ -0,0 +1,175 @@ +/** + * @fileoverview Dictation lands in the session it was started for. + * + * `VoiceInput` (voice-input.js) used to read `app.activeSessionId` when the + * transcript ARRIVED, and again when the green send button or the compose + * overlay's Send was pressed. Both happen seconds after recording started, so a + * user who switched tabs in between had their dictation typed into the other + * session. The target is now captured in `start()` (through `_focusedPane()`, + * so a second terminal pane can claim it later) and every send path uses it. + * + * Loaded via `vm` with a stubbed `app` (no jsdom). + */ +import { readFileSync } from 'node:fs'; +import { resolve } from 'node:path'; +import vm from 'node:vm'; +import { describe, expect, it, vi } from 'vitest'; + +const voiceSource = readFileSync(resolve(import.meta.dirname, '../src/web/public/voice-input.js'), 'utf8'); + +type Voice = { + start: () => void; + _insertText: (text: string) => void; + _resolveProvider: () => string; + _startWebSpeech: () => void; + _showComposeOverlay: (text: string) => void; + _voiceSendHandler: (() => void) | null; + _targetSessionId: string | null; +}; + +function load(opts: { insertMode?: string; localEcho?: boolean; focused?: string } = {}) { + const sendInput = vi.fn(async () => {}); + const sendInputAsync = vi.fn(); + const appendText = vi.fn(); + const showToast = vi.fn(); + const gear = { + classList: { contains: () => false, add: vi.fn(), remove: vi.fn() }, + innerHTML: '', + title: '', + getAttribute: () => null, + setAttribute: vi.fn(), + removeAttribute: vi.fn(), + addEventListener: vi.fn(), + removeEventListener: vi.fn(), + }; + const app: Record = { + activeSessionId: 'session-a', + sessions: new Map([ + ['session-a', {}], + ['session-b', {}], + ]), + sendInput, + _sendInputAsync: sendInputAsync, + showToast, + terminal: { focus: vi.fn() }, + _localEchoEnabled: !!opts.localEcho, + _localEchoOverlay: opts.localEcho ? { appendText, pendingText: '', clear: vi.fn() } : null, + }; + if (opts.focused) app._focusedPane = () => ({ sessionId: opts.focused }); + const context = vm.createContext({ + console, + setTimeout: (fn: () => void) => fn(), + clearTimeout: () => {}, + setInterval: () => 0, + clearInterval: () => {}, + app, + localStorage: { + getItem: (key: string) => + key === 'codeman-voice-settings' ? JSON.stringify({ insertMode: opts.insertMode || 'direct' }) : null, + setItem: () => {}, + }, + document: { + querySelector: (sel: string) => (sel === '.btn-settings' ? gear : null), + createElement: () => ({}), + body: { appendChild: () => {} }, + }, + window: {}, + navigator: {}, + }); + vm.runInContext(`${voiceSource}\nglobalThis.__VoiceInput = VoiceInput;`, context); + const voice = (context as unknown as { __VoiceInput: Voice }).__VoiceInput; + // Recording itself is out of scope: start() only has to pick the target. + voice._resolveProvider = () => 'webspeech'; + voice._startWebSpeech = vi.fn(); + return { voice, app, sendInput, sendInputAsync, appendText, showToast, gear }; +} + +describe('dictation target', () => { + it('sends to the session recording started in, even after a tab switch', () => { + const { voice, app, sendInput, sendInputAsync } = load(); + voice.start(); + app.activeSessionId = 'session-b'; // user switched tabs while speaking + + voice._insertText('fix the login bug'); + + expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'fix the login bug', { useMux: true }); + expect(sendInput).not.toHaveBeenCalled(); + }); + + it('keeps the existing path when the target is still the active session', () => { + const { voice, sendInput, sendInputAsync } = load(); + voice.start(); + + voice._insertText('hello'); + + expect(sendInput).toHaveBeenCalledWith('hello'); + expect(sendInputAsync).not.toHaveBeenCalled(); + }); + + it('never types another session dictation into the active local-echo overlay', () => { + const { voice, app, appendText, sendInputAsync } = load({ localEcho: true }); + voice.start(); + app.activeSessionId = 'session-b'; + + voice._insertText('for session a'); + + expect(appendText).not.toHaveBeenCalled(); + expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'for session a', { useMux: true }); + }); + + it('still uses the local-echo overlay for the active session', () => { + const { voice, appendText, sendInput } = load({ localEcho: true }); + voice.start(); + + voice._insertText('typed locally'); + + expect(appendText).toHaveBeenCalledWith('typed locally'); + expect(sendInput).not.toHaveBeenCalled(); + }); + + it('takes its target from the focused pane when one is reported', () => { + const { voice, sendInputAsync } = load({ focused: 'session-b' }); + voice.start(); + + voice._insertText('into pane b'); + + expect(sendInputAsync).toHaveBeenCalledWith('session-b', 'into pane b', { useMux: true }); + }); + + it('the green send button sends Enter to the dictation target', () => { + const { voice, app, gear, sendInput, sendInputAsync } = load(); + voice.start(); + voice._insertText('ship it'); + app.activeSessionId = 'session-b'; + const handler = gear.addEventListener.mock.calls.find((c: unknown[]) => c[0] === 'click')?.[1] as () => void; + + handler(); + + expect(sendInputAsync).toHaveBeenLastCalledWith('session-a', '\r', { useMux: true }); + // Only the original insert went through sendInput (target was active then). + expect(sendInput).toHaveBeenCalledTimes(1); + }); + + it('drops dictation for a session that closed meanwhile, with a toast', () => { + const { voice, app, sendInput, sendInputAsync, showToast } = load(); + voice.start(); + app.activeSessionId = 'session-b'; + (app.sessions as Map).delete('session-a'); + + voice._insertText('too late'); + + expect(sendInput).not.toHaveBeenCalled(); + expect(sendInputAsync).not.toHaveBeenCalled(); + expect(showToast).toHaveBeenCalledWith('That session has closed; dictation not sent', 'warning'); + }); + + it('refuses to start with no session at all', () => { + const { voice, app, showToast } = load(); + app.activeSessionId = null; + + voice.start(); + + expect(showToast).toHaveBeenCalledWith('No active session', 'warning'); + expect(voice._targetSessionId).toBeNull(); + }); +});