fix(input): image paste and dictation land in the session they started in

Both read activeSessionId at the END of an async gap, so switching tabs
in between sent the input to the wrong session:

- An image upload inserted its paths with sendInput(), which re-reads
  activeSessionId after the uploads finish. It now inserts into the
  session the batch was uploaded to, through the same durable queue.
- Voice dictation read the target when the transcript arrived and again
  when the send button or the compose overlay's Send was pressed. The
  target is now captured in start() (via _focusedPane()), the local-echo
  overlay is only used when that target is the active session, and a
  target that closed meanwhile gets a toast instead of a 404.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Codeman maintainer
2026-10-06 09:09:51 +02:00
parent ead3d34411
commit e2f56dc077
4 changed files with 259 additions and 13 deletions
+6 -2
View File
@@ -185,8 +185,12 @@ Object.assign(CodemanApp.prototype, {
const paths = results.filter(Boolean); const paths = results.filter(Boolean);
if (paths.length > 0 && options.insert !== false) { if (paths.length > 0 && options.insert !== false) {
// Insert all paths in one shot, space-separated, in selection order. // Insert all paths in one shot, space-separated, in selection order, into
await this.sendInput(paths.join(' ')); // the session the batch was uploaded TO. Not sendInput(): it re-reads
// activeSessionId, and after the awaits above that is whatever tab the
// user switched to mid-upload, so the paths landed in the wrong session.
// Same delivery sendInput() uses (durable queue, useMux for the POST path).
this._sendInputAsync(sessionId, paths.join(' '), { useMux: true });
} }
// Final status: successes, plus any failures / cap so nothing is silent. // Final status: successes, plus any failures / cap so nothing is silent.
+48 -10
View File
@@ -554,6 +554,11 @@ const VoiceInput = {
_analyserSource: null, // MediaStreamSource for level meter _analyserSource: null, // MediaStreamSource for level meter
_audioContext: null, // AudioContext for level meter _audioContext: null, // AudioContext for level meter
_levelAnimFrame: null, // rAF handle for level meter _levelAnimFrame: null, // rAF handle for level meter
// The session dictation was started FOR, captured in start(). Transcripts
// arrive seconds later and the green send button / compose overlay can be
// used later still; reading app.activeSessionId at that point sent the text
// to whatever tab the user had switched to in the meantime.
_targetSessionId: null,
init() { init() {
this._initRecognition(); this._initRecognition();
@@ -663,10 +668,12 @@ const VoiceInput = {
start() { start() {
if (this.isRecording) return; if (this.isRecording) return;
if (!app.activeSessionId) { const target = app._focusedPane?.()?.sessionId || app.activeSessionId;
if (!target) {
app.showToast('No active session', 'warning'); app.showToast('No active session', 'warning');
return; return;
} }
this._targetSessionId = target;
this._retryCount = 0; this._retryCount = 0;
const provider = this._resolveProvider(); const provider = this._resolveProvider();
@@ -959,8 +966,31 @@ const VoiceInput = {
this.stop(); this.stop();
}, },
/** The session this dictation belongs to (see _targetSessionId). */
_targetSession() {
return this._targetSessionId || app.activeSessionId;
},
/**
* Send text to the dictation's own session. The active session keeps going
* through app.sendInput() exactly as before; any other session goes straight
* to the durable queue with the same useMux flag sendInput() passes.
*/
_sendToTarget(target, text) {
if (target === app.activeSessionId) return app.sendInput(text);
// Closed while dictating: say so instead of queueing text for a session
// that will only answer 404 (and never typing it into some other tab).
if (app.sessions && !app.sessions.has(target)) {
app.showToast?.('That session has closed; dictation not sent', 'warning');
return Promise.resolve();
}
app._sendInputAsync(target, text, { useMux: true });
return Promise.resolve();
},
_insertText(text) { _insertText(text) {
if (!app.activeSessionId || !text.trim()) return; const target = this._targetSession();
if (!target || !text.trim()) return;
const trimmed = text.trim(); const trimmed = text.trim();
const mode = this._getDeepgramConfig().insertMode || 'direct'; const mode = this._getDeepgramConfig().insertMode || 'direct';
@@ -975,14 +1005,17 @@ const VoiceInput = {
this._showComposeOverlay(trimmed); this._showComposeOverlay(trimmed);
} }
} else { } else {
// Direct mode: inject into local echo overlay if available, else send to PTY // Direct mode: inject into local echo overlay if available, else send to PTY.
if (app._localEchoEnabled && app._localEchoOverlay) { // The overlay belongs to the ACTIVE session's terminal, so text dictated
// for any other session must not be typed into it.
const isActive = target === app.activeSessionId;
if (isActive && app._localEchoEnabled && app._localEchoOverlay) {
app._localEchoOverlay.appendText(trimmed); app._localEchoOverlay.appendText(trimmed);
} else { } else {
app.sendInput(trimmed).catch(() => {}); this._sendToTarget(target, trimmed).catch(() => {});
} }
this._showVoiceSendBtn(); this._showVoiceSendBtn();
setTimeout(() => { if (app.terminal) app.terminal.focus(); }, 150); setTimeout(() => { if (isActive && app.terminal) app.terminal.focus(); }, 150);
} }
}, },
@@ -1008,10 +1041,15 @@ const VoiceInput = {
// Click handler // Click handler
this._voiceSendHandler = () => { this._voiceSendHandler = () => {
if (!app.activeSessionId) return; const target = this._targetSession();
if (!target) return;
// Simulate Enter key: if local echo is active, flush its buffer + send \r; // Simulate Enter key: if local echo is active, flush its buffer + send \r;
// otherwise just send \r directly to the PTY // otherwise just send \r directly to the PTY. Both the overlay and the
if (app._localEchoEnabled && app._localEchoOverlay) { // predictions belong to the ACTIVE session's terminal, so a dictation
// for another session just sends its Enter there.
if (target !== app.activeSessionId) {
this._sendToTarget(target, '\r').catch(() => {});
} else if (app._localEchoEnabled && app._localEchoOverlay) {
const text = app._localEchoOverlay.pendingText || ''; const text = app._localEchoOverlay.pendingText || '';
app._localEchoOverlay.clear(); app._localEchoOverlay.clear();
app._localEchoOverlay.suppressBufferDetection(); app._localEchoOverlay.suppressBufferDetection();
@@ -1065,7 +1103,7 @@ const VoiceInput = {
const send = () => { const send = () => {
const val = textarea.value.trim(); const val = textarea.value.trim();
overlay.remove(); overlay.remove();
if (val) app.sendInput(val + '\r').catch(() => {}); if (val) this._sendToTarget(this._targetSession(), val + '\r').catch(() => {});
}; };
const cancel = () => overlay.remove(); const cancel = () => overlay.remove();
const newInput = () => { const newInput = () => {
+30 -1
View File
@@ -143,6 +143,7 @@ function loadImageInputApp() {
app.activeSessionId = 'session-1'; app.activeSessionId = 'session-1';
app.showToast = vi.fn(); app.showToast = vi.fn();
app.sendInput = vi.fn(async () => {}); app.sendInput = vi.fn(async () => {});
app._sendInputAsync = vi.fn();
app._normalizeImageForUpload = vi.fn(async (file) => file); app._normalizeImageForUpload = vi.fn(async (file) => file);
app._uploadPasteImage = vi.fn(async (_sessionId, file: { path: string }) => file.path); app._uploadPasteImage = vi.fn(async (_sessionId, file: { path: string }) => file.path);
return app as Record<string, any>; return app as Record<string, any>;
@@ -199,6 +200,7 @@ describe('image upload insertion policy', () => {
expect(Array.from(paths)).toEqual(['/tmp/first.png', '/tmp/second.png']); expect(Array.from(paths)).toEqual(['/tmp/first.png', '/tmp/second.png']);
expect(app.sendInput).not.toHaveBeenCalled(); expect(app.sendInput).not.toHaveBeenCalled();
expect(app._sendInputAsync).not.toHaveBeenCalled();
}); });
it('preserves terminal insertion by default', async () => { it('preserves terminal insertion by default', async () => {
@@ -207,6 +209,33 @@ describe('image upload insertion policy', () => {
const paths = await app._uploadAndInsertImages([{ path: '/tmp/legacy.png' }]); const paths = await app._uploadAndInsertImages([{ path: '/tmp/legacy.png' }]);
expect(Array.from(paths)).toEqual(['/tmp/legacy.png']); expect(Array.from(paths)).toEqual(['/tmp/legacy.png']);
expect(app.sendInput).toHaveBeenCalledWith('/tmp/legacy.png'); // The same delivery sendInput() uses (durable queue, useMux for the POST
// fallback), but addressed to the session the batch was uploaded to.
expect(app._sendInputAsync).toHaveBeenCalledWith('session-1', '/tmp/legacy.png', { useMux: true });
});
it('inserts into the session the upload started in, even after a tab switch mid-upload', async () => {
const app = loadImageInputApp();
// The user switches tabs while the upload is in flight. sendInput() re-read
// activeSessionId after the awaits, so the paths used to land in session-2.
app._uploadPasteImage = vi.fn(async (_sessionId, file: { path: string }) => {
app.activeSessionId = 'session-2';
return file.path;
});
await app._uploadAndInsertImages([{ path: '/tmp/shot.png' }]);
expect(app._uploadPasteImage).toHaveBeenCalledWith('session-1', { path: '/tmp/shot.png' });
expect(app._sendInputAsync).toHaveBeenCalledWith('session-1', '/tmp/shot.png', { useMux: true });
expect(app.sendInput).not.toHaveBeenCalled();
});
it('uploads to and inserts into an explicitly named session', async () => {
const app = loadImageInputApp();
await app._uploadAndInsertImages([{ path: '/tmp/pane-b.png' }], { sessionId: 'session-b' });
expect(app._uploadPasteImage).toHaveBeenCalledWith('session-b', { path: '/tmp/pane-b.png' });
expect(app._sendInputAsync).toHaveBeenCalledWith('session-b', '/tmp/pane-b.png', { useMux: true });
}); });
}); });
+175
View File
@@ -0,0 +1,175 @@
/**
* @fileoverview Dictation lands in the session it was started for.
*
* `VoiceInput` (voice-input.js) used to read `app.activeSessionId` when the
* transcript ARRIVED, and again when the green send button or the compose
* overlay's Send was pressed. Both happen seconds after recording started, so a
* user who switched tabs in between had their dictation typed into the other
* session. The target is now captured in `start()` (through `_focusedPane()`,
* so a second terminal pane can claim it later) and every send path uses it.
*
* Loaded via `vm` with a stubbed `app` (no jsdom).
*/
import { readFileSync } from 'node:fs';
import { resolve } from 'node:path';
import vm from 'node:vm';
import { describe, expect, it, vi } from 'vitest';
const voiceSource = readFileSync(resolve(import.meta.dirname, '../src/web/public/voice-input.js'), 'utf8');
type Voice = {
start: () => void;
_insertText: (text: string) => void;
_resolveProvider: () => string;
_startWebSpeech: () => void;
_showComposeOverlay: (text: string) => void;
_voiceSendHandler: (() => void) | null;
_targetSessionId: string | null;
};
function load(opts: { insertMode?: string; localEcho?: boolean; focused?: string } = {}) {
const sendInput = vi.fn(async () => {});
const sendInputAsync = vi.fn();
const appendText = vi.fn();
const showToast = vi.fn();
const gear = {
classList: { contains: () => false, add: vi.fn(), remove: vi.fn() },
innerHTML: '',
title: '',
getAttribute: () => null,
setAttribute: vi.fn(),
removeAttribute: vi.fn(),
addEventListener: vi.fn(),
removeEventListener: vi.fn(),
};
const app: Record<string, unknown> = {
activeSessionId: 'session-a',
sessions: new Map([
['session-a', {}],
['session-b', {}],
]),
sendInput,
_sendInputAsync: sendInputAsync,
showToast,
terminal: { focus: vi.fn() },
_localEchoEnabled: !!opts.localEcho,
_localEchoOverlay: opts.localEcho ? { appendText, pendingText: '', clear: vi.fn() } : null,
};
if (opts.focused) app._focusedPane = () => ({ sessionId: opts.focused });
const context = vm.createContext({
console,
setTimeout: (fn: () => void) => fn(),
clearTimeout: () => {},
setInterval: () => 0,
clearInterval: () => {},
app,
localStorage: {
getItem: (key: string) =>
key === 'codeman-voice-settings' ? JSON.stringify({ insertMode: opts.insertMode || 'direct' }) : null,
setItem: () => {},
},
document: {
querySelector: (sel: string) => (sel === '.btn-settings' ? gear : null),
createElement: () => ({}),
body: { appendChild: () => {} },
},
window: {},
navigator: {},
});
vm.runInContext(`${voiceSource}\nglobalThis.__VoiceInput = VoiceInput;`, context);
const voice = (context as unknown as { __VoiceInput: Voice }).__VoiceInput;
// Recording itself is out of scope: start() only has to pick the target.
voice._resolveProvider = () => 'webspeech';
voice._startWebSpeech = vi.fn();
return { voice, app, sendInput, sendInputAsync, appendText, showToast, gear };
}
describe('dictation target', () => {
it('sends to the session recording started in, even after a tab switch', () => {
const { voice, app, sendInput, sendInputAsync } = load();
voice.start();
app.activeSessionId = 'session-b'; // user switched tabs while speaking
voice._insertText('fix the login bug');
expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'fix the login bug', { useMux: true });
expect(sendInput).not.toHaveBeenCalled();
});
it('keeps the existing path when the target is still the active session', () => {
const { voice, sendInput, sendInputAsync } = load();
voice.start();
voice._insertText('hello');
expect(sendInput).toHaveBeenCalledWith('hello');
expect(sendInputAsync).not.toHaveBeenCalled();
});
it('never types another session dictation into the active local-echo overlay', () => {
const { voice, app, appendText, sendInputAsync } = load({ localEcho: true });
voice.start();
app.activeSessionId = 'session-b';
voice._insertText('for session a');
expect(appendText).not.toHaveBeenCalled();
expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'for session a', { useMux: true });
});
it('still uses the local-echo overlay for the active session', () => {
const { voice, appendText, sendInput } = load({ localEcho: true });
voice.start();
voice._insertText('typed locally');
expect(appendText).toHaveBeenCalledWith('typed locally');
expect(sendInput).not.toHaveBeenCalled();
});
it('takes its target from the focused pane when one is reported', () => {
const { voice, sendInputAsync } = load({ focused: 'session-b' });
voice.start();
voice._insertText('into pane b');
expect(sendInputAsync).toHaveBeenCalledWith('session-b', 'into pane b', { useMux: true });
});
it('the green send button sends Enter to the dictation target', () => {
const { voice, app, gear, sendInput, sendInputAsync } = load();
voice.start();
voice._insertText('ship it');
app.activeSessionId = 'session-b';
const handler = gear.addEventListener.mock.calls.find((c: unknown[]) => c[0] === 'click')?.[1] as () => void;
handler();
expect(sendInputAsync).toHaveBeenLastCalledWith('session-a', '\r', { useMux: true });
// Only the original insert went through sendInput (target was active then).
expect(sendInput).toHaveBeenCalledTimes(1);
});
it('drops dictation for a session that closed meanwhile, with a toast', () => {
const { voice, app, sendInput, sendInputAsync, showToast } = load();
voice.start();
app.activeSessionId = 'session-b';
(app.sessions as Map<string, unknown>).delete('session-a');
voice._insertText('too late');
expect(sendInput).not.toHaveBeenCalled();
expect(sendInputAsync).not.toHaveBeenCalled();
expect(showToast).toHaveBeenCalledWith('That session has closed; dictation not sent', 'warning');
});
it('refuses to start with no session at all', () => {
const { voice, app, showToast } = load();
app.activeSessionId = null;
voice.start();
expect(showToast).toHaveBeenCalledWith('No active session', 'warning');
expect(voice._targetSessionId).toBeNull();
});
});