mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-09 16:59:43 +02:00
fix(input): image paste and dictation land in the session they started in
Both read activeSessionId at the END of an async gap, so switching tabs in between sent the input to the wrong session: - An image upload inserted its paths with sendInput(), which re-reads activeSessionId after the uploads finish. It now inserts into the session the batch was uploaded to, through the same durable queue. - Voice dictation read the target when the transcript arrived and again when the send button or the compose overlay's Send was pressed. The target is now captured in start() (via _focusedPane()), the local-echo overlay is only used when that target is the active session, and a target that closed meanwhile gets a toast instead of a 404. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,175 @@
|
||||
/**
|
||||
* @fileoverview Dictation lands in the session it was started for.
|
||||
*
|
||||
* `VoiceInput` (voice-input.js) used to read `app.activeSessionId` when the
|
||||
* transcript ARRIVED, and again when the green send button or the compose
|
||||
* overlay's Send was pressed. Both happen seconds after recording started, so a
|
||||
* user who switched tabs in between had their dictation typed into the other
|
||||
* session. The target is now captured in `start()` (through `_focusedPane()`,
|
||||
* so a second terminal pane can claim it later) and every send path uses it.
|
||||
*
|
||||
* Loaded via `vm` with a stubbed `app` (no jsdom).
|
||||
*/
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { resolve } from 'node:path';
|
||||
import vm from 'node:vm';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
const voiceSource = readFileSync(resolve(import.meta.dirname, '../src/web/public/voice-input.js'), 'utf8');
|
||||
|
||||
type Voice = {
|
||||
start: () => void;
|
||||
_insertText: (text: string) => void;
|
||||
_resolveProvider: () => string;
|
||||
_startWebSpeech: () => void;
|
||||
_showComposeOverlay: (text: string) => void;
|
||||
_voiceSendHandler: (() => void) | null;
|
||||
_targetSessionId: string | null;
|
||||
};
|
||||
|
||||
function load(opts: { insertMode?: string; localEcho?: boolean; focused?: string } = {}) {
|
||||
const sendInput = vi.fn(async () => {});
|
||||
const sendInputAsync = vi.fn();
|
||||
const appendText = vi.fn();
|
||||
const showToast = vi.fn();
|
||||
const gear = {
|
||||
classList: { contains: () => false, add: vi.fn(), remove: vi.fn() },
|
||||
innerHTML: '',
|
||||
title: '',
|
||||
getAttribute: () => null,
|
||||
setAttribute: vi.fn(),
|
||||
removeAttribute: vi.fn(),
|
||||
addEventListener: vi.fn(),
|
||||
removeEventListener: vi.fn(),
|
||||
};
|
||||
const app: Record<string, unknown> = {
|
||||
activeSessionId: 'session-a',
|
||||
sessions: new Map([
|
||||
['session-a', {}],
|
||||
['session-b', {}],
|
||||
]),
|
||||
sendInput,
|
||||
_sendInputAsync: sendInputAsync,
|
||||
showToast,
|
||||
terminal: { focus: vi.fn() },
|
||||
_localEchoEnabled: !!opts.localEcho,
|
||||
_localEchoOverlay: opts.localEcho ? { appendText, pendingText: '', clear: vi.fn() } : null,
|
||||
};
|
||||
if (opts.focused) app._focusedPane = () => ({ sessionId: opts.focused });
|
||||
const context = vm.createContext({
|
||||
console,
|
||||
setTimeout: (fn: () => void) => fn(),
|
||||
clearTimeout: () => {},
|
||||
setInterval: () => 0,
|
||||
clearInterval: () => {},
|
||||
app,
|
||||
localStorage: {
|
||||
getItem: (key: string) =>
|
||||
key === 'codeman-voice-settings' ? JSON.stringify({ insertMode: opts.insertMode || 'direct' }) : null,
|
||||
setItem: () => {},
|
||||
},
|
||||
document: {
|
||||
querySelector: (sel: string) => (sel === '.btn-settings' ? gear : null),
|
||||
createElement: () => ({}),
|
||||
body: { appendChild: () => {} },
|
||||
},
|
||||
window: {},
|
||||
navigator: {},
|
||||
});
|
||||
vm.runInContext(`${voiceSource}\nglobalThis.__VoiceInput = VoiceInput;`, context);
|
||||
const voice = (context as unknown as { __VoiceInput: Voice }).__VoiceInput;
|
||||
// Recording itself is out of scope: start() only has to pick the target.
|
||||
voice._resolveProvider = () => 'webspeech';
|
||||
voice._startWebSpeech = vi.fn();
|
||||
return { voice, app, sendInput, sendInputAsync, appendText, showToast, gear };
|
||||
}
|
||||
|
||||
describe('dictation target', () => {
|
||||
it('sends to the session recording started in, even after a tab switch', () => {
|
||||
const { voice, app, sendInput, sendInputAsync } = load();
|
||||
voice.start();
|
||||
app.activeSessionId = 'session-b'; // user switched tabs while speaking
|
||||
|
||||
voice._insertText('fix the login bug');
|
||||
|
||||
expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'fix the login bug', { useMux: true });
|
||||
expect(sendInput).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('keeps the existing path when the target is still the active session', () => {
|
||||
const { voice, sendInput, sendInputAsync } = load();
|
||||
voice.start();
|
||||
|
||||
voice._insertText('hello');
|
||||
|
||||
expect(sendInput).toHaveBeenCalledWith('hello');
|
||||
expect(sendInputAsync).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('never types another session dictation into the active local-echo overlay', () => {
|
||||
const { voice, app, appendText, sendInputAsync } = load({ localEcho: true });
|
||||
voice.start();
|
||||
app.activeSessionId = 'session-b';
|
||||
|
||||
voice._insertText('for session a');
|
||||
|
||||
expect(appendText).not.toHaveBeenCalled();
|
||||
expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'for session a', { useMux: true });
|
||||
});
|
||||
|
||||
it('still uses the local-echo overlay for the active session', () => {
|
||||
const { voice, appendText, sendInput } = load({ localEcho: true });
|
||||
voice.start();
|
||||
|
||||
voice._insertText('typed locally');
|
||||
|
||||
expect(appendText).toHaveBeenCalledWith('typed locally');
|
||||
expect(sendInput).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('takes its target from the focused pane when one is reported', () => {
|
||||
const { voice, sendInputAsync } = load({ focused: 'session-b' });
|
||||
voice.start();
|
||||
|
||||
voice._insertText('into pane b');
|
||||
|
||||
expect(sendInputAsync).toHaveBeenCalledWith('session-b', 'into pane b', { useMux: true });
|
||||
});
|
||||
|
||||
it('the green send button sends Enter to the dictation target', () => {
|
||||
const { voice, app, gear, sendInput, sendInputAsync } = load();
|
||||
voice.start();
|
||||
voice._insertText('ship it');
|
||||
app.activeSessionId = 'session-b';
|
||||
const handler = gear.addEventListener.mock.calls.find((c: unknown[]) => c[0] === 'click')?.[1] as () => void;
|
||||
|
||||
handler();
|
||||
|
||||
expect(sendInputAsync).toHaveBeenLastCalledWith('session-a', '\r', { useMux: true });
|
||||
// Only the original insert went through sendInput (target was active then).
|
||||
expect(sendInput).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('drops dictation for a session that closed meanwhile, with a toast', () => {
|
||||
const { voice, app, sendInput, sendInputAsync, showToast } = load();
|
||||
voice.start();
|
||||
app.activeSessionId = 'session-b';
|
||||
(app.sessions as Map<string, unknown>).delete('session-a');
|
||||
|
||||
voice._insertText('too late');
|
||||
|
||||
expect(sendInput).not.toHaveBeenCalled();
|
||||
expect(sendInputAsync).not.toHaveBeenCalled();
|
||||
expect(showToast).toHaveBeenCalledWith('That session has closed; dictation not sent', 'warning');
|
||||
});
|
||||
|
||||
it('refuses to start with no session at all', () => {
|
||||
const { voice, app, showToast } = load();
|
||||
app.activeSessionId = null;
|
||||
|
||||
voice.start();
|
||||
|
||||
expect(showToast).toHaveBeenCalledWith('No active session', 'warning');
|
||||
expect(voice._targetSessionId).toBeNull();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user