mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-10 01:09:43 +02:00
With the tile grid open the main terminal is parked (display: none), but local echo stays on, so direct-mode dictation for the focused tile's session (which is activeSessionId) was appended to the main terminal's hidden local-echo overlay. Nothing appeared in the tile, Enter in the tile submitted without the dictated text, and the stranded text was later flushed into whichever tile had focus when the grid closed, or dropped. - _insertText: skip the overlay while _tilesOwnTerminal() is true, so the text goes through _sendToTarget to the session itself. - The post-insert refocus gives the keyboard to the focused tile (only if it is still the dictation target) instead of the parked main terminal. Outside the grid it still focuses the main terminal, so split view keeps its behaviour even when Pane B took focus mid-dictation. - Green send button: with tiles open, send only Enter to the target and leave the parked overlay and main-terminal predictions alone. The gate is _tilesOwnTerminal(), not _focusedPane().isPrimary: in split view a target equal to activeSessionId is Pane A with a visible overlay, and focus read at transcript time could otherwise push Pane A's dictation past its own unflushed overlay text. Tests: tile-grid dictation and green-send cases (both fail without the fix) plus a split-view pin in test/voice-input-target.test.ts. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
257 lines
9.7 KiB
TypeScript
257 lines
9.7 KiB
TypeScript
/**
|
|
* @fileoverview Dictation lands in the session it was started for.
|
|
*
|
|
* `VoiceInput` (voice-input.js) used to read `app.activeSessionId` when the
|
|
* transcript ARRIVED, and again when the green send button or the compose
|
|
* overlay's Send was pressed. Both happen seconds after recording started, so a
|
|
* user who switched tabs in between had their dictation typed into the other
|
|
* session. The target is now captured in `start()` (through `_focusedPane()`,
|
|
* so a second terminal pane can claim it later) and every send path uses it.
|
|
*
|
|
* With the tile grid open the main terminal (and with it the local-echo
|
|
* overlay) is parked with `display: none`, so direct-mode dictation used to be
|
|
* typed into an invisible overlay and never reached the focused tile. While
|
|
* `_tilesOwnTerminal()` is true the overlay is skipped and the keyboard goes
|
|
* back to the focused tile, not the parked main terminal.
|
|
*
|
|
* Loaded via `vm` with a stubbed `app` (no jsdom).
|
|
*/
|
|
import { readFileSync } from 'node:fs';
|
|
import { resolve } from 'node:path';
|
|
import vm from 'node:vm';
|
|
import { describe, expect, it, vi } from 'vitest';
|
|
|
|
const voiceSource = readFileSync(resolve(import.meta.dirname, '../src/web/public/voice-input.js'), 'utf8');
|
|
|
|
type Voice = {
|
|
start: () => void;
|
|
_insertText: (text: string) => void;
|
|
_resolveProvider: () => string;
|
|
_startWebSpeech: () => void;
|
|
_showComposeOverlay: (text: string) => void;
|
|
_voiceSendHandler: (() => void) | null;
|
|
_targetSessionId: string | null;
|
|
};
|
|
|
|
type Pane = { sessionId: string; isPrimary?: boolean; terminal?: { focus: () => void } };
|
|
|
|
function load(opts: { insertMode?: string; localEcho?: boolean; focused?: string; tiles?: boolean; pane?: Pane } = {}) {
|
|
const sendInput = vi.fn(async () => {});
|
|
const sendInputAsync = vi.fn();
|
|
const appendText = vi.fn();
|
|
const overlayClear = vi.fn();
|
|
const showToast = vi.fn();
|
|
const gear = {
|
|
classList: { contains: () => false, add: vi.fn(), remove: vi.fn() },
|
|
innerHTML: '',
|
|
title: '',
|
|
getAttribute: () => null,
|
|
setAttribute: vi.fn(),
|
|
removeAttribute: vi.fn(),
|
|
addEventListener: vi.fn(),
|
|
removeEventListener: vi.fn(),
|
|
};
|
|
const app: Record<string, unknown> = {
|
|
activeSessionId: 'session-a',
|
|
sessions: new Map([
|
|
['session-a', {}],
|
|
['session-b', {}],
|
|
]),
|
|
sendInput,
|
|
_sendInputAsync: sendInputAsync,
|
|
showToast,
|
|
terminal: { focus: vi.fn() },
|
|
_localEchoEnabled: !!opts.localEcho,
|
|
_localEchoOverlay: opts.localEcho
|
|
? { appendText, pendingText: '', clear: overlayClear, suppressBufferDetection: vi.fn() }
|
|
: null,
|
|
_predictiveEcho: { clearPredictions: vi.fn() },
|
|
};
|
|
if (opts.focused) app._focusedPane = () => ({ sessionId: opts.focused });
|
|
// A full pane record, as terminal-ui.js _focusedPane() returns it. Reassign
|
|
// app._focusedPane in a test to move focus between panes.
|
|
if (opts.pane) {
|
|
const pane = opts.pane;
|
|
app._focusedPane = () => pane;
|
|
}
|
|
if (opts.tiles) app._tilesOwnTerminal = () => true;
|
|
const context = vm.createContext({
|
|
console,
|
|
setTimeout: (fn: () => void) => fn(),
|
|
clearTimeout: () => {},
|
|
setInterval: () => 0,
|
|
clearInterval: () => {},
|
|
app,
|
|
localStorage: {
|
|
getItem: (key: string) =>
|
|
key === 'codeman-voice-settings' ? JSON.stringify({ insertMode: opts.insertMode || 'direct' }) : null,
|
|
setItem: () => {},
|
|
},
|
|
document: {
|
|
querySelector: (sel: string) => (sel === '.btn-settings' ? gear : null),
|
|
createElement: () => ({}),
|
|
body: { appendChild: () => {} },
|
|
},
|
|
window: {},
|
|
navigator: {},
|
|
});
|
|
vm.runInContext(`${voiceSource}\nglobalThis.__VoiceInput = VoiceInput;`, context);
|
|
const voice = (context as unknown as { __VoiceInput: Voice }).__VoiceInput;
|
|
// Recording itself is out of scope: start() only has to pick the target.
|
|
voice._resolveProvider = () => 'webspeech';
|
|
voice._startWebSpeech = vi.fn();
|
|
return { voice, app, sendInput, sendInputAsync, appendText, overlayClear, showToast, gear };
|
|
}
|
|
|
|
function clickGreenSend(gear: { addEventListener: { mock: { calls: unknown[][] } } }) {
|
|
const handler = gear.addEventListener.mock.calls.find((c: unknown[]) => c[0] === 'click')?.[1] as () => void;
|
|
handler();
|
|
}
|
|
|
|
describe('dictation target', () => {
|
|
it('sends to the session recording started in, even after a tab switch', () => {
|
|
const { voice, app, sendInput, sendInputAsync } = load();
|
|
voice.start();
|
|
app.activeSessionId = 'session-b'; // user switched tabs while speaking
|
|
|
|
voice._insertText('fix the login bug');
|
|
|
|
expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'fix the login bug', { useMux: true });
|
|
expect(sendInput).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('keeps the existing path when the target is still the active session', () => {
|
|
const { voice, sendInput, sendInputAsync } = load();
|
|
voice.start();
|
|
|
|
voice._insertText('hello');
|
|
|
|
expect(sendInput).toHaveBeenCalledWith('hello');
|
|
expect(sendInputAsync).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('never types another session dictation into the active local-echo overlay', () => {
|
|
const { voice, app, appendText, sendInputAsync } = load({ localEcho: true });
|
|
voice.start();
|
|
app.activeSessionId = 'session-b';
|
|
|
|
voice._insertText('for session a');
|
|
|
|
expect(appendText).not.toHaveBeenCalled();
|
|
expect(sendInputAsync).toHaveBeenCalledWith('session-a', 'for session a', { useMux: true });
|
|
});
|
|
|
|
it('still uses the local-echo overlay for the active session', () => {
|
|
const { voice, appendText, sendInput } = load({ localEcho: true });
|
|
voice.start();
|
|
|
|
voice._insertText('typed locally');
|
|
|
|
expect(appendText).toHaveBeenCalledWith('typed locally');
|
|
expect(sendInput).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('takes its target from the focused pane when one is reported', () => {
|
|
const { voice, sendInputAsync } = load({ focused: 'session-b' });
|
|
voice.start();
|
|
|
|
voice._insertText('into pane b');
|
|
|
|
expect(sendInputAsync).toHaveBeenCalledWith('session-b', 'into pane b', { useMux: true });
|
|
});
|
|
|
|
it('the green send button sends Enter to the dictation target', () => {
|
|
const { voice, app, gear, sendInput, sendInputAsync } = load();
|
|
voice.start();
|
|
voice._insertText('ship it');
|
|
app.activeSessionId = 'session-b';
|
|
const handler = gear.addEventListener.mock.calls.find((c: unknown[]) => c[0] === 'click')?.[1] as () => void;
|
|
|
|
handler();
|
|
|
|
expect(sendInputAsync).toHaveBeenLastCalledWith('session-a', '\r', { useMux: true });
|
|
// Only the original insert went through sendInput (target was active then).
|
|
expect(sendInput).toHaveBeenCalledTimes(1);
|
|
});
|
|
|
|
it('drops dictation for a session that closed meanwhile, with a toast', () => {
|
|
const { voice, app, sendInput, sendInputAsync, showToast } = load();
|
|
voice.start();
|
|
app.activeSessionId = 'session-b';
|
|
(app.sessions as Map<string, unknown>).delete('session-a');
|
|
|
|
voice._insertText('too late');
|
|
|
|
expect(sendInput).not.toHaveBeenCalled();
|
|
expect(sendInputAsync).not.toHaveBeenCalled();
|
|
expect(showToast).toHaveBeenCalledWith('That session has closed; dictation not sent', 'warning');
|
|
});
|
|
|
|
it('with the tile grid open, dictation goes to the focused tile, not the parked overlay', () => {
|
|
const tileTerminal = { focus: vi.fn() };
|
|
const { voice, app, appendText, sendInput, sendInputAsync } = load({
|
|
localEcho: true,
|
|
tiles: true,
|
|
pane: { sessionId: 'session-a', isPrimary: false, terminal: tileTerminal },
|
|
});
|
|
voice.start();
|
|
|
|
voice._insertText('into the tile');
|
|
|
|
expect(appendText).not.toHaveBeenCalled();
|
|
expect(sendInput).toHaveBeenCalledWith('into the tile');
|
|
expect(sendInputAsync).not.toHaveBeenCalled();
|
|
// The keyboard goes back to the tile; the main terminal is display: none.
|
|
expect(tileTerminal.focus).toHaveBeenCalled();
|
|
expect((app.terminal as { focus: ReturnType<typeof vi.fn> }).focus).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('with the tile grid open, the green send button sends only Enter to the tile session', () => {
|
|
const tileTerminal = { focus: vi.fn() };
|
|
const { voice, app, gear, sendInput, overlayClear } = load({
|
|
localEcho: true,
|
|
tiles: true,
|
|
pane: { sessionId: 'session-a', isPrimary: false, terminal: tileTerminal },
|
|
});
|
|
voice.start();
|
|
voice._insertText('ship it');
|
|
// Something stale in the parked main overlay must not ride along.
|
|
(app._localEchoOverlay as { pendingText: string }).pendingText = 'stale main-terminal text';
|
|
|
|
clickGreenSend(gear);
|
|
|
|
expect(overlayClear).not.toHaveBeenCalled();
|
|
expect(sendInput).not.toHaveBeenCalledWith('stale main-terminal text');
|
|
expect(sendInput).toHaveBeenLastCalledWith('\r');
|
|
// The parked main terminal's predictions are not this pane's.
|
|
expect(
|
|
(app._predictiveEcho as { clearPredictions: ReturnType<typeof vi.fn> }).clearPredictions
|
|
).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('split view is unchanged: a Pane A dictation keeps the overlay and refocuses the main terminal', () => {
|
|
const paneB = { focus: vi.fn() };
|
|
const { voice, app, appendText, sendInput } = load({ localEcho: true, focused: 'session-a' });
|
|
voice.start();
|
|
// The user clicked into Pane B while speaking.
|
|
app._focusedPane = () => ({ sessionId: 'session-b', isPrimary: false, terminal: paneB });
|
|
|
|
voice._insertText('for pane a');
|
|
|
|
expect(appendText).toHaveBeenCalledWith('for pane a');
|
|
expect(sendInput).not.toHaveBeenCalled();
|
|
expect((app.terminal as { focus: ReturnType<typeof vi.fn> }).focus).toHaveBeenCalled();
|
|
expect(paneB.focus).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('refuses to start with no session at all', () => {
|
|
const { voice, app, showToast } = load();
|
|
app.activeSessionId = null;
|
|
|
|
voice.start();
|
|
|
|
expect(showToast).toHaveBeenCalledWith('No active session', 'warning');
|
|
expect(voice._targetSessionId).toBeNull();
|
|
});
|
|
});
|