feat: add the TUI raw-mode key parser

Decodes printable UTF-8, the control keys, arrows in both CSI and SS3 forms and
SGR mouse reports out of a byte stream that can tear anywhere, so a sequence
split across two reads decodes the same as one that arrives whole.

A lone ESC cannot be told from the start of an arrow key by looking at bytes,
so the parser holds it and the caller resolves it with flush() once its
disambiguation timer fires. Unknown sequences are swallowed: a stray CSI must
never reach a prompt composer as typed text.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Codeman maintainer
2026-08-22 14:13:57 +02:00
parent 64cf8384f2
commit 5d7fdb528b
2 changed files with 432 additions and 0 deletions
+222
View File
@@ -0,0 +1,222 @@
/**
* @fileoverview Pure byte-stream to input-event parser for raw-mode stdin.
*
* Stateful (a sequence can arrive split across reads, and a UTF-8 character can
* be split mid-code-point) but pure: it owns a byte buffer and nothing else, no
* stdin, no timers. The one timing decision a terminal forces on us stays with
* the caller: a lone ESC is indistinguishable from the start of an arrow key
* until something either follows it or does not, so the parser HOLDS a trailing
* ESC and the caller calls `flush()` after ~30ms of silence to turn it into an
* Escape event.
*
* Unknown sequences are swallowed rather than leaked as text: a stray
* `CSI 200~` must never end up typed into a prompt composer.
*
* @module tui/tui-keys
*/
/** Keys with a name rather than a character. */
export type TuiNamedKey =
| 'up'
| 'down'
| 'left'
| 'right'
| 'home'
| 'end'
| 'pageup'
| 'pagedown'
| 'delete'
| 'insert';
export type TuiMouseKind = 'press' | 'release' | 'wheel-up' | 'wheel-down';
/** Discriminated union, exhaustive-switch friendly (see `utils/assertNever`). */
export type TuiInputEvent =
| { type: 'char'; value: string }
| { type: 'enter' }
| { type: 'tab' }
| { type: 'backspace' }
| { type: 'escape' }
| { type: 'ctrl'; key: string }
| { type: 'key'; name: TuiNamedKey }
| { type: 'mouse'; kind: TuiMouseKind; x: number; y: number; button: number };
export interface TuiKeyParser {
/** Decode a chunk. Incomplete tails are held for the next call. */
feed(chunk: Buffer | string): TuiInputEvent[];
/** Resolve a held ESC (the caller's disambiguation timer fired). */
flush(): TuiInputEvent[];
/** Bytes currently held back. Exposed for the ESC timer and for tests. */
pending(): number;
}
/**
* An unterminated sequence longer than this is not a sequence: the held bytes
* are dropped whole, so a garbage burst can neither wedge the parser nor leak
* its bytes into a prompt as typed characters.
*/
const MAX_PENDING_BYTES = 64;
/** Bytes in a UTF-8 sequence given its lead byte; 0 for a byte that cannot lead one. */
function utf8SequenceLength(lead: number): number {
if (lead < 0x80) return 1;
if (lead >= 0xc2 && lead <= 0xdf) return 2;
if (lead >= 0xe0 && lead <= 0xef) return 3;
if (lead >= 0xf0 && lead <= 0xf4) return 4;
return 0;
}
const CSI_FINAL_KEYS: Record<string, TuiNamedKey> = {
A: 'up',
B: 'down',
C: 'right',
D: 'left',
H: 'home',
F: 'end',
};
/** `CSI <n> ~` keys, by their first numeric parameter. */
const CSI_TILDE_KEYS: Record<number, TuiNamedKey> = {
1: 'home',
2: 'insert',
3: 'delete',
4: 'end',
5: 'pageup',
6: 'pagedown',
7: 'home',
8: 'end',
};
/** Result of trying to parse one sequence off the front of the buffer. */
type ParseStep = { consumed: number; events: TuiInputEvent[] } | 'incomplete';
const NOTHING: TuiInputEvent[] = [];
export function createKeyParser(): TuiKeyParser {
let buf: Buffer = Buffer.alloc(0);
/** Parse the CSI/SS3 sequence that starts at buf[0] === ESC. */
const parseEscape = (): ParseStep => {
if (buf.length < 2) return 'incomplete';
const second = buf[1];
// SS3 (`ESC O <final>`): the arrows/Home/End of application-cursor mode.
if (second === 0x4f) {
if (buf.length < 3) return 'incomplete';
const name = CSI_FINAL_KEYS[String.fromCharCode(buf[2])];
return { consumed: 3, events: name ? [{ type: 'key', name }] : NOTHING };
}
// Anything that is not a CSI is a lone ESC as far as we are concerned; the
// next byte then parses on its own (so Alt+x reads as Escape then `x`).
if (second !== 0x5b) return { consumed: 1, events: [{ type: 'escape' }] };
let j = 2;
while (j < buf.length && buf[j] >= 0x30 && buf[j] <= 0x3f) j++;
while (j < buf.length && buf[j] >= 0x20 && buf[j] <= 0x2f) j++;
if (j >= buf.length) return 'incomplete';
const final = String.fromCharCode(buf[j]);
const params = buf.subarray(2, j).toString('latin1');
const consumed = j + 1;
// X10 mouse (`CSI M` + 3 raw bytes): swallowed, but its payload bytes must
// be consumed or they would surface as typed characters.
if (params === '' && final === 'M') {
if (buf.length < consumed + 3) return 'incomplete';
return { consumed: consumed + 3, events: NOTHING };
}
if (params.startsWith('<') && (final === 'M' || final === 'm')) {
return { consumed, events: parseSgrMouse(params.slice(1), final) };
}
if (final === '~') {
const name = CSI_TILDE_KEYS[Number.parseInt(params, 10)];
return { consumed, events: name ? [{ type: 'key', name }] : NOTHING };
}
// Modified arrows (`CSI 1;5A`) carry the same final byte; the modifier is
// dropped rather than exposed, since nothing in the keymap wants it yet.
const named = CSI_FINAL_KEYS[final];
return { consumed, events: named ? [{ type: 'key', name: named }] : NOTHING };
};
const parseSgrMouse = (params: string, final: string): TuiInputEvent[] => {
const parts = params.split(';');
if (parts.length < 3) return NOTHING;
const button = Number.parseInt(parts[0], 10);
const x = Number.parseInt(parts[1], 10);
const y = Number.parseInt(parts[2], 10);
if (!Number.isFinite(button) || !Number.isFinite(x) || !Number.isFinite(y)) return NOTHING;
if (button >= 64) {
// 64 = wheel up, 65 = wheel down (the low bit is the direction).
const kind: TuiMouseKind = (button & 1) === 1 ? 'wheel-down' : 'wheel-up';
return [{ type: 'mouse', kind, x, y, button }];
}
// Motion reports (bit 32) would fire on every pixel of a drag; nothing in
// the keymap consumes them, so they are swallowed here rather than upstream.
if ((button & 32) === 32) return NOTHING;
return [{ type: 'mouse', kind: final === 'M' ? 'press' : 'release', x, y, button }];
};
/** Parse one non-escape byte (or one UTF-8 character) off the front. */
const parseByte = (): ParseStep => {
const b = buf[0];
if (b === 0x0d || b === 0x0a) return { consumed: 1, events: [{ type: 'enter' }] };
if (b === 0x09) return { consumed: 1, events: [{ type: 'tab' }] };
if (b === 0x7f || b === 0x08) return { consumed: 1, events: [{ type: 'backspace' }] };
if (b === 0x00) return { consumed: 1, events: [{ type: 'ctrl', key: '@' }] };
if (b >= 0x01 && b <= 0x1a) {
return { consumed: 1, events: [{ type: 'ctrl', key: String.fromCharCode(b + 0x60) }] };
}
if (b >= 0x1c && b <= 0x1f) {
return { consumed: 1, events: [{ type: 'ctrl', key: String.fromCharCode(b + 0x40) }] };
}
const length = utf8SequenceLength(b);
if (length === 0) return { consumed: 1, events: NOTHING };
if (buf.length < length) return 'incomplete';
const value = buf.subarray(0, length).toString('utf8');
// A lead byte followed by junk decodes to U+FFFD; that is corruption on the
// wire, not something to type into a composer. Only the bad lead byte is
// dropped, so whatever valid input followed it still decodes.
if (value.includes('�')) return { consumed: 1, events: NOTHING };
return { consumed: length, events: [{ type: 'char', value }] };
};
/** Drain the buffer, stopping at the first incomplete sequence. */
const drain = (events: TuiInputEvent[]): void => {
while (buf.length > 0) {
const step = buf[0] === 0x1b ? parseEscape() : parseByte();
if (step === 'incomplete') {
if (buf.length > MAX_PENDING_BYTES) buf = Buffer.alloc(0);
return;
}
for (const event of step.events) events.push(event);
buf = buf.subarray(step.consumed);
}
};
return {
feed(chunk: Buffer | string): TuiInputEvent[] {
const bytes = typeof chunk === 'string' ? Buffer.from(chunk, 'utf8') : chunk;
buf = buf.length === 0 ? Buffer.from(bytes) : Buffer.concat([buf, bytes]);
const events: TuiInputEvent[] = [];
drain(events);
return events;
},
flush(): TuiInputEvent[] {
const events: TuiInputEvent[] = [];
if (buf.length > 0 && buf[0] === 0x1b) {
events.push({ type: 'escape' });
buf = buf.subarray(1);
drain(events);
}
return events;
},
pending(): number {
return buf.length;
},
};
}
+210
View File
@@ -0,0 +1,210 @@
/**
* @fileoverview Unit tests for the raw-mode key parser.
*
* The two failure modes that matter are covered explicitly: a sequence that
* arrives split across reads must decode identically at EVERY split position
* (a terminal is free to break a chunk anywhere), and an unknown sequence must
* be swallowed rather than leaked as typed text.
*/
import { describe, it, expect } from 'vitest';
import { createKeyParser, type TuiInputEvent } from '../../src/tui/tui-keys.js';
/** Feed a whole sequence in one go. */
function decode(input: string | Buffer): TuiInputEvent[] {
return createKeyParser().feed(input);
}
/** Feed the same bytes split at `at`, so a torn read must not change the result. */
function decodeSplit(bytes: Buffer, at: number): TuiInputEvent[] {
const parser = createKeyParser();
return [...parser.feed(bytes.subarray(0, at)), ...parser.feed(bytes.subarray(at))];
}
describe('printable input', () => {
it('emits one event per code point', () => {
expect(decode('ab')).toEqual([
{ type: 'char', value: 'a' },
{ type: 'char', value: 'b' },
]);
});
it('decodes multi-byte UTF-8', () => {
expect(decode('é中')).toEqual([
{ type: 'char', value: 'é' },
{ type: 'char', value: '中' },
]);
expect(decode('\u{1f600}')).toEqual([{ type: 'char', value: '\u{1f600}' }]);
});
it('holds a UTF-8 character split across chunks', () => {
const bytes = Buffer.from('中', 'utf8');
const parser = createKeyParser();
expect(parser.feed(bytes.subarray(0, 1))).toEqual([]);
expect(parser.pending()).toBe(1);
expect(parser.feed(bytes.subarray(1, 2))).toEqual([]);
expect(parser.feed(bytes.subarray(2))).toEqual([{ type: 'char', value: '中' }]);
expect(parser.pending()).toBe(0);
});
it('decodes a 4-byte character at every split position', () => {
const bytes = Buffer.from('\u{1f600}', 'utf8');
for (let at = 0; at <= bytes.length; at++) {
expect(decodeSplit(bytes, at)).toEqual([{ type: 'char', value: '\u{1f600}' }]);
}
});
it('swallows invalid UTF-8 rather than typing a replacement character', () => {
expect(decode(Buffer.from([0xc3, 0x28]))).toEqual([{ type: 'char', value: '(' }]);
});
});
describe('control keys', () => {
it('maps Enter, Tab and Backspace', () => {
expect(decode('\r')).toEqual([{ type: 'enter' }]);
expect(decode('\n')).toEqual([{ type: 'enter' }]);
expect(decode('\t')).toEqual([{ type: 'tab' }]);
expect(decode('\x7f')).toEqual([{ type: 'backspace' }]);
expect(decode('\x08')).toEqual([{ type: 'backspace' }]);
});
it('maps Ctrl+letter, keeping Ctrl+I and Ctrl+M as Tab and Enter', () => {
expect(decode('\x03')).toEqual([{ type: 'ctrl', key: 'c' }]);
expect(decode('\x17')).toEqual([{ type: 'ctrl', key: 'w' }]);
expect(decode('\x01')).toEqual([{ type: 'ctrl', key: 'a' }]);
expect(decode('\x09')).toEqual([{ type: 'tab' }]);
expect(decode('\x0d')).toEqual([{ type: 'enter' }]);
expect(decode('\x00')).toEqual([{ type: 'ctrl', key: '@' }]);
});
});
describe('escape sequences', () => {
it('decodes CSI arrows, Home and End', () => {
expect(decode('\x1b[A')).toEqual([{ type: 'key', name: 'up' }]);
expect(decode('\x1b[B')).toEqual([{ type: 'key', name: 'down' }]);
expect(decode('\x1b[C')).toEqual([{ type: 'key', name: 'right' }]);
expect(decode('\x1b[D')).toEqual([{ type: 'key', name: 'left' }]);
expect(decode('\x1b[H')).toEqual([{ type: 'key', name: 'home' }]);
expect(decode('\x1b[F')).toEqual([{ type: 'key', name: 'end' }]);
});
it('decodes the SS3 variants application-cursor mode sends', () => {
expect(decode('\x1bOA')).toEqual([{ type: 'key', name: 'up' }]);
expect(decode('\x1bOD')).toEqual([{ type: 'key', name: 'left' }]);
expect(decode('\x1bOH')).toEqual([{ type: 'key', name: 'home' }]);
expect(decode('\x1bOP')).toEqual([]);
});
it('decodes the numbered CSI keys', () => {
expect(decode('\x1b[2~')).toEqual([{ type: 'key', name: 'insert' }]);
expect(decode('\x1b[3~')).toEqual([{ type: 'key', name: 'delete' }]);
expect(decode('\x1b[5~')).toEqual([{ type: 'key', name: 'pageup' }]);
expect(decode('\x1b[6~')).toEqual([{ type: 'key', name: 'pagedown' }]);
expect(decode('\x1b[1~')).toEqual([{ type: 'key', name: 'home' }]);
expect(decode('\x1b[4~')).toEqual([{ type: 'key', name: 'end' }]);
});
it('ignores modifiers on an arrow rather than dropping the key', () => {
expect(decode('\x1b[1;5A')).toEqual([{ type: 'key', name: 'up' }]);
});
it('swallows unknown sequences instead of leaking them as text', () => {
expect(decode('\x1b[Z')).toEqual([]);
expect(decode('\x1b[999~')).toEqual([]);
expect(decode('\x1b[?1049h')).toEqual([]);
expect(decode('\x1b[200~hi\x1b[201~')).toEqual([
{ type: 'char', value: 'h' },
{ type: 'char', value: 'i' },
]);
});
it('consumes the payload of an X10 mouse report', () => {
expect(decode('\x1b[M !!x')).toEqual([{ type: 'char', value: 'x' }]);
});
it('reads ESC followed by a letter as Escape then that letter', () => {
expect(decode('\x1bx')).toEqual([{ type: 'escape' }, { type: 'char', value: 'x' }]);
});
});
describe('lone escape', () => {
it('holds a trailing ESC until the caller flushes', () => {
const parser = createKeyParser();
expect(parser.feed('\x1b')).toEqual([]);
expect(parser.pending()).toBe(1);
expect(parser.flush()).toEqual([{ type: 'escape' }]);
expect(parser.pending()).toBe(0);
});
it('completes the sequence instead when the rest arrives', () => {
const parser = createKeyParser();
expect(parser.feed('\x1b')).toEqual([]);
expect(parser.feed('[A')).toEqual([{ type: 'key', name: 'up' }]);
expect(parser.flush()).toEqual([]);
});
it('turns a half-typed sequence into Escape plus its characters', () => {
const parser = createKeyParser();
expect(parser.feed('\x1b[')).toEqual([]);
expect(parser.flush()).toEqual([{ type: 'escape' }, { type: 'char', value: '[' }]);
});
});
describe('SGR mouse', () => {
it('decodes press and release with 1-based coordinates', () => {
expect(decode('\x1b[<0;12;34M')).toEqual([{ type: 'mouse', kind: 'press', x: 12, y: 34, button: 0 }]);
expect(decode('\x1b[<0;12;34m')).toEqual([{ type: 'mouse', kind: 'release', x: 12, y: 34, button: 0 }]);
});
it('decodes the wheel', () => {
expect(decode('\x1b[<64;3;4M')).toEqual([{ type: 'mouse', kind: 'wheel-up', x: 3, y: 4, button: 64 }]);
expect(decode('\x1b[<65;3;4M')).toEqual([{ type: 'mouse', kind: 'wheel-down', x: 3, y: 4, button: 65 }]);
});
it('swallows drag/motion reports', () => {
expect(decode('\x1b[<32;5;6M')).toEqual([]);
});
it('swallows a malformed report', () => {
expect(decode('\x1b[<0;12M')).toEqual([]);
});
});
describe('torn reads', () => {
const cases: Array<[string, TuiInputEvent[]]> = [
['\x1b[A', [{ type: 'key', name: 'up' }]],
['\x1b[6~', [{ type: 'key', name: 'pagedown' }]],
['\x1b[<64;3;4M', [{ type: 'mouse', kind: 'wheel-up', x: 3, y: 4, button: 64 }]],
['\x1bOB', [{ type: 'key', name: 'down' }]],
['\x1b[1;5C', [{ type: 'key', name: 'right' }]],
];
for (const [sequence, expected] of cases) {
it(`decodes ${JSON.stringify(sequence)} at every split position`, () => {
const bytes = Buffer.from(sequence, 'utf8');
for (let at = 0; at <= bytes.length; at++) {
expect(decodeSplit(bytes, at)).toEqual(expected);
}
});
}
it('decodes a mixed burst split anywhere', () => {
const bytes = Buffer.from('a\x1b[Bx\r\x1b[<65;1;1M', 'utf8');
const expected: TuiInputEvent[] = [
{ type: 'char', value: 'a' },
{ type: 'key', name: 'down' },
{ type: 'char', value: 'x' },
{ type: 'enter' },
{ type: 'mouse', kind: 'wheel-down', x: 1, y: 1, button: 65 },
];
for (let at = 0; at <= bytes.length; at++) {
expect(decodeSplit(bytes, at)).toEqual(expected);
}
});
it('drops a garbage burst whole instead of wedging or leaking it', () => {
const parser = createKeyParser();
expect(parser.feed(`\x1b[${'1'.repeat(200)}`)).toEqual([]);
expect(parser.pending()).toBe(0);
expect(parser.feed('\x1b[A')).toEqual([{ type: 'key', name: 'up' }]);
});
});