diff --git a/src/web/response-viewer-transcript.ts b/src/web/response-viewer-transcript.ts new file mode 100644 index 00000000..1404aa81 --- /dev/null +++ b/src/web/response-viewer-transcript.ts @@ -0,0 +1,311 @@ +export type ResponseViewerTranscriptKind = 'prompt' | 'response' | 'status' | 'tool'; + +export interface ResponseViewerTranscriptBlock { + kind: ResponseViewerTranscriptKind; + label: 'Prompt' | 'Response' | 'Status' | 'Tool'; + text: string; +} + +const EXTERNAL_CLI_MODES = new Set(['codex', 'gemini', 'opencode', 'antigravity']); + +function isPromptLine(line: string): boolean { + return /^\s*›\s*/.test(line); +} + +function normalizeDividerStatusLine(line: string): string | null { + const trimmed = line.trim(); + const matched = trimmed.match(/^[─-]+\s*(.+?)\s*[─-]{3,}$/); + if (!matched) return null; + return matched[1]?.trim() || null; +} + +function isDividerOnlyLine(line: string): boolean { + return /^[\s─-]{8,}$/.test(line.trim()); +} + +// COD-227: unambiguous Codex tool-call markers. These only ever appear as internal +// activity, never as ordinary assistant prose, so they are always a Tool block. +function isToolActivityMarker(line: string): boolean { + return /^\s*[•*-]\s+(Calling|Called)\b/.test(line.trim()); +} + +// COD-227: action verbs that ALSO occur in ordinary assistant prose (e.g. +// "• Created COD-226: …"). These are a Tool header only when corroborated by a +// box-drawing result tree on the next non-blank line (see the caller); the verb +// alone is not sufficient. +function isToolVerbBullet(line: string): boolean { + return /^\s*[•*-]\s+(Explored|Viewed|Read|Edited|Updated|Created|Deleted|Ran|Searched|Opened|Listed|Found|Applied|Patched|Used|Wrote|Executed|Modified|Analyzed|Compared|Fetched|Installed)\b/.test( + line.trim() + ); +} + +// A genuine Codex tool block renders its result as a box-drawing tree (└ │ ├). +function isBoxDrawingLine(line: string): boolean { + return /^[│├└]/.test(line.trim()); +} + +// Look past blank lines from `fromIndex + 1` for the next non-blank line and report +// whether it is a box-drawing tool-result line — the signal that a verb bullet is a +// real tool block rather than assistant prose that happens to start with a verb. +function nextNonBlankIsBoxDrawing(lines: string[], fromIndex: number): boolean { + for (let j = fromIndex + 1; j < lines.length; j += 1) { + const trimmed = (lines[j] || '').trim(); + if (!trimmed) continue; + return isBoxDrawingLine(trimmed); + } + return false; +} + +function isToolContinuationLine(line: string, currentKind: ResponseViewerTranscriptKind | null): boolean { + if (currentKind !== 'tool') return false; + const trimmed = line.trimEnd(); + if (!trimmed) return true; + return /^\s*[│├└]/.test(trimmed) || /^\s{2,}\S/.test(line); +} + +function isStatusLine(line: string, mode: string): boolean { + if (!EXTERNAL_CLI_MODES.has(mode)) return false; + const trimmed = line.trim(); + if (!trimmed) return false; + if (normalizeDividerStatusLine(trimmed)) return true; + if (/^(model|directory):\s+/i.test(trimmed)) return true; + if (/\bContext\b.*\bleft\b/i.test(trimmed)) return true; + if (/\b\/model to change\b/i.test(trimmed)) return true; + if (/\bReady\b/i.test(trimmed) && /·/.test(trimmed)) return true; + if (/^(gpt|o\d|claude|gemini)\b/i.test(trimmed) && /·/.test(trimmed)) return true; + if (/^Tip:/i.test(trimmed)) return true; + if (/^\s*[•*-]\s+(Working|Thinking|Loading|Starting\b|Waiting\b)/i.test(trimmed)) return true; + return false; +} + +function isStandaloneMarkdownLine(line: string): boolean { + const trimmed = line.trim(); + if (!trimmed) return false; + if (/^#{1,6}\s/.test(trimmed)) return true; + if (/^>\s/.test(trimmed)) return true; + if (/^(```|~~~)/.test(trimmed)) return true; + if (/^[-*+]\s/.test(trimmed)) return true; + if (/^\d+[.)]\s/.test(trimmed)) return true; + if (/^\|/.test(trimmed)) return true; + if (/^\s{4,}\S/.test(line)) return true; + return false; +} + +function shouldJoinWrappedLine(previous: string, next: string): boolean { + const prev = previous.trimEnd(); + const curr = next.trim(); + if (!prev || !curr) return false; + if (/[.!?]$/.test(prev)) return false; + if (/[:;]$/.test(prev)) return false; + if (isStandaloneMarkdownLine(curr)) return false; + if (/^[a-z(]/.test(curr)) return true; + if (prev.length >= 72 && /^[A-Za-z0-9"'(]/.test(curr)) return true; + return false; +} + +function normalizeWrappedText(lines: string[]): string { + const out: string[] = []; + let paragraph = ''; + + const flushParagraph = () => { + if (!paragraph) return; + out.push(paragraph); + paragraph = ''; + }; + + for (const rawLine of lines) { + const line = rawLine.trimEnd(); + const trimmed = line.trim(); + if (!trimmed) { + flushParagraph(); + if (out[out.length - 1] !== '') out.push(''); + continue; + } + if (isStandaloneMarkdownLine(line)) { + flushParagraph(); + out.push(trimmed); + continue; + } + if (!paragraph) { + paragraph = trimmed; + continue; + } + if (shouldJoinWrappedLine(paragraph, trimmed)) { + paragraph += ` ${trimmed}`; + continue; + } + flushParagraph(); + paragraph = trimmed; + } + + flushParagraph(); + return out + .join('\n') + .replace(/\n{3,}/g, '\n\n') + .trim(); +} + +function cleanTerminalTranscript(buffer: string): string { + // Stripping ANSI/OSC/DCS escape sequences and stray C0/C1 control bytes + // legitimately requires control characters in these patterns. + /* eslint-disable no-control-regex */ + return String(buffer || '') + .replace(/\x1b\[[\x30-\x3F]*[\x20-\x2F]*[\x40-\x7E]/g, '') + .replace(/\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)/g, '') + .replace(/\x1b[PX^_][^\x1b]*\x1b\\/g, '') + .replace(/\x1b[NO()][A-Z0-9]?/g, '') + .replace(/\x1b[>=<78cDEHM]/g, '') + .replace(/[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]/g, '') + .replace(/\r\n/g, '\n') + .replace(/\r/g, '\n') + .replace(/[ \t]+$/gm, '') + .trim(); + /* eslint-enable no-control-regex */ +} + +function trimLeadingStartup(lines: string[]): string[] { + let index = 0; + while (index < lines.length) { + const line = lines[index] || ''; + const trimmed = line.trim(); + if (!trimmed) { + index += 1; + continue; + } + if (/^[╭╰│─].*[╮╯│]?$/.test(trimmed)) { + index += 1; + continue; + } + if (/^>_\s*OpenAI Codex/i.test(trimmed)) { + index += 1; + continue; + } + if (/^(model|directory):\s+/i.test(trimmed)) { + index += 1; + continue; + } + break; + } + return lines.slice(index); +} + +function pushBlock( + blocks: ResponseViewerTranscriptBlock[], + kind: ResponseViewerTranscriptKind | null, + lines: string[] +): void { + if (!kind || lines.length === 0) return; + const normalizedLines = + kind === 'status' + ? lines.map((line) => normalizeDividerStatusLine(line) || line.trim()).filter((line) => line.length > 0) + : lines; + const text = + kind === 'tool' || kind === 'status' + ? normalizedLines + .join('\n') + .replace(/\n{3,}/g, '\n\n') + .trim() + : normalizeWrappedText(normalizedLines); + if (!text) return; + const label = (kind.charAt(0).toUpperCase() + kind.slice(1)) as ResponseViewerTranscriptBlock['label']; + blocks.push({ kind, label, text }); +} + +export function isExternalCliTranscriptMode(mode: string | null | undefined): boolean { + return EXTERNAL_CLI_MODES.has(String(mode || '')); +} + +export function parseExternalCliTranscript( + buffer: string, + mode: string | null | undefined +): ResponseViewerTranscriptBlock[] { + const resolvedMode = String(mode || ''); + if (!isExternalCliTranscriptMode(resolvedMode)) return []; + + const cleaned = cleanTerminalTranscript(buffer); + if (!cleaned) return []; + + const lines = trimLeadingStartup(cleaned.split('\n')); + const blocks: ResponseViewerTranscriptBlock[] = []; + let currentKind: ResponseViewerTranscriptKind | null = null; + let currentLines: string[] = []; + + const flush = () => { + pushBlock(blocks, currentKind, currentLines); + currentKind = null; + currentLines = []; + }; + + for (let i = 0; i < lines.length; i += 1) { + const line = lines[i] ?? ''; + // COD-226: within a prompt, the 2-space Codex gutter is authoritative. Blank + // lines and gutter-indented (2+ leading spaces) continuation lines stay in the + // Prompt block ahead of every structural detector below, so multiline prompts — + // including bullets, indented dividers, and literal › lines — are not + // misclassified as responses. Only a non-blank, non-gutter line ends the prompt + // and falls through (a column-0 › then opens a NEW prompt). + if (currentKind === 'prompt') { + if (!line.trim()) { + currentLines.push(''); + continue; + } + if (/^ {2,}\S/.test(line)) { + currentLines.push(line); + continue; + } + } + + if (isDividerOnlyLine(line)) { + flush(); + continue; + } + + if (isPromptLine(line)) { + flush(); + currentKind = 'prompt'; + currentLines = [line.replace(/^\s*›\s*/, '').trim()]; + continue; + } + + // COD-227: • Calling / • Called are always tool markers; the other action verbs + // are a tool header only when a box-drawing result tree follows on the next + // non-blank line — otherwise the verb bullet is ordinary assistant prose. + if (isToolActivityMarker(line) || (isToolVerbBullet(line) && nextNonBlankIsBoxDrawing(lines, i))) { + if (currentKind !== 'tool') flush(); + currentKind = 'tool'; + currentLines.push(line.trimEnd()); + continue; + } + + if (isToolContinuationLine(line, currentKind)) { + currentLines.push(line.trimEnd()); + continue; + } + + if (isStatusLine(line, resolvedMode)) { + if (currentKind !== 'status') flush(); + currentKind = 'status'; + currentLines.push(line.trim()); + continue; + } + + if (!line.trim()) { + currentLines.push(''); + continue; + } + + if (currentKind !== 'response') flush(); + currentKind = 'response'; + currentLines.push(line); + } + + flush(); + return blocks.filter((block) => block.text.trim().length > 0); +} + +export function getLastTranscriptResponse(blocks: ResponseViewerTranscriptBlock[]): string { + for (let i = blocks.length - 1; i >= 0; i -= 1) { + if (blocks[i]?.kind === 'response') return blocks[i].text; + } + return ''; +} diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index 98ea4e7c..aa012142 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -131,6 +131,11 @@ import { toSessionDocker, } from '../../docker-hosts.js'; import { LRUMap } from '../../utils/lru-map.js'; +import { + getLastTranscriptResponse, + isExternalCliTranscriptMode, + parseExternalCliTranscript, +} from '../response-viewer-transcript.js'; // Path to linked-cases registry (same file used by case-routes resolveCasePath) const LINKED_CASES_FILE = dataPath('linked-cases.json'); @@ -1898,6 +1903,22 @@ export function registerSessionRoutes( return await readCodexLastResponse(session, codexQuery.context === 'full'); } + // OpenCode / Gemini / Antigravity render their own TUIs and write no Claude + // transcript, so the scan below finds nothing and the response viewer renders + // permanently empty for them. Segment the terminal buffer instead — the pane + // IS the transcript for these CLIs. Codex is already handled above, where a + // real rollout file is the better source. + if (isExternalCliTranscriptMode(session.mode)) { + const externalQuery = req.query as { context?: string }; + const blocks = parseExternalCliTranscript(session.terminalBuffer, session.mode); + return { + text: getLastTranscriptResponse(blocks), + timestamp: '', + hasContext: blocks.length > 0, + messages: externalQuery.context === 'full' ? blocks : undefined, + }; + } + // Scan ~/.claude/projects/*/ for the transcript file const projectsDir = join(process.env.HOME || '/tmp', '.claude', 'projects'); diff --git a/test/response-viewer-transcript.test.ts b/test/response-viewer-transcript.test.ts new file mode 100644 index 00000000..818c69d7 --- /dev/null +++ b/test/response-viewer-transcript.test.ts @@ -0,0 +1,323 @@ +import { describe, expect, it } from 'vitest'; +import { getLastTranscriptResponse, parseExternalCliTranscript } from '../src/web/response-viewer-transcript.js'; + +describe('response viewer transcript parser', () => { + it('extracts structured COD transcript blocks and keeps labeled status/tool entries', () => { + const transcript = ` +╭──────────────────────────────────────────────────────╮ +│ >_ OpenAI Codex (v0.143.0) │ +│ model: gpt-5.4 medium /model to change │ +│ directory: /mnt/c/Users/aakhter/.../kb │ +╰──────────────────────────────────────────────────────╯ + +Tip: Use /side to start a side conversation in a temporary fork without polluting the main thread. + +› i need to create a naming recommendation for project helix + +gpt-5.4 medium · kb · main · Ready · Context 100% left + +• Explored + └ Read SKILL.md + +Subject: Naming Recommendation for Project Helix + +The product helps customers move virtual machine workloads between hypervisors while keeping operations +stable. It improves visibility into application dependencies, supports pre-migration validation, and reduces +the risk +and effort involved in moving workloads. + +› Improve documentation in @filename + +gpt-5.4 medium · kb · main · Ready · Context 92% left +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + + expect(blocks.map((block) => block.kind)).toEqual([ + 'status', + 'prompt', + 'status', + 'tool', + 'response', + 'prompt', + 'status', + ]); + expect(blocks[0]?.label).toBe('Status'); + expect(blocks[3]?.label).toBe('Tool'); + expect(blocks[4]?.text).toContain('reduces the risk and effort involved'); + expect(blocks[4]?.text).not.toContain('reduces\nthe risk'); + }); + + it('returns the most recent completed response when the last block is a new prompt', () => { + const transcript = ` +› say again + +Final polished answer +With two lines + +› Improve documentation in @filename + +gpt-5.4 medium · kb · main · Ready · Context 92% left +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + + expect(getLastTranscriptResponse(blocks)).toBe('Final polished answer\nWith two lines'); + }); + + it('drops box-drawing dividers from the last response and treats worked-for lines as status', () => { + const transcript = ` +──────────────────────────────────────────────────────────────────────── + +• The launcher supports CODEMAN_APP_DIR, so I can deploy this exact worktree without merging it back first. + +──────────────────────────────────────────────────────────────────────── + +• Deployed. + +The local Codeman service is now running from the worktree at app/.worktrees/cod-215-response-viewer on +commit 8bbcf77bafbdbf01a34653386c256b685898ad06. + +Verified: + +- process pid: 2708947 +- HTTPS health: https://127.0.0.1:3000/ returned 401 as expected + +One detail: I had to restart it outside the sandbox because the sandboxed launch path could not see tmux. + +─ Worked for 1m 51s ─────────────────────────────────────────────────── + +› what are these lines in the middle column? +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + + expect(getLastTranscriptResponse(blocks)).toContain('• Deployed.'); + expect(getLastTranscriptResponse(blocks)).not.toContain('────────────────'); + expect(getLastTranscriptResponse(blocks)).not.toContain('Worked for 1m 51s'); + expect(blocks.some((block) => block.kind === 'status' && block.text === 'Worked for 1m 51s')).toBe(true); + }); + + // COD-226: multiline prompt continuations (2-space Codex gutter) must stay in the + // Prompt block, not be misclassified as Response. The gutter is authoritative ahead + // of every structural detector (divider, prompt-marker, tool, status). + describe('COD-226 multiline prompt gutter is authoritative', () => { + it('keeps bullet/prose continuations and internal blank lines in the Prompt block', () => { + const transcript = ` +› i think we can improve this slide. or maybe a follow on slide. here is what i'm thinking: + * left hand side: current AI stack (frontier model, cloud hosted, sovereign concerns) + right hand -> future enterprise stack: frontier model (optional, cloud), on-prem model router, OSS models + + make the point that the right hand side addresses the concerns. + +gpt-5.4 medium · kb · main · Ready · Context 100% left + +• Explored + └ Read SKILL.md + +Here is the actual assistant answer that starts at column zero and is a real response. + +› next prompt +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + const promptBlocks = blocks.filter((b) => b.kind === 'prompt'); + + // Exactly two prompts, and the first holds the whole multiline prompt. + expect(promptBlocks).toHaveLength(2); + expect(promptBlocks[0]?.text).toContain('left hand side'); + expect(promptBlocks[0]?.text).toContain('right hand'); + expect(promptBlocks[0]?.text).toContain('make the point that the right hand side'); + + // The continuation must NOT have leaked into a Response block. + const responseBlocks = blocks.filter((b) => b.kind === 'response'); + expect(responseBlocks.some((b) => b.text.includes('make the point'))).toBe(false); + expect(getLastTranscriptResponse(blocks)).toContain('actual assistant answer'); + expect(getLastTranscriptResponse(blocks)).not.toContain('make the point'); + }); + + it('does not let an indented divider inside a prompt flush the Prompt block', () => { + const transcript = ` +› compare these two layouts + first layout uses a single column + ──────────────────────────── + second layout uses two columns + +The response begins here at column zero. +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + const promptBlocks = blocks.filter((b) => b.kind === 'prompt'); + + expect(promptBlocks).toHaveLength(1); + expect(promptBlocks[0]?.text).toContain('first layout'); + expect(promptBlocks[0]?.text).toContain('second layout uses two columns'); + expect(getLastTranscriptResponse(blocks)).toBe('The response begins here at column zero.'); + }); + + it('treats a gutter-indented literal › as prompt content, not a new prompt', () => { + const transcript = ` +› here is my question about the ui + › should this arrow start a new prompt? + no it should not — it is part of my question + +Answer at column zero. +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + const promptBlocks = blocks.filter((b) => b.kind === 'prompt'); + + // The gutter-indented › must NOT open a second prompt. + expect(promptBlocks).toHaveLength(1); + expect(promptBlocks[0]?.text).toContain('here is my question'); + expect(promptBlocks[0]?.text).toContain('no it should not'); + expect(getLastTranscriptResponse(blocks)).toBe('Answer at column zero.'); + }); + + // The two exact live roadmap-tab examples from the ticket (AC: use both verbatim). + it('live example 1: two-line prompt keeps the gutter continuation in the Prompt block', () => { + const transcript = ` +› create uid for each feature so it's easy to ref. + for the p200 - why is diffentiation only 2/5? +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + const promptBlocks = blocks.filter((b) => b.kind === 'prompt'); + + expect(promptBlocks).toHaveLength(1); + expect(promptBlocks[0]?.text).toContain('create uid for each feature'); + expect(promptBlocks[0]?.text).toContain('for the p200 - why is diffentiation only 2/5?'); + // The continuation must not have leaked into a Response block. + expect(blocks.some((b) => b.kind === 'response')).toBe(false); + }); + + it('live example 2: five-line prompt (bullet + prose + blank + prose) stays one Prompt block', () => { + const transcript = ` +› i think we can improve this slide. or maybe a follow on slide. here is what i'm thinking: + * left hand side... current AI stack: (frontier model), large component cloud hosted, soverign concerns etc. + right hand -> future-enterprise-stack: frontier-model (optional, in cloud), on-prem: model router, OSS models, rest of stack (gpus, data, etc.) + + make the point that the right hand side addresses the concerns. +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + const promptBlocks = blocks.filter((b) => b.kind === 'prompt'); + + expect(promptBlocks).toHaveLength(1); + expect(promptBlocks[0]?.text).toContain('left hand side'); + expect(promptBlocks[0]?.text).toContain('right hand -> future-enterprise-stack'); + expect(promptBlocks[0]?.text).toContain('make the point that the right hand side addresses the concerns'); + expect(blocks.some((b) => b.kind === 'response')).toBe(false); + }); + + it('still separates a single-line prompt from a column-zero response (no regression)', () => { + const transcript = ` +› say again + +Final polished answer at column zero. + +› next +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + const promptBlocks = blocks.filter((b) => b.kind === 'prompt'); + + expect(promptBlocks).toHaveLength(2); + expect(promptBlocks[0]?.text).toBe('say again'); + expect(getLastTranscriptResponse(blocks)).toBe('Final polished answer at column zero.'); + }); + }); + + // COD-227: Last Response must return the final assistant answer, not tool logs. + // A response bullet beginning with a tool-like verb (• Created …) must not be + // classified as Tool, and genuine • Calling / • Called blocks must be classified + // as Tool. Disambiguator: a verb-bullet is a tool header only when followed by a + // box-drawing result tree (└│├); Calling/Called are always tool markers. + describe('COD-227 tool-header vs response disambiguation', () => { + const MINIMAL_REPRO = ` +› new jira issue + +• The fresh read shows a formatting problem. + +• Calling +└ atlassian.jira_update_issue({}) + +• Called atlassian.jira_get_issue({}) +└ { result: true } + +• Created COD-226: View Response → More misclassifies multiline prompt continuations as responses. + + It includes: + + - Two concrete failures + - Regression-test criteria +`.trim(); + + it('returns the final • Created … answer, not the Jira Calling/Called tool log', () => { + const blocks = parseExternalCliTranscript(MINIMAL_REPRO, 'codex'); + const last = getLastTranscriptResponse(blocks); + + expect(last).toContain('Created COD-226'); + expect(last).toContain('Two concrete failures'); + // Must exclude tool invocations, raw results, and earlier commentary. + expect(last).not.toContain('atlassian.jira'); + expect(last).not.toContain('Calling'); + expect(last).not.toContain('Called'); + expect(last).not.toContain('fresh read'); + }); + + it('classifies genuine • Calling / • Called (with box-drawing results) as Tool', () => { + const blocks = parseExternalCliTranscript(MINIMAL_REPRO, 'codex'); + const toolText = blocks + .filter((b) => b.kind === 'tool') + .map((b) => b.text) + .join('\n'); + + expect(toolText).toContain('Calling'); + expect(toolText).toContain('Called'); + expect(toolText).toContain('atlassian.jira_update_issue'); + // The final answer must not have been swallowed into the tool block. + expect(toolText).not.toContain('Created COD-226'); + }); + + it('labels the repro chronologically: Prompt, Response (commentary), Tool, Response (final)', () => { + const blocks = parseExternalCliTranscript(MINIMAL_REPRO, 'codex'); + + expect(blocks.map((b) => b.kind)).toEqual(['prompt', 'response', 'tool', 'response']); + expect(blocks[1]?.text).toContain('fresh read'); + expect(blocks[3]?.text).toContain('Created COD-226'); + }); + + it('does not classify a verb-prefixed prose bullet as Tool when no result tree follows', () => { + const transcript = ` +› do it + +• Created COD-999: a brand new issue with a descriptive title. + + Follow-up prose that belongs to the same answer. +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + + expect(blocks.some((b) => b.kind === 'tool')).toBe(false); + expect(getLastTranscriptResponse(blocks)).toContain('Created COD-999'); + expect(getLastTranscriptResponse(blocks)).toContain('Follow-up prose'); + }); + + it('does not regress a genuine verb tool block that has a box-drawing continuation', () => { + const transcript = ` +› look around + +• Explored + └ Read SKILL.md + +Here is the assistant answer at column zero. +`.trim(); + + const blocks = parseExternalCliTranscript(transcript, 'codex'); + + expect(blocks.some((b) => b.kind === 'tool' && b.text.includes('Explored'))).toBe(true); + expect(getLastTranscriptResponse(blocks)).toBe('Here is the assistant answer at column zero.'); + }); + }); +}); diff --git a/test/routes/session-routes-external-cli-last-response.test.ts b/test/routes/session-routes-external-cli-last-response.test.ts new file mode 100644 index 00000000..9853ea1c --- /dev/null +++ b/test/routes/session-routes-external-cli-last-response.test.ts @@ -0,0 +1,160 @@ +/** + * @fileoverview Tests for the external-CLI branch of GET /api/sessions/:id/last-response. + * + * Uses app.inject() — no real HTTP ports needed. + * Port: N/A (app.inject doesn't open ports) + * + * OpenCode / Gemini / Antigravity render their own TUIs and never write a Claude + * transcript under ~/.claude/projects, so before this branch existed the handler + * fell through to the Claude scan, found nothing, and the response viewer was + * permanently empty for those modes. These tests pin: + * - the pane buffer is segmented and the LAST response is returned + * - ?context=full carries the parsed blocks, and the short form omits them + * - a pane that has produced no output reports hasContext: false rather than 404ing + * - Claude mode still takes the Claude path (regression guard) + */ + +import { describe, it, expect, beforeEach } from 'vitest'; +import Fastify, { type FastifyInstance } from 'fastify'; +import fastifyCookie from '@fastify/cookie'; +import { createMockRouteContext, createMockSession, type MockRouteContext } from '../mocks/index.js'; +import { ApiErrorCode, httpStatusForErrorCode } from '../../src/types.js'; +import { registerSessionRoutes } from '../../src/web/routes/session-routes.js'; + +interface LocalHarness { + app: FastifyInstance; + ctx: MockRouteContext; +} + +/** Mirror of the production uniform-envelope hook (server.ts), as in the sibling suites. */ +async function createEnvelopeHarness( + registerFn: (app: FastifyInstance, ctx: MockRouteContext) => void +): Promise { + const app = Fastify({ logger: false }); + await app.register(fastifyCookie); + + const ctx = createMockRouteContext(); + registerFn(app, ctx); + + app.addHook('preSerialization', (req, reply, payload: unknown, done) => { + if (!req.url.startsWith('/api')) return done(null, payload); + if (payload === null || typeof payload !== 'object') return done(null, payload); + if (Buffer.isBuffer(payload) || typeof (payload as { pipe?: unknown }).pipe === 'function') { + return done(null, payload); + } + const p = payload as { success?: unknown; errorCode?: unknown }; + if (p.success === false) { + if (reply.statusCode === 200 && typeof p.errorCode === 'string') { + reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode)); + } + return done(null, payload); + } + if (p.success === true) return done(null, payload); + return done(null, { success: true, data: payload }); + }); + + await app.ready(); + return { app, ctx }; +} + +// A pane as one of these CLIs actually leaves it: banner, a `›` prompt line, a +// status divider, a tool-activity marker, then the assistant's prose. +const PANE = ` +╭──────────────────────────────────────────────────────╮ +│ >_ OpenCode │ +│ directory: /workspace/project │ +╰──────────────────────────────────────────────────────╯ + +› summarise the retry logic + +model · project · main · Ready · Context 100% left + +• Explored + └ Read src/retry.ts + +The retry helper backs off exponentially and gives up after five attempts. + +› now document it + +model · project · main · Ready · Context 92% left + +• Called write_file + +Documented the helper in docs/retry.md, including the five-attempt ceiling. +`.trim(); + +describe('GET /api/sessions/:id/last-response — external CLI panes', () => { + let harness: LocalHarness; + let session: ReturnType; + + beforeEach(async () => { + harness = await createEnvelopeHarness(registerSessionRoutes); + session = harness.ctx._session; + }); + + async function lastResponse(full = false) { + const res = await harness.app.inject({ + method: 'GET', + url: `/api/sessions/${session.id}/last-response${full ? '?context=full' : ''}`, + }); + return { res, body: JSON.parse(res.body) }; + } + + for (const mode of ['opencode', 'gemini', 'antigravity'] as const) { + it(`returns the last assistant response from the ${mode} pane buffer`, async () => { + session.mode = mode; + session.terminalBuffer = PANE; + + const { res, body } = await lastResponse(); + + expect(res.statusCode).toBe(200); + // The LAST response, not the first — the viewer shows the current turn. + expect(body.data.text).toContain('Documented the helper in docs/retry.md'); + expect(body.data.text).not.toContain('backs off exponentially'); + expect(body.data.hasContext).toBe(true); + // Short form stays short: blocks only travel under ?context=full. + expect(body.data.messages).toBeUndefined(); + }); + } + + it('carries the parsed blocks under ?context=full', async () => { + session.mode = 'opencode'; + session.terminalBuffer = PANE; + + const { body } = await lastResponse(true); + + const kinds = body.data.messages.map((block: { kind: string }) => block.kind); + expect(kinds).toContain('prompt'); + expect(kinds).toContain('response'); + expect(kinds).toContain('tool'); + // Both user turns survive segmentation, so the viewer can show the exchange. + const prompts = body.data.messages.filter((b: { kind: string }) => b.kind === 'prompt'); + expect(prompts).toHaveLength(2); + expect(prompts[1].text).toContain('now document it'); + }); + + it('reports hasContext false for a pane that has produced no output', async () => { + session.mode = 'gemini'; + // Session created but nothing rendered yet. Any non-prompt line counts as + // prose to the parser, so the empty pane is the honest no-context case. + session.terminalBuffer = ''; + + const { res, body } = await lastResponse(); + + expect(res.statusCode).toBe(200); + expect(body.data.text).toBe(''); + expect(body.data.hasContext).toBe(false); + }); + + it('leaves Claude mode on the Claude transcript path', async () => { + // Regression guard: a claude pane must NOT be segmented off its terminal + // buffer, or a real transcript would be shadowed by scraped pane text. + session.mode = 'claude'; + session.terminalBuffer = PANE; + + const { res, body } = await lastResponse(); + + expect(res.statusCode).toBe(200); + expect(body.data.text).not.toContain('Documented the helper'); + }); +});