Merge pull request #326 from aakhter/response-viewer-external-cli

Response viewer is empty for OpenCode / Gemini / Antigravity sessions
This commit is contained in:
Ark0N
2026-08-21 02:10:29 +02:00
committed by GitHub
4 changed files with 815 additions and 0 deletions
+323
View File
@@ -0,0 +1,323 @@
import { describe, expect, it } from 'vitest';
import { getLastTranscriptResponse, parseExternalCliTranscript } from '../src/web/response-viewer-transcript.js';
describe('response viewer transcript parser', () => {
it('extracts structured COD transcript blocks and keeps labeled status/tool entries', () => {
const transcript = `
╭──────────────────────────────────────────────────────╮
│ >_ OpenAI Codex (v0.143.0) │
│ model: gpt-5.4 medium /model to change │
│ directory: /mnt/c/Users/aakhter/.../kb │
╰──────────────────────────────────────────────────────╯
Tip: Use /side to start a side conversation in a temporary fork without polluting the main thread.
› i need to create a naming recommendation for project helix
gpt-5.4 medium · kb · main · Ready · Context 100% left
• Explored
└ Read SKILL.md
Subject: Naming Recommendation for Project Helix
The product helps customers move virtual machine workloads between hypervisors while keeping operations
stable. It improves visibility into application dependencies, supports pre-migration validation, and reduces
the risk
and effort involved in moving workloads.
› Improve documentation in @filename
gpt-5.4 medium · kb · main · Ready · Context 92% left
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
expect(blocks.map((block) => block.kind)).toEqual([
'status',
'prompt',
'status',
'tool',
'response',
'prompt',
'status',
]);
expect(blocks[0]?.label).toBe('Status');
expect(blocks[3]?.label).toBe('Tool');
expect(blocks[4]?.text).toContain('reduces the risk and effort involved');
expect(blocks[4]?.text).not.toContain('reduces\nthe risk');
});
it('returns the most recent completed response when the last block is a new prompt', () => {
const transcript = `
› say again
Final polished answer
With two lines
› Improve documentation in @filename
gpt-5.4 medium · kb · main · Ready · Context 92% left
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
expect(getLastTranscriptResponse(blocks)).toBe('Final polished answer\nWith two lines');
});
it('drops box-drawing dividers from the last response and treats worked-for lines as status', () => {
const transcript = `
────────────────────────────────────────────────────────────────────────
• The launcher supports CODEMAN_APP_DIR, so I can deploy this exact worktree without merging it back first.
────────────────────────────────────────────────────────────────────────
• Deployed.
The local Codeman service is now running from the worktree at app/.worktrees/cod-215-response-viewer on
commit 8bbcf77bafbdbf01a34653386c256b685898ad06.
Verified:
- process pid: 2708947
- HTTPS health: https://127.0.0.1:3000/ returned 401 as expected
One detail: I had to restart it outside the sandbox because the sandboxed launch path could not see tmux.
─ Worked for 1m 51s ───────────────────────────────────────────────────
› what are these lines in the middle column?
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
expect(getLastTranscriptResponse(blocks)).toContain('• Deployed.');
expect(getLastTranscriptResponse(blocks)).not.toContain('────────────────');
expect(getLastTranscriptResponse(blocks)).not.toContain('Worked for 1m 51s');
expect(blocks.some((block) => block.kind === 'status' && block.text === 'Worked for 1m 51s')).toBe(true);
});
// COD-226: multiline prompt continuations (2-space Codex gutter) must stay in the
// Prompt block, not be misclassified as Response. The gutter is authoritative ahead
// of every structural detector (divider, prompt-marker, tool, status).
describe('COD-226 multiline prompt gutter is authoritative', () => {
it('keeps bullet/prose continuations and internal blank lines in the Prompt block', () => {
const transcript = `
› i think we can improve this slide. or maybe a follow on slide. here is what i'm thinking:
* left hand side: current AI stack (frontier model, cloud hosted, sovereign concerns)
right hand -> future enterprise stack: frontier model (optional, cloud), on-prem model router, OSS models
make the point that the right hand side addresses the concerns.
gpt-5.4 medium · kb · main · Ready · Context 100% left
• Explored
└ Read SKILL.md
Here is the actual assistant answer that starts at column zero and is a real response.
› next prompt
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
const promptBlocks = blocks.filter((b) => b.kind === 'prompt');
// Exactly two prompts, and the first holds the whole multiline prompt.
expect(promptBlocks).toHaveLength(2);
expect(promptBlocks[0]?.text).toContain('left hand side');
expect(promptBlocks[0]?.text).toContain('right hand');
expect(promptBlocks[0]?.text).toContain('make the point that the right hand side');
// The continuation must NOT have leaked into a Response block.
const responseBlocks = blocks.filter((b) => b.kind === 'response');
expect(responseBlocks.some((b) => b.text.includes('make the point'))).toBe(false);
expect(getLastTranscriptResponse(blocks)).toContain('actual assistant answer');
expect(getLastTranscriptResponse(blocks)).not.toContain('make the point');
});
it('does not let an indented divider inside a prompt flush the Prompt block', () => {
const transcript = `
› compare these two layouts
first layout uses a single column
────────────────────────────
second layout uses two columns
The response begins here at column zero.
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
const promptBlocks = blocks.filter((b) => b.kind === 'prompt');
expect(promptBlocks).toHaveLength(1);
expect(promptBlocks[0]?.text).toContain('first layout');
expect(promptBlocks[0]?.text).toContain('second layout uses two columns');
expect(getLastTranscriptResponse(blocks)).toBe('The response begins here at column zero.');
});
it('treats a gutter-indented literal › as prompt content, not a new prompt', () => {
const transcript = `
› here is my question about the ui
› should this arrow start a new prompt?
no it should not — it is part of my question
Answer at column zero.
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
const promptBlocks = blocks.filter((b) => b.kind === 'prompt');
// The gutter-indented › must NOT open a second prompt.
expect(promptBlocks).toHaveLength(1);
expect(promptBlocks[0]?.text).toContain('here is my question');
expect(promptBlocks[0]?.text).toContain('no it should not');
expect(getLastTranscriptResponse(blocks)).toBe('Answer at column zero.');
});
// The two exact live roadmap-tab examples from the ticket (AC: use both verbatim).
it('live example 1: two-line prompt keeps the gutter continuation in the Prompt block', () => {
const transcript = `
› create uid for each feature so it's easy to ref.
for the p200 - why is diffentiation only 2/5?
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
const promptBlocks = blocks.filter((b) => b.kind === 'prompt');
expect(promptBlocks).toHaveLength(1);
expect(promptBlocks[0]?.text).toContain('create uid for each feature');
expect(promptBlocks[0]?.text).toContain('for the p200 - why is diffentiation only 2/5?');
// The continuation must not have leaked into a Response block.
expect(blocks.some((b) => b.kind === 'response')).toBe(false);
});
it('live example 2: five-line prompt (bullet + prose + blank + prose) stays one Prompt block', () => {
const transcript = `
› i think we can improve this slide. or maybe a follow on slide. here is what i'm thinking:
* left hand side... current AI stack: (frontier model), large component cloud hosted, soverign concerns etc.
right hand -> future-enterprise-stack: frontier-model (optional, in cloud), on-prem: model router, OSS models, rest of stack (gpus, data, etc.)
make the point that the right hand side addresses the concerns.
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
const promptBlocks = blocks.filter((b) => b.kind === 'prompt');
expect(promptBlocks).toHaveLength(1);
expect(promptBlocks[0]?.text).toContain('left hand side');
expect(promptBlocks[0]?.text).toContain('right hand -> future-enterprise-stack');
expect(promptBlocks[0]?.text).toContain('make the point that the right hand side addresses the concerns');
expect(blocks.some((b) => b.kind === 'response')).toBe(false);
});
it('still separates a single-line prompt from a column-zero response (no regression)', () => {
const transcript = `
› say again
Final polished answer at column zero.
› next
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
const promptBlocks = blocks.filter((b) => b.kind === 'prompt');
expect(promptBlocks).toHaveLength(2);
expect(promptBlocks[0]?.text).toBe('say again');
expect(getLastTranscriptResponse(blocks)).toBe('Final polished answer at column zero.');
});
});
// COD-227: Last Response must return the final assistant answer, not tool logs.
// A response bullet beginning with a tool-like verb (• Created …) must not be
// classified as Tool, and genuine • Calling / • Called blocks must be classified
// as Tool. Disambiguator: a verb-bullet is a tool header only when followed by a
// box-drawing result tree (└│├); Calling/Called are always tool markers.
describe('COD-227 tool-header vs response disambiguation', () => {
const MINIMAL_REPRO = `
› new jira issue
• The fresh read shows a formatting problem.
• Calling
└ atlassian.jira_update_issue({})
• Called atlassian.jira_get_issue({})
└ { result: true }
• Created COD-226: View Response → More misclassifies multiline prompt continuations as responses.
It includes:
- Two concrete failures
- Regression-test criteria
`.trim();
it('returns the final • Created … answer, not the Jira Calling/Called tool log', () => {
const blocks = parseExternalCliTranscript(MINIMAL_REPRO, 'codex');
const last = getLastTranscriptResponse(blocks);
expect(last).toContain('Created COD-226');
expect(last).toContain('Two concrete failures');
// Must exclude tool invocations, raw results, and earlier commentary.
expect(last).not.toContain('atlassian.jira');
expect(last).not.toContain('Calling');
expect(last).not.toContain('Called');
expect(last).not.toContain('fresh read');
});
it('classifies genuine • Calling / • Called (with box-drawing results) as Tool', () => {
const blocks = parseExternalCliTranscript(MINIMAL_REPRO, 'codex');
const toolText = blocks
.filter((b) => b.kind === 'tool')
.map((b) => b.text)
.join('\n');
expect(toolText).toContain('Calling');
expect(toolText).toContain('Called');
expect(toolText).toContain('atlassian.jira_update_issue');
// The final answer must not have been swallowed into the tool block.
expect(toolText).not.toContain('Created COD-226');
});
it('labels the repro chronologically: Prompt, Response (commentary), Tool, Response (final)', () => {
const blocks = parseExternalCliTranscript(MINIMAL_REPRO, 'codex');
expect(blocks.map((b) => b.kind)).toEqual(['prompt', 'response', 'tool', 'response']);
expect(blocks[1]?.text).toContain('fresh read');
expect(blocks[3]?.text).toContain('Created COD-226');
});
it('does not classify a verb-prefixed prose bullet as Tool when no result tree follows', () => {
const transcript = `
› do it
• Created COD-999: a brand new issue with a descriptive title.
Follow-up prose that belongs to the same answer.
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
expect(blocks.some((b) => b.kind === 'tool')).toBe(false);
expect(getLastTranscriptResponse(blocks)).toContain('Created COD-999');
expect(getLastTranscriptResponse(blocks)).toContain('Follow-up prose');
});
it('does not regress a genuine verb tool block that has a box-drawing continuation', () => {
const transcript = `
› look around
• Explored
└ Read SKILL.md
Here is the assistant answer at column zero.
`.trim();
const blocks = parseExternalCliTranscript(transcript, 'codex');
expect(blocks.some((b) => b.kind === 'tool' && b.text.includes('Explored'))).toBe(true);
expect(getLastTranscriptResponse(blocks)).toBe('Here is the assistant answer at column zero.');
});
});
});
@@ -0,0 +1,160 @@
/**
* @fileoverview Tests for the external-CLI branch of GET /api/sessions/:id/last-response.
*
* Uses app.inject() — no real HTTP ports needed.
* Port: N/A (app.inject doesn't open ports)
*
* OpenCode / Gemini / Antigravity render their own TUIs and never write a Claude
* transcript under ~/.claude/projects, so before this branch existed the handler
* fell through to the Claude scan, found nothing, and the response viewer was
* permanently empty for those modes. These tests pin:
* - the pane buffer is segmented and the LAST response is returned
* - ?context=full carries the parsed blocks, and the short form omits them
* - a pane that has produced no output reports hasContext: false rather than 404ing
* - Claude mode still takes the Claude path (regression guard)
*/
import { describe, it, expect, beforeEach } from 'vitest';
import Fastify, { type FastifyInstance } from 'fastify';
import fastifyCookie from '@fastify/cookie';
import { createMockRouteContext, createMockSession, type MockRouteContext } from '../mocks/index.js';
import { ApiErrorCode, httpStatusForErrorCode } from '../../src/types.js';
import { registerSessionRoutes } from '../../src/web/routes/session-routes.js';
interface LocalHarness {
app: FastifyInstance;
ctx: MockRouteContext;
}
/** Mirror of the production uniform-envelope hook (server.ts), as in the sibling suites. */
async function createEnvelopeHarness(
registerFn: (app: FastifyInstance, ctx: MockRouteContext) => void
): Promise<LocalHarness> {
const app = Fastify({ logger: false });
await app.register(fastifyCookie);
const ctx = createMockRouteContext();
registerFn(app, ctx);
app.addHook('preSerialization', (req, reply, payload: unknown, done) => {
if (!req.url.startsWith('/api')) return done(null, payload);
if (payload === null || typeof payload !== 'object') return done(null, payload);
if (Buffer.isBuffer(payload) || typeof (payload as { pipe?: unknown }).pipe === 'function') {
return done(null, payload);
}
const p = payload as { success?: unknown; errorCode?: unknown };
if (p.success === false) {
if (reply.statusCode === 200 && typeof p.errorCode === 'string') {
reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode));
}
return done(null, payload);
}
if (p.success === true) return done(null, payload);
return done(null, { success: true, data: payload });
});
await app.ready();
return { app, ctx };
}
// A pane as one of these CLIs actually leaves it: banner, a `›` prompt line, a
// status divider, a tool-activity marker, then the assistant's prose.
const PANE = `
╭──────────────────────────────────────────────────────╮
│ >_ OpenCode │
│ directory: /workspace/project │
╰──────────────────────────────────────────────────────╯
› summarise the retry logic
model · project · main · Ready · Context 100% left
• Explored
└ Read src/retry.ts
The retry helper backs off exponentially and gives up after five attempts.
› now document it
model · project · main · Ready · Context 92% left
• Called write_file
Documented the helper in docs/retry.md, including the five-attempt ceiling.
`.trim();
describe('GET /api/sessions/:id/last-response — external CLI panes', () => {
let harness: LocalHarness;
let session: ReturnType<typeof createMockSession>;
beforeEach(async () => {
harness = await createEnvelopeHarness(registerSessionRoutes);
session = harness.ctx._session;
});
async function lastResponse(full = false) {
const res = await harness.app.inject({
method: 'GET',
url: `/api/sessions/${session.id}/last-response${full ? '?context=full' : ''}`,
});
return { res, body: JSON.parse(res.body) };
}
for (const mode of ['opencode', 'gemini', 'antigravity'] as const) {
it(`returns the last assistant response from the ${mode} pane buffer`, async () => {
session.mode = mode;
session.terminalBuffer = PANE;
const { res, body } = await lastResponse();
expect(res.statusCode).toBe(200);
// The LAST response, not the first — the viewer shows the current turn.
expect(body.data.text).toContain('Documented the helper in docs/retry.md');
expect(body.data.text).not.toContain('backs off exponentially');
expect(body.data.hasContext).toBe(true);
// Short form stays short: blocks only travel under ?context=full.
expect(body.data.messages).toBeUndefined();
});
}
it('carries the parsed blocks under ?context=full', async () => {
session.mode = 'opencode';
session.terminalBuffer = PANE;
const { body } = await lastResponse(true);
const kinds = body.data.messages.map((block: { kind: string }) => block.kind);
expect(kinds).toContain('prompt');
expect(kinds).toContain('response');
expect(kinds).toContain('tool');
// Both user turns survive segmentation, so the viewer can show the exchange.
const prompts = body.data.messages.filter((b: { kind: string }) => b.kind === 'prompt');
expect(prompts).toHaveLength(2);
expect(prompts[1].text).toContain('now document it');
});
it('reports hasContext false for a pane that has produced no output', async () => {
session.mode = 'gemini';
// Session created but nothing rendered yet. Any non-prompt line counts as
// prose to the parser, so the empty pane is the honest no-context case.
session.terminalBuffer = '';
const { res, body } = await lastResponse();
expect(res.statusCode).toBe(200);
expect(body.data.text).toBe('');
expect(body.data.hasContext).toBe(false);
});
it('leaves Claude mode on the Claude transcript path', async () => {
// Regression guard: a claude pane must NOT be segmented off its terminal
// buffer, or a real transcript would be shadowed by scraped pane text.
session.mode = 'claude';
session.terminalBuffer = PANE;
const { res, body } = await lastResponse();
expect(res.statusCode).toBe(200);
expect(body.data.text).not.toContain('Documented the helper');
});
});