mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
feat(deepseek): read dsh session transcripts for last-response
`GET /api/sessions/:id/last-response` is how an agent (and the Response Viewer) reads what a worker said. DeepSeek was falling through to the pane segmenter with the other external CLIs, which for this mode is not merely coarse but wrong: dsh-TUI paints a full-screen splash, so a `last-response` call on a fresh dsh session answered with its ASCII-art logo -- and anything polling for a worker's first reply reads that as a reply. dsh does not belong in that group. It writes a structured JSONL transcript per session, so read it. Four things in that file shaped the reader, all measured against real transcripts on disk: 1. dsh appends ONE ZSTD FRAME PER WRITE, and Node's zlib zstd decoder (one-shot and streaming alike) stops at the first frame end: a real 56-line transcript decoded as 1 line / 158 bytes -- the session header alone, i.e. a silent truncation that reads as "nothing said yet" forever. `zstdFrameRanges()` walks frame and block headers to find exact boundaries; splitting on the 4-byte magic would corrupt everything after a magic sequence occurring inside compressed data. zstd is resolved at RUNTIME because it landed in Node 22.15 while the project floor is 22.0, so an older Node keeps the pane behaviour. 2. Every turn also records a plugin-sourced `user/message` (the runtime context snapshot), which must not render as the user's own words. 3. A turn that ends in an error carries the provider's message; it is surfaced as `Turn error: …` (and a non-error early stop as `Turn ended: …`) rather than as an empty string, which an agent reads as "still thinking" through fifteen polls. 4. Reply text is assembled per (turn, step): a finalized message wins and the streamed deltas fill in only for a step that never finalized, so a partial answer is readable mid-turn and never doubled. "Finalized" is tracked as a set of steps rather than as non-empty text, because a step whose whole reply was reasoning strips to '' at the `</think>` boundary and would otherwise resurrect the raw deltas in its place. Session-to-transcript pairing is by the transcript's own header `cwd` plus a boot window against the session's createdAt, never by reproducing dsh's directory mangling (already two forms on disk) and never by newest-mtime alone -- mtime alone handed a freshly spawned worker its predecessor's answer in the same case directory. An empty result still wins over the pane; only a Node that cannot decode zstd falls back to it.
This commit is contained in:
@@ -0,0 +1,361 @@
|
||||
/**
|
||||
* Reading a DeepSeek Harness session transcript.
|
||||
*
|
||||
* Two of these assertions exist because the obvious implementation was measured
|
||||
* to be wrong against real files:
|
||||
*
|
||||
* - dsh appends ONE ZSTD FRAME PER WRITE, and Node's `zlib` zstd decoder stops
|
||||
* at the first frame end. A 56-line transcript decoded as 1 line / 158 bytes,
|
||||
* which reads as "the worker never answered" rather than as an error. The
|
||||
* multi-frame fixtures below are the guard.
|
||||
* - a fresh worker in a case directory that had been used before answered its
|
||||
* first `last-response` with the PREVIOUS session's reply. Session-to-
|
||||
* transcript pairing is therefore its own describe block.
|
||||
*/
|
||||
import { describe, expect, it, beforeAll, afterAll } from 'vitest';
|
||||
import { mkdtemp, mkdir, writeFile, rm, utimes } from 'node:fs/promises';
|
||||
import { join } from 'node:path';
|
||||
import { tmpdir } from 'node:os';
|
||||
import * as zlib from 'node:zlib';
|
||||
|
||||
import {
|
||||
decodeZstdFrames,
|
||||
findDeepSeekTranscript,
|
||||
parseDeepSeekTranscript,
|
||||
readDeepSeekLastResponse,
|
||||
resolveDeepSeekHome,
|
||||
zstdFrameRanges,
|
||||
zstdSupported,
|
||||
} from '../src/deepseek-transcript.js';
|
||||
|
||||
const zstdCompressSync = (zlib as unknown as { zstdCompressSync?: (b: Buffer) => Buffer }).zstdCompressSync;
|
||||
|
||||
/** Compress each line into its own frame — exactly how dsh appends. */
|
||||
function framed(lines: string[]): Buffer {
|
||||
if (!zstdCompressSync) throw new Error('zstd unavailable');
|
||||
return Buffer.concat(lines.map((line) => zstdCompressSync(Buffer.from(`${line}\n`, 'utf8'))));
|
||||
}
|
||||
|
||||
const sessionHeader = (cwd: string, createdAt: number, id = 'aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee') =>
|
||||
JSON.stringify({ type: 'session', version: 0, id, createdAt, cwd, delegationDepth: 0 });
|
||||
|
||||
const userPrompt = (text: string) =>
|
||||
JSON.stringify({
|
||||
type: 'user/message',
|
||||
data: { content: [{ type: 'text', text }], source: { kind: 'user' }, role: 'user' },
|
||||
});
|
||||
|
||||
const pluginContext = (text: string) =>
|
||||
JSON.stringify({
|
||||
type: 'user/message',
|
||||
data: { content: [{ type: 'text', text }], source: { kind: 'plugin', plugin: '@deepseek-ai/dsh-system-prompt' } },
|
||||
});
|
||||
|
||||
const assistantMessage = (text: string, turn = 1, step = 1, time = 1_700_000_000_000) =>
|
||||
JSON.stringify({
|
||||
type: 'assistant/message',
|
||||
time,
|
||||
data: { turn, step, message: { role: 'assistant', content: [{ type: 'text', text }] } },
|
||||
});
|
||||
|
||||
const turnEnd = (turn: number, reason: Record<string, unknown>, time = 1_700_000_000_001) =>
|
||||
JSON.stringify({ type: 'turn/end', time, data: { turn, reason } });
|
||||
|
||||
describe.skipIf(!zstdSupported())('zstd frame walking', () => {
|
||||
it('decodes every frame, not just the first (the silent-truncation bug)', () => {
|
||||
const lines = Array.from({ length: 40 }, (_, i) => JSON.stringify({ type: 'noise', seq: i }));
|
||||
const buf = framed(lines);
|
||||
|
||||
// The one-shot decoder is what this module had to replace.
|
||||
const oneShot = (zlib as unknown as { zstdDecompressSync?: (b: Buffer) => Buffer }).zstdDecompressSync!(buf);
|
||||
expect(oneShot.toString('utf8').trim().split('\n')).toHaveLength(1);
|
||||
|
||||
expect(zstdFrameRanges(buf)).toHaveLength(40);
|
||||
expect(decodeZstdFrames(buf).trim().split('\n')).toHaveLength(40);
|
||||
});
|
||||
|
||||
it('round-trips a single-frame file', () => {
|
||||
const buf = framed(['{"type":"session"}']);
|
||||
expect(decodeZstdFrames(buf)).toBe('{"type":"session"}\n');
|
||||
});
|
||||
|
||||
it('passes an uncompressed transcript straight through', () => {
|
||||
const plain = Buffer.from('{"type":"session"}\n{"type":"turn/start"}\n', 'utf8');
|
||||
expect(decodeZstdFrames(plain)).toBe('{"type":"session"}\n{"type":"turn/start"}\n');
|
||||
});
|
||||
|
||||
it('keeps the whole frames before a torn tail instead of failing the read', () => {
|
||||
const buf = framed(['{"a":1}', '{"b":2}', '{"c":3}']);
|
||||
const torn = buf.subarray(0, buf.length - 4);
|
||||
const decoded = decodeZstdFrames(torn);
|
||||
expect(decoded).toContain('{"a":1}');
|
||||
expect(decoded).toContain('{"b":2}');
|
||||
expect(decoded).not.toContain('{"c":3}');
|
||||
});
|
||||
|
||||
it('refuses to walk a buffer that is not zstd', () => {
|
||||
expect(zstdFrameRanges(Buffer.from('not zstd at all', 'utf8'))).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseDeepSeekTranscript', () => {
|
||||
it('returns the last turn text and skips plugin-injected context', () => {
|
||||
const raw = [
|
||||
sessionHeader('/w', 1),
|
||||
userPrompt('what is 2+2?'),
|
||||
pluginContext('Current runtime context. This snapshot supersedes earlier snapshots.'),
|
||||
assistantMessage('4.'),
|
||||
turnEnd(1, { kind: 'completed' }),
|
||||
].join('\n');
|
||||
|
||||
const result = parseDeepSeekTranscript(raw, { blocks: true });
|
||||
expect(result.text).toBe('4.');
|
||||
expect(result.cwd).toBe('/w');
|
||||
expect(result.blocks.filter((b) => b.kind === 'prompt').map((b) => b.text)).toEqual(['what is 2+2?']);
|
||||
expect(result.blocks.some((b) => b.text.includes('runtime context'))).toBe(false);
|
||||
});
|
||||
|
||||
it('drops a leaked reasoning prefix at the closing tag', () => {
|
||||
const raw = [
|
||||
sessionHeader('/w', 1),
|
||||
assistantMessage('I should read the file first.</think>\n\nThe add function is wrong.'),
|
||||
turnEnd(1, { kind: 'completed' }),
|
||||
].join('\n');
|
||||
expect(parseDeepSeekTranscript(raw).text).toBe('The add function is wrong.');
|
||||
});
|
||||
|
||||
it('renders tool calls and tool results as tool blocks', () => {
|
||||
const raw = [
|
||||
sessionHeader('/w', 1),
|
||||
JSON.stringify({
|
||||
type: 'assistant/message',
|
||||
data: {
|
||||
turn: 1,
|
||||
step: 1,
|
||||
message: {
|
||||
role: 'assistant',
|
||||
content: [
|
||||
{ type: 'text', text: 'Reading it.' },
|
||||
{ type: 'tool-call', id: 'c1', name: 'read', arguments: '{"file_path":"calc.py"}' },
|
||||
],
|
||||
},
|
||||
},
|
||||
}),
|
||||
JSON.stringify({
|
||||
type: 'tool/result',
|
||||
data: {
|
||||
turn: 1,
|
||||
step: 1,
|
||||
message: {
|
||||
content: [{ type: 'tool-result', toolCallId: 'c1', content: [{ type: 'text', text: 'def add' }] }],
|
||||
},
|
||||
},
|
||||
}),
|
||||
assistantMessage('It subtracts instead of adding.', 1, 2),
|
||||
turnEnd(1, { kind: 'completed' }),
|
||||
].join('\n');
|
||||
|
||||
const result = parseDeepSeekTranscript(raw, { blocks: true });
|
||||
expect(result.blocks.filter((b) => b.kind === 'tool').map((b) => b.text)).toEqual([
|
||||
'read({"file_path":"calc.py"})',
|
||||
'def add',
|
||||
]);
|
||||
// Both steps of the turn read back, in order.
|
||||
expect(result.text).toBe('Reading it.\n\nIt subtracts instead of adding.');
|
||||
});
|
||||
|
||||
it('uses streamed deltas only for a step the model never finalized', () => {
|
||||
const raw = [
|
||||
sessionHeader('/w', 1),
|
||||
JSON.stringify({
|
||||
type: 'assistant/chunk',
|
||||
data: { turn: 1, step: 1, chunk: { type: 'text-delta', index: 0, text: 'Par' } },
|
||||
}),
|
||||
JSON.stringify({
|
||||
type: 'text-chunks',
|
||||
seq: null,
|
||||
data: { turn: 1, step: 1, index: 0, texts: ['is is ', 'the'] },
|
||||
}),
|
||||
JSON.stringify({
|
||||
type: 'assistant/chunk',
|
||||
data: { turn: 1, step: 1, chunk: { type: 'text-delta', index: 0, text: ' capital.' } },
|
||||
}),
|
||||
].join('\n');
|
||||
// Still streaming: the partial answer is readable.
|
||||
expect(parseDeepSeekTranscript(raw).text).toBe('Paris is the capital.');
|
||||
|
||||
// Once finalized, the deltas must not be appended a second time.
|
||||
const finalized = `${raw}\n${assistantMessage('Paris is the capital.')}\n${turnEnd(1, { kind: 'completed' })}`;
|
||||
expect(parseDeepSeekTranscript(finalized).text).toBe('Paris is the capital.');
|
||||
});
|
||||
|
||||
it('does not resurrect raw deltas for a step whose reply was all reasoning', () => {
|
||||
// Measured on a real conversation: step 1 finalized as reasoning only, so
|
||||
// its text stripped to '' and the (unstripped) deltas took its place,
|
||||
// putting `</think>` and the monologue back in front of the caller.
|
||||
const raw = [
|
||||
sessionHeader('/w', 1),
|
||||
JSON.stringify({
|
||||
type: 'assistant/chunk',
|
||||
data: { turn: 1, step: 1, chunk: { type: 'text-delta', index: 0, text: "I'll read the file.</think>\n\n" } },
|
||||
}),
|
||||
assistantMessage("I'll read the file.</think>\n\n", 1, 1),
|
||||
assistantMessage('The add function is wrong.', 1, 2),
|
||||
turnEnd(1, { kind: 'completed' }),
|
||||
].join('\n');
|
||||
expect(parseDeepSeekTranscript(raw).text).toBe('The add function is wrong.');
|
||||
});
|
||||
|
||||
it('answers a failed turn with its error rather than the previous turn text', () => {
|
||||
const raw = [
|
||||
sessionHeader('/w', 1),
|
||||
assistantMessage('First answer.', 1),
|
||||
turnEnd(1, { kind: 'completed' }),
|
||||
turnEnd(2, { kind: 'error', error: { message: '400: model does not support tools', code: 'INVALID_REQUEST' } }),
|
||||
].join('\n');
|
||||
expect(parseDeepSeekTranscript(raw).text).toBe('Turn error: 400: model does not support tools');
|
||||
});
|
||||
|
||||
it('calls an early stop an ending, not an error, and keeps the text it did produce', () => {
|
||||
const raw = [sessionHeader('/w', 1), assistantMessage('Most'), turnEnd(1, { kind: 'max-tokens' })].join('\n');
|
||||
const result = parseDeepSeekTranscript(raw, { blocks: true });
|
||||
expect(result.text).toBe('Most');
|
||||
expect(result.blocks.map((b) => b.text)).toContain('Turn ended: max-tokens');
|
||||
});
|
||||
|
||||
it('survives a torn last line', () => {
|
||||
const raw = [sessionHeader('/w', 1), assistantMessage('Complete.'), '{"type":"turn/e'].join('\n');
|
||||
expect(parseDeepSeekTranscript(raw).text).toBe('Complete.');
|
||||
});
|
||||
|
||||
it('is empty for a session that has said nothing', () => {
|
||||
expect(parseDeepSeekTranscript(sessionHeader('/w', 1)).text).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
describe('resolveDeepSeekHome', () => {
|
||||
it('prefers the session override over the environment', () => {
|
||||
const previous = process.env.DSH_HOME;
|
||||
process.env.DSH_HOME = '/from-env';
|
||||
try {
|
||||
expect(resolveDeepSeekHome({ deepSeekHomeOverride: '/from-session' })).toBe('/from-session');
|
||||
expect(resolveDeepSeekHome({})).toBe('/from-env');
|
||||
} finally {
|
||||
if (previous === undefined) delete process.env.DSH_HOME;
|
||||
else process.env.DSH_HOME = previous;
|
||||
}
|
||||
});
|
||||
|
||||
it('falls back to ~/.dsh', () => {
|
||||
const previous = process.env.DSH_HOME;
|
||||
delete process.env.DSH_HOME;
|
||||
try {
|
||||
expect(resolveDeepSeekHome({})).toMatch(/\.dsh$/);
|
||||
} finally {
|
||||
if (previous !== undefined) process.env.DSH_HOME = previous;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe.skipIf(!zstdSupported())('session-to-transcript pairing', () => {
|
||||
let dshHome: string;
|
||||
const workspace = '/home/tester/cases/worker-1';
|
||||
const sessionStart = 1_800_000_000_000;
|
||||
|
||||
/** Write a transcript for `cwd`, created at `createdAt`, mtime `mtime`. */
|
||||
async function writeTranscript(name: string, cwd: string, createdAt: number, mtime: number, answer?: string) {
|
||||
const dir = join(dshHome, 'sessions', '--home-tester-cases-worker-1--', name);
|
||||
await mkdir(dir, { recursive: true });
|
||||
const lines = [sessionHeader(cwd, createdAt, name)];
|
||||
if (answer) lines.push(assistantMessage(answer), turnEnd(1, { kind: 'completed' }));
|
||||
const path = join(dir, 'session.jsonl.zstd');
|
||||
await writeFile(path, framed(lines));
|
||||
await utimes(path, new Date(mtime), new Date(mtime));
|
||||
return path;
|
||||
}
|
||||
|
||||
beforeAll(async () => {
|
||||
dshHome = await mkdtemp(join(tmpdir(), 'dsh-home-'));
|
||||
});
|
||||
afterAll(async () => {
|
||||
await rm(dshHome, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("never hands a fresh session its predecessor's answer", async () => {
|
||||
await writeTranscript('older', workspace, sessionStart - 600_000, sessionStart - 590_000, 'stale answer');
|
||||
const found = await findDeepSeekTranscript({ dshHome, workingDir: workspace, startedAt: sessionStart });
|
||||
expect(found).toBeNull();
|
||||
|
||||
const result = await readDeepSeekLastResponse({
|
||||
workingDir: workspace,
|
||||
createdAt: sessionStart,
|
||||
deepSeekHomeOverride: dshHome,
|
||||
});
|
||||
expect(result).not.toBeNull();
|
||||
expect(result?.text).toBe('');
|
||||
});
|
||||
|
||||
it('pairs on the boot window even when a sibling wrote more recently', async () => {
|
||||
await writeTranscript('mine', workspace, sessionStart + 2_000, sessionStart + 2_000, 'my answer');
|
||||
await writeTranscript('sibling', workspace, sessionStart + 300_000, sessionStart + 400_000, 'sibling answer');
|
||||
|
||||
const found = await findDeepSeekTranscript({ dshHome, workingDir: workspace, startedAt: sessionStart });
|
||||
expect(found).toContain('/mine/');
|
||||
|
||||
const result = await readDeepSeekLastResponse({
|
||||
workingDir: workspace,
|
||||
createdAt: new Date(sessionStart),
|
||||
deepSeekHomeOverride: dshHome,
|
||||
});
|
||||
expect(result?.text).toBe('my answer');
|
||||
});
|
||||
|
||||
it('ignores a transcript recorded for another workspace', async () => {
|
||||
const other = await mkdtemp(join(tmpdir(), 'dsh-home-'));
|
||||
try {
|
||||
const dir = join(other, 'sessions', '--home-tester-cases-worker-1--', 'foreign');
|
||||
await mkdir(dir, { recursive: true });
|
||||
await writeFile(
|
||||
join(dir, 'session.jsonl.zstd'),
|
||||
framed([sessionHeader('/somewhere/else', sessionStart + 1_000), assistantMessage('not yours')])
|
||||
);
|
||||
const found = await findDeepSeekTranscript({ dshHome: other, workingDir: workspace, startedAt: sessionStart });
|
||||
expect(found).toBeNull();
|
||||
} finally {
|
||||
await rm(other, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('reads a transcript created later in the session (a /new conversation)', async () => {
|
||||
const later = await mkdtemp(join(tmpdir(), 'dsh-home-'));
|
||||
try {
|
||||
const dir = join(later, 'sessions', '--home-tester-cases-worker-1--', 'after-new');
|
||||
await mkdir(dir, { recursive: true });
|
||||
await writeFile(
|
||||
join(dir, 'session.jsonl.zstd'),
|
||||
framed([
|
||||
sessionHeader(workspace, sessionStart + 1_800_000),
|
||||
assistantMessage('after /new'),
|
||||
turnEnd(1, { kind: 'completed' }),
|
||||
])
|
||||
);
|
||||
const result = await readDeepSeekLastResponse({
|
||||
workingDir: workspace,
|
||||
createdAt: sessionStart,
|
||||
deepSeekHomeOverride: later,
|
||||
});
|
||||
expect(result?.text).toBe('after /new');
|
||||
} finally {
|
||||
await rm(later, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('reports an unreadable home as empty, not as an error', async () => {
|
||||
const result = await readDeepSeekLastResponse({
|
||||
workingDir: workspace,
|
||||
createdAt: sessionStart,
|
||||
deepSeekHomeOverride: join(tmpdir(), 'dsh-home-that-does-not-exist'),
|
||||
});
|
||||
expect(result).toEqual({ text: '', timestamp: '', blocks: [] });
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user