harden: scope entrypoint detection to message lines, add fallback coverage

Two follow-ups from reviewing the entrypoint-filter and head-buffer
fixes before submitting them upstream:

1. extractTranscriptEntrypoint() scanned any line containing the
   substring "entrypoint", not specifically the first "type":"user"/
   "type":"assistant" message line (unlike its sibling
   extractFirstUserPrompt, which does scope to type). A transcript
   that started under an older Claude Code version (no entrypoint
   field) and got resumed under a newer one mid-conversation could
   pick up the field from a much later message than the true first
   one, misattributing the session's origin. Scoped it to match.

2. Added a regression test proving the tail-read fallback still
   engages correctly when bookkeeping accumulation exceeds even the
   new 128KB head window, not just the 16KB it previously blanked at.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
timkjr
2026-08-05 11:11:18 -05:00
co-authored by Claude Sonnet 5
parent 18b473f0e4
commit 251706be3b
2 changed files with 87 additions and 1 deletions
+11 -1
View File
@@ -2462,6 +2462,13 @@ export function registerSessionRoutes(
* an SDK/automated invocation. Used to exclude non-interactive transcripts
* (CI review bots, etc.) from the resumable history list — they were never
* something a user can resume into.
*
* Scoped to `"type":"user"`/`"type":"assistant"` lines specifically, mirroring
* `extractFirstUserPrompt`'s type check, rather than any line that happens to
* contain the substring "entrypoint". A transcript that started under an older
* Claude Code version (no entrypoint field) and got resumed under a newer one
* mid-conversation could otherwise pick up the field from a much later message
* than the true first one, misattributing the session's origin.
*/
function extractTranscriptEntrypoint(text: string): string | undefined {
let start = 0;
@@ -2469,10 +2476,13 @@ export function registerSessionRoutes(
const end = text.indexOf('\n', start);
const line = end === -1 ? text.slice(start) : text.slice(start, end);
start = end === -1 ? text.length : end + 1;
if (!line.includes('"type":"user"') && !line.includes('"type":"assistant"')) continue;
if (!line.includes('"entrypoint"')) continue;
try {
const entry = JSON.parse(line);
if (typeof entry.entrypoint === 'string') return entry.entrypoint;
if ((entry.type === 'user' || entry.type === 'assistant') && typeof entry.entrypoint === 'string') {
return entry.entrypoint;
}
} catch {
// Malformed/truncated line — skip
}
+76
View File
@@ -1396,6 +1396,44 @@ describe('session-routes', () => {
expect(ids).not.toContain(sdkId);
});
it('attributes entrypoint from the true first message, not a later one or a bookkeeping line', async () => {
// A transcript that started under an older Claude Code version (no
// entrypoint field) and got resumed under a newer one mid-conversation
// could otherwise pick up entrypoint from a later message, misattributing
// the session's origin. Scanning must anchor on "type":"user"/"assistant"
// lines specifically, not any line that happens to mention "entrypoint".
const home = process.env.HOME as string;
const projPath = join(home, '.claude', 'projects', 'proj-entrypoint-scoping-test');
await mkdir(projPath, { recursive: true });
const sessionId = '77777777-7777-7777-7777-777777777777';
// True first message: no entrypoint field (old-version transcript).
const firstLine =
JSON.stringify({ type: 'user', message: { role: 'user', content: 'the genuine first message' } }) + '\n';
// A bookkeeping line that (hypothetically) mentions entrypoint outside a
// real message record — must not be mistaken for message metadata.
const bookkeepingLine = JSON.stringify({ type: 'mode', mode: 'normal', entrypoint: 'sdk-py' }) + '\n';
// A later message, after the resume, that DOES carry entrypoint: 'cli' —
// this is what the (fixed) scan should find, since it's the first
// user/assistant line that actually carries the field.
const laterLine =
JSON.stringify({ type: 'user', entrypoint: 'cli', message: { role: 'user', content: 'a later message' } }) +
'\n';
// scanProjectDir skips files under 4000 bytes.
const body = firstLine + bookkeepingLine + laterLine;
await writeFile(join(projPath, `${sessionId}.jsonl`), body + '#'.repeat(4200 - body.length));
const res = await harness.app.inject({
method: 'GET',
url: '/api/history/sessions?projectKey=proj-entrypoint-scoping-test',
});
expect(res.statusCode).toBe(200);
const ids = JSON.parse(res.body).data.sessions.map((s: { sessionId: string }) => s.sessionId);
// entrypoint: 'cli' (from the later message) — shown, not excluded.
expect(ids).toContain(sessionId);
});
it('finds the real first prompt past a large run of pre-message bookkeeping lines', async () => {
// A session restarted many times over a long conversation accumulates a batch
// of small bookkeeping lines (mode/permission-mode/last-prompt/queue-operation)
@@ -1429,6 +1467,44 @@ describe('session-routes', () => {
expect(row).toBeDefined();
expect(row.firstPrompt).toBe('the real first message');
});
it('still falls back to the tail read when bookkeeping alone exceeds the new 128KB head window', async () => {
// Raising the head buffer to 128KB helps most restart-heavy sessions, but an
// even more extreme case (many more restarts) can still exceed it. The
// existing tail-read fallback must stay correctly wired to the new
// threshold (headBuf.length, not the old hardcoded 65536) rather than being
// silently skipped because the size comparison no longer means what it used
// to. The real message here sits near the end of the file, well inside the
// 32KB tail window, so a working fallback finds it; a broken one leaves the
// row blank exactly like the bug this whole fix addresses.
const home = process.env.HOME as string;
const projPath = join(home, '.claude', 'projects', 'proj-tail-fallback-test');
await mkdir(projPath, { recursive: true });
const sessionId = '88888888-8888-8888-8888-888888888888';
const bookkeepingLine = JSON.stringify({ type: 'mode', mode: 'normal', sessionId }) + '\n';
// Comfortably past the new 128KB head window (was 16KB), so the head read
// never reaches a single "type":"user"/"assistant"/"summary" line.
const prefix = bookkeepingLine.repeat(Math.ceil(140000 / bookkeepingLine.length));
const realLine =
JSON.stringify({
type: 'user',
entrypoint: 'cli',
message: { role: 'user', content: 'found via tail fallback' },
}) + '\n';
expect(prefix.length).toBeGreaterThan(131072);
await writeFile(join(projPath, `${sessionId}.jsonl`), prefix + realLine);
const res = await harness.app.inject({
method: 'GET',
url: '/api/history/sessions?projectKey=proj-tail-fallback-test',
});
expect(res.statusCode).toBe(200);
const row = JSON.parse(res.body).data.sessions.find((s: { sessionId: string }) => s.sessionId === sessionId);
expect(row).toBeDefined();
expect(row.firstPrompt).toBe('found via tail fallback');
});
});
// ========== POST /api/sessions (with resumeSessionId) ==========