COD-9 add cross-session search backend (GET /api/search) v1

Bounded federated search over in-memory stores (sessions/cases, run-summary
events, file paths). Zod-validated query (q 1-200 chars, types csv, limit 1-60),
grouped session->event->file with exact-match-first + recency tiebreak, total
cap 60 + per-group cap 25, snippet cap 200, path-safety (relativePath only).
Frontend search box (history panel) deferred to next cycle; resume/history-prompt
text matching deferred to v1.1 (lives in large on-disk files, out of v1 bounded scope).

New: src/search-service.ts (pure core), src/types/search.ts, src/web/routes/search-routes.ts.
Tests: test/search-service.test.ts (14), test/routes/search-routes.test.ts (10).
This commit is contained in:
Aamer Akhter
2026-06-19 12:35:16 -04:00
parent 1255e28f6f
commit 95df96e06a
9 changed files with 925 additions and 0 deletions
+207
View File
@@ -0,0 +1,207 @@
/**
* @fileoverview Pure cross-session federated search core (COD-9).
*
* `searchSources()` is the testable heart of `GET /api/search`: it takes a
* normalized query plus already-collected, in-memory source data and returns
* grouped, ranked, and capped results. It performs NO I/O — the route wrapper
* (`src/web/routes/search-routes.ts`) is responsible for harvesting the source
* arrays from the live server stores (sessions, run-summary trackers, attachment
* histories) in a bounded way before calling this.
*
* v1 scope (do not expand here): three sources — sessions/cases, run-summary
* events, file paths. Terminal-buffer scanning and any persisted index are
* explicitly deferred.
*
* Ranking: results are grouped by source type in the fixed order
* sessions → events → files. Within each group, exact (case-insensitive)
* name/path matches come first, then recency (newest timestamp first) as the
* tiebreak. There is no relevance-scoring pass in v1.
*
* Safety: file results only ever expose a workspace-relative path — server-
* private absolute paths are never placed in a result. Per-group and total caps
* bound the output so a broad query cannot return an unbounded payload.
*
* Key exports:
* - searchSources() — the pure core.
* - SEARCH_TOTAL_CAP / SEARCH_PER_GROUP_CAP — the output bounds.
* - SearchSources and the *Input row types — the source-data contract.
*/
import type { SearchResult, SearchResultGroup, SearchResponseData, SearchSourceType } from './types/search.js';
/** Maximum results returned across all groups combined. */
export const SEARCH_TOTAL_CAP = 60;
/** Maximum results returned within any single source group. */
export const SEARCH_PER_GROUP_CAP = 25;
/** Maximum characters in a result snippet. */
export const SEARCH_SNIPPET_MAX = 200;
/** A live-session row harvested for the session/case source. */
export interface SessionSearchInput {
sessionId: string;
sessionName: string;
workingDir: string;
/** Recency timestamp (e.g. lastActivityAt or createdAt). */
timestamp: number;
}
/** A run-summary timeline event harvested for the event source. */
export interface EventSearchInput {
sessionId: string;
sessionName: string;
eventId: string;
title: string;
details: string;
timestamp: number;
}
/** A per-session attachment harvested for the file source. */
export interface FileSearchInput {
sessionId: string;
sessionName: string;
fileName: string;
/** Workspace-relative path, if known. Absolute/external paths are never passed in. */
relativePath: string | undefined;
timestamp: number;
/** Attachment history item id, used as the jump-to target. */
itemId: string;
}
/** The full set of in-memory source data the pure core searches over. */
export interface SearchSources {
sessions: SessionSearchInput[];
events: EventSearchInput[];
files: FileSearchInput[];
}
/** Fixed group/render order. */
const GROUP_ORDER: SearchSourceType[] = ['session', 'event', 'file'];
function truncate(text: string, max = SEARCH_SNIPPET_MAX): string {
const trimmed = text.trim().replace(/\s+/g, ' ');
return trimmed.length > max ? trimmed.slice(0, max - 1) + '…' : trimmed;
}
/**
* Sort a group's results: exact matches first, then newest timestamp first.
* Stable for equal keys.
*/
function sortGroup(rows: SearchResult[]): SearchResult[] {
return rows
.map((result, index) => ({ result, index }))
.sort((a, b) => {
if (a.result.exactMatch !== b.result.exactMatch) {
return a.result.exactMatch ? -1 : 1;
}
if (a.result.timestamp !== b.result.timestamp) {
return b.result.timestamp - a.result.timestamp;
}
return a.index - b.index;
})
.map((r) => r.result);
}
/**
* Search the provided in-memory sources for `query`.
*
* @param query Raw query string (already length-validated by the route). Blank
* queries return an empty result set.
* @param sources Harvested, bounded source arrays.
*/
export function searchSources(query: string, sources: SearchSources): SearchResponseData {
const needle = query.trim().toLowerCase();
if (needle.length === 0) {
return { query: query.trim(), groups: [], totalResults: 0, truncated: false };
}
const contains = (s: string | undefined): boolean => !!s && s.toLowerCase().includes(needle);
const isExact = (s: string | undefined): boolean => !!s && s.toLowerCase() === needle;
// -- Source: sessions/cases --
const sessionRows: SearchResult[] = [];
for (const s of sources.sessions) {
if (contains(s.sessionName) || contains(s.workingDir) || contains(s.sessionId)) {
sessionRows.push({
type: 'session',
sessionId: s.sessionId,
sessionName: s.sessionName,
timestamp: s.timestamp,
snippet: truncate(s.workingDir ? `${s.sessionName} — ${s.workingDir}` : s.sessionName),
exactMatch: isExact(s.sessionName),
jumpTo: { kind: 'session', sessionId: s.sessionId },
});
}
}
// -- Source: run-summary events --
const eventRows: SearchResult[] = [];
for (const e of sources.events) {
if (contains(e.title) || contains(e.details)) {
const snippetBase = e.details && contains(e.details) ? `${e.title}: ${e.details}` : e.title;
eventRows.push({
type: 'event',
sessionId: e.sessionId,
sessionName: e.sessionName,
timestamp: e.timestamp,
snippet: truncate(snippetBase),
exactMatch: isExact(e.title),
jumpTo: { kind: 'run-summary', sessionId: e.sessionId, targetId: e.eventId },
});
}
}
// -- Source: file paths --
const fileRows: SearchResult[] = [];
for (const f of sources.files) {
if (contains(f.fileName) || contains(f.relativePath)) {
fileRows.push({
type: 'file',
sessionId: f.sessionId,
sessionName: f.sessionName,
timestamp: f.timestamp,
snippet: truncate(f.relativePath ?? f.fileName),
// Exact match keys off the safe path (or filename) — never an absolute path.
exactMatch: isExact(f.relativePath) || isExact(f.fileName),
jumpTo: {
kind: 'file-preview',
sessionId: f.sessionId,
targetId: f.itemId,
// Only ever expose a relative path; absolute/external paths are not passed in.
relativePath: f.relativePath,
},
});
}
}
const byType: Record<SearchSourceType, SearchResult[]> = {
session: sortGroup(sessionRows),
event: sortGroup(eventRows),
file: sortGroup(fileRows),
};
const groups: SearchResultGroup[] = [];
let total = 0;
let truncated = false;
for (const type of GROUP_ORDER) {
const all = byType[type];
if (all.length === 0) continue;
// Per-group cap.
let capped = all.slice(0, SEARCH_PER_GROUP_CAP);
if (all.length > capped.length) truncated = true;
// Total cap (never exceed the global budget).
const remaining = SEARCH_TOTAL_CAP - total;
if (capped.length > remaining) {
capped = capped.slice(0, Math.max(0, remaining));
truncated = true;
}
if (capped.length === 0) continue;
groups.push({ type, results: capped });
total += capped.length;
}
return { query: query.trim(), groups, totalResults: total, truncated };
}
+1
View File
@@ -68,3 +68,4 @@ export * from './plan.js';
export * from './orchestrator.js';
export * from './update.js';
export * from './workflow-run.js';
export * from './search.js';
+77
View File
@@ -0,0 +1,77 @@
/**
* @fileoverview Cross-session federated search types (COD-9).
*
* Defines the typed shapes for `GET /api/search` — a bounded, in-memory
* federated search across three v1 sources: live sessions/cases, run-summary
* timeline events, and per-session attachment file paths. Terminal-buffer scans
* and any persisted index are explicitly out of scope for v1.
*
* Key exports:
* - SearchSourceType — the federated source kinds, also the group order key.
* - SearchResult — a single typed result card (source, session id/name,
* timestamp, snippet, jump-to action target).
* - SearchJumpTarget — where the frontend should navigate when a card is opened.
* - SearchResponseData — grouped result payload returned in the ApiResponse envelope.
*
* No I/O, no dependencies on other domain modules. The pure search core lives
* in `src/search-service.ts`; the route wrapper in `src/web/routes/search-routes.ts`.
*/
/** Federated source kinds. Group/render order is sessions → events → files. */
export type SearchSourceType = 'session' | 'event' | 'file';
/** Where the frontend should jump when a result card is activated. */
export interface SearchJumpTarget {
/** Kind of navigation target. */
kind: 'session' | 'run-summary' | 'file-preview';
/** Owning Codeman session id (always present — every result is session-scoped). */
sessionId: string;
/**
* Secondary identifier for the target:
* - kind 'run-summary': the run-summary event id
* - kind 'file-preview': the attachment history item id
* - kind 'session': undefined (the sessionId is sufficient)
*/
targetId?: string;
/**
* Workspace-relative path for file-preview targets. Never an absolute path —
* server-private external paths are intentionally omitted to avoid leakage.
*/
relativePath?: string;
}
/** A single typed search result card. */
export interface SearchResult {
/** Which federated source produced this result. */
type: SearchSourceType;
/** Owning Codeman session id. */
sessionId: string;
/** Display name of the owning session / case. */
sessionName: string;
/** Millisecond timestamp used for recency ranking and display. */
timestamp: number;
/** Short, already-truncated snippet describing the match. */
snippet: string;
/** True when the query matched the primary name/path exactly (case-insensitive). */
exactMatch: boolean;
/** Navigation target for the jump-to action. */
jumpTo: SearchJumpTarget;
}
/** A group of results for one source type, in render order. */
export interface SearchResultGroup {
type: SearchSourceType;
results: SearchResult[];
}
/** Payload returned as `data` inside the standard ApiResponse envelope. */
export interface SearchResponseData {
/** The normalized query that was executed. */
query: string;
/** Results grouped by source type, ordered sessions → events → files. */
groups: SearchResultGroup[];
/** Total number of results across all groups (after caps applied). */
totalResults: number;
/** True if any group or the total was capped (more matches existed). */
truncated: boolean;
}
+1
View File
@@ -17,4 +17,5 @@ export { registerRalphRoutes } from './ralph-routes.js';
export { registerPlanRoutes } from './plan-routes.js';
export { registerOrchestratorRoutes } from './orchestrator-routes.js';
export { registerClipboardRoutes } from './clipboard-routes.js';
export { registerSearchRoutes } from './search-routes.js';
export { registerWsRoutes } from './ws-routes.js';
+162
View File
@@ -0,0 +1,162 @@
/**
* @fileoverview Cross-session federated search route (COD-9).
*
* Registers `GET /api/search?q=&types=&limit=` — a bounded, in-memory search
* across three v1 sources, returned in the standard ApiResponse envelope:
* 1. sessions/cases — name, working directory, session id
* 2. run-summary events — event title/details (from the live run-summary trackers)
* 3. file paths — per-session attachment history (workspace-relative paths only)
*
* This route is a THIN wrapper: it harvests the source arrays from the live
* server stores (held on the route context) in a bounded way, then delegates
* grouping/ranking/capping to the pure `searchSources()` core in
* `src/search-service.ts`. Terminal-buffer scanning and any persisted index are
* out of scope for v1.
*
* Safety: query input is Zod-validated (length-bounded `q`, allowlisted `types`,
* numeric `limit`); only workspace-relative file paths are ever exposed (the
* server-private `externalPath` on attachment history is never read here); and
* the pure core enforces a per-group and total result cap so a broad query
* cannot return an unbounded payload. No terminal output is read.
*
* Endpoints: GET /api/search
*/
import { FastifyInstance } from 'fastify';
import { parseBody } from '../route-helpers.js';
import { SearchQuerySchema } from '../schemas.js';
import {
searchSources,
type SearchSources,
type SessionSearchInput,
type EventSearchInput,
type FileSearchInput,
} from '../../search-service.js';
import type { SearchSourceType } from '../../types/search.js';
import type { SessionPort, InfraPort } from '../ports/index.js';
/**
* Per-source harvest caps. These bound how much in-memory data we hand to the
* pure core BEFORE it applies its own result caps — they keep the harvest itself
* cheap on large deployments (e.g. 50 sessions × many events). They are
* deliberately well above the result caps so ranking still sees enough candidates.
*/
const MAX_EVENTS_PER_SESSION = 500;
interface SessionLike {
id: string;
name: string;
workingDir: string;
lastActivityAt?: number;
createdAt?: number;
attachmentHistory?: Array<{
id: string;
fileName: string;
relativePath?: string;
timestamp?: number;
mtimeMs?: number;
}>;
}
/**
* Harvest the three source arrays from the live in-memory stores. Reads only
* bounded, already-loaded data — no disk I/O, no terminal buffers.
*/
function harvestSources(ctx: SessionPort & InfraPort): SearchSources {
const sessions: SessionSearchInput[] = [];
const events: EventSearchInput[] = [];
const files: FileSearchInput[] = [];
for (const raw of ctx.sessions.values()) {
const s = raw as unknown as SessionLike;
const sessionName = s.name ?? '';
const timestamp = s.lastActivityAt ?? s.createdAt ?? 0;
sessions.push({
sessionId: s.id,
sessionName,
workingDir: s.workingDir ?? '',
timestamp,
});
// Files: per-session attachment history. Only the workspace-relative path is
// surfaced; the server-private externalPath is intentionally never read.
const history = s.attachmentHistory ?? [];
for (const item of history) {
files.push({
sessionId: s.id,
sessionName,
fileName: item.fileName,
relativePath: item.relativePath,
timestamp: item.timestamp ?? item.mtimeMs ?? timestamp,
itemId: item.id,
});
}
}
// Events: from the live run-summary trackers, keyed by session id.
for (const [sessionId, tracker] of ctx.runSummaryTrackers) {
const session = ctx.sessions.get(sessionId) as unknown as SessionLike | undefined;
const sessionName = session?.name ?? '';
const summary = tracker.getSummary();
// Newest events are most relevant; cap the per-session harvest.
const evts = summary.events.slice(-MAX_EVENTS_PER_SESSION);
for (const e of evts) {
events.push({
sessionId,
sessionName,
eventId: e.id,
title: e.title,
details: e.details ?? '',
timestamp: e.timestamp,
});
}
}
return { sessions, events, files };
}
export function registerSearchRoutes(app: FastifyInstance, ctx: SessionPort & InfraPort): void {
app.get('/api/search', async (req) => {
// Zod-validate the query. parseBody throws a structured 400 on failure.
const { q, types, limit } = parseBody(SearchQuerySchema, req.query);
const allowed: Set<SearchSourceType> | null = types
? new Set(
types
.split(',')
.map((t) => t.trim())
.filter(Boolean) as SearchSourceType[]
)
: null;
const sources = harvestSources(ctx);
// Apply the optional source-type filter before searching so excluded
// sources never contribute to (or consume budget in) the result set.
const filtered: SearchSources = {
sessions: !allowed || allowed.has('session') ? sources.sessions : [],
events: !allowed || allowed.has('event') ? sources.events : [],
files: !allowed || allowed.has('file') ? sources.files : [],
};
const result = searchSources(q, filtered);
// Optional caller-supplied total cap (always on top of the core's hard caps).
if (limit !== undefined && result.totalResults > limit) {
let remaining = limit;
const cappedGroups = [];
for (const group of result.groups) {
if (remaining <= 0) break;
const slice = group.results.slice(0, remaining);
remaining -= slice.length;
cappedGroups.push({ type: group.type, results: slice });
}
result.groups = cappedGroups;
result.totalResults = limit;
result.truncated = true;
}
return { success: true, data: result };
});
}
+32
View File
@@ -708,3 +708,35 @@ export const OrchestratorStartSchema = z.object({
export const OrchestratorRejectSchema = z.object({
feedback: z.string().min(1).max(10000),
});
// ========== Cross-Session Search (COD-9) ==========
/** Valid federated source kinds for `GET /api/search?types=`. */
export const SEARCH_SOURCE_TYPES = ['session', 'event', 'file'] as const;
/**
* GET /api/search query validation.
*
* Query params arrive as strings: `q` is bounded (1..200 chars), `types` is an
* optional comma-separated allowlisted CSV, and `limit` is an optional coerced
* integer clamped to 1..60. Validation is the first line of defense — a missing
* or oversized `q`, an unknown type, or a non-numeric limit is rejected with 400.
*/
export const SearchQuerySchema = z.object({
q: z.string().trim().min(1, 'Query is required').max(200, 'Query too long (max 200 chars)'),
types: z
.string()
.max(100)
.optional()
.refine(
(v) =>
v === undefined ||
v
.split(',')
.map((t) => t.trim())
.filter(Boolean)
.every((t) => (SEARCH_SOURCE_TYPES as readonly string[]).includes(t)),
{ message: 'Invalid types value' }
),
limit: z.coerce.number().int().min(1).max(60).optional(),
});
+2
View File
@@ -149,6 +149,7 @@ import {
registerRalphRoutes,
registerPlanRoutes,
registerClipboardRoutes,
registerSearchRoutes,
registerOrchestratorRoutes,
registerWsRoutes,
} from './routes/index.js';
@@ -869,6 +870,7 @@ export class WebServer extends EventEmitter {
registerRalphRoutes(this.app, ctx);
registerPlanRoutes(this.app, ctx);
registerClipboardRoutes(this.app, ctx);
registerSearchRoutes(this.app, ctx);
registerOrchestratorRoutes(this.app, ctx);
registerWsRoutes(this.app, ctx, () => this.getHostPolicy());
}