mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-02 13:39:41 +02:00
COD-9 add cross-session search backend (GET /api/search) v1
Bounded federated search over in-memory stores (sessions/cases, run-summary events, file paths). Zod-validated query (q 1-200 chars, types csv, limit 1-60), grouped session->event->file with exact-match-first + recency tiebreak, total cap 60 + per-group cap 25, snippet cap 200, path-safety (relativePath only). Frontend search box (history panel) deferred to next cycle; resume/history-prompt text matching deferred to v1.1 (lives in large on-disk files, out of v1 bounded scope). New: src/search-service.ts (pure core), src/types/search.ts, src/web/routes/search-routes.ts. Tests: test/search-service.test.ts (14), test/routes/search-routes.test.ts (10).
This commit is contained in:
@@ -0,0 +1,207 @@
|
||||
/**
|
||||
* @fileoverview Pure cross-session federated search core (COD-9).
|
||||
*
|
||||
* `searchSources()` is the testable heart of `GET /api/search`: it takes a
|
||||
* normalized query plus already-collected, in-memory source data and returns
|
||||
* grouped, ranked, and capped results. It performs NO I/O — the route wrapper
|
||||
* (`src/web/routes/search-routes.ts`) is responsible for harvesting the source
|
||||
* arrays from the live server stores (sessions, run-summary trackers, attachment
|
||||
* histories) in a bounded way before calling this.
|
||||
*
|
||||
* v1 scope (do not expand here): three sources — sessions/cases, run-summary
|
||||
* events, file paths. Terminal-buffer scanning and any persisted index are
|
||||
* explicitly deferred.
|
||||
*
|
||||
* Ranking: results are grouped by source type in the fixed order
|
||||
* sessions → events → files. Within each group, exact (case-insensitive)
|
||||
* name/path matches come first, then recency (newest timestamp first) as the
|
||||
* tiebreak. There is no relevance-scoring pass in v1.
|
||||
*
|
||||
* Safety: file results only ever expose a workspace-relative path — server-
|
||||
* private absolute paths are never placed in a result. Per-group and total caps
|
||||
* bound the output so a broad query cannot return an unbounded payload.
|
||||
*
|
||||
* Key exports:
|
||||
* - searchSources() — the pure core.
|
||||
* - SEARCH_TOTAL_CAP / SEARCH_PER_GROUP_CAP — the output bounds.
|
||||
* - SearchSources and the *Input row types — the source-data contract.
|
||||
*/
|
||||
|
||||
import type { SearchResult, SearchResultGroup, SearchResponseData, SearchSourceType } from './types/search.js';
|
||||
|
||||
/** Maximum results returned across all groups combined. */
|
||||
export const SEARCH_TOTAL_CAP = 60;
|
||||
/** Maximum results returned within any single source group. */
|
||||
export const SEARCH_PER_GROUP_CAP = 25;
|
||||
/** Maximum characters in a result snippet. */
|
||||
export const SEARCH_SNIPPET_MAX = 200;
|
||||
|
||||
/** A live-session row harvested for the session/case source. */
|
||||
export interface SessionSearchInput {
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
workingDir: string;
|
||||
/** Recency timestamp (e.g. lastActivityAt or createdAt). */
|
||||
timestamp: number;
|
||||
}
|
||||
|
||||
/** A run-summary timeline event harvested for the event source. */
|
||||
export interface EventSearchInput {
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
eventId: string;
|
||||
title: string;
|
||||
details: string;
|
||||
timestamp: number;
|
||||
}
|
||||
|
||||
/** A per-session attachment harvested for the file source. */
|
||||
export interface FileSearchInput {
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
fileName: string;
|
||||
/** Workspace-relative path, if known. Absolute/external paths are never passed in. */
|
||||
relativePath: string | undefined;
|
||||
timestamp: number;
|
||||
/** Attachment history item id, used as the jump-to target. */
|
||||
itemId: string;
|
||||
}
|
||||
|
||||
/** The full set of in-memory source data the pure core searches over. */
|
||||
export interface SearchSources {
|
||||
sessions: SessionSearchInput[];
|
||||
events: EventSearchInput[];
|
||||
files: FileSearchInput[];
|
||||
}
|
||||
|
||||
/** Fixed group/render order. */
|
||||
const GROUP_ORDER: SearchSourceType[] = ['session', 'event', 'file'];
|
||||
|
||||
function truncate(text: string, max = SEARCH_SNIPPET_MAX): string {
|
||||
const trimmed = text.trim().replace(/\s+/g, ' ');
|
||||
return trimmed.length > max ? trimmed.slice(0, max - 1) + '…' : trimmed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Sort a group's results: exact matches first, then newest timestamp first.
|
||||
* Stable for equal keys.
|
||||
*/
|
||||
function sortGroup(rows: SearchResult[]): SearchResult[] {
|
||||
return rows
|
||||
.map((result, index) => ({ result, index }))
|
||||
.sort((a, b) => {
|
||||
if (a.result.exactMatch !== b.result.exactMatch) {
|
||||
return a.result.exactMatch ? -1 : 1;
|
||||
}
|
||||
if (a.result.timestamp !== b.result.timestamp) {
|
||||
return b.result.timestamp - a.result.timestamp;
|
||||
}
|
||||
return a.index - b.index;
|
||||
})
|
||||
.map((r) => r.result);
|
||||
}
|
||||
|
||||
/**
|
||||
* Search the provided in-memory sources for `query`.
|
||||
*
|
||||
* @param query Raw query string (already length-validated by the route). Blank
|
||||
* queries return an empty result set.
|
||||
* @param sources Harvested, bounded source arrays.
|
||||
*/
|
||||
export function searchSources(query: string, sources: SearchSources): SearchResponseData {
|
||||
const needle = query.trim().toLowerCase();
|
||||
if (needle.length === 0) {
|
||||
return { query: query.trim(), groups: [], totalResults: 0, truncated: false };
|
||||
}
|
||||
|
||||
const contains = (s: string | undefined): boolean => !!s && s.toLowerCase().includes(needle);
|
||||
const isExact = (s: string | undefined): boolean => !!s && s.toLowerCase() === needle;
|
||||
|
||||
// -- Source: sessions/cases --
|
||||
const sessionRows: SearchResult[] = [];
|
||||
for (const s of sources.sessions) {
|
||||
if (contains(s.sessionName) || contains(s.workingDir) || contains(s.sessionId)) {
|
||||
sessionRows.push({
|
||||
type: 'session',
|
||||
sessionId: s.sessionId,
|
||||
sessionName: s.sessionName,
|
||||
timestamp: s.timestamp,
|
||||
snippet: truncate(s.workingDir ? `${s.sessionName} — ${s.workingDir}` : s.sessionName),
|
||||
exactMatch: isExact(s.sessionName),
|
||||
jumpTo: { kind: 'session', sessionId: s.sessionId },
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// -- Source: run-summary events --
|
||||
const eventRows: SearchResult[] = [];
|
||||
for (const e of sources.events) {
|
||||
if (contains(e.title) || contains(e.details)) {
|
||||
const snippetBase = e.details && contains(e.details) ? `${e.title}: ${e.details}` : e.title;
|
||||
eventRows.push({
|
||||
type: 'event',
|
||||
sessionId: e.sessionId,
|
||||
sessionName: e.sessionName,
|
||||
timestamp: e.timestamp,
|
||||
snippet: truncate(snippetBase),
|
||||
exactMatch: isExact(e.title),
|
||||
jumpTo: { kind: 'run-summary', sessionId: e.sessionId, targetId: e.eventId },
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// -- Source: file paths --
|
||||
const fileRows: SearchResult[] = [];
|
||||
for (const f of sources.files) {
|
||||
if (contains(f.fileName) || contains(f.relativePath)) {
|
||||
fileRows.push({
|
||||
type: 'file',
|
||||
sessionId: f.sessionId,
|
||||
sessionName: f.sessionName,
|
||||
timestamp: f.timestamp,
|
||||
snippet: truncate(f.relativePath ?? f.fileName),
|
||||
// Exact match keys off the safe path (or filename) — never an absolute path.
|
||||
exactMatch: isExact(f.relativePath) || isExact(f.fileName),
|
||||
jumpTo: {
|
||||
kind: 'file-preview',
|
||||
sessionId: f.sessionId,
|
||||
targetId: f.itemId,
|
||||
// Only ever expose a relative path; absolute/external paths are not passed in.
|
||||
relativePath: f.relativePath,
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const byType: Record<SearchSourceType, SearchResult[]> = {
|
||||
session: sortGroup(sessionRows),
|
||||
event: sortGroup(eventRows),
|
||||
file: sortGroup(fileRows),
|
||||
};
|
||||
|
||||
const groups: SearchResultGroup[] = [];
|
||||
let total = 0;
|
||||
let truncated = false;
|
||||
|
||||
for (const type of GROUP_ORDER) {
|
||||
const all = byType[type];
|
||||
if (all.length === 0) continue;
|
||||
|
||||
// Per-group cap.
|
||||
let capped = all.slice(0, SEARCH_PER_GROUP_CAP);
|
||||
if (all.length > capped.length) truncated = true;
|
||||
|
||||
// Total cap (never exceed the global budget).
|
||||
const remaining = SEARCH_TOTAL_CAP - total;
|
||||
if (capped.length > remaining) {
|
||||
capped = capped.slice(0, Math.max(0, remaining));
|
||||
truncated = true;
|
||||
}
|
||||
if (capped.length === 0) continue;
|
||||
|
||||
groups.push({ type, results: capped });
|
||||
total += capped.length;
|
||||
}
|
||||
|
||||
return { query: query.trim(), groups, totalResults: total, truncated };
|
||||
}
|
||||
@@ -68,3 +68,4 @@ export * from './plan.js';
|
||||
export * from './orchestrator.js';
|
||||
export * from './update.js';
|
||||
export * from './workflow-run.js';
|
||||
export * from './search.js';
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
/**
|
||||
* @fileoverview Cross-session federated search types (COD-9).
|
||||
*
|
||||
* Defines the typed shapes for `GET /api/search` — a bounded, in-memory
|
||||
* federated search across three v1 sources: live sessions/cases, run-summary
|
||||
* timeline events, and per-session attachment file paths. Terminal-buffer scans
|
||||
* and any persisted index are explicitly out of scope for v1.
|
||||
*
|
||||
* Key exports:
|
||||
* - SearchSourceType — the federated source kinds, also the group order key.
|
||||
* - SearchResult — a single typed result card (source, session id/name,
|
||||
* timestamp, snippet, jump-to action target).
|
||||
* - SearchJumpTarget — where the frontend should navigate when a card is opened.
|
||||
* - SearchResponseData — grouped result payload returned in the ApiResponse envelope.
|
||||
*
|
||||
* No I/O, no dependencies on other domain modules. The pure search core lives
|
||||
* in `src/search-service.ts`; the route wrapper in `src/web/routes/search-routes.ts`.
|
||||
*/
|
||||
|
||||
/** Federated source kinds. Group/render order is sessions → events → files. */
|
||||
export type SearchSourceType = 'session' | 'event' | 'file';
|
||||
|
||||
/** Where the frontend should jump when a result card is activated. */
|
||||
export interface SearchJumpTarget {
|
||||
/** Kind of navigation target. */
|
||||
kind: 'session' | 'run-summary' | 'file-preview';
|
||||
/** Owning Codeman session id (always present — every result is session-scoped). */
|
||||
sessionId: string;
|
||||
/**
|
||||
* Secondary identifier for the target:
|
||||
* - kind 'run-summary': the run-summary event id
|
||||
* - kind 'file-preview': the attachment history item id
|
||||
* - kind 'session': undefined (the sessionId is sufficient)
|
||||
*/
|
||||
targetId?: string;
|
||||
/**
|
||||
* Workspace-relative path for file-preview targets. Never an absolute path —
|
||||
* server-private external paths are intentionally omitted to avoid leakage.
|
||||
*/
|
||||
relativePath?: string;
|
||||
}
|
||||
|
||||
/** A single typed search result card. */
|
||||
export interface SearchResult {
|
||||
/** Which federated source produced this result. */
|
||||
type: SearchSourceType;
|
||||
/** Owning Codeman session id. */
|
||||
sessionId: string;
|
||||
/** Display name of the owning session / case. */
|
||||
sessionName: string;
|
||||
/** Millisecond timestamp used for recency ranking and display. */
|
||||
timestamp: number;
|
||||
/** Short, already-truncated snippet describing the match. */
|
||||
snippet: string;
|
||||
/** True when the query matched the primary name/path exactly (case-insensitive). */
|
||||
exactMatch: boolean;
|
||||
/** Navigation target for the jump-to action. */
|
||||
jumpTo: SearchJumpTarget;
|
||||
}
|
||||
|
||||
/** A group of results for one source type, in render order. */
|
||||
export interface SearchResultGroup {
|
||||
type: SearchSourceType;
|
||||
results: SearchResult[];
|
||||
}
|
||||
|
||||
/** Payload returned as `data` inside the standard ApiResponse envelope. */
|
||||
export interface SearchResponseData {
|
||||
/** The normalized query that was executed. */
|
||||
query: string;
|
||||
/** Results grouped by source type, ordered sessions → events → files. */
|
||||
groups: SearchResultGroup[];
|
||||
/** Total number of results across all groups (after caps applied). */
|
||||
totalResults: number;
|
||||
/** True if any group or the total was capped (more matches existed). */
|
||||
truncated: boolean;
|
||||
}
|
||||
@@ -17,4 +17,5 @@ export { registerRalphRoutes } from './ralph-routes.js';
|
||||
export { registerPlanRoutes } from './plan-routes.js';
|
||||
export { registerOrchestratorRoutes } from './orchestrator-routes.js';
|
||||
export { registerClipboardRoutes } from './clipboard-routes.js';
|
||||
export { registerSearchRoutes } from './search-routes.js';
|
||||
export { registerWsRoutes } from './ws-routes.js';
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
/**
|
||||
* @fileoverview Cross-session federated search route (COD-9).
|
||||
*
|
||||
* Registers `GET /api/search?q=&types=&limit=` — a bounded, in-memory search
|
||||
* across three v1 sources, returned in the standard ApiResponse envelope:
|
||||
* 1. sessions/cases — name, working directory, session id
|
||||
* 2. run-summary events — event title/details (from the live run-summary trackers)
|
||||
* 3. file paths — per-session attachment history (workspace-relative paths only)
|
||||
*
|
||||
* This route is a THIN wrapper: it harvests the source arrays from the live
|
||||
* server stores (held on the route context) in a bounded way, then delegates
|
||||
* grouping/ranking/capping to the pure `searchSources()` core in
|
||||
* `src/search-service.ts`. Terminal-buffer scanning and any persisted index are
|
||||
* out of scope for v1.
|
||||
*
|
||||
* Safety: query input is Zod-validated (length-bounded `q`, allowlisted `types`,
|
||||
* numeric `limit`); only workspace-relative file paths are ever exposed (the
|
||||
* server-private `externalPath` on attachment history is never read here); and
|
||||
* the pure core enforces a per-group and total result cap so a broad query
|
||||
* cannot return an unbounded payload. No terminal output is read.
|
||||
*
|
||||
* Endpoints: GET /api/search
|
||||
*/
|
||||
|
||||
import { FastifyInstance } from 'fastify';
|
||||
import { parseBody } from '../route-helpers.js';
|
||||
import { SearchQuerySchema } from '../schemas.js';
|
||||
import {
|
||||
searchSources,
|
||||
type SearchSources,
|
||||
type SessionSearchInput,
|
||||
type EventSearchInput,
|
||||
type FileSearchInput,
|
||||
} from '../../search-service.js';
|
||||
import type { SearchSourceType } from '../../types/search.js';
|
||||
import type { SessionPort, InfraPort } from '../ports/index.js';
|
||||
|
||||
/**
|
||||
* Per-source harvest caps. These bound how much in-memory data we hand to the
|
||||
* pure core BEFORE it applies its own result caps — they keep the harvest itself
|
||||
* cheap on large deployments (e.g. 50 sessions × many events). They are
|
||||
* deliberately well above the result caps so ranking still sees enough candidates.
|
||||
*/
|
||||
const MAX_EVENTS_PER_SESSION = 500;
|
||||
|
||||
interface SessionLike {
|
||||
id: string;
|
||||
name: string;
|
||||
workingDir: string;
|
||||
lastActivityAt?: number;
|
||||
createdAt?: number;
|
||||
attachmentHistory?: Array<{
|
||||
id: string;
|
||||
fileName: string;
|
||||
relativePath?: string;
|
||||
timestamp?: number;
|
||||
mtimeMs?: number;
|
||||
}>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Harvest the three source arrays from the live in-memory stores. Reads only
|
||||
* bounded, already-loaded data — no disk I/O, no terminal buffers.
|
||||
*/
|
||||
function harvestSources(ctx: SessionPort & InfraPort): SearchSources {
|
||||
const sessions: SessionSearchInput[] = [];
|
||||
const events: EventSearchInput[] = [];
|
||||
const files: FileSearchInput[] = [];
|
||||
|
||||
for (const raw of ctx.sessions.values()) {
|
||||
const s = raw as unknown as SessionLike;
|
||||
const sessionName = s.name ?? '';
|
||||
const timestamp = s.lastActivityAt ?? s.createdAt ?? 0;
|
||||
|
||||
sessions.push({
|
||||
sessionId: s.id,
|
||||
sessionName,
|
||||
workingDir: s.workingDir ?? '',
|
||||
timestamp,
|
||||
});
|
||||
|
||||
// Files: per-session attachment history. Only the workspace-relative path is
|
||||
// surfaced; the server-private externalPath is intentionally never read.
|
||||
const history = s.attachmentHistory ?? [];
|
||||
for (const item of history) {
|
||||
files.push({
|
||||
sessionId: s.id,
|
||||
sessionName,
|
||||
fileName: item.fileName,
|
||||
relativePath: item.relativePath,
|
||||
timestamp: item.timestamp ?? item.mtimeMs ?? timestamp,
|
||||
itemId: item.id,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Events: from the live run-summary trackers, keyed by session id.
|
||||
for (const [sessionId, tracker] of ctx.runSummaryTrackers) {
|
||||
const session = ctx.sessions.get(sessionId) as unknown as SessionLike | undefined;
|
||||
const sessionName = session?.name ?? '';
|
||||
const summary = tracker.getSummary();
|
||||
// Newest events are most relevant; cap the per-session harvest.
|
||||
const evts = summary.events.slice(-MAX_EVENTS_PER_SESSION);
|
||||
for (const e of evts) {
|
||||
events.push({
|
||||
sessionId,
|
||||
sessionName,
|
||||
eventId: e.id,
|
||||
title: e.title,
|
||||
details: e.details ?? '',
|
||||
timestamp: e.timestamp,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return { sessions, events, files };
|
||||
}
|
||||
|
||||
export function registerSearchRoutes(app: FastifyInstance, ctx: SessionPort & InfraPort): void {
|
||||
app.get('/api/search', async (req) => {
|
||||
// Zod-validate the query. parseBody throws a structured 400 on failure.
|
||||
const { q, types, limit } = parseBody(SearchQuerySchema, req.query);
|
||||
|
||||
const allowed: Set<SearchSourceType> | null = types
|
||||
? new Set(
|
||||
types
|
||||
.split(',')
|
||||
.map((t) => t.trim())
|
||||
.filter(Boolean) as SearchSourceType[]
|
||||
)
|
||||
: null;
|
||||
|
||||
const sources = harvestSources(ctx);
|
||||
|
||||
// Apply the optional source-type filter before searching so excluded
|
||||
// sources never contribute to (or consume budget in) the result set.
|
||||
const filtered: SearchSources = {
|
||||
sessions: !allowed || allowed.has('session') ? sources.sessions : [],
|
||||
events: !allowed || allowed.has('event') ? sources.events : [],
|
||||
files: !allowed || allowed.has('file') ? sources.files : [],
|
||||
};
|
||||
|
||||
const result = searchSources(q, filtered);
|
||||
|
||||
// Optional caller-supplied total cap (always on top of the core's hard caps).
|
||||
if (limit !== undefined && result.totalResults > limit) {
|
||||
let remaining = limit;
|
||||
const cappedGroups = [];
|
||||
for (const group of result.groups) {
|
||||
if (remaining <= 0) break;
|
||||
const slice = group.results.slice(0, remaining);
|
||||
remaining -= slice.length;
|
||||
cappedGroups.push({ type: group.type, results: slice });
|
||||
}
|
||||
result.groups = cappedGroups;
|
||||
result.totalResults = limit;
|
||||
result.truncated = true;
|
||||
}
|
||||
|
||||
return { success: true, data: result };
|
||||
});
|
||||
}
|
||||
@@ -708,3 +708,35 @@ export const OrchestratorStartSchema = z.object({
|
||||
export const OrchestratorRejectSchema = z.object({
|
||||
feedback: z.string().min(1).max(10000),
|
||||
});
|
||||
|
||||
// ========== Cross-Session Search (COD-9) ==========
|
||||
|
||||
/** Valid federated source kinds for `GET /api/search?types=`. */
|
||||
export const SEARCH_SOURCE_TYPES = ['session', 'event', 'file'] as const;
|
||||
|
||||
/**
|
||||
* GET /api/search query validation.
|
||||
*
|
||||
* Query params arrive as strings: `q` is bounded (1..200 chars), `types` is an
|
||||
* optional comma-separated allowlisted CSV, and `limit` is an optional coerced
|
||||
* integer clamped to 1..60. Validation is the first line of defense — a missing
|
||||
* or oversized `q`, an unknown type, or a non-numeric limit is rejected with 400.
|
||||
*/
|
||||
export const SearchQuerySchema = z.object({
|
||||
q: z.string().trim().min(1, 'Query is required').max(200, 'Query too long (max 200 chars)'),
|
||||
types: z
|
||||
.string()
|
||||
.max(100)
|
||||
.optional()
|
||||
.refine(
|
||||
(v) =>
|
||||
v === undefined ||
|
||||
v
|
||||
.split(',')
|
||||
.map((t) => t.trim())
|
||||
.filter(Boolean)
|
||||
.every((t) => (SEARCH_SOURCE_TYPES as readonly string[]).includes(t)),
|
||||
{ message: 'Invalid types value' }
|
||||
),
|
||||
limit: z.coerce.number().int().min(1).max(60).optional(),
|
||||
});
|
||||
|
||||
@@ -149,6 +149,7 @@ import {
|
||||
registerRalphRoutes,
|
||||
registerPlanRoutes,
|
||||
registerClipboardRoutes,
|
||||
registerSearchRoutes,
|
||||
registerOrchestratorRoutes,
|
||||
registerWsRoutes,
|
||||
} from './routes/index.js';
|
||||
@@ -869,6 +870,7 @@ export class WebServer extends EventEmitter {
|
||||
registerRalphRoutes(this.app, ctx);
|
||||
registerPlanRoutes(this.app, ctx);
|
||||
registerClipboardRoutes(this.app, ctx);
|
||||
registerSearchRoutes(this.app, ctx);
|
||||
registerOrchestratorRoutes(this.app, ctx);
|
||||
registerWsRoutes(this.app, ctx, () => this.getHostPolicy());
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user