From a92826b66daae4f5f308b99b97ecacfecc8f250f Mon Sep 17 00:00:00 2001 From: arkon Date: Fri, 30 Jan 2026 12:15:09 +0100 Subject: [PATCH] refactor: simplify ralph wizard from 9 agents to 2 Remove execution layer that never actually controlled execution: - execution-bridge.ts (model param was ignored) - group-scheduler.ts (task-tool mode never used) - model-selector.ts (recommendations were display-only) - context-manager.ts (never invoked) - execution-limits.ts Remove redundant agent prompts (overlapping outputs): - requirements-analyst, architecture-planner, risk-analyst - testing-specialist, verification (merged into planner.ts) - execution-optimizer (output was ignored) - final-review (scores were cosmetic) Simplify plan-orchestrator.ts from 2400 LOC to 520 LOC: - Before: 9 agents, 6 phases, ~40-60 minutes - After: 2 agents (research + planner), ~18 minutes Remove /api/execution/* endpoints and related server code. Co-Authored-By: Claude Opus 4.5 --- src/config/execution-limits.ts | 93 -- src/context-manager.ts | 333 ---- src/execution-bridge.ts | 846 ---------- src/group-scheduler.ts | 551 ------ src/model-selector.ts | 320 ---- src/plan-orchestrator.ts | 2415 +++------------------------ src/prompts/architecture-planner.ts | 32 - src/prompts/execution-optimizer.ts | 134 -- src/prompts/final-review.ts | 100 -- src/prompts/index.ts | 8 +- src/prompts/planner.ts | 83 + src/prompts/requirements-analyst.ts | 31 - src/prompts/risk-analyst.ts | 32 - src/prompts/testing-specialist.ts | 76 - src/prompts/verification.ts | 96 -- src/types.ts | 35 +- src/web/server.ts | 342 ---- test/execution-bridge.test.ts | 270 --- 18 files changed, 329 insertions(+), 5468 deletions(-) delete mode 100644 src/config/execution-limits.ts delete mode 100644 src/context-manager.ts delete mode 100644 src/execution-bridge.ts delete mode 100644 src/group-scheduler.ts delete mode 100644 src/model-selector.ts delete mode 100644 src/prompts/architecture-planner.ts delete mode 100644 src/prompts/execution-optimizer.ts delete mode 100644 src/prompts/final-review.ts create mode 100644 src/prompts/planner.ts delete mode 100644 src/prompts/requirements-analyst.ts delete mode 100644 src/prompts/risk-analyst.ts delete mode 100644 src/prompts/testing-specialist.ts delete mode 100644 src/prompts/verification.ts delete mode 100644 test/execution-bridge.test.ts diff --git a/src/config/execution-limits.ts b/src/config/execution-limits.ts deleted file mode 100644 index e3a8ad6d..00000000 --- a/src/config/execution-limits.ts +++ /dev/null @@ -1,93 +0,0 @@ -/** - * @fileoverview Centralized limits for the Execution Bridge system. - * - * These constants define resource limits for parallel task execution, - * group scheduling, and model selection. - * - * @module config/execution-limits - */ - -// ============================================================================ -// Parallel Execution Limits -// ============================================================================ - -/** - * Maximum tasks to execute in parallel within a single group. - * Higher values increase throughput but may strain system resources. - */ -export const MAX_PARALLEL_TASKS_PER_GROUP = 5; - -/** - * Default timeout for an execution group in milliseconds (30 minutes). - * If all tasks in a group don't complete within this time, stragglers are cancelled. - */ -export const GROUP_TIMEOUT_MS = 30 * 60 * 1000; - -/** - * Maximum retry attempts for a failed task before marking as permanently failed. - */ -export const MAX_TASK_RETRIES = 2; - -/** - * Delay between task retries in milliseconds (10 seconds). - */ -export const TASK_RETRY_DELAY_MS = 10 * 1000; - -// ============================================================================ -// Model Selection Limits -// ============================================================================ - -/** - * Default model when no recommendation is provided. - */ -export const DEFAULT_MODEL: 'opus' | 'sonnet' | 'haiku' = 'sonnet'; - -/** - * Token threshold for switching execution modes. - * Tasks with estimated tokens below this use task-tool mode. - * Tasks with estimated tokens above this use session mode. - */ -export const TOKEN_THRESHOLD_FOR_SESSION_MODE = 50000; - -/** - * Token threshold for low-complexity tasks (haiku-appropriate). - */ -export const TOKEN_THRESHOLD_HAIKU = 15000; - -// ============================================================================ -// Context Management Limits -// ============================================================================ - -/** - * Delay between /clear and /init commands in milliseconds. - */ -export const CONTEXT_REFRESH_DELAY_MS = 2000; - -/** - * Maximum pending context refresh operations. - */ -export const MAX_PENDING_CONTEXT_REFRESHES = 10; - -// ============================================================================ -// Execution Bridge Limits -// ============================================================================ - -/** - * Maximum execution groups to track in history. - */ -export const MAX_EXECUTION_HISTORY = 50; - -/** - * Polling interval for execution progress in milliseconds. - */ -export const EXECUTION_POLL_INTERVAL_MS = 1000; - -/** - * Maximum total tasks in a single execution plan. - */ -export const MAX_TASKS_PER_PLAN = 100; - -/** - * Grace period after group completion before cleanup in milliseconds. - */ -export const GROUP_CLEANUP_DELAY_MS = 5000; diff --git a/src/context-manager.ts b/src/context-manager.ts deleted file mode 100644 index 67dd6b54..00000000 --- a/src/context-manager.ts +++ /dev/null @@ -1,333 +0,0 @@ -/** - * @fileoverview Context Manager - Handles fresh context requirements. - * - * Manages context refresh operations for tasks that require starting - * with a clean context window. Supports both /clear + /init sequences - * and new session spawning. - * - * @module context-manager - */ - -import { EventEmitter } from 'node:events'; -import { - CONTEXT_REFRESH_DELAY_MS, - MAX_PENDING_CONTEXT_REFRESHES, -} from './config/execution-limits.js'; - -// ========== Types ========== - -/** Methods for refreshing context */ -export type ContextRefreshMethod = 'clear-init' | 'new-session'; - -/** Status of a context refresh operation */ -export type ContextRefreshStatus = 'pending' | 'clearing' | 'initializing' | 'completed' | 'failed'; - -/** - * Request for a context refresh. - */ -export interface ContextRefreshRequest { - /** Task ID requesting refresh */ - taskId: string; - /** Session ID to refresh */ - sessionId: string; - /** Preferred refresh method */ - method: ContextRefreshMethod; - /** Working directory (for new-session method) */ - workingDir?: string; - /** Optional init prompt to use */ - initPrompt?: string; -} - -/** - * Result of a context refresh operation. - */ -export interface ContextRefreshResult { - /** Request that was processed */ - request: ContextRefreshRequest; - /** Final status */ - status: ContextRefreshStatus; - /** New session ID (if method was new-session) */ - newSessionId?: string; - /** Error message if failed */ - error?: string; - /** Duration in milliseconds */ - durationMs: number; -} - -/** - * Tracked refresh operation. - */ -interface TrackedRefresh { - request: ContextRefreshRequest; - status: ContextRefreshStatus; - startedAt: number; - completedAt?: number; - error?: string; - newSessionId?: string; -} - -// ========== Events ========== - -export interface ContextManagerEvents { - /** Context refresh started */ - refreshStarted: (data: { taskId: string; sessionId: string; method: ContextRefreshMethod }) => void; - /** Context refresh completed */ - refreshCompleted: (result: ContextRefreshResult) => void; - /** Context refresh failed */ - refreshFailed: (data: { taskId: string; sessionId: string; error: string }) => void; -} - -// ========== Session Writer Interface ========== - -/** - * Interface for writing to sessions. - * Injected to avoid circular dependencies. - */ -export interface SessionWriter { - /** Write text to session */ - writeToSession(sessionId: string, text: string): void; - /** Create a new session */ - createSession?(workingDir: string, name?: string): Promise<{ sessionId: string }>; -} - -// ========== Context Manager ========== - -/** - * ContextManager - Handles context refresh operations. - * - * When a task requires fresh context (requiresFreshContext=true), - * this manager coordinates the refresh using one of two methods: - * - * 1. clear-init: Send /clear followed by /init to existing session - * 2. new-session: Spawn a completely new session (more expensive) - */ -export class ContextManager extends EventEmitter { - private _pending: Map = new Map(); - private _sessionWriter: SessionWriter | null = null; - private _refreshDelayMs: number; - - constructor(refreshDelayMs: number = CONTEXT_REFRESH_DELAY_MS) { - super(); - this._refreshDelayMs = refreshDelayMs; - } - - /** - * Set the session writer for performing actual operations. - */ - setSessionWriter(writer: SessionWriter): void { - this._sessionWriter = writer; - } - - /** - * Get number of pending refresh operations. - */ - get pendingCount(): number { - return this._pending.size; - } - - /** - * Check if a refresh is pending for a session. - */ - hasPendingRefresh(sessionId: string): boolean { - for (const tracked of this._pending.values()) { - if (tracked.request.sessionId === sessionId && - (tracked.status === 'pending' || tracked.status === 'clearing' || tracked.status === 'initializing')) { - return true; - } - } - return false; - } - - /** - * Request a context refresh for a task. - * - * @returns Promise that resolves when refresh completes - */ - async requestRefresh(request: ContextRefreshRequest): Promise { - if (!this._sessionWriter) { - return { - request, - status: 'failed', - error: 'No session writer configured', - durationMs: 0, - }; - } - - // Check pending limit - if (this._pending.size >= MAX_PENDING_CONTEXT_REFRESHES) { - return { - request, - status: 'failed', - error: `Max pending refreshes (${MAX_PENDING_CONTEXT_REFRESHES}) exceeded`, - durationMs: 0, - }; - } - - // Track the refresh - const tracked: TrackedRefresh = { - request, - status: 'pending', - startedAt: Date.now(), - }; - this._pending.set(request.taskId, tracked); - - this.emit('refreshStarted', { - taskId: request.taskId, - sessionId: request.sessionId, - method: request.method, - }); - - try { - if (request.method === 'clear-init') { - await this.performClearInit(tracked); - } else { - await this.performNewSession(tracked); - } - - tracked.status = 'completed'; - tracked.completedAt = Date.now(); - - const result: ContextRefreshResult = { - request, - status: 'completed', - newSessionId: tracked.newSessionId, - durationMs: tracked.completedAt - tracked.startedAt, - }; - - this.emit('refreshCompleted', result); - return result; - - } catch (err) { - tracked.status = 'failed'; - tracked.error = err instanceof Error ? err.message : String(err); - tracked.completedAt = Date.now(); - - const result: ContextRefreshResult = { - request, - status: 'failed', - error: tracked.error, - durationMs: tracked.completedAt - tracked.startedAt, - }; - - this.emit('refreshFailed', { - taskId: request.taskId, - sessionId: request.sessionId, - error: tracked.error, - }); - - return result; - - } finally { - // Clean up after a delay - setTimeout(() => { - this._pending.delete(request.taskId); - }, 5000); - } - } - - /** - * Perform /clear + /init sequence. - */ - private async performClearInit(tracked: TrackedRefresh): Promise { - if (!this._sessionWriter) throw new Error('No session writer'); - - const { sessionId, initPrompt } = tracked.request; - - // Send /clear - tracked.status = 'clearing'; - this._sessionWriter.writeToSession(sessionId, '/clear\r'); - - // Wait for clear to process - await this.delay(this._refreshDelayMs); - - // Send /init (or custom init prompt) - tracked.status = 'initializing'; - const prompt = initPrompt || '/init'; - this._sessionWriter.writeToSession(sessionId, prompt + '\r'); - - // Wait for init to complete - await this.delay(this._refreshDelayMs); - } - - /** - * Spawn a new session for complete context isolation. - */ - private async performNewSession(tracked: TrackedRefresh): Promise { - if (!this._sessionWriter?.createSession) { - throw new Error('Session creation not supported'); - } - - const { workingDir, taskId } = tracked.request; - if (!workingDir) { - throw new Error('Working directory required for new-session method'); - } - - tracked.status = 'initializing'; - - const result = await this._sessionWriter.createSession(workingDir, `task-${taskId}`); - tracked.newSessionId = result.sessionId; - } - - /** - * Cancel a pending refresh. - */ - cancelRefresh(taskId: string): boolean { - const tracked = this._pending.get(taskId); - if (tracked && tracked.status === 'pending') { - tracked.status = 'failed'; - tracked.error = 'Cancelled'; - tracked.completedAt = Date.now(); - this._pending.delete(taskId); - return true; - } - return false; - } - - /** - * Get status of all pending refreshes. - */ - getPendingStatus(): Array<{ taskId: string; sessionId: string; status: ContextRefreshStatus; elapsedMs: number }> { - const now = Date.now(); - return Array.from(this._pending.values()).map(tracked => ({ - taskId: tracked.request.taskId, - sessionId: tracked.request.sessionId, - status: tracked.status, - elapsedMs: now - tracked.startedAt, - })); - } - - /** - * Clear all pending operations (for cleanup). - */ - clearAll(): void { - this._pending.clear(); - } - - private delay(ms: number): Promise { - return new Promise(resolve => setTimeout(resolve, ms)); - } -} - -// ========== Singleton ========== - -let managerInstance: ContextManager | null = null; - -/** - * Get or create the singleton ContextManager instance. - */ -export function getContextManager(): ContextManager { - if (!managerInstance) { - managerInstance = new ContextManager(); - } - return managerInstance; -} - -/** - * Reset the singleton (for testing). - */ -export function resetContextManager(): void { - if (managerInstance) { - managerInstance.clearAll(); - } - managerInstance = null; -} diff --git a/src/execution-bridge.ts b/src/execution-bridge.ts deleted file mode 100644 index d071f5ec..00000000 --- a/src/execution-bridge.ts +++ /dev/null @@ -1,846 +0,0 @@ -/** - * @fileoverview Execution Bridge - Coordinates parallel task execution. - * - * The central coordinator that: - * - Loads optimized plans from PlanOrchestrator - * - Converts plans to executable groups via GroupScheduler - * - Manages parallel execution within groups - * - Coordinates model selection via ModelSelector - * - Handles fresh context requirements via ContextManager - * - Tracks overall execution progress - * - Integrates with SpawnOrchestrator for session-based execution - * - * @module execution-bridge - */ - -import { EventEmitter } from 'node:events'; -import { - GroupScheduler, - getGroupScheduler, - type ExecutionGroup, - type GroupTask, - type ExecutionSchedule, -} from './group-scheduler.js'; -import { - ModelSelector, - getModelSelector, - type ModelConfig, - type ModelSelection, - type ExecutionMode, -} from './model-selector.js'; -import { - ContextManager, - getContextManager, - type SessionWriter, -} from './context-manager.js'; -import { - MAX_PARALLEL_TASKS_PER_GROUP, - GROUP_TIMEOUT_MS, - MAX_TASK_RETRIES, - TASK_RETRY_DELAY_MS, - EXECUTION_POLL_INTERVAL_MS, - MAX_EXECUTION_HISTORY, -} from './config/execution-limits.js'; - -// ========== Types ========== - -/** Overall execution status */ -export type ExecutionStatus = 'idle' | 'loading' | 'running' | 'paused' | 'completed' | 'partial' | 'failed' | 'cancelled'; - -/** - * Execution progress for UI display. - */ -export interface ExecutionProgress { - /** Current status */ - status: ExecutionStatus; - /** Current group being executed */ - currentGroup: number | null; - /** Total groups */ - totalGroups: number; - /** Completed groups */ - completedGroups: number; - /** Total tasks */ - totalTasks: number; - /** Completed tasks */ - completedTasks: number; - /** Failed tasks */ - failedTasks: number; - /** Tasks currently running */ - runningTasks: number; - /** Elapsed time in milliseconds */ - elapsedMs: number; - /** Estimated remaining time (if available) */ - estimatedRemainingMs?: number; -} - -/** - * Task assignment for spawning. - */ -export interface TaskAssignment { - /** Task from the schedule */ - task: GroupTask; - /** Selected model */ - model: ModelSelection; - /** Execution mode */ - executionMode: ExecutionMode; - /** Session ID (if session mode) */ - sessionId?: string; - /** Whether fresh context was requested */ - freshContextRequested: boolean; -} - -/** - * Plan item from PlanOrchestrator (input format). - */ -export interface PlanItem { - id: string; - title: string; - description: string; - parallelGroup?: number; - agentType?: string; - recommendedModel?: string; - requiresFreshContext?: boolean; - estimatedTokens?: number; - inputFiles?: string[]; - outputFiles?: string[]; - dependencies?: string[]; -} - -/** - * Execution history entry. - */ -export interface ExecutionHistoryEntry { - /** Unique execution ID */ - id: string; - /** When execution started */ - startedAt: number; - /** When execution ended */ - endedAt?: number; - /** Final status */ - status: ExecutionStatus; - /** Task counts */ - totalTasks: number; - completedTasks: number; - failedTasks: number; - /** Total cost estimate */ - estimatedCost?: number; -} - -// ========== Events ========== - -export interface ExecutionBridgeEvents { - /** Plan loaded and schedule built */ - planLoaded: (schedule: ExecutionSchedule) => void; - /** Execution started */ - started: () => void; - /** Execution paused */ - paused: () => void; - /** Execution resumed */ - resumed: () => void; - /** Execution completed */ - completed: (result: { status: ExecutionStatus; stats: ExecutionProgress }) => void; - /** Execution cancelled */ - cancelled: (reason: string) => void; - /** Group started */ - groupStarted: (data: { groupNumber: number; taskCount: number; executionMode: ExecutionMode }) => void; - /** Group completed */ - groupCompleted: (data: { groupNumber: number; status: string; completedCount: number; failedCount: number }) => void; - /** Task assigned to execution */ - taskAssigned: (assignment: TaskAssignment) => void; - /** Task completed */ - taskCompleted: (data: { taskId: string; groupNumber: number; durationMs: number }) => void; - /** Task failed */ - taskFailed: (data: { taskId: string; groupNumber: number; error: string; willRetry: boolean }) => void; - /** Fresh context triggered */ - freshContext: (data: { taskId: string; method: string; success: boolean }) => void; - /** Model selected for task */ - modelSelected: (data: { taskId: string; model: string; reason: string; optimizerSuggested?: string }) => void; - /** Progress update */ - progress: (progress: ExecutionProgress) => void; -} - -// ========== Spawn Interface ========== - -/** - * Interface for spawning agents. - * Injected to avoid circular dependencies. - */ -export interface AgentSpawner { - /** Spawn an agent with specific model */ - spawnAgentWithModel( - taskId: string, - workingDir: string, - prompt: string, - model: string, - options?: { requiresFreshContext?: boolean } - ): Promise<{ sessionId: string }>; - /** Use Task tool for lightweight execution */ - useTaskTool( - sessionId: string, - taskId: string, - prompt: string, - model: string - ): Promise; - /** Check if task is complete */ - isTaskComplete(taskId: string): boolean; - /** Get task result */ - getTaskResult(taskId: string): { success: boolean; output?: string; error?: string } | null; -} - -// ========== Execution Bridge ========== - -/** - * ExecutionBridge - Coordinates parallel task execution. - * - * This is the main entry point for the optimized execution system. - * It bridges the gap between PlanOrchestrator's output and actual - * task execution, making use of the optimizer's metadata. - */ -export class ExecutionBridge extends EventEmitter { - private _scheduler: GroupScheduler; - private _modelSelector: ModelSelector; - private _contextManager: ContextManager; - - private _status: ExecutionStatus = 'idle'; - private _executionId: string | null = null; - private _startedAt: number | null = null; - private _pausedAt: number | null = null; - - private _agentSpawner: AgentSpawner | null = null; - private _sessionWriter: SessionWriter | null = null; - - private _pollTimer: NodeJS.Timeout | null = null; - private _groupTimeoutTimers: Map = new Map(); - private _retryTimers: Map = new Map(); - private _runningTasks: Map = new Map(); - - private _history: ExecutionHistoryEntry[] = []; - private _workingDir: string = process.cwd(); - - constructor(modelConfig?: Partial) { - super(); - this._scheduler = getGroupScheduler(); - this._modelSelector = getModelSelector(modelConfig); - this._contextManager = getContextManager(); - - // Forward scheduler events - this._scheduler.on('groupStarted', group => { - this.emit('groupStarted', { - groupNumber: group.groupNumber, - taskCount: group.tasks.length, - executionMode: group.executionMode, - }); - }); - - this._scheduler.on('groupCompleted', group => { - // Clear the group timeout timer to prevent memory leak - const timer = this._groupTimeoutTimers.get(group.groupNumber); - if (timer) { - clearTimeout(timer); - this._groupTimeoutTimers.delete(group.groupNumber); - } - this.emit('groupCompleted', { - groupNumber: group.groupNumber, - status: group.status, - completedCount: group.completedCount, - failedCount: group.failedCount, - }); - }); - - this._scheduler.on('taskStatusChanged', data => { - if (data.newStatus === 'completed') { - const runningInfo = this._runningTasks.get(data.taskId); - const durationMs = runningInfo ? Date.now() - runningInfo.startedAt : 0; - this._runningTasks.delete(data.taskId); - this.emit('taskCompleted', { taskId: data.taskId, groupNumber: data.groupNumber, durationMs }); - } - }); - } - - /** - * Get current execution status. - */ - get status(): ExecutionStatus { - return this._status; - } - - /** - * Get current execution progress. - */ - getProgress(): ExecutionProgress { - const stats = this._scheduler.getStats(); - const schedule = this._scheduler.schedule; - - return { - status: this._status, - currentGroup: schedule?.currentGroupIndex ?? null, - totalGroups: stats.totalGroups, - completedGroups: stats.completedGroups, - totalTasks: stats.totalTasks, - completedTasks: stats.completedTasks, - failedTasks: stats.failedTasks, - runningTasks: this._runningTasks.size, - elapsedMs: this.calculateElapsedMs(), - }; - } - - /** - * Calculate elapsed time, accounting for pauses. - */ - private calculateElapsedMs(): number { - if (!this._startedAt) return 0; - if (this._pausedAt) { - return this._pausedAt - this._startedAt; - } - return Date.now() - this._startedAt; - } - - /** - * Set the agent spawner for session-based execution. - */ - setAgentSpawner(spawner: AgentSpawner): void { - this._agentSpawner = spawner; - } - - /** - * Set the session writer for context management. - */ - setSessionWriter(writer: SessionWriter): void { - this._sessionWriter = writer; - this._contextManager.setSessionWriter(writer); - } - - /** - * Set working directory for execution. - */ - setWorkingDir(dir: string): void { - this._workingDir = dir; - } - - /** - * Update model configuration. - */ - updateModelConfig(config: Partial): void { - this._modelSelector.updateConfig(config); - } - - /** - * Get current model configuration. - */ - getModelConfig(): ModelConfig { - return this._modelSelector.config; - } - - /** - * Load a plan and build execution schedule. - */ - loadPlan(items: PlanItem[]): ExecutionSchedule { - if (this._status === 'running') { - throw new Error('Cannot load plan while execution is running'); - } - - this._status = 'loading'; - const schedule = this._scheduler.buildSchedule(items); - this._status = 'idle'; - - this.emit('planLoaded', schedule); - return schedule; - } - - /** - * Start execution of the loaded plan. - */ - async start(): Promise { - const schedule = this._scheduler.schedule; - if (!schedule) { - throw new Error('No plan loaded'); - } - - if (this._status === 'running') { - return; - } - - if (!this._agentSpawner) { - throw new Error('No agent spawner configured'); - } - - this._status = 'running'; - this._executionId = `exec-${Date.now()}`; - this._startedAt = Date.now(); - - // Add to history - this._history.unshift({ - id: this._executionId, - startedAt: this._startedAt, - status: 'running', - totalTasks: schedule.totalTasks, - completedTasks: 0, - failedTasks: 0, - }); - - // Trim history - if (this._history.length > MAX_EXECUTION_HISTORY) { - this._history = this._history.slice(0, MAX_EXECUTION_HISTORY); - } - - this.emit('started'); - - // Start the execution loop - this.startExecutionLoop(); - } - - /** - * Pause execution. - */ - pause(): void { - if (this._status !== 'running') return; - - this._status = 'paused'; - this._pausedAt = Date.now(); - this.stopExecutionLoop(); - this.emit('paused'); - } - - /** - * Resume paused execution. - */ - resume(): void { - if (this._status !== 'paused') return; - - this._status = 'running'; - this._pausedAt = null; - this.startExecutionLoop(); - this.emit('resumed'); - } - - /** - * Cancel execution. - */ - async cancel(reason: string = 'User cancelled'): Promise { - if (this._status === 'idle' || this._status === 'completed' || this._status === 'cancelled') { - return; - } - - this._status = 'cancelled'; - this.stopExecutionLoop(); - - // Clear group timeouts - for (const timer of this._groupTimeoutTimers.values()) { - clearTimeout(timer); - } - this._groupTimeoutTimers.clear(); - - // Update history - this.updateHistoryEntry('cancelled'); - - this.emit('cancelled', reason); - } - - /** - * Get execution history. - */ - getHistory(): ExecutionHistoryEntry[] { - return [...this._history]; - } - - /** - * Start the main execution loop. - */ - private startExecutionLoop(): void { - if (this._pollTimer) return; - - this._pollTimer = setInterval(() => { - this.tick().catch(err => { - console.error('[execution-bridge] Tick error:', err); - }); - }, EXECUTION_POLL_INTERVAL_MS); - - // Run immediately - this.tick().catch(err => { - console.error('[execution-bridge] Initial tick error:', err); - }); - } - - /** - * Stop the execution loop. - */ - private stopExecutionLoop(): void { - if (this._pollTimer) { - clearInterval(this._pollTimer); - this._pollTimer = null; - } - } - - /** - * Main execution tick. - */ - private async tick(): Promise { - if (this._status !== 'running') return; - - const schedule = this._scheduler.schedule; - if (!schedule) return; - - // Check if we're done - if (schedule.status === 'completed' || schedule.status === 'partial' || schedule.status === 'failed') { - this.handleExecutionComplete(); - return; - } - - // Process current group or start next one - await this.processGroups(); - - // Emit progress - this.emit('progress', this.getProgress()); - } - - /** - * Process groups - start new ones or continue existing. - */ - private async processGroups(): Promise { - const schedule = this._scheduler.schedule; - if (!schedule) return; - - // Find current running group - const runningGroup = schedule.groups.find(g => g.status === 'running'); - - if (runningGroup) { - // Continue processing current group - await this.processGroup(runningGroup); - } else { - // Try to start next group - const nextGroup = this._scheduler.getNextReadyGroup(); - if (nextGroup) { - await this.startGroup(nextGroup); - } - } - } - - /** - * Start executing a group. - */ - private async startGroup(group: ExecutionGroup): Promise { - this._scheduler.startGroup(group.groupNumber); - - // Set up group timeout - const timeoutTimer = setTimeout(() => { - this.handleGroupTimeout(group.groupNumber); - }, GROUP_TIMEOUT_MS); - this._groupTimeoutTimers.set(group.groupNumber, timeoutTimer); - - // Start initial tasks - await this.processGroup(group); - } - - /** - * Process tasks within a group. - */ - private async processGroup(group: ExecutionGroup): Promise { - if (!this._agentSpawner) return; - - // Get ready tasks - const readyTasks = this._scheduler.getReadyTasksInGroup(group.groupNumber); - if (readyTasks.length === 0) return; - - // Limit parallel tasks - const slotsAvailable = MAX_PARALLEL_TASKS_PER_GROUP - this._runningTasks.size; - if (slotsAvailable <= 0) return; - - const tasksToStart = readyTasks.slice(0, slotsAvailable); - - for (const task of tasksToStart) { - await this.assignTask(task, group); - } - } - - /** - * Assign a task for execution. - */ - private async assignTask(task: GroupTask, group: ExecutionGroup): Promise { - if (!this._agentSpawner) return; - - // Select model - const modelSelection = this._modelSelector.selectModel(task.id, { - estimatedTokens: task.estimatedTokens, - agentType: task.agentType, - recommendedModel: task.recommendedModel, - outputFiles: task.outputFiles, - inputFiles: task.inputFiles, - }); - - this.emit('modelSelected', { - taskId: task.id, - model: modelSelection.model, - reason: modelSelection.reason, - optimizerSuggested: modelSelection.optimizerRecommendation, - }); - - // Handle fresh context if required - let freshContextRequested = false; - if (task.requiresFreshContext && this._sessionWriter) { - freshContextRequested = true; - // Context refresh will be handled by the spawner - } - - // Mark task as running - this._scheduler.updateTaskStatus(task.id, 'running'); - this._runningTasks.set(task.id, { startedAt: Date.now() }); - - const assignment: TaskAssignment = { - task, - model: modelSelection, - executionMode: group.executionMode, - freshContextRequested, - }; - - this.emit('taskAssigned', assignment); - - // Execute based on mode - try { - if (group.executionMode === 'session') { - const result = await this._agentSpawner.spawnAgentWithModel( - task.id, - this._workingDir, - task.description, - modelSelection.model, - { requiresFreshContext: task.requiresFreshContext } - ); - assignment.sessionId = result.sessionId; - this._runningTasks.set(task.id, { startedAt: Date.now(), sessionId: result.sessionId }); - } else { - // task-tool mode - would use Task tool in main session - // For now, fall back to session mode - const result = await this._agentSpawner.spawnAgentWithModel( - task.id, - this._workingDir, - task.description, - modelSelection.model - ); - assignment.sessionId = result.sessionId; - this._runningTasks.set(task.id, { startedAt: Date.now(), sessionId: result.sessionId }); - } - - if (task.requiresFreshContext) { - this.emit('freshContext', { - taskId: task.id, - method: 'session', - success: true, - }); - } - - } catch (err) { - const error = err instanceof Error ? err.message : String(err); - await this.handleTaskFailure(task, group.groupNumber, error); - } - } - - /** - * Handle task failure. - */ - private async handleTaskFailure(task: GroupTask, groupNumber: number, error: string): Promise { - task.retryCount++; - const willRetry = task.retryCount < MAX_TASK_RETRIES; - - this.emit('taskFailed', { - taskId: task.id, - groupNumber, - error, - willRetry, - }); - - if (willRetry) { - // Schedule retry with tracked timer - const retryTimer = setTimeout(() => { - this._retryTimers.delete(task.id); - task.status = 'pending'; - task.error = undefined; - this._runningTasks.delete(task.id); - }, TASK_RETRY_DELAY_MS); - this._retryTimers.set(task.id, retryTimer); - } else { - // Mark as permanently failed - this._scheduler.updateTaskStatus(task.id, 'failed', error); - this._runningTasks.delete(task.id); - - // Mark dependent tasks as blocked - this._scheduler.markDependentTasksBlocked(task.id); - } - } - - /** - * Mark a task as complete (called externally when agent finishes). - */ - markTaskComplete(taskId: string): void { - const schedule = this._scheduler.schedule; - if (!schedule) return; - - const groupNum = this.findTaskGroup(taskId); - if (groupNum === null) return; - - // Clear any pending retry timer for this task - const retryTimer = this._retryTimers.get(taskId); - if (retryTimer) { - clearTimeout(retryTimer); - this._retryTimers.delete(taskId); - } - - this._scheduler.updateTaskStatus(taskId, 'completed'); - this._runningTasks.delete(taskId); - } - - /** - * Mark a task as failed (called externally when agent fails). - */ - markTaskFailed(taskId: string, error: string): void { - const schedule = this._scheduler.schedule; - if (!schedule) return; - - const groupNum = this.findTaskGroup(taskId); - if (groupNum === null) return; - - // Get the task to check retry count - for (const group of schedule.groups) { - const task = group.tasks.find(t => t.id === taskId); - if (task) { - this.handleTaskFailure(task, group.groupNumber, error); - return; - } - } - } - - /** - * Find which group a task belongs to. - */ - private findTaskGroup(taskId: string): number | null { - const schedule = this._scheduler.schedule; - if (!schedule) return null; - - for (const group of schedule.groups) { - if (group.tasks.some(t => t.id === taskId)) { - return group.groupNumber; - } - } - return null; - } - - /** - * Handle group timeout. - */ - private handleGroupTimeout(groupNumber: number): void { - const schedule = this._scheduler.schedule; - if (!schedule) return; - - const group = schedule.groups.find(g => g.groupNumber === groupNumber); - if (!group || group.status !== 'running') return; - - console.warn(`[execution-bridge] Group ${groupNumber} timed out`); - - // Mark running tasks as failed - for (const task of group.tasks) { - if (task.status === 'running') { - this._scheduler.updateTaskStatus(task.id, 'failed', 'Group timeout'); - this._runningTasks.delete(task.id); - } else if (task.status === 'pending') { - this._scheduler.updateTaskStatus(task.id, 'skipped', 'Group timeout'); - } - } - - this._groupTimeoutTimers.delete(groupNumber); - } - - /** - * Handle execution complete. - */ - private handleExecutionComplete(): void { - this.stopExecutionLoop(); - - // Clear timers - for (const timer of this._groupTimeoutTimers.values()) { - clearTimeout(timer); - } - this._groupTimeoutTimers.clear(); - - const schedule = this._scheduler.schedule; - if (!schedule) return; - - // Map schedule status to execution status - if (schedule.status === 'completed') { - this._status = 'completed'; - } else if (schedule.status === 'partial') { - this._status = 'partial'; - } else { - this._status = 'failed'; - } - - // Update history - this.updateHistoryEntry(this._status); - - this.emit('completed', { - status: this._status, - stats: this.getProgress(), - }); - } - - /** - * Update history entry for current execution. - */ - private updateHistoryEntry(status: ExecutionStatus): void { - if (!this._executionId) return; - - const entry = this._history.find(e => e.id === this._executionId); - if (entry) { - entry.status = status; - entry.endedAt = Date.now(); - entry.completedTasks = this._scheduler.getStats().completedTasks; - entry.failedTasks = this._scheduler.getStats().failedTasks; - } - } - - /** - * Reset the bridge for a new execution. - */ - reset(): void { - this.stopExecutionLoop(); - - for (const timer of this._groupTimeoutTimers.values()) { - clearTimeout(timer); - } - this._groupTimeoutTimers.clear(); - - // Clear any pending retry timers - for (const timer of this._retryTimers.values()) { - clearTimeout(timer); - } - this._retryTimers.clear(); - - this._scheduler.reset(); - this._runningTasks.clear(); - this._status = 'idle'; - this._executionId = null; - this._startedAt = null; - this._pausedAt = null; - } -} - -// ========== Singleton ========== - -let bridgeInstance: ExecutionBridge | null = null; - -/** - * Get or create the singleton ExecutionBridge instance. - */ -export function getExecutionBridge(modelConfig?: Partial): ExecutionBridge { - if (!bridgeInstance) { - bridgeInstance = new ExecutionBridge(modelConfig); - } - return bridgeInstance; -} - -/** - * Reset the singleton (for testing). - */ -export function resetExecutionBridge(): void { - if (bridgeInstance) { - bridgeInstance.reset(); - } - bridgeInstance = null; -} diff --git a/src/group-scheduler.ts b/src/group-scheduler.ts deleted file mode 100644 index 23c74c95..00000000 --- a/src/group-scheduler.ts +++ /dev/null @@ -1,551 +0,0 @@ -/** - * @fileoverview Group Scheduler - Topological ordering and dependency management. - * - * Builds execution schedules from parallel groups, manages dependencies - * between groups, and tracks group-level completion status. - * - * @module group-scheduler - */ - -import { EventEmitter } from 'node:events'; -import type { ExecutionMode } from './model-selector.js'; -import type { ModelTier, AgentType } from './model-selector.js'; - -// ========== Types ========== - -/** Status of a task within a group */ -export type GroupTaskStatus = 'pending' | 'running' | 'completed' | 'failed' | 'blocked' | 'skipped'; - -/** Status of an execution group */ -export type ExecutionGroupStatus = 'pending' | 'ready' | 'running' | 'completed' | 'partial' | 'failed'; - -/** - * A task within an execution group. - */ -export interface GroupTask { - /** Task ID from the plan */ - id: string; - /** Task title/description */ - title: string; - /** Full task description */ - description: string; - /** Parallel group number */ - parallelGroup: number; - /** Agent type for model selection */ - agentType: AgentType; - /** Recommended model */ - recommendedModel?: ModelTier; - /** Whether task requires fresh context */ - requiresFreshContext: boolean; - /** Estimated token usage */ - estimatedTokens?: number; - /** Input files (read-only) */ - inputFiles?: string[]; - /** Output files (will be modified) */ - outputFiles?: string[]; - /** Current status */ - status: GroupTaskStatus; - /** Task dependencies (other task IDs) */ - dependencies: string[]; - /** Error message if failed */ - error?: string; - /** Retry count */ - retryCount: number; -} - -/** - * An execution group containing tasks that can run in parallel. - */ -export interface ExecutionGroup { - /** Group number (from parallelGroup) */ - groupNumber: number; - /** Tasks in this group */ - tasks: GroupTask[]; - /** Group status */ - status: ExecutionGroupStatus; - /** Execution mode for this group */ - executionMode: ExecutionMode; - /** Rationale for execution mode choice */ - executionModeRationale: string; - /** Groups that must complete before this one */ - dependsOnGroups: number[]; - /** When group started executing */ - startedAt?: number; - /** When group completed */ - completedAt?: number; - /** Number of tasks completed */ - completedCount: number; - /** Number of tasks failed */ - failedCount: number; - /** Number of tasks skipped due to dependency failures */ - skippedCount: number; -} - -/** - * Full execution schedule. - */ -export interface ExecutionSchedule { - /** Ordered groups (by dependency order) */ - groups: ExecutionGroup[]; - /** Total task count */ - totalTasks: number; - /** Completed task count */ - completedTasks: number; - /** Failed task count */ - failedTasks: number; - /** Current executing group index (-1 if not started) */ - currentGroupIndex: number; - /** Overall status */ - status: 'pending' | 'running' | 'completed' | 'partial' | 'failed'; -} - -// ========== Events ========== - -export interface GroupSchedulerEvents { - /** Schedule built from plan */ - scheduleBuilt: (schedule: ExecutionSchedule) => void; - /** Group started executing */ - groupStarted: (group: ExecutionGroup) => void; - /** Group completed (fully or partially) */ - groupCompleted: (group: ExecutionGroup) => void; - /** Task status changed */ - taskStatusChanged: (data: { taskId: string; groupNumber: number; oldStatus: GroupTaskStatus; newStatus: GroupTaskStatus }) => void; -} - -// ========== Group Scheduler ========== - -/** - * GroupScheduler - Manages execution order and dependencies. - * - * Responsibilities: - * - Build topologically ordered groups from plan items - * - Track group dependencies (lower groups must complete first) - * - Handle partial failures (continue with independent tasks) - * - Determine group execution mode (session vs task-tool) - */ -export class GroupScheduler extends EventEmitter { - private _schedule: ExecutionSchedule | null = null; - private _taskToGroup: Map = new Map(); - - constructor() { - super(); - } - - /** - * Get current schedule. - */ - get schedule(): ExecutionSchedule | null { - return this._schedule; - } - - /** - * Build execution schedule from plan items. - * - * @param items - Array of plan items with parallelGroup assignments - * @returns Built execution schedule - */ - buildSchedule(items: Array<{ - id: string; - title: string; - description: string; - parallelGroup?: number; - agentType?: string; - recommendedModel?: string; - requiresFreshContext?: boolean; - estimatedTokens?: number; - inputFiles?: string[]; - outputFiles?: string[]; - dependencies?: string[]; - }>): ExecutionSchedule { - // Group items by parallel group - const groupMap = new Map(); - - for (const item of items) { - const groupNum = item.parallelGroup ?? 0; - const task: GroupTask = { - id: item.id, - title: item.title, - description: item.description, - parallelGroup: groupNum, - agentType: (item.agentType as AgentType) ?? 'general', - recommendedModel: item.recommendedModel as ModelTier | undefined, - requiresFreshContext: item.requiresFreshContext ?? false, - estimatedTokens: item.estimatedTokens, - inputFiles: item.inputFiles, - outputFiles: item.outputFiles, - status: 'pending', - dependencies: item.dependencies ?? [], - retryCount: 0, - }; - - if (!groupMap.has(groupNum)) { - groupMap.set(groupNum, []); - } - groupMap.get(groupNum)!.push(task); - this._taskToGroup.set(item.id, groupNum); - } - - // Sort groups by number - const sortedGroupNums = Array.from(groupMap.keys()).sort((a, b) => a - b); - - // Build execution groups - const groups: ExecutionGroup[] = sortedGroupNums.map(groupNum => { - const tasks = groupMap.get(groupNum)!; - - // Determine which groups this one depends on - const dependsOnGroups = new Set(); - for (const task of tasks) { - for (const depId of task.dependencies) { - const depGroup = this._taskToGroup.get(depId); - if (depGroup !== undefined && depGroup !== groupNum && depGroup < groupNum) { - dependsOnGroups.add(depGroup); - } - } - } - - // Determine execution mode based on task characteristics - const { mode, rationale } = this.determineGroupExecutionMode(tasks); - - return { - groupNumber: groupNum, - tasks, - status: 'pending', - executionMode: mode, - executionModeRationale: rationale, - dependsOnGroups: Array.from(dependsOnGroups).sort((a, b) => a - b), - completedCount: 0, - failedCount: 0, - skippedCount: 0, - }; - }); - - this._schedule = { - groups, - totalTasks: items.length, - completedTasks: 0, - failedTasks: 0, - currentGroupIndex: -1, - status: 'pending', - }; - - this.emit('scheduleBuilt', this._schedule); - return this._schedule; - } - - /** - * Determine execution mode for a group based on task characteristics. - */ - private determineGroupExecutionMode(tasks: GroupTask[]): { mode: ExecutionMode; rationale: string } { - // High token estimate → session mode - const highTokenTask = tasks.find(t => t.estimatedTokens && t.estimatedTokens > 50000); - if (highTokenTask) { - return { - mode: 'session', - rationale: `Task ${highTokenTask.id} has high token estimate (${highTokenTask.estimatedTokens})`, - }; - } - - // Complex agent types → session mode - const complexTask = tasks.find(t => t.agentType === 'implement' || t.agentType === 'review'); - if (complexTask) { - return { - mode: 'session', - rationale: `Task ${complexTask.id} has complex agent type (${complexTask.agentType})`, - }; - } - - // Multiple output files in any task → session mode - const multiOutputTask = tasks.find(t => t.outputFiles && t.outputFiles.length > 2); - if (multiOutputTask) { - return { - mode: 'session', - rationale: `Task ${multiOutputTask.id} has multiple output files (${multiOutputTask.outputFiles!.length})`, - }; - } - - // Fresh context required → session mode - const freshContextTask = tasks.find(t => t.requiresFreshContext); - if (freshContextTask) { - return { - mode: 'session', - rationale: `Task ${freshContextTask.id} requires fresh context`, - }; - } - - // All low-token explore tasks → task-tool mode - const allLowToken = tasks.every(t => !t.estimatedTokens || t.estimatedTokens < 15000); - const allExplore = tasks.every(t => t.agentType === 'explore' || t.agentType === 'general'); - if (allLowToken && allExplore) { - return { - mode: 'task-tool', - rationale: 'All tasks are low-token explore/general tasks', - }; - } - - // Default to session mode for reliability - return { - mode: 'session', - rationale: 'Default to session mode for reliability', - }; - } - - /** - * Get the next group ready for execution. - */ - getNextReadyGroup(): ExecutionGroup | null { - if (!this._schedule) return null; - - for (const group of this._schedule.groups) { - if (group.status === 'pending' && this.areGroupDependenciesSatisfied(group)) { - group.status = 'ready'; - return group; - } - } - - return null; - } - - /** - * Check if a group's dependencies are satisfied. - */ - areGroupDependenciesSatisfied(group: ExecutionGroup): boolean { - if (!this._schedule) return false; - - for (const depGroupNum of group.dependsOnGroups) { - const depGroup = this._schedule.groups.find(g => g.groupNumber === depGroupNum); - if (!depGroup) continue; - - // Dependency must be completed (fully or partially) - if (depGroup.status !== 'completed' && depGroup.status !== 'partial') { - return false; - } - } - - return true; - } - - /** - * Mark a group as started. - */ - startGroup(groupNumber: number): void { - if (!this._schedule) return; - - const group = this._schedule.groups.find(g => g.groupNumber === groupNumber); - if (!group) return; - - group.status = 'running'; - group.startedAt = Date.now(); - this._schedule.status = 'running'; - this._schedule.currentGroupIndex = this._schedule.groups.indexOf(group); - - this.emit('groupStarted', group); - } - - /** - * Update task status within a group. - */ - updateTaskStatus(taskId: string, status: GroupTaskStatus, error?: string): void { - if (!this._schedule) return; - - const groupNum = this._taskToGroup.get(taskId); - if (groupNum === undefined) return; - - const group = this._schedule.groups.find(g => g.groupNumber === groupNum); - if (!group) return; - - const task = group.tasks.find(t => t.id === taskId); - if (!task) return; - - const oldStatus = task.status; - task.status = status; - if (error) task.error = error; - - // Update group counters - if (status === 'completed') { - group.completedCount++; - this._schedule.completedTasks++; - } else if (status === 'failed') { - group.failedCount++; - this._schedule.failedTasks++; - } else if (status === 'skipped') { - group.skippedCount++; - } - - this.emit('taskStatusChanged', { taskId, groupNumber: groupNum, oldStatus, newStatus: status }); - - // Check if group is complete - this.checkGroupCompletion(group); - } - - /** - * Mark tasks blocked by a failed dependency. - */ - markDependentTasksBlocked(failedTaskId: string): void { - if (!this._schedule) return; - - for (const group of this._schedule.groups) { - for (const task of group.tasks) { - if (task.dependencies.includes(failedTaskId) && task.status === 'pending') { - this.updateTaskStatus(task.id, 'skipped', `Blocked by failed task ${failedTaskId}`); - } - } - } - } - - /** - * Get tasks ready to execute in a group. - */ - getReadyTasksInGroup(groupNumber: number): GroupTask[] { - if (!this._schedule) return []; - - const group = this._schedule.groups.find(g => g.groupNumber === groupNumber); - if (!group) return []; - - return group.tasks.filter(task => { - if (task.status !== 'pending') return false; - - // Check if task dependencies are satisfied (within and across groups) - for (const depId of task.dependencies) { - // Check if dependency is in the same group - const sameGroupDep = group.tasks.find(t => t.id === depId); - if (sameGroupDep && sameGroupDep.status !== 'completed') { - return false; - } - - // Check if dependency is in a different group - const depGroupNum = this._taskToGroup.get(depId); - if (depGroupNum !== undefined && depGroupNum !== groupNumber) { - const depGroup = this._schedule!.groups.find(g => g.groupNumber === depGroupNum); - const depTask = depGroup?.tasks.find(t => t.id === depId); - if (depTask && depTask.status !== 'completed') { - return false; - } - } - } - - return true; - }); - } - - /** - * Check if a group has completed (all tasks done, failed, or skipped). - */ - private checkGroupCompletion(group: ExecutionGroup): void { - const pendingOrRunning = group.tasks.filter( - t => t.status === 'pending' || t.status === 'running' - ); - - if (pendingOrRunning.length > 0) return; - - group.completedAt = Date.now(); - - // Determine final status - if (group.failedCount === 0 && group.skippedCount === 0) { - group.status = 'completed'; - } else if (group.completedCount > 0) { - group.status = 'partial'; - } else { - group.status = 'failed'; - } - - this.emit('groupCompleted', group); - - // Check if all groups are done - this.checkScheduleCompletion(); - } - - /** - * Check if entire schedule has completed. - */ - private checkScheduleCompletion(): void { - if (!this._schedule) return; - - const pendingOrRunning = this._schedule.groups.filter( - g => g.status === 'pending' || g.status === 'ready' || g.status === 'running' - ); - - if (pendingOrRunning.length > 0) return; - - // Determine final status - const failedGroups = this._schedule.groups.filter(g => g.status === 'failed'); - const partialGroups = this._schedule.groups.filter(g => g.status === 'partial'); - - if (failedGroups.length === this._schedule.groups.length) { - this._schedule.status = 'failed'; - } else if (failedGroups.length > 0 || partialGroups.length > 0) { - this._schedule.status = 'partial'; - } else { - this._schedule.status = 'completed'; - } - } - - /** - * Get schedule statistics. - */ - getStats(): { - totalGroups: number; - completedGroups: number; - failedGroups: number; - partialGroups: number; - totalTasks: number; - completedTasks: number; - failedTasks: number; - skippedTasks: number; - } { - if (!this._schedule) { - return { - totalGroups: 0, - completedGroups: 0, - failedGroups: 0, - partialGroups: 0, - totalTasks: 0, - completedTasks: 0, - failedTasks: 0, - skippedTasks: 0, - }; - } - - return { - totalGroups: this._schedule.groups.length, - completedGroups: this._schedule.groups.filter(g => g.status === 'completed').length, - failedGroups: this._schedule.groups.filter(g => g.status === 'failed').length, - partialGroups: this._schedule.groups.filter(g => g.status === 'partial').length, - totalTasks: this._schedule.totalTasks, - completedTasks: this._schedule.completedTasks, - failedTasks: this._schedule.failedTasks, - skippedTasks: this._schedule.groups.reduce((sum, g) => sum + g.skippedCount, 0), - }; - } - - /** - * Reset the scheduler. - */ - reset(): void { - this._schedule = null; - this._taskToGroup.clear(); - } -} - -// ========== Singleton ========== - -let schedulerInstance: GroupScheduler | null = null; - -/** - * Get or create the singleton GroupScheduler instance. - */ -export function getGroupScheduler(): GroupScheduler { - if (!schedulerInstance) { - schedulerInstance = new GroupScheduler(); - } - return schedulerInstance; -} - -/** - * Reset the singleton (for testing). - */ -export function resetGroupScheduler(): void { - if (schedulerInstance) { - schedulerInstance.reset(); - } - schedulerInstance = null; -} diff --git a/src/model-selector.ts b/src/model-selector.ts deleted file mode 100644 index 6ead048a..00000000 --- a/src/model-selector.ts +++ /dev/null @@ -1,320 +0,0 @@ -/** - * @fileoverview Model Selector - Routes tasks to appropriate Claude models. - * - * Handles model selection based on: - * - User-configured default model - * - Optimizer recommendations (advisory) - * - Agent type mappings - * - Per-task overrides - * - * @module model-selector - */ - -import { EventEmitter } from 'node:events'; -import { - DEFAULT_MODEL, - TOKEN_THRESHOLD_HAIKU, - TOKEN_THRESHOLD_FOR_SESSION_MODE, -} from './config/execution-limits.js'; - -// ========== Types ========== - -/** Supported Claude model tiers */ -export type ModelTier = 'opus' | 'sonnet' | 'haiku'; - -/** Agent types that influence model selection */ -export type AgentType = 'explore' | 'implement' | 'test' | 'review' | 'general'; - -/** Execution mode for tasks */ -export type ExecutionMode = 'session' | 'task-tool'; - -/** - * User-configurable model settings. - * Stored in settings.json and editable via App Settings. - */ -export interface ModelConfig { - /** User's preferred default model */ - defaultModel: ModelTier; - /** Whether to show optimizer recommendations in UI (advisory only) */ - showRecommendations: boolean; - /** Override map for specific agent types */ - agentTypeOverrides: Partial>; -} - -/** - * Model selection result with reasoning. - */ -export interface ModelSelection { - /** The model to use */ - model: ModelTier; - /** Why this model was selected */ - reason: string; - /** What the optimizer recommended (if different) */ - optimizerRecommendation?: ModelTier; - /** Was user default used? */ - usedUserDefault: boolean; -} - -/** - * Execution mode selection result. - */ -export interface ExecutionModeSelection { - /** How to execute this task */ - mode: ExecutionMode; - /** Why this mode was selected */ - rationale: string; -} - -/** - * Task characteristics for selection decisions. - */ -export interface TaskCharacteristics { - /** Estimated token usage */ - estimatedTokens?: number; - /** Agent type */ - agentType?: AgentType; - /** Optimizer's recommended model */ - recommendedModel?: ModelTier; - /** Files task will modify */ - outputFiles?: string[]; - /** Files task will read */ - inputFiles?: string[]; - /** Task complexity hint */ - complexity?: 'low' | 'medium' | 'high'; -} - -// ========== Events ========== - -export interface ModelSelectorEvents { - /** Emitted when model is selected */ - modelSelected: (data: { taskId: string; selection: ModelSelection }) => void; - /** Emitted when config changes */ - configUpdated: (config: ModelConfig) => void; -} - -// ========== Default Configuration ========== - -/** - * Default optimizer recommendations by agent type. - * These are advisory - user default always wins. - */ -const DEFAULT_RECOMMENDATIONS: Record = { - explore: 'haiku', - implement: 'sonnet', - test: 'sonnet', - review: 'opus', - general: 'sonnet', -}; - -/** - * Creates default model configuration. - */ -export function createDefaultModelConfig(): ModelConfig { - return { - defaultModel: DEFAULT_MODEL, - showRecommendations: true, - agentTypeOverrides: {}, - }; -} - -// ========== Model Selector ========== - -/** - * ModelSelector - Manages model selection for task execution. - * - * User preferences always take precedence. Optimizer recommendations - * are shown in the UI for awareness but don't override user settings. - */ -export class ModelSelector extends EventEmitter { - private _config: ModelConfig; - - constructor(config?: Partial) { - super(); - this._config = { ...createDefaultModelConfig(), ...config }; - } - - /** - * Get current configuration. - */ - get config(): ModelConfig { - return { ...this._config }; - } - - /** - * Update configuration. - */ - updateConfig(config: Partial): void { - Object.assign(this._config, config); - this.emit('configUpdated', this._config); - } - - /** - * Select model for a task. - * - * Priority order: - * 1. User's agent type override (if set) - * 2. User's default model - * 3. Optimizer recommendation (only if no user preference) - */ - selectModel(taskId: string, characteristics: TaskCharacteristics): ModelSelection { - const { agentType, recommendedModel } = characteristics; - - let model: ModelTier; - let reason: string; - let usedUserDefault = false; - - // Check for agent type override - if (agentType && this._config.agentTypeOverrides[agentType]) { - model = this._config.agentTypeOverrides[agentType]!; - reason = `User override for ${agentType} tasks`; - } else { - // Use user's default model - model = this._config.defaultModel; - reason = 'User default model'; - usedUserDefault = true; - } - - // Determine what optimizer would have recommended - const optimizerRecommendation = recommendedModel || - (agentType ? DEFAULT_RECOMMENDATIONS[agentType] : undefined); - - const selection: ModelSelection = { - model, - reason, - usedUserDefault, - }; - - // Include optimizer recommendation if different (for UI display) - if (optimizerRecommendation && optimizerRecommendation !== model) { - selection.optimizerRecommendation = optimizerRecommendation; - if (this._config.showRecommendations) { - selection.reason += ` (optimizer suggested ${optimizerRecommendation})`; - } - } - - this.emit('modelSelected', { taskId, selection }); - return selection; - } - - /** - * Select execution mode for a task. - * - * Decision based on task characteristics: - * - High token estimate → session mode (needs full context) - * - Complex agent types → session mode (dedicated Claude) - * - Low token estimate → task-tool mode (efficient) - * - Multiple output files → session mode (avoid conflicts) - * - Read-only tasks → task-tool mode (no side effects) - */ - selectExecutionMode(characteristics: TaskCharacteristics): ExecutionModeSelection { - const { estimatedTokens, agentType, outputFiles, inputFiles, complexity } = characteristics; - - // High token estimate needs session mode - if (estimatedTokens && estimatedTokens > TOKEN_THRESHOLD_FOR_SESSION_MODE) { - return { - mode: 'session', - rationale: `High token estimate (${estimatedTokens} > ${TOKEN_THRESHOLD_FOR_SESSION_MODE})`, - }; - } - - // Complex agent types benefit from session mode - if (agentType === 'implement' || agentType === 'review') { - return { - mode: 'session', - rationale: `Complex agent type (${agentType}) benefits from dedicated context`, - }; - } - - // Multiple output files need session mode to avoid conflicts - if (outputFiles && outputFiles.length > 2) { - return { - mode: 'session', - rationale: `Multiple output files (${outputFiles.length}) - avoid conflicts`, - }; - } - - // High complexity tasks need session mode - if (complexity === 'high') { - return { - mode: 'session', - rationale: 'High complexity task needs dedicated context', - }; - } - - // Low token tasks can use task-tool mode - if (estimatedTokens && estimatedTokens < TOKEN_THRESHOLD_HAIKU) { - return { - mode: 'task-tool', - rationale: `Low token estimate (${estimatedTokens} < ${TOKEN_THRESHOLD_HAIKU})`, - }; - } - - // Explore tasks typically work well with task-tool - if (agentType === 'explore') { - return { - mode: 'task-tool', - rationale: 'Explore tasks efficient with shared context', - }; - } - - // Read-only tasks (input files only) can use task-tool - if (inputFiles && inputFiles.length > 0 && (!outputFiles || outputFiles.length === 0)) { - return { - mode: 'task-tool', - rationale: 'Read-only task (no output files)', - }; - } - - // Default to session mode for safety - return { - mode: 'session', - rationale: 'Default to session mode for reliability', - }; - } - - /** - * Get optimizer's recommendation for an agent type (for UI display). - */ - getOptimizerRecommendation(agentType: AgentType): ModelTier { - return DEFAULT_RECOMMENDATIONS[agentType]; - } - - /** - * Get model cost multiplier (relative to sonnet). - * Used for cost estimation in UI. - */ - getModelCostMultiplier(model: ModelTier): number { - switch (model) { - case 'opus': return 5.0; // ~5x more expensive - case 'sonnet': return 1.0; // baseline - case 'haiku': return 0.04; // ~25x cheaper - default: { - // Exhaustive check - if new ModelTier added, this will catch it - const _exhaustive: never = model; - console.warn(`[ModelSelector] Unknown model tier: ${_exhaustive}, using sonnet multiplier`); - return 1.0; - } - } - } -} - -// ========== Singleton ========== - -let selectorInstance: ModelSelector | null = null; - -/** - * Get or create the singleton ModelSelector instance. - */ -export function getModelSelector(config?: Partial): ModelSelector { - if (!selectorInstance) { - selectorInstance = new ModelSelector(config); - } - return selectorInstance; -} - -/** - * Reset the singleton (for testing). - */ -export function resetModelSelector(): void { - selectorInstance = null; -} diff --git a/src/plan-orchestrator.ts b/src/plan-orchestrator.ts index 54a1ef0d..7c2da24c 100644 --- a/src/plan-orchestrator.ts +++ b/src/plan-orchestrator.ts @@ -1,32 +1,27 @@ /** - * @fileoverview Enhanced plan generation using subagent orchestration. + * @fileoverview Simplified plan generation with 2 agents instead of 9. * - * This module implements a multi-phase plan generation system that leverages - * Claude's subagent capabilities for parallel analysis and verification. + * Previous architecture (removed): + * - 4 parallel analysis agents (requirements, architecture, testing, risks) + * - Synthesis phase + * - Verification agent + * - Execution optimizer agent (output was ignored anyway) + * - Final review agent * - * Architecture: - * 1. Phase 1 (Parallel Analysis): Spawn 4 specialist subagents simultaneously - * 2. Phase 2 (Synthesis): Merge and deduplicate outputs - * 3. Phase 3 (Verification): Final review and priority assignment + * New architecture: + * 1. Research Agent - gather context (optional, can fail) + * 2. Planner Agent - single agent generates complete TDD plan * - * @see https://code.claude.com/docs/en/sub-agents * @module plan-orchestrator */ import { Session } from './session.js'; import { ScreenManager } from './screen-manager.js'; -import { existsSync, mkdirSync, writeFileSync, readFileSync } from 'node:fs'; -import { join, dirname } from 'node:path'; +import { existsSync, mkdirSync, writeFileSync } from 'node:fs'; +import { join } from 'node:path'; import { RESEARCH_AGENT_PROMPT, - REQUIREMENTS_ANALYST_PROMPT, - ARCHITECTURE_PLANNER_PROMPT, - TESTING_SPECIALIST_PROMPT, - RISK_ANALYST_PROMPT, - CODE_REVIEWER_PROMPT, - VERIFICATION_PROMPT, - EXECUTION_OPTIMIZER_PROMPT, - FINAL_REVIEW_PROMPT, + PLANNER_PROMPT, } from './prompts/index.js'; // ============================================================================ @@ -40,78 +35,30 @@ export type PlanPhase = 'setup' | 'test' | 'impl' | 'verify' | 'review'; export type PlanTaskStatus = 'pending' | 'in_progress' | 'completed' | 'failed' | 'blocked'; /** - * Enhanced plan item with verification, dependencies, and execution tracking. - * Supports TDD workflow, failure tracking, and plan versioning. + * Plan item with TDD structure. */ export interface PlanItem { - /** Unique identifier (e.g., "P0-001") */ id?: string; - /** Task description */ content: string; - /** Criticality level */ priority: 'P0' | 'P1' | 'P2' | null; - /** Which subagent generated this item */ source?: string; - /** Why this task is needed */ rationale?: string; - /** Legacy numeric phase (1-4) */ - phase?: number; - - // === Verification === - /** How to know it's done (e.g., "npm test passes", "endpoint returns 200") */ verificationCriteria?: string; - /** Command to run for verification (e.g., "npm test -- --grep='auth'") */ testCommand?: string; - - // === Dependencies === - /** IDs of tasks that must complete first */ dependencies?: string[]; - - // === Execution tracking === - /** Current execution status */ status?: PlanTaskStatus; - /** How many times attempted */ attempts?: number; - /** Most recent failure reason */ lastError?: string; - /** Timestamp of completion */ completedAt?: number; - - // === Metadata === - /** Estimated complexity */ complexity?: 'low' | 'medium' | 'high'; - /** How to undo if needed */ - rollbackStrategy?: string; - /** Plan version this belongs to */ - version?: number; - /** TDD phase category */ tddPhase?: PlanPhase; - /** ID of paired test/impl task */ pairedWith?: string; - /** Checklist items for review tasks (tddPhase: 'review') */ reviewChecklist?: string[]; - - // === Claude Code Execution Optimization === - /** Group ID for tasks that can run in parallel */ - parallelGroup?: string; - /** Recommended Claude Code agent type */ - agentType?: 'explore' | 'implement' | 'test' | 'review' | 'general'; - /** Whether this task benefits from a fresh context */ - requiresFreshContext?: boolean; - /** Estimated token usage for this task */ - estimatedTokens?: number; - /** Recommended model for this task */ - recommendedModel?: 'opus' | 'sonnet' | 'haiku'; - /** Files this task will likely read */ - inputFiles?: string[]; - /** Files this task will likely modify */ - outputFiles?: string[]; } export interface ResearchResult { success: boolean; findings: { - /** External resources discovered (GitHub repos, docs, tutorials) */ externalResources: Array<{ type: 'github' | 'documentation' | 'tutorial' | 'article' | 'stackoverflow'; url?: string; @@ -119,79 +66,28 @@ export interface ResearchResult { relevance: string; keyInsights: string[]; }>; - /** Existing codebase patterns relevant to the task */ codebasePatterns: Array<{ pattern: string; location: string; relevance: string; }>; - /** Technical approach recommendations based on research */ technicalRecommendations: string[]; - /** Potential challenges identified from research */ potentialChallenges: string[]; - /** Libraries or tools recommended */ recommendedTools: Array<{ name: string; purpose: string; reason: string; }>; }; - /** Enhanced task description with research context */ enrichedTaskDescription: string; error?: string; durationMs: number; } -export interface SubagentResult { - agentType: 'requirements' | 'architecture' | 'testing' | 'risks'; - items: Array<{ - category: string; - content: string; - rationale?: string; - }>; - success: boolean; - error?: string; - durationMs: number; -} - -export interface SynthesisResult { +export interface PlannerResult { items: PlanItem[]; - stats: { - totalFromSubagents: number; - afterDedup: number; - sourceBreakdown: Record; - }; -} - -export interface VerificationResult { - validatedPlan: PlanItem[]; gaps: string[]; warnings: string[]; - qualityScore: number; -} - -export interface ParallelGroup { - id: string; - tasks: string[]; - rationale: string; - estimatedDuration?: string; - totalTokens?: number; -} - -export interface ExecutionStrategy { - totalParallelGroups: number; - sequentialBlockers: string[]; - freshContextPoints: string[]; - estimatedTotalTokens: number; - estimatedAgentSpawns: number; - criticalPath: string[]; - optimizationNotes: string[]; -} - -export interface ExecutionOptimizerResult { - optimizedPlan: PlanItem[]; - parallelGroups: ParallelGroup[]; - executionStrategy: ExecutionStrategy; } export interface DetailedPlanResult { @@ -200,26 +96,19 @@ export interface DetailedPlanResult { costUsd?: number; metadata?: { researchResult?: ResearchResult; - subagentResults: SubagentResult[]; - synthesisStats: SynthesisResult['stats']; - verificationGaps: string[]; - verificationWarnings: string[]; - qualityScore: number; + plannerGaps: string[]; + plannerWarnings: string[]; totalDurationMs: number; - parallelGroups?: ParallelGroup[]; - executionStrategy?: ExecutionStrategy; - finalReview?: FinalReviewResult; }; error?: string; } export type ProgressCallback = (phase: string, detail: string) => void; -/** Event types for plan subagent visibility */ export interface PlanSubagentEvent { type: 'started' | 'progress' | 'completed' | 'failed'; agentId: string; - agentType: 'research' | 'requirements' | 'architecture' | 'testing' | 'risks' | 'verification' | 'execution' | 'final-review'; + agentType: 'research' | 'planner'; model: string; status: string; detail?: string; @@ -228,92 +117,31 @@ export interface PlanSubagentEvent { error?: string; } -/** Result from the final review agent */ -export interface FinalReviewResult { - overallAssessment: 'ready' | 'needs-revision' | 'major-issues'; - scores: { - logic: number; - completeness: number; - coherence: number; - feasibility: number; - overall: number; - }; - summary: string; - issues: Array<{ - severity: 'warning' | 'error'; - issue: string; - affectedTasks: string[]; - suggestion: string; - }>; - missingTasks: Array<{ - content: string; - reason: string; - insertAfter?: string; - priority: 'P0' | 'P1' | 'P2'; - }>; - recommendations: string[]; -} - export type SubagentCallback = (event: PlanSubagentEvent) => void; // ============================================================================ // JSON Repair Helper // ============================================================================ -/** - * Attempt to parse JSON with automatic repair for common LLM output issues. - * Handles: trailing commas, unescaped newlines in strings, truncated output. - */ -// eslint-disable-next-line @typescript-eslint/no-explicit-any -function tryParseJSON(jsonString: string): { success: boolean; data?: any; error?: string } { - // First, try direct parse +function tryParseJSON(jsonString: string): { success: boolean; data?: unknown; error?: string } { try { return { success: true, data: JSON.parse(jsonString) }; } catch (firstError) { - // Try repairs let repaired = jsonString; - - // 1. Remove trailing commas before ] or } repaired = repaired.replace(/,(\s*[\]}])/g, '$1'); - - // 2. Fix unescaped newlines in strings (common LLM issue) - // Match strings and escape newlines within them repaired = repaired.replace(/"([^"\\]|\\.)*"/g, (match) => { return match.replace(/\n/g, '\\n').replace(/\r/g, '\\r').replace(/\t/g, '\\t'); }); - // 3. Try to close unclosed arrays/objects (truncated output) - const openBrackets = (repaired.match(/\[/g) || []).length; - const closeBrackets = (repaired.match(/\]/g) || []).length; const openBraces = (repaired.match(/\{/g) || []).length; const closeBraces = (repaired.match(/\}/g) || []).length; - - // Add missing closing brackets for (let i = 0; i < openBraces - closeBraces; i++) { - // Find last incomplete object and close it repaired = repaired.replace(/,\s*$/, '') + '}'; } - for (let i = 0; i < openBrackets - closeBrackets; i++) { - repaired = repaired.replace(/,\s*$/, '') + ']'; - } - // Try parsing repaired JSON try { return { success: true, data: JSON.parse(repaired) }; - } catch (secondError) { - // 4. Last resort: try to extract valid items from array - // Find all complete objects in the array - const objectMatches = jsonString.match(/\{[^{}]*\}/g); - if (objectMatches && objectMatches.length > 0) { - try { - const items = objectMatches.map(obj => JSON.parse(obj)); - console.warn(`[JSON Repair] Extracted ${items.length} items from malformed array`); - return { success: true, data: items }; - } catch { - // Give up - } - } - + } catch { const errMsg = firstError instanceof Error ? firstError.message : String(firstError); return { success: false, error: errMsg }; } @@ -324,16 +152,7 @@ function tryParseJSON(jsonString: string): { success: boolean; data?: any; error // Constants // ============================================================================ -const RESEARCH_TIMEOUT_MS = 600000; // 10 minutes for research (may need extensive web search) -const SUBAGENT_TIMEOUT_MS = 480000; // 8 minutes per subagent (Opus needs time for complex analysis) -const VERIFICATION_TIMEOUT_MS = 600000; // 10 minutes for verification (Opus + large plans) -const MODEL_RESEARCH = 'opus'; // Best model for research (needs reasoning for web search) -const MODEL_ANALYSIS = 'opus'; // Best model for thorough analysis -const MODEL_VERIFICATION = 'opus'; // Best model for verification - -// Note: Prompts are now in src/prompts/*.ts for easier editing -// Suppress unused variable warning for CODE_REVIEWER_PROMPT (reserved for future use) -void CODE_REVIEWER_PROMPT; +const MODEL = 'opus'; // ============================================================================ // Main Orchestrator Class @@ -353,690 +172,105 @@ export class PlanOrchestrator { this.outputDir = outputDir; } - /** - * Save agent prompt and result to the output directory. - * Creates a folder for each agent with prompt.md and result.json files. - */ - private saveAgentOutput( - agentType: string, - prompt: string, - result: unknown, - durationMs: number - ): void { + private saveAgentOutput(agentType: string, prompt: string, result: unknown, durationMs: number): void { if (!this.outputDir) return; try { - // Ensure output directory exists if (!existsSync(this.outputDir)) { mkdirSync(this.outputDir, { recursive: true }); } - // Create agent folder const agentDir = join(this.outputDir, agentType); if (!existsSync(agentDir)) { mkdirSync(agentDir, { recursive: true }); } - // Save prompt const promptPath = join(agentDir, 'prompt.md'); - const promptContent = `# ${agentType} Agent Prompt + writeFileSync(promptPath, `# ${agentType} Agent Prompt\n\nGenerated: ${new Date().toISOString()}\nDuration: ${(durationMs / 1000).toFixed(1)}s\n\n## Task\n${this.taskDescription}\n\n## Prompt\n${prompt}\n`, 'utf-8'); -Generated: ${new Date().toISOString()} -Duration: ${(durationMs / 1000).toFixed(1)}s - -## Task Description -${this.taskDescription} - -## Prompt -${prompt} -`; - writeFileSync(promptPath, promptContent, 'utf-8'); - - // Save result const resultPath = join(agentDir, 'result.json'); writeFileSync(resultPath, JSON.stringify(result, null, 2), 'utf-8'); - - console.log(`[PlanOrchestrator] Saved ${agentType} output to ${agentDir}`); } catch (err) { - console.error(`[PlanOrchestrator] Failed to save ${agentType} output:`, err); + console.warn(`[PlanOrchestrator] Failed to save ${agentType} output:`, err); } } - /** - * Save the final combined plan result. - */ private saveFinalResult(result: DetailedPlanResult): void { if (!this.outputDir) return; try { - // Ensure output directory exists if (!existsSync(this.outputDir)) { mkdirSync(this.outputDir, { recursive: true }); } - // Save final result + // Save final result JSON const resultPath = join(this.outputDir, 'final-result.json'); writeFileSync(resultPath, JSON.stringify(result, null, 2), 'utf-8'); - // Also save a human-readable summary - const summaryPath = join(this.outputDir, 'summary.md'); - const summary = this.generateReadableSummary(result); - writeFileSync(summaryPath, summary, 'utf-8'); - - console.log(`[PlanOrchestrator] Saved final result to ${this.outputDir}`); - - // Update the case CLAUDE.md with research context links - this.updateCaseClaudeMd(); - - // Update the project's main CLAUDE.md with Ralph Loop context - this.updateProjectClaudeMd(); + // Save human-readable summary + if (result.success && result.items) { + const summaryPath = join(this.outputDir, 'summary.md'); + const summary = this.generateSummary(result); + writeFileSync(summaryPath, summary, 'utf-8'); + } } catch (err) { - console.error('[PlanOrchestrator] Failed to save final result:', err); + console.warn('[PlanOrchestrator] Failed to save final result:', err); } } - /** - * Update the case folder's CLAUDE.md with links to research and analysis files. - * This allows Ralph Loop to `/init` and understand the available knowledge base. - */ - private updateCaseClaudeMd(): void { - if (!this.outputDir) return; + private generateSummary(result: DetailedPlanResult): string { + const items = result.items || []; + const p0 = items.filter(i => i.priority === 'P0'); + const p1 = items.filter(i => i.priority === 'P1'); + const p2 = items.filter(i => i.priority === 'P2'); - try { - // outputDir is like /path/to/case/ralph-wizard/ - // CLAUDE.md is in the parent (case folder) - const caseDir = dirname(this.outputDir); - const claudeMdPath = join(caseDir, 'CLAUDE.md'); + let md = `# Plan Summary\n\n`; + md += `Generated: ${new Date().toISOString()}\n`; + md += `Total Tasks: ${items.length} (P0: ${p0.length}, P1: ${p1.length}, P2: ${p2.length})\n\n`; - // Get relative path from case folder to ralph-wizard - const wizardRelPath = 'ralph-wizard'; + md += `## Task\n${this.taskDescription}\n\n`; - // Build the research context section - const contextSection = ` -## Ralph Wizard Knowledge Base - -The following research and analysis files were generated for this task. -Use \`/init\` to load this context, then read specific files as needed. - -### Research Phase -- \`${wizardRelPath}/research/result.json\` - External resources, codebase patterns, recommendations -- \`${wizardRelPath}/research/prompt.md\` - Research agent instructions - -### Analysis Phase -- \`${wizardRelPath}/requirements/result.json\` - Requirements analysis -- \`${wizardRelPath}/architecture/result.json\` - Architecture planning -- \`${wizardRelPath}/testing/result.json\` - Testing strategy -- \`${wizardRelPath}/risks/result.json\` - Risk analysis - -### Synthesis & Optimization -- \`${wizardRelPath}/verification/result.json\` - Plan verification and quality scoring -- \`${wizardRelPath}/execution-optimizer/result.json\` - Parallelization strategy -- \`${wizardRelPath}/final-review/result.json\` - Final review and completeness check - -### Summary -- \`${wizardRelPath}/summary.md\` - Human-readable plan summary -- \`${wizardRelPath}/final-result.json\` - Complete plan with all metadata - -**Key Research Insights**: Check \`${wizardRelPath}/research/result.json\` for: -- External GitHub repos and documentation links -- Existing codebase patterns to follow -- Technical recommendations from research -- Potential challenges to watch for -- Recommended tools and libraries -`; - - // Read existing CLAUDE.md or create new one - let existingContent = ''; - if (existsSync(claudeMdPath)) { - existingContent = readFileSync(claudeMdPath, 'utf-8'); - - // Remove any existing Ralph Wizard section to avoid duplicates - const sectionStart = existingContent.indexOf('## Ralph Wizard Knowledge Base'); - if (sectionStart !== -1) { - // Find the next ## heading or end of file - const nextSectionMatch = existingContent.slice(sectionStart + 1).match(/\n## /); - const sectionEnd = nextSectionMatch - ? sectionStart + 1 + nextSectionMatch.index! - : existingContent.length; - existingContent = existingContent.slice(0, sectionStart) + existingContent.slice(sectionEnd); - } + const addSection = (title: string, tasks: PlanItem[]) => { + if (tasks.length === 0) return; + md += `## ${title}\n\n`; + for (const t of tasks) { + const phase = t.tddPhase ? ` [${t.tddPhase}]` : ''; + md += `- **${t.id}**${phase}: ${t.content}\n`; + if (t.verificationCriteria) md += ` - Verify: ${t.verificationCriteria}\n`; } - - // Append the new section - const newContent = existingContent.trimEnd() + '\n' + contextSection; - writeFileSync(claudeMdPath, newContent, 'utf-8'); - - console.log(`[PlanOrchestrator] Updated CLAUDE.md with research context at ${claudeMdPath}`); - } catch (err) { - console.error('[PlanOrchestrator] Failed to update CLAUDE.md:', err); - } - } - - /** - * Update the project's main CLAUDE.md with Ralph Loop context. - * This ensures Claude instances running in the project directory know about the current task, - * where to find the plan files, and how to track todos. - */ - private updateProjectClaudeMd(): void { - if (!this.outputDir || !this.workingDir) return; - - try { - const projectClaudeMdPath = join(this.workingDir, 'CLAUDE.md'); - const caseDir = dirname(this.outputDir); - - // Build the Ralph Loop context section - const contextSection = ` -## Active Ralph Loop Task - -**Current Task**: ${this.taskDescription} - -**Case Folder**: \`${caseDir}\` - -### Key Files -- **Plan Summary**: \`${caseDir}/ralph-wizard/summary.md\` - Human-readable plan overview -- **Todo Items**: \`${caseDir}/ralph-wizard/final-result.json\` - Contains \`items\` array with all todo tasks -- **Research**: \`${caseDir}/ralph-wizard/research/result.json\` - External resources and codebase patterns - -### How to Work on This Task -1. Read the plan summary to understand the overall approach -2. Check \`final-result.json\` for the todo items array - each item has \`id\`, \`title\`, \`description\`, \`priority\` -3. Work through items in priority order (critical → high → medium → low) -4. Use \`COMPLETION_PHRASE\` when the entire task is complete - -### Research Insights -Check \`${caseDir}/ralph-wizard/research/result.json\` for: -- External GitHub repos and documentation links to reference -- Existing codebase patterns to follow -- Technical recommendations from the research phase -`; - - // Read existing CLAUDE.md or create new one - let existingContent = ''; - if (existsSync(projectClaudeMdPath)) { - existingContent = readFileSync(projectClaudeMdPath, 'utf-8'); - - // Remove any existing Ralph Loop section to avoid duplicates - const sectionStart = existingContent.indexOf('## Active Ralph Loop Task'); - if (sectionStart !== -1) { - // Find the next ## heading or end of file - const nextSectionMatch = existingContent.slice(sectionStart + 1).match(/\n## /); - const sectionEnd = nextSectionMatch - ? sectionStart + 1 + nextSectionMatch.index! - : existingContent.length; - existingContent = existingContent.slice(0, sectionStart) + existingContent.slice(sectionEnd); - } - } - - // Append the new section - const newContent = existingContent.trimEnd() + '\n' + contextSection; - writeFileSync(projectClaudeMdPath, newContent, 'utf-8'); - - console.log(`[PlanOrchestrator] Updated project CLAUDE.md with Ralph Loop context at ${projectClaudeMdPath}`); - } catch (err) { - console.error('[PlanOrchestrator] Failed to update project CLAUDE.md:', err); - } - } - - /** - * Generate a human-readable summary of the plan. - */ - private generateReadableSummary(result: DetailedPlanResult): string { - const lines: string[] = [ - '# Ralph Wizard Plan Summary', - '', - `Generated: ${new Date().toISOString()}`, - '', - '## Task Description', - this.taskDescription, - '', - ]; - - if (result.metadata?.qualityScore) { - lines.push(`## Quality Score: ${Math.round(result.metadata.qualityScore * 100)}%`); - lines.push(''); - } - - // Include research findings if available - if (result.metadata?.researchResult?.success) { - const research = result.metadata.researchResult; - lines.push('## Research Findings'); - lines.push(''); - - if (research.findings.externalResources.length > 0) { - lines.push('### External Resources'); - for (const resource of research.findings.externalResources) { - lines.push(`- **${resource.title}** (${resource.type})`); - if (resource.url) lines.push(` - URL: ${resource.url}`); - if (resource.keyInsights.length > 0) { - lines.push(` - Insights: ${resource.keyInsights.join('; ')}`); - } - } - lines.push(''); - } - - if (research.findings.technicalRecommendations.length > 0) { - lines.push('### Technical Recommendations'); - for (const rec of research.findings.technicalRecommendations) { - lines.push(`- ${rec}`); - } - lines.push(''); - } - - if (research.findings.potentialChallenges.length > 0) { - lines.push('### Potential Challenges'); - for (const challenge of research.findings.potentialChallenges) { - lines.push(`- ${challenge}`); - } - lines.push(''); - } - - if (research.findings.recommendedTools.length > 0) { - lines.push('### Recommended Tools'); - for (const tool of research.findings.recommendedTools) { - lines.push(`- **${tool.name}**: ${tool.purpose}`); - } - lines.push(''); - } - } - - if (result.metadata?.finalReview) { - const review = result.metadata.finalReview; - lines.push('## Final Review'); - lines.push(`- Assessment: ${review.overallAssessment}`); - lines.push(`- Logic: ${Math.round(review.scores.logic * 100)}%`); - lines.push(`- Completeness: ${Math.round(review.scores.completeness * 100)}%`); - lines.push(`- Coherence: ${Math.round(review.scores.coherence * 100)}%`); - lines.push(`- Feasibility: ${Math.round(review.scores.feasibility * 100)}%`); - lines.push(''); - if (review.summary) { - lines.push(review.summary); - lines.push(''); - } - } - - if (result.items) { - lines.push('## Plan Items'); - lines.push(''); - for (const item of result.items) { - const priority = item.priority || 'P1'; - const phase = item.tddPhase ? ` [${item.tddPhase}]` : ''; - lines.push(`### ${item.id || 'task'}: ${item.content}`); - lines.push(`- Priority: ${priority}${phase}`); - if (item.verificationCriteria) { - lines.push(`- Verification: ${item.verificationCriteria}`); - } - if (item.dependencies?.length) { - lines.push(`- Dependencies: ${item.dependencies.join(', ')}`); - } - if (item.parallelGroup) { - lines.push(`- Parallel Group: ${item.parallelGroup}`); - } - if (item.recommendedModel) { - lines.push(`- Recommended Model: ${item.recommendedModel}`); - } - lines.push(''); - } - } - - if (result.metadata?.executionStrategy) { - const strategy = result.metadata.executionStrategy; - lines.push('## Execution Strategy'); - lines.push(`- Parallel Groups: ${strategy.totalParallelGroups}`); - lines.push(`- Estimated Agent Spawns: ${strategy.estimatedAgentSpawns}`); - lines.push(`- Estimated Total Tokens: ${strategy.estimatedTotalTokens?.toLocaleString()}`); - lines.push(''); - if (strategy.optimizationNotes?.length) { - lines.push('### Optimization Notes'); - for (const note of strategy.optimizationNotes) { - lines.push(`- ${note}`); - } - lines.push(''); - } - } - - return lines.join('\n'); - } - - /** - * Format research findings as context for other agents. - */ - private formatResearchContext(research: ResearchResult): string { - if (!research.success) return ''; - - const lines: string[] = [ - '## RESEARCH FINDINGS (from prior research phase)', - '', - ]; - - // External resources - if (research.findings.externalResources.length > 0) { - lines.push('### External Resources Discovered'); - for (const resource of research.findings.externalResources) { - lines.push(`- **${resource.title}** (${resource.type})`); - if (resource.url) lines.push(` URL: ${resource.url}`); - lines.push(` Relevance: ${resource.relevance}`); - if (resource.keyInsights.length > 0) { - lines.push(` Key insights: ${resource.keyInsights.join('; ')}`); - } - } - lines.push(''); - } - - // Codebase patterns - if (research.findings.codebasePatterns.length > 0) { - lines.push('### Existing Codebase Patterns'); - for (const pattern of research.findings.codebasePatterns) { - lines.push(`- **${pattern.pattern}** at ${pattern.location}`); - lines.push(` Relevance: ${pattern.relevance}`); - } - lines.push(''); - } - - // Technical recommendations - if (research.findings.technicalRecommendations.length > 0) { - lines.push('### Technical Recommendations'); - for (const rec of research.findings.technicalRecommendations) { - lines.push(`- ${rec}`); - } - lines.push(''); - } - - // Potential challenges - if (research.findings.potentialChallenges.length > 0) { - lines.push('### Potential Challenges'); - for (const challenge of research.findings.potentialChallenges) { - lines.push(`- ${challenge}`); - } - lines.push(''); - } - - // Recommended tools - if (research.findings.recommendedTools.length > 0) { - lines.push('### Recommended Tools/Libraries'); - for (const tool of research.findings.recommendedTools) { - lines.push(`- **${tool.name}**: ${tool.purpose} (${tool.reason})`); - } - lines.push(''); - } - - return lines.join('\n'); - } - - /** - * Run the research agent to gather context before analysis. - */ - private async runResearchAgent( - taskDescription: string, - onProgress?: ProgressCallback, - onSubagent?: SubagentCallback - ): Promise { - const agentId = `plan-research-${Date.now()}`; - const startTime = Date.now(); - - // Default result if research fails - const defaultResult: ResearchResult = { - success: false, - findings: { - externalResources: [], - codebasePatterns: [], - technicalRecommendations: [], - potentialChallenges: [], - recommendedTools: [], - }, - enrichedTaskDescription: taskDescription, - error: 'Research skipped', - durationMs: 0, + md += '\n'; }; - // Check if already cancelled - if (this.cancelled) { - return { ...defaultResult, error: 'Cancelled' }; - } + addSection('P0 - Critical', p0); + addSection('P1 - Required', p1); + addSection('P2 - Enhancement', p2); - // Emit started event - onSubagent?.({ - type: 'started', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'running', - detail: 'Researching external resources and codebase patterns...', - }); - - const session = new Session({ - workingDir: this.workingDir, - screenManager: this.screenManager, - useScreen: false, - mode: 'claude', - }); - - // Track this session for cancellation - this.runningSessions.add(session); - - try { - const prompt = RESEARCH_AGENT_PROMPT - .replace('{TASK}', taskDescription) - .replace('{WORKING_DIR}', this.workingDir); - - onProgress?.('research', 'Starting research agent (local project + web search)...'); - - // Periodic progress updates showing elapsed time with contextual messages - const researchPhases = [ - { min: 0, msg: 'Reading CLAUDE.md and exploring project structure...' }, - { min: 30, msg: 'Analyzing existing code patterns and conventions...' }, - { min: 60, msg: 'Searching for relevant documentation and examples...' }, - { min: 120, msg: 'Synthesizing local and external findings...' }, - { min: 180, msg: 'Compiling research results (complex tasks take time)...' }, - { min: 300, msg: 'Almost done - finalizing enriched task description...' }, - ]; - - const progressInterval = setInterval(() => { - const elapsedSec = Math.floor((Date.now() - startTime) / 1000); - const timeoutSec = Math.floor(RESEARCH_TIMEOUT_MS / 1000); - - // Find appropriate phase message based on elapsed time - let phaseMsg = researchPhases[0].msg; - for (const phase of researchPhases) { - if (elapsedSec >= phase.min) { - phaseMsg = phase.msg; - } - } - - const progressMsg = `${phaseMsg} (${elapsedSec}s / ${timeoutSec}s)`; - onProgress?.('research', progressMsg); - onSubagent?.({ - type: 'progress', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'running', - detail: progressMsg, - }); - console.log(`[PlanOrchestrator] Research agent: ${elapsedSec}s - ${phaseMsg}`); - }, 15000); // Update every 15 seconds - - let result: string; - try { - const response = await Promise.race([ - session.runPrompt(prompt, { model: MODEL_RESEARCH }), - this.timeoutWithContext(RESEARCH_TIMEOUT_MS, 'Research agent', startTime), - ]); - result = response.result; - } finally { - clearInterval(progressInterval); - } - - // Check if cancelled during execution - if (this.cancelled) { - onSubagent?.({ - type: 'failed', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'cancelled', - error: 'Cancelled', - durationMs: Date.now() - startTime, - }); - return { ...defaultResult, error: 'Cancelled', durationMs: Date.now() - startTime }; - } - - // Parse JSON from result with repair for common LLM issues - const jsonMatch = result.match(/\{[\s\S]*\}/); - if (!jsonMatch) { - console.warn('[PlanOrchestrator] Research agent returned no JSON'); - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'completed', - detail: 'Research completed with limited findings', - durationMs: Date.now() - startTime, - }); - return { ...defaultResult, durationMs: Date.now() - startTime }; - } - - const parseResult = tryParseJSON(jsonMatch[0]); - if (!parseResult.success) { - console.warn('[PlanOrchestrator] Research agent JSON parse failed:', parseResult.error); - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'completed', - detail: 'Research completed with parse issues', - durationMs: Date.now() - startTime, - }); - return { ...defaultResult, error: parseResult.error, durationMs: Date.now() - startTime }; - } - - const parsed = parseResult.data; - - // Parse external resources - const externalResources = (Array.isArray(parsed.externalResources) ? parsed.externalResources : []).map((r: Record) => ({ - type: (['github', 'documentation', 'tutorial', 'article', 'stackoverflow'].includes(String(r.type)) - ? r.type : 'article') as 'github' | 'documentation' | 'tutorial' | 'article' | 'stackoverflow', - url: r.url ? String(r.url) : undefined, - title: String(r.title || 'Untitled'), - relevance: String(r.relevance || ''), - keyInsights: Array.isArray(r.keyInsights) ? r.keyInsights.map(String) : [], - })); - - // Parse codebase patterns - const codebasePatterns = (Array.isArray(parsed.codebasePatterns) ? parsed.codebasePatterns : []).map((p: Record) => ({ - pattern: String(p.pattern || ''), - location: String(p.location || ''), - relevance: String(p.relevance || ''), - })); - - // Parse other fields - const technicalRecommendations = Array.isArray(parsed.technicalRecommendations) - ? parsed.technicalRecommendations.map(String) - : []; - - const potentialChallenges = Array.isArray(parsed.potentialChallenges) - ? parsed.potentialChallenges.map(String) - : []; - - const recommendedTools = (Array.isArray(parsed.recommendedTools) ? parsed.recommendedTools : []).map((t: Record) => ({ - name: String(t.name || ''), - purpose: String(t.purpose || ''), - reason: String(t.reason || ''), - })); - - const enrichedTaskDescription = parsed.enrichedTaskDescription - ? String(parsed.enrichedTaskDescription) - : taskDescription; - - const durationMs = Date.now() - startTime; - - const researchResult: ResearchResult = { - success: true, - findings: { - externalResources, - codebasePatterns, - technicalRecommendations, - potentialChallenges, - recommendedTools, - }, - enrichedTaskDescription, - durationMs, - }; - - onProgress?.('research', `Research complete (${externalResources.length} resources, ${codebasePatterns.length} patterns)`); - - // Emit completed event - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'completed', - itemCount: externalResources.length + codebasePatterns.length, - durationMs, - }); - - // Save research output to file - this.saveAgentOutput('research', prompt, { ...researchResult, rawResponse: result }, durationMs); - - return researchResult; - } catch (err) { - const errorMsg = err instanceof Error ? err.message : String(err); - console.error('[PlanOrchestrator] Research agent failed:', errorMsg); - onSubagent?.({ - type: 'failed', - agentId, - agentType: 'research', - model: MODEL_RESEARCH, - status: 'failed', - error: errorMsg, - durationMs: Date.now() - startTime, - }); - return { ...defaultResult, error: errorMsg, durationMs: Date.now() - startTime }; - } finally { - // Remove from tracking and clean up - this.runningSessions.delete(session); - try { - await session.stop(); - } catch { - // Ignore cleanup errors + if (result.metadata?.plannerWarnings?.length) { + md += `## Warnings\n\n`; + for (const w of result.metadata.plannerWarnings) { + md += `- ${w}\n`; } } + + return md; } - /** - * Cancel all running subagent sessions. - * Call this when the client disconnects or user clicks Stop. - */ - async cancel(): Promise { + cancel(): void { this.cancelled = true; - console.log(`[PlanOrchestrator] Cancelling ${this.runningSessions.size} running sessions...`); - - const stopPromises = Array.from(this.runningSessions).map(async (session) => { + for (const session of this.runningSessions) { try { - await session.stop(); - } catch { - // Ignore cleanup errors - } - }); - - await Promise.all(stopPromises); + session.stop(); + } catch {} + } this.runningSessions.clear(); - console.log('[PlanOrchestrator] All sessions cancelled'); } /** - * Generate a detailed implementation plan using subagent orchestration. + * Generate a detailed TDD plan. * - * Phases: - * 0. Research - Gather external resources and codebase context - * 1. Spawn 4 specialist subagents in parallel for analysis - * 2. Synthesize their outputs into a unified plan - * 3. Run verification subagent for quality assurance - * 4. Ensure all impl tasks have review tasks - * 5. Run execution optimizer for Claude Code optimization - * 6. Run final review for holistic validation + * Flow: + * 1. Research Agent (optional - failures don't block) + * 2. Planner Agent (generates complete TDD plan) */ async generateDetailedPlan( taskDescription: string, @@ -1046,156 +280,53 @@ Check \`${caseDir}/ralph-wizard/research/result.json\` for: const startTime = Date.now(); let totalCost = 0; - // Store task description for file saving this.taskDescription = taskDescription; try { - // Phase 0: Research - Gather context - onProgress?.('research', 'Running research agent to gather context...'); + // Phase 1: Research (optional) + onProgress?.('research', 'Running research agent...'); const researchResult = await this.runResearchAgent(taskDescription, onProgress, onSubagent); + totalCost += researchResult.success ? 0.01 : 0; - totalCost += researchResult.success ? 0.01 : 0; // Research cost estimate - - // Use enriched task description if research was successful const effectiveTaskDescription = researchResult.success ? researchResult.enrichedTaskDescription : taskDescription; - // Format research context for other agents const researchContext = this.formatResearchContext(researchResult); - // Phase 1: Parallel Analysis (with research context) - onProgress?.('parallel-analysis', 'Spawning analysis subagents...'); - const subagentResults = await this.runParallelAnalysis( + // Phase 2: Planner (main agent) + onProgress?.('planning', 'Running planner agent...'); + const plannerResult = await this.runPlannerAgent( effectiveTaskDescription, - onProgress, - onSubagent, - researchContext - ); - - totalCost += subagentResults.reduce((sum, r) => sum + (r.success ? 0.002 : 0), 0); // Estimate - - // Check if we got enough results to continue - const successfulResults = subagentResults.filter(r => r.success); - if (successfulResults.length < 2) { - return { - success: false, - error: `Only ${successfulResults.length} subagents succeeded. Falling back to standard generation.`, - }; - } - - // Phase 2: Synthesis - onProgress?.('synthesis', 'Synthesizing subagent outputs...'); - const synthesisResult = this.synthesizeResults(subagentResults); - - // Phase 3: Verification (use enriched task description from research) - onProgress?.('verification', 'Running verification subagent...'); - const verificationResult = await this.runVerification( - effectiveTaskDescription, - synthesisResult.items, + researchContext, onProgress, onSubagent ); - totalCost += 0.01; // Verification cost estimate - - // Phase 4: Ensure all impl tasks have review tasks - onProgress?.('review-injection', 'Ensuring review tasks for all implementations...'); - const planWithReviews = this.ensureReviewTasks(verificationResult.validatedPlan); - const reviewsAdded = planWithReviews.length - verificationResult.validatedPlan.length; - if (reviewsAdded > 0) { - onProgress?.('review-injection', `Added ${reviewsAdded} auto-review task(s)`); - } - - // Phase 5: Execution Optimization for Claude Code (optional - failures don't block plan) - let executionResult: Awaited>; - try { - onProgress?.('execution-optimization', 'Running execution optimizer for Claude Code...'); - executionResult = await this.runExecutionOptimizer( - effectiveTaskDescription, - planWithReviews, - onProgress, - onSubagent - ); - totalCost += 0.008; // Execution optimizer cost estimate - } catch (err) { - const errMsg = err instanceof Error ? err.message : String(err); - console.warn('[PlanOrchestrator] Execution optimizer failed (continuing without):', errMsg); - onProgress?.('execution-optimization', `Skipped (${errMsg.slice(0, 50)})`); - executionResult = { + if (!plannerResult.success) { + return { success: false, - optimizedPlan: planWithReviews, - parallelGroups: [], - executionStrategy: { - totalParallelGroups: 0, - sequentialBlockers: [], - freshContextPoints: [], - estimatedTotalTokens: planWithReviews.length * 20000, - estimatedAgentSpawns: Math.ceil(planWithReviews.length / 3), - criticalPath: [], - optimizationNotes: [`Skipped due to error: ${errMsg}`], - }, + error: plannerResult.error || 'Planner failed', }; } - // Use optimized plan if successful, otherwise fall back to review plan - const optimizedPlan = executionResult.success ? executionResult.optimizedPlan : planWithReviews; - - // Phase 6: Final Review - Holistic validation (optional - failures don't block plan) - let finalReviewResult: FinalReviewResult; - try { - onProgress?.('final-review', 'Running final review for holistic validation...'); - finalReviewResult = await this.runFinalReview( - effectiveTaskDescription, - optimizedPlan, - onProgress, - onSubagent - ); - totalCost += 0.01; // Final review cost estimate - } catch (err) { - const errMsg = err instanceof Error ? err.message : String(err); - console.warn('[PlanOrchestrator] Final review failed (continuing without):', errMsg); - onProgress?.('final-review', `Skipped (${errMsg.slice(0, 50)})`); - finalReviewResult = { - overallAssessment: 'ready', - scores: { logic: 0.8, completeness: 0.8, coherence: 0.8, feasibility: 0.8, overall: 0.8 }, - summary: `Review skipped: ${errMsg}`, - issues: [], - missingTasks: [], - recommendations: [], - }; - } - - // Apply any missing tasks from final review - let finalPlan = optimizedPlan; - if (finalReviewResult.missingTasks.length > 0) { - finalPlan = this.applyMissingTasks(optimizedPlan, finalReviewResult.missingTasks); - onProgress?.('final-review', `Added ${finalReviewResult.missingTasks.length} missing task(s)`); - } + totalCost += 0.01; const totalDurationMs = Date.now() - startTime; const result: DetailedPlanResult = { success: true, - items: finalPlan, + items: plannerResult.items, costUsd: totalCost, metadata: { researchResult: researchResult.success ? researchResult : undefined, - subagentResults, - synthesisStats: synthesisResult.stats, - verificationGaps: verificationResult.gaps, - verificationWarnings: verificationResult.warnings, - qualityScore: finalReviewResult.scores.overall || verificationResult.qualityScore, + plannerGaps: plannerResult.gaps || [], + plannerWarnings: plannerResult.warnings || [], totalDurationMs, - parallelGroups: executionResult.parallelGroups, - executionStrategy: executionResult.executionStrategy, - finalReview: finalReviewResult, }, }; - // Save final result to output directory this.saveFinalResult(result); - return result; } catch (err) { return { @@ -1205,1250 +336,190 @@ Check \`${caseDir}/ralph-wizard/research/result.json\` for: } } - /** - * Run all 4 analysis subagents in parallel. - */ - private async runParallelAnalysis( + private formatResearchContext(research: ResearchResult): string { + if (!research.success) return ''; + + const parts: string[] = ['## Research Context\n']; + + if (research.findings.externalResources.length > 0) { + parts.push('### External Resources'); + for (const r of research.findings.externalResources.slice(0, 5)) { + parts.push(`- ${r.title}${r.url ? ` (${r.url})` : ''}`); + if (r.keyInsights.length > 0) { + parts.push(` Key insights: ${r.keyInsights.slice(0, 3).join(', ')}`); + } + } + parts.push(''); + } + + if (research.findings.codebasePatterns.length > 0) { + parts.push('### Existing Codebase Patterns'); + for (const p of research.findings.codebasePatterns.slice(0, 5)) { + parts.push(`- ${p.pattern} at ${p.location}`); + } + parts.push(''); + } + + if (research.findings.technicalRecommendations.length > 0) { + parts.push('### Recommendations'); + for (const r of research.findings.technicalRecommendations.slice(0, 5)) { + parts.push(`- ${r}`); + } + parts.push(''); + } + + return parts.join('\n'); + } + + private async runResearchAgent( taskDescription: string, + _onProgress?: ProgressCallback, + onSubagent?: SubagentCallback + ): Promise { + const agentId = `research-${Date.now()}`; + const startTime = Date.now(); + + if (this.cancelled) { + return { success: false, findings: { externalResources: [], codebasePatterns: [], technicalRecommendations: [], potentialChallenges: [], recommendedTools: [] }, enrichedTaskDescription: taskDescription, error: 'Cancelled', durationMs: 0 }; + } + + onSubagent?.({ type: 'started', agentId, agentType: 'research', model: MODEL, status: 'running', detail: 'Researching...' }); + + const session = new Session({ + workingDir: this.workingDir, + screenManager: this.screenManager, + useScreen: false, + mode: 'claude', + }); + + this.runningSessions.add(session); + + try { + const prompt = RESEARCH_AGENT_PROMPT.replace('{TASK}', taskDescription); + + const progressInterval = setInterval(() => { + const elapsed = Math.floor((Date.now() - startTime) / 1000); + onSubagent?.({ type: 'progress', agentId, agentType: 'research', model: MODEL, status: 'running', detail: `${elapsed}s elapsed` }); + }, 30000); + + const { result: response } = await session.runPrompt(prompt, { model: MODEL }); + + clearInterval(progressInterval); + this.runningSessions.delete(session); + + const durationMs = Date.now() - startTime; + + // Extract JSON from response + const jsonMatch = response.match(/\{[\s\S]*\}/); + if (!jsonMatch) { + onSubagent?.({ type: 'failed', agentId, agentType: 'research', model: MODEL, status: 'failed', error: 'No JSON found', durationMs }); + return { success: false, findings: { externalResources: [], codebasePatterns: [], technicalRecommendations: [], potentialChallenges: [], recommendedTools: [] }, enrichedTaskDescription: taskDescription, error: 'No JSON in response', durationMs }; + } + + const parsed = tryParseJSON(jsonMatch[0]); + if (!parsed.success) { + onSubagent?.({ type: 'failed', agentId, agentType: 'research', model: MODEL, status: 'failed', error: parsed.error, durationMs }); + return { success: false, findings: { externalResources: [], codebasePatterns: [], technicalRecommendations: [], potentialChallenges: [], recommendedTools: [] }, enrichedTaskDescription: taskDescription, error: parsed.error, durationMs }; + } + + const data = parsed.data as Record; + const result: ResearchResult = { + success: true, + findings: { + externalResources: Array.isArray(data.externalResources) ? data.externalResources : [], + codebasePatterns: Array.isArray(data.codebasePatterns) ? data.codebasePatterns : [], + technicalRecommendations: Array.isArray(data.technicalRecommendations) ? data.technicalRecommendations : [], + potentialChallenges: Array.isArray(data.potentialChallenges) ? data.potentialChallenges : [], + recommendedTools: Array.isArray(data.recommendedTools) ? data.recommendedTools : [], + }, + enrichedTaskDescription: typeof data.enrichedTaskDescription === 'string' ? data.enrichedTaskDescription : taskDescription, + durationMs, + }; + + this.saveAgentOutput('research', prompt, result, durationMs); + onSubagent?.({ type: 'completed', agentId, agentType: 'research', model: MODEL, status: 'completed', durationMs }); + + return result; + } catch (err) { + this.runningSessions.delete(session); + const durationMs = Date.now() - startTime; + const error = err instanceof Error ? err.message : String(err); + onSubagent?.({ type: 'failed', agentId, agentType: 'research', model: MODEL, status: 'failed', error, durationMs }); + return { success: false, findings: { externalResources: [], codebasePatterns: [], technicalRecommendations: [], potentialChallenges: [], recommendedTools: [] }, enrichedTaskDescription: taskDescription, error, durationMs }; + } + } + + private async runPlannerAgent( + taskDescription: string, + researchContext: string, onProgress?: ProgressCallback, - onSubagent?: SubagentCallback, - researchContext: string = '' - ): Promise { - // Inject research context into each prompt - const injectContext = (prompt: string) => - prompt + onSubagent?: SubagentCallback + ): Promise<{ success: boolean; items?: PlanItem[]; gaps?: string[]; warnings?: string[]; error?: string }> { + const agentId = `planner-${Date.now()}`; + const startTime = Date.now(); + + if (this.cancelled) { + return { success: false, error: 'Cancelled' }; + } + + onSubagent?.({ type: 'started', agentId, agentType: 'planner', model: MODEL, status: 'running', detail: 'Generating plan...' }); + + const session = new Session({ + workingDir: this.workingDir, + screenManager: this.screenManager, + useScreen: false, + mode: 'claude', + }); + + this.runningSessions.add(session); + + try { + const prompt = PLANNER_PROMPT .replace('{TASK}', taskDescription) .replace('{RESEARCH_CONTEXT}', researchContext || ''); - const subagents: Array<{ - type: SubagentResult['agentType']; - prompt: string; - }> = [ - { type: 'requirements', prompt: injectContext(REQUIREMENTS_ANALYST_PROMPT) }, - { type: 'architecture', prompt: injectContext(ARCHITECTURE_PLANNER_PROMPT) }, - { type: 'testing', prompt: injectContext(TESTING_SPECIALIST_PROMPT) }, - { type: 'risks', prompt: injectContext(RISK_ANALYST_PROMPT) }, - ]; - - // Run all subagents in parallel - const promises = subagents.map(({ type, prompt }) => - this.runSubagent(type, prompt, onProgress, onSubagent) - ); - - return Promise.all(promises); - } - - /** - * Run a single analysis subagent. - */ - private async runSubagent( - agentType: SubagentResult['agentType'], - prompt: string, - onProgress?: ProgressCallback, - onSubagent?: SubagentCallback - ): Promise { - const agentId = `plan-${agentType}-${Date.now()}`; - - // Check if already cancelled - if (this.cancelled) { - return { - agentType, - items: [], - success: false, - error: 'Cancelled', - durationMs: 0, - }; - } - - const startTime = Date.now(); - - // Emit started event - onSubagent?.({ - type: 'started', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'running', - detail: `Analyzing ${agentType}...`, - }); - - const session = new Session({ - workingDir: this.workingDir, - screenManager: this.screenManager, - useScreen: false, - mode: 'claude', - }); - - // Track this session for cancellation - this.runningSessions.add(session); - - try { - onProgress?.('subagent', `Running ${agentType} analysis...`); - - // Periodic progress updates showing elapsed time const progressInterval = setInterval(() => { - const elapsedSec = Math.floor((Date.now() - startTime) / 1000); - const timeoutSec = Math.floor(SUBAGENT_TIMEOUT_MS / 1000); - console.log(`[PlanOrchestrator] ${agentType} agent: ${elapsedSec}s elapsed (timeout: ${timeoutSec}s)`); - onSubagent?.({ - type: 'progress', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'running', - detail: `${agentType} analysis... (${elapsedSec}s / ${timeoutSec}s)`, - }); - }, 30000); // Update every 30 seconds + const elapsed = Math.floor((Date.now() - startTime) / 1000); + onSubagent?.({ type: 'progress', agentId, agentType: 'planner', model: MODEL, status: 'running', detail: `${elapsed}s elapsed` }); + }, 30000); - let result: string; - try { - const response = await Promise.race([ - session.runPrompt(prompt, { model: MODEL_ANALYSIS }), - this.timeoutWithContext(SUBAGENT_TIMEOUT_MS, `${agentType} agent`, startTime), - ]); - result = response.result; - } finally { - clearInterval(progressInterval); - } + const { result: response } = await session.runPrompt(prompt, { model: MODEL }); - // Check if cancelled during execution - if (this.cancelled) { - onSubagent?.({ - type: 'failed', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'cancelled', - error: 'Cancelled', - durationMs: Date.now() - startTime, - }); - return { - agentType, - items: [], - success: false, - error: 'Cancelled', - durationMs: Date.now() - startTime, - }; - } - - // Parse JSON from result with repair for common LLM issues - const jsonMatch = result.match(/\[[\s\S]*\]/); - if (!jsonMatch) { - onSubagent?.({ - type: 'failed', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'failed', - error: 'No JSON array found in response', - durationMs: Date.now() - startTime, - }); - return { - agentType, - items: [], - success: false, - error: 'No JSON array found in response', - durationMs: Date.now() - startTime, - }; - } - - const parseResult = tryParseJSON(jsonMatch[0]); - if (!parseResult.success) { - console.error(`[PlanOrchestrator] ${agentType} JSON parse failed:`, parseResult.error); - onSubagent?.({ - type: 'failed', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'failed', - error: `JSON parse error: ${parseResult.error}`, - durationMs: Date.now() - startTime, - }); - return { - agentType, - items: [], - success: false, - error: `JSON parse error: ${parseResult.error}`, - durationMs: Date.now() - startTime, - }; - } - - const parsed = parseResult.data; - if (!Array.isArray(parsed)) { - onSubagent?.({ - type: 'failed', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'failed', - error: 'Response is not an array', - durationMs: Date.now() - startTime, - }); - return { - agentType, - items: [], - success: false, - error: 'Response is not an array', - durationMs: Date.now() - startTime, - }; - } - - const items = parsed.map((item: unknown) => { - if (typeof item !== 'object' || item === null) { - return { category: 'unknown', content: String(item) }; - } - const obj = item as Record; - return { - category: String(obj.category || 'general'), - content: String(obj.content || ''), - rationale: obj.rationale ? String(obj.rationale) : undefined, - }; - }); + clearInterval(progressInterval); + this.runningSessions.delete(session); const durationMs = Date.now() - startTime; - onProgress?.('subagent', `${agentType} complete (${items.length} items)`); - - // Emit completed event - onSubagent?.({ - type: 'completed', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'completed', - itemCount: items.length, - durationMs, - }); - - // Save agent output to file - this.saveAgentOutput(agentType, prompt, { items, rawResponse: result }, durationMs); - - return { - agentType, - items, - success: true, - durationMs, - }; - } catch (err) { - const errorMsg = err instanceof Error ? err.message : String(err); - onSubagent?.({ - type: 'failed', - agentId, - agentType, - model: MODEL_ANALYSIS, - status: 'failed', - error: errorMsg, - durationMs: Date.now() - startTime, - }); - return { - agentType, - items: [], - success: false, - error: errorMsg, - durationMs: Date.now() - startTime, - }; - } finally { - // Remove from tracking and clean up - this.runningSessions.delete(session); - try { - await session.stop(); - } catch { - // Ignore cleanup errors - } - } - } - - /** - * Synthesize results from all subagents into a unified plan. - */ - private synthesizeResults(subagentResults: SubagentResult[]): SynthesisResult { - const allItems: PlanItem[] = []; - const sourceBreakdown: Record = {}; - - // Collect items from all successful subagents - for (const result of subagentResults) { - if (!result.success) continue; - - sourceBreakdown[result.agentType] = result.items.length; - - for (const item of result.items) { - allItems.push({ - content: item.content, - priority: null, // Will be assigned by verification - source: result.agentType, - rationale: item.rationale, - phase: this.determinePhase(item.content, result.agentType), - }); - } - } - - const totalFromSubagents = allItems.length; - - // Deduplicate similar items - const deduped = this.deduplicateItems(allItems); - - // Sort by phase - deduped.sort((a, b) => (a.phase || 4) - (b.phase || 4)); - - return { - items: deduped, - stats: { - totalFromSubagents, - afterDedup: deduped.length, - sourceBreakdown, - }, - }; - } - - /** - * Determine the implementation phase for an item based on keywords. - */ - private determinePhase(content: string, source: string): number { - const lower = content.toLowerCase(); - - // Phase 1: Foundation - if ( - lower.includes('create') && (lower.includes('project') || lower.includes('directory')) || - lower.includes('setup') || - lower.includes('initialize') || - lower.includes('configure') || - lower.includes('install') || - lower.includes('define') && (lower.includes('interface') || lower.includes('type')) - ) { - return 1; - } - - // Phase 2: Tests (before implementation) - if ( - source === 'testing' || - lower.includes('test') || - lower.includes('verify') && !lower.includes('final') - ) { - return 2; - } - - // Phase 3: Implementation - if ( - lower.includes('implement') || - lower.includes('build') || - lower.includes('add') || - lower.includes('create') && !lower.includes('test') - ) { - return 3; - } - - // Phase 4: Integration & Verification - if ( - lower.includes('integrate') || - lower.includes('connect') || - lower.includes('final') || - lower.includes('run') && lower.includes('suite') - ) { - return 4; - } - - return 3; // Default to implementation phase - } - - /** - * Deduplicate similar items using fuzzy matching. - */ - private deduplicateItems(items: PlanItem[]): PlanItem[] { - const result: PlanItem[] = []; - - for (const item of items) { - const isDuplicate = result.some(existing => - this.isSimilar(existing.content, item.content) - ); - - if (!isDuplicate) { - result.push(item); - } - } - - return result; - } - - /** - * Check if two strings are similar (>60% word overlap). - */ - private isSimilar(a: string, b: string): boolean { - const wordsA = new Set(a.toLowerCase().split(/\s+/).filter(w => w.length > 3)); - const wordsB = new Set(b.toLowerCase().split(/\s+/).filter(w => w.length > 3)); - - if (wordsA.size === 0 || wordsB.size === 0) return false; - - let overlap = 0; - for (const word of wordsA) { - if (wordsB.has(word)) overlap++; - } - - const similarity = overlap / Math.min(wordsA.size, wordsB.size); - return similarity > 0.6; - } - - /** - * Run the verification subagent to validate and prioritize the plan. - */ - private async runVerification( - taskDescription: string, - synthesizedItems: PlanItem[], - onProgress?: ProgressCallback, - onSubagent?: SubagentCallback - ): Promise { - const agentId = `plan-verification-${Date.now()}`; - - // Check if already cancelled - if (this.cancelled) { - return this.fallbackVerification(synthesizedItems); - } - - // Emit started event - onSubagent?.({ - type: 'started', - agentId, - agentType: 'verification', - model: MODEL_VERIFICATION, - status: 'running', - detail: 'Validating and prioritizing plan...', - }); - - const startTime = Date.now(); - - const session = new Session({ - workingDir: this.workingDir, - screenManager: this.screenManager, - useScreen: false, - mode: 'claude', - }); - - // Track this session for cancellation - this.runningSessions.add(session); - - try { - // Format plan for verification - const planText = synthesizedItems - .map((item, idx) => `${idx + 1}. [Phase ${item.phase}] ${item.content}`) - .join('\n'); - - const prompt = VERIFICATION_PROMPT - .replace('{TASK}', taskDescription) - .replace('{PLAN}', planText); - - onProgress?.('verification', 'Validating plan quality...'); - - // Periodic progress updates showing elapsed time - const progressInterval = setInterval(() => { - const elapsedSec = Math.floor((Date.now() - startTime) / 1000); - const timeoutSec = Math.floor(VERIFICATION_TIMEOUT_MS / 1000); - console.log(`[PlanOrchestrator] verification agent: ${elapsedSec}s elapsed (timeout: ${timeoutSec}s)`); - onSubagent?.({ - type: 'progress', - agentId, - agentType: 'verification', - model: MODEL_VERIFICATION, - status: 'running', - detail: `Verifying plan... (${elapsedSec}s / ${timeoutSec}s)`, - }); - }, 30000); // Update every 30 seconds - - let result: string; - try { - const response = await Promise.race([ - session.runPrompt(prompt, { model: MODEL_VERIFICATION }), - this.timeoutWithContext(VERIFICATION_TIMEOUT_MS, 'verification agent', startTime), - ]); - result = response.result; - } finally { - clearInterval(progressInterval); - } - - // Check if cancelled during execution - if (this.cancelled) { - return this.fallbackVerification(synthesizedItems); - } - - // Parse JSON from result with repair for common LLM issues - const jsonMatch = result.match(/\{[\s\S]*\}/); + // Extract JSON from response + const jsonMatch = response.match(/\{[\s\S]*\}/); if (!jsonMatch) { - // Fallback: return items with default priorities - return this.fallbackVerification(synthesizedItems); + onSubagent?.({ type: 'failed', agentId, agentType: 'planner', model: MODEL, status: 'failed', error: 'No JSON found', durationMs }); + return { success: false, error: 'No JSON in response' }; } - const parseResult = tryParseJSON(jsonMatch[0]); - if (!parseResult.success) { - console.warn('[PlanOrchestrator] Verification JSON parse failed:', parseResult.error); - return this.fallbackVerification(synthesizedItems); + const parsed = tryParseJSON(jsonMatch[0]); + if (!parsed.success) { + onSubagent?.({ type: 'failed', agentId, agentType: 'planner', model: MODEL, status: 'failed', error: parsed.error, durationMs }); + return { success: false, error: parsed.error }; } - const parsed = parseResult.data; + const data = parsed.data as Record; + const items: PlanItem[] = Array.isArray(data.items) ? data.items : []; + const gaps: string[] = Array.isArray(data.gaps) ? data.gaps : []; + const warnings: string[] = Array.isArray(data.warnings) ? data.warnings : []; - const validatedPlan: PlanItem[] = (parsed.validatedPlan || []).map((item: unknown, idx: number) => { - if (typeof item !== 'object' || item === null) { - return { content: String(item), priority: 'P1' as const, id: `task-${idx}` }; - } - const obj = item as Record; - let priority: PlanItem['priority'] = null; - if (obj.priority === 'P0' || obj.priority === 'P1' || obj.priority === 'P2') { - priority = obj.priority; - } + this.saveAgentOutput('planner', prompt, { items, gaps, warnings }, durationMs); + onSubagent?.({ type: 'completed', agentId, agentType: 'planner', model: MODEL, status: 'completed', itemCount: items.length, durationMs }); - // Parse TDD phase (now includes 'review') - let tddPhase: PlanItem['tddPhase']; - if (obj.tddPhase === 'setup' || obj.tddPhase === 'test' || obj.tddPhase === 'impl' || obj.tddPhase === 'verify' || obj.tddPhase === 'review') { - tddPhase = obj.tddPhase; - } + onProgress?.('planning', `Generated ${items.length} tasks`); - // Parse complexity - let complexity: PlanItem['complexity']; - if (obj.complexity === 'low' || obj.complexity === 'medium' || obj.complexity === 'high') { - complexity = obj.complexity; - } - - // Parse reviewChecklist for review tasks - let reviewChecklist: string[] | undefined; - if (Array.isArray(obj.reviewChecklist)) { - reviewChecklist = obj.reviewChecklist.map(String); - } - - return { - id: obj.id ? String(obj.id) : `task-${idx}`, - content: String(obj.content || ''), - priority, - rationale: obj.rationale ? String(obj.rationale) : undefined, - // Enhanced fields - verificationCriteria: obj.verificationCriteria ? String(obj.verificationCriteria) : undefined, - testCommand: obj.testCommand ? String(obj.testCommand) : undefined, - tddPhase, - pairedWith: obj.pairedWith ? String(obj.pairedWith) : undefined, - dependencies: Array.isArray(obj.dependencies) ? obj.dependencies.map(String) : undefined, - complexity, - reviewChecklist, - // Execution tracking defaults - status: 'pending' as PlanTaskStatus, - attempts: 0, - version: 1, - }; - }); - - const durationMs = Date.now() - startTime; - - onProgress?.('verification', `Verification complete (quality: ${Math.round((parsed.qualityScore || 0.8) * 100)}%)`); - - // Emit completed event - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'verification', - model: MODEL_VERIFICATION, - status: 'completed', - itemCount: validatedPlan.length, - durationMs, - }); - - const verificationResult = { - validatedPlan, - gaps: Array.isArray(parsed.gaps) ? parsed.gaps.map(String) : [], - warnings: Array.isArray(parsed.warnings) ? parsed.warnings.map(String) : [], - qualityScore: typeof parsed.qualityScore === 'number' ? parsed.qualityScore : 0.8, - }; - - // Save agent output to file - this.saveAgentOutput('verification', prompt, { ...verificationResult, rawResponse: result }, durationMs); - - return verificationResult; + return { success: true, items, gaps, warnings }; } catch (err) { - console.error('[PlanOrchestrator] Verification failed:', err); - onSubagent?.({ - type: 'failed', - agentId, - agentType: 'verification', - model: MODEL_VERIFICATION, - status: 'failed', - error: err instanceof Error ? err.message : String(err), - durationMs: Date.now() - startTime, - }); - return this.fallbackVerification(synthesizedItems); - } finally { - // Remove from tracking and clean up this.runningSessions.delete(session); - try { - await session.stop(); - } catch { - // Ignore cleanup errors - } - } - } - - /** - * Fallback verification when the verification subagent fails. - * Assigns heuristic priorities and adds basic verification criteria. - * Note: This is a silent fallback - no warning shown to user since the result is still useful. - */ - private fallbackVerification(items: PlanItem[]): VerificationResult { - console.log('[PlanOrchestrator] Using heuristic priorities (verification subagent timed out or failed)'); - return { - validatedPlan: items.map((item, idx) => ({ - ...item, - id: item.id || `task-${idx}`, - priority: item.phase === 1 ? 'P0' as const : - item.phase === 4 ? 'P2' as const : 'P1' as const, - // Add default verification criteria based on content - verificationCriteria: item.verificationCriteria || - this.inferVerificationCriteria(item.content), - status: 'pending' as PlanTaskStatus, - attempts: 0, - version: 1, - })), - gaps: [], - warnings: [], // Silent fallback - heuristics are good enough, no need to alarm user - qualityScore: 0.75, // Slightly higher since fallback still produces useful results - }; - } - - /** - * Infer verification criteria from task content. - */ - private inferVerificationCriteria(content: string): string { - const lower = content.toLowerCase(); - - if (lower.includes('test')) { - return 'Tests pass without errors'; - } - if (lower.includes('implement') || lower.includes('create') || lower.includes('add')) { - return 'Code compiles, no type errors'; - } - if (lower.includes('fix') || lower.includes('debug')) { - return 'Issue is resolved, tests pass'; - } - if (lower.includes('refactor')) { - return 'Code refactored, all tests still pass'; - } - if (lower.includes('document') || lower.includes('readme')) { - return 'Documentation exists and is accurate'; - } - if (lower.includes('config') || lower.includes('setup')) { - return 'Configuration is valid, app starts'; - } - - return 'Task completed successfully'; - } - - /** - * Run the execution optimizer to enhance the plan for Claude Code execution. - * Adds parallel groups, agent type recommendations, and fresh context points. - */ - private async runExecutionOptimizer( - taskDescription: string, - items: PlanItem[], - onProgress?: ProgressCallback, - onSubagent?: SubagentCallback - ): Promise<{ - success: boolean; - optimizedPlan: PlanItem[]; - parallelGroups: ParallelGroup[]; - executionStrategy: ExecutionStrategy; - }> { - const agentId = `plan-execution-${Date.now()}`; - const startTime = Date.now(); - - // Default fallback result - const defaultResult = { - success: false, - optimizedPlan: items, - parallelGroups: [], - executionStrategy: { - totalParallelGroups: 0, - sequentialBlockers: [], - freshContextPoints: [], - estimatedTotalTokens: items.length * 20000, - estimatedAgentSpawns: Math.ceil(items.length / 3), - criticalPath: items.slice(0, 5).map(i => i.id || 'unknown'), - optimizationNotes: ['Fallback: no optimization applied'], - }, - }; - - // Check if already cancelled - if (this.cancelled) { - return defaultResult; - } - - // Emit started event - onSubagent?.({ - type: 'started', - agentId, - agentType: 'execution', - model: MODEL_VERIFICATION, - status: 'running', - detail: 'Optimizing plan for Claude Code execution...', - }); - - const session = new Session({ - workingDir: this.workingDir, - screenManager: this.screenManager, - useScreen: false, - mode: 'claude', - }); - - // Track this session for cancellation - this.runningSessions.add(session); - - try { - // Format plan for optimization - const planText = JSON.stringify(items, null, 2); - - const prompt = EXECUTION_OPTIMIZER_PROMPT - .replace('{TASK}', taskDescription) - .replace('{PLAN}', planText); - - onProgress?.('execution-optimization', 'Analyzing parallelization opportunities...'); - - // Periodic progress updates showing elapsed time - const progressInterval = setInterval(() => { - const elapsedSec = Math.floor((Date.now() - startTime) / 1000); - const timeoutSec = Math.floor(VERIFICATION_TIMEOUT_MS / 1000); - console.log(`[PlanOrchestrator] execution optimizer: ${elapsedSec}s elapsed (timeout: ${timeoutSec}s)`); - onSubagent?.({ - type: 'progress', - agentId, - agentType: 'execution', - model: MODEL_VERIFICATION, - status: 'running', - detail: `Optimizing execution... (${elapsedSec}s / ${timeoutSec}s)`, - }); - }, 30000); // Update every 30 seconds - - let result: string; - try { - const response = await Promise.race([ - session.runPrompt(prompt, { model: MODEL_VERIFICATION }), - this.timeoutWithContext(VERIFICATION_TIMEOUT_MS, 'execution optimizer', startTime), - ]); - result = response.result; - } finally { - clearInterval(progressInterval); - } - - // Check if cancelled during execution - if (this.cancelled) { - return defaultResult; - } - - // Parse JSON from result - const jsonMatch = result.match(/\{[\s\S]*\}/); - if (!jsonMatch) { - console.warn('[PlanOrchestrator] Execution optimizer returned no JSON, using defaults'); - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'execution', - model: MODEL_VERIFICATION, - status: 'completed', - detail: 'Using default execution strategy', - itemCount: items.length, - durationMs: Date.now() - startTime, - }); - return defaultResult; - } - - const parseResult = tryParseJSON(jsonMatch[0]); - if (!parseResult.success) { - console.warn('[PlanOrchestrator] Execution optimizer JSON parse failed:', parseResult.error); - return defaultResult; - } - - const parsed = parseResult.data; - - // Parse optimized plan items - const optimizedPlan: PlanItem[] = (parsed.optimizedPlan || []).map((item: Record, idx: number) => { - // Parse agent type - let agentType: PlanItem['agentType']; - if (['explore', 'implement', 'test', 'review', 'general'].includes(String(item.agentType))) { - agentType = item.agentType as PlanItem['agentType']; - } - - // Parse recommended model - let recommendedModel: PlanItem['recommendedModel']; - if (['opus', 'sonnet', 'haiku'].includes(String(item.recommendedModel))) { - recommendedModel = item.recommendedModel as PlanItem['recommendedModel']; - } - - // Find original item to preserve fields - const originalItem = items.find(i => i.id === item.id) || items[idx] || {}; - - return { - ...originalItem, - id: item.id ? String(item.id) : originalItem.id || `task-${idx}`, - content: String(item.content || originalItem.content || ''), - priority: item.priority as PlanItem['priority'] || originalItem.priority, - // Preserve existing fields - tddPhase: item.tddPhase as PlanItem['tddPhase'] || originalItem.tddPhase, - verificationCriteria: item.verificationCriteria ? String(item.verificationCriteria) : originalItem.verificationCriteria, - testCommand: item.testCommand ? String(item.testCommand) : originalItem.testCommand, - dependencies: Array.isArray(item.dependencies) ? item.dependencies.map(String) : originalItem.dependencies, - pairedWith: item.pairedWith ? String(item.pairedWith) : originalItem.pairedWith, - complexity: item.complexity as PlanItem['complexity'] || originalItem.complexity, - reviewChecklist: Array.isArray(item.reviewChecklist) ? item.reviewChecklist.map(String) : originalItem.reviewChecklist, - // New execution optimization fields - parallelGroup: item.parallelGroup ? String(item.parallelGroup) : undefined, - agentType, - requiresFreshContext: Boolean(item.requiresFreshContext), - estimatedTokens: typeof item.estimatedTokens === 'number' ? item.estimatedTokens : undefined, - recommendedModel, - inputFiles: Array.isArray(item.inputFiles) ? item.inputFiles.map(String) : undefined, - outputFiles: Array.isArray(item.outputFiles) ? item.outputFiles.map(String) : undefined, - // Preserve execution tracking - status: originalItem.status || 'pending' as PlanTaskStatus, - attempts: originalItem.attempts || 0, - version: originalItem.version || 1, - }; - }); - - // Parse parallel groups - const parallelGroups: ParallelGroup[] = (parsed.parallelGroups || []).map((group: Record) => ({ - id: String(group.id || ''), - tasks: Array.isArray(group.tasks) ? group.tasks.map(String) : [], - rationale: String(group.rationale || ''), - estimatedDuration: group.estimatedDuration ? String(group.estimatedDuration) : undefined, - totalTokens: typeof group.totalTokens === 'number' ? group.totalTokens : undefined, - })); - - // Parse execution strategy - const strategy = parsed.executionStrategy || {}; - const executionStrategy: ExecutionStrategy = { - totalParallelGroups: typeof strategy.totalParallelGroups === 'number' ? strategy.totalParallelGroups : parallelGroups.length, - sequentialBlockers: Array.isArray(strategy.sequentialBlockers) ? strategy.sequentialBlockers.map(String) : [], - freshContextPoints: Array.isArray(strategy.freshContextPoints) ? strategy.freshContextPoints.map(String) : [], - estimatedTotalTokens: typeof strategy.estimatedTotalTokens === 'number' ? strategy.estimatedTotalTokens : optimizedPlan.length * 20000, - estimatedAgentSpawns: typeof strategy.estimatedAgentSpawns === 'number' ? strategy.estimatedAgentSpawns : parallelGroups.length, - criticalPath: Array.isArray(strategy.criticalPath) ? strategy.criticalPath.map(String) : [], - optimizationNotes: Array.isArray(strategy.optimizationNotes) ? strategy.optimizationNotes.map(String) : [], - }; - const durationMs = Date.now() - startTime; - - onProgress?.('execution-optimization', `Optimization complete (${parallelGroups.length} parallel groups)`); - - // Emit completed event - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'execution', - model: MODEL_VERIFICATION, - status: 'completed', - itemCount: optimizedPlan.length, - durationMs, - }); - - const executionResult = { - success: true, - optimizedPlan, - parallelGroups, - executionStrategy, - }; - - // Save agent output to file - this.saveAgentOutput('execution-optimizer', prompt, { ...executionResult, rawResponse: result }, durationMs); - - return executionResult; - } catch (err) { - console.error('[PlanOrchestrator] Execution optimizer failed:', err); - onSubagent?.({ - type: 'failed', - agentId, - agentType: 'execution', - model: MODEL_VERIFICATION, - status: 'failed', - error: err instanceof Error ? err.message : String(err), - durationMs: Date.now() - startTime, - }); - return defaultResult; - } finally { - // Remove from tracking and clean up - this.runningSessions.delete(session); - try { - await session.stop(); - } catch { - // Ignore cleanup errors - } + const error = err instanceof Error ? err.message : String(err); + onSubagent?.({ type: 'failed', agentId, agentType: 'planner', model: MODEL, status: 'failed', error, durationMs }); + return { success: false, error }; } } - - /** - * Run the final review agent for holistic plan validation. - */ - private async runFinalReview( - taskDescription: string, - items: PlanItem[], - onProgress?: ProgressCallback, - onSubagent?: SubagentCallback - ): Promise { - const agentId = `plan-final-review-${Date.now()}`; - const startTime = Date.now(); - - // Default fallback result - const defaultResult: FinalReviewResult = { - overallAssessment: 'ready', - scores: { logic: 0.8, completeness: 0.8, coherence: 0.8, feasibility: 0.8, overall: 0.8 }, - summary: 'Plan review completed with default assessment.', - issues: [], - missingTasks: [], - recommendations: [], - }; - - // Check if already cancelled - if (this.cancelled) { - return defaultResult; - } - - // Emit started event - onSubagent?.({ - type: 'started', - agentId, - agentType: 'final-review', - model: MODEL_VERIFICATION, - status: 'running', - detail: 'Performing holistic plan review...', - }); - - const session = new Session({ - workingDir: this.workingDir, - screenManager: this.screenManager, - useScreen: false, - mode: 'claude', - }); - - // Track this session for cancellation - this.runningSessions.add(session); - - try { - // Format plan for review - const planText = JSON.stringify(items, null, 2); - - const prompt = FINAL_REVIEW_PROMPT - .replace('{TASK}', taskDescription) - .replace('{PLAN}', planText); - - onProgress?.('final-review', 'Analyzing overall plan coherence...'); - - // Periodic progress updates showing elapsed time - const progressInterval = setInterval(() => { - const elapsedSec = Math.floor((Date.now() - startTime) / 1000); - const timeoutSec = Math.floor(VERIFICATION_TIMEOUT_MS / 1000); - console.log(`[PlanOrchestrator] final review: ${elapsedSec}s elapsed (timeout: ${timeoutSec}s)`); - onSubagent?.({ - type: 'progress', - agentId, - agentType: 'final-review', - model: MODEL_VERIFICATION, - status: 'running', - detail: `Final review... (${elapsedSec}s / ${timeoutSec}s)`, - }); - }, 30000); // Update every 30 seconds - - let result: string; - try { - const response = await Promise.race([ - session.runPrompt(prompt, { model: MODEL_VERIFICATION }), - this.timeoutWithContext(VERIFICATION_TIMEOUT_MS, 'final review', startTime), - ]); - result = response.result; - } finally { - clearInterval(progressInterval); - } - - // Check if cancelled during execution - if (this.cancelled) { - return defaultResult; - } - - // Parse JSON from result with repair for common LLM issues - const jsonMatch = result.match(/\{[\s\S]*\}/); - if (!jsonMatch) { - console.warn('[PlanOrchestrator] Final review returned no JSON, using defaults'); - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'final-review', - model: MODEL_VERIFICATION, - status: 'completed', - detail: 'Using default assessment', - itemCount: items.length, - durationMs: Date.now() - startTime, - }); - return defaultResult; - } - - const parseResult = tryParseJSON(jsonMatch[0]); - if (!parseResult.success) { - console.warn('[PlanOrchestrator] Final review JSON parse failed:', parseResult.error); - return defaultResult; - } - - const parsed = parseResult.data; - - // Parse scores - const scores = { - logic: typeof parsed.logicScore === 'number' ? parsed.logicScore : 0.8, - completeness: typeof parsed.completenessScore === 'number' ? parsed.completenessScore : 0.8, - coherence: typeof parsed.coherenceScore === 'number' ? parsed.coherenceScore : 0.8, - feasibility: typeof parsed.feasibilityScore === 'number' ? parsed.feasibilityScore : 0.8, - overall: typeof parsed.overallScore === 'number' ? parsed.overallScore : 0.8, - }; - - // Parse assessment - let overallAssessment: FinalReviewResult['overallAssessment'] = 'ready'; - if (parsed.overallAssessment === 'needs-revision' || parsed.overallAssessment === 'major-issues') { - overallAssessment = parsed.overallAssessment; - } - - // Parse issues - const issues = (Array.isArray(parsed.logicIssues) ? parsed.logicIssues : []).map((issue: Record) => ({ - severity: (issue.severity === 'error' ? 'error' : 'warning') as 'warning' | 'error', - issue: String(issue.issue || ''), - affectedTasks: Array.isArray(issue.affectedTasks) ? issue.affectedTasks.map(String) : [], - suggestion: String(issue.suggestion || ''), - })); - - // Parse missing tasks - const missingTasks = (Array.isArray(parsed.missingTasks) ? parsed.missingTasks : []).map((task: Record) => ({ - content: String(task.content || ''), - reason: String(task.reason || ''), - insertAfter: task.insertAfter ? String(task.insertAfter) : undefined, - priority: (['P0', 'P1', 'P2'].includes(String(task.priority)) ? task.priority : 'P1') as 'P0' | 'P1' | 'P2', - })); - - // Parse recommendations - const recommendations = Array.isArray(parsed.finalRecommendations) - ? parsed.finalRecommendations.map(String) - : []; - - const reviewResult: FinalReviewResult = { - overallAssessment, - scores, - summary: String(parsed.summary || 'Plan review completed.'), - issues, - missingTasks, - recommendations, - }; - - const durationMs = Date.now() - startTime; - - onProgress?.('final-review', `Review complete (${overallAssessment}, score: ${Math.round(scores.overall * 100)}%)`); - - // Emit completed event - onSubagent?.({ - type: 'completed', - agentId, - agentType: 'final-review', - model: MODEL_VERIFICATION, - status: 'completed', - itemCount: items.length, - durationMs, - }); - - // Save agent output to file - this.saveAgentOutput('final-review', prompt, { ...reviewResult, rawResponse: result }, durationMs); - - return reviewResult; - } catch (err) { - console.error('[PlanOrchestrator] Final review failed:', err); - onSubagent?.({ - type: 'failed', - agentId, - agentType: 'final-review', - model: MODEL_VERIFICATION, - status: 'failed', - error: err instanceof Error ? err.message : String(err), - durationMs: Date.now() - startTime, - }); - return defaultResult; - } finally { - // Remove from tracking and clean up - this.runningSessions.delete(session); - try { - await session.stop(); - } catch { - // Ignore cleanup errors - } - } - } - - /** - * Apply missing tasks identified by the final review. - */ - private applyMissingTasks( - items: PlanItem[], - missingTasks: FinalReviewResult['missingTasks'] - ): PlanItem[] { - const result = [...items]; - - for (const missing of missingTasks) { - // Find insertion point - let insertIndex = result.length; // Default: append at end - if (missing.insertAfter) { - const afterIndex = result.findIndex(item => item.id === missing.insertAfter); - if (afterIndex !== -1) { - insertIndex = afterIndex + 1; - } - } - - // Create new task - const newId = `${missing.priority}-${String(result.length + 1).padStart(3, '0')}`; - const newTask: PlanItem = { - id: newId, - content: missing.content, - priority: missing.priority, - rationale: missing.reason, - status: 'pending', - attempts: 0, - version: 1, - source: 'final-review', - }; - - // Insert at position - result.splice(insertIndex, 0, newTask); - } - - return result; - } - - /** - * Create a timeout promise with context about what timed out. - */ - private timeoutWithContext(ms: number, context: string, startTime: number): Promise { - return new Promise((_, reject) => { - setTimeout(() => { - const elapsedSec = Math.floor((Date.now() - startTime) / 1000); - const timeoutSec = Math.floor(ms / 1000); - reject(new Error(`${context} timed out after ${elapsedSec}s (limit: ${timeoutSec}s)`)); - }, ms); - }); - } - - /** - * Inject review tasks for any implementation tasks that don't have them. - * Called as a post-processing step to ensure the test → impl → review cycle is complete. - * - * @param items - The validated plan items - * @returns Updated plan items with review tasks added - */ - ensureReviewTasks(items: PlanItem[]): PlanItem[] { - const result: PlanItem[] = []; - const implTasksNeedingReview: Map = new Map(); - - // First pass: identify impl tasks and their existing review pairs - const reviewPairs = new Set(); - for (const item of items) { - if (item.tddPhase === 'review' && item.pairedWith) { - reviewPairs.add(item.pairedWith); - } - } - - // Second pass: collect impl tasks without review pairs - for (const item of items) { - if (item.tddPhase === 'impl' && item.id && !reviewPairs.has(item.id)) { - implTasksNeedingReview.set(item.id, item); - } - } - - // Third pass: build result with injected review tasks - for (const item of items) { - result.push(item); - - // If this is an impl task needing review, inject one after it - if (item.id && implTasksNeedingReview.has(item.id)) { - const reviewId = this.generateReviewId(item.id); - const reviewTask = this.createReviewTask(item, reviewId); - result.push(reviewTask); - } - } - - return result; - } - - /** - * Generate a review task ID from an impl task ID. - * P0-002 → P0-002-R, task-5 → task-5-R - */ - private generateReviewId(implId: string): string { - return `${implId}-R`; - } - - /** - * Create a review task for an implementation task. - */ - private createReviewTask(implTask: PlanItem, reviewId: string): PlanItem { - const reviewChecklist = this.generateReviewChecklist(implTask.content); - - return { - id: reviewId, - content: `Review: ${implTask.content}`, - priority: implTask.priority, - tddPhase: 'review', - pairedWith: implTask.id, - dependencies: implTask.id ? [implTask.id] : [], - verificationCriteria: 'Code review complete, no issues found or all issues addressed', - reviewChecklist, - status: 'pending', - attempts: 0, - version: implTask.version || 1, - complexity: 'low', - }; - } - - /** - * Generate a review checklist based on the implementation task content. - */ - private generateReviewChecklist(content: string): string[] { - const lower = content.toLowerCase(); - const checklist: string[] = []; - - // Always include these - checklist.push('Code compiles without errors'); - checklist.push('No TypeScript/linting warnings'); - - // Security checks for certain patterns - if (lower.includes('auth') || lower.includes('login') || lower.includes('password')) { - checklist.push('Input validation implemented'); - checklist.push('No sensitive data in logs'); - checklist.push('Secure session handling'); - } - - if (lower.includes('api') || lower.includes('endpoint') || lower.includes('route')) { - checklist.push('Request validation'); - checklist.push('Error responses do not leak internals'); - checklist.push('Rate limiting considered'); - } - - if (lower.includes('database') || lower.includes('query') || lower.includes('sql')) { - checklist.push('Parameterized queries used'); - checklist.push('No N+1 query issues'); - } - - if (lower.includes('file') || lower.includes('path') || lower.includes('upload')) { - checklist.push('Path traversal prevented'); - checklist.push('File type validation'); - } - - if (lower.includes('user') || lower.includes('input')) { - checklist.push('XSS prevention'); - checklist.push('Input sanitization'); - } - - // Performance checks - if (lower.includes('loop') || lower.includes('iterate') || lower.includes('array')) { - checklist.push('Algorithm complexity is appropriate'); - } - - // Error handling - checklist.push('Error cases handled'); - checklist.push('Meaningful error messages'); - - // Code quality - checklist.push('Code is readable and maintainable'); - checklist.push('No duplicate logic'); - - return checklist; - } } diff --git a/src/prompts/architecture-planner.ts b/src/prompts/architecture-planner.ts deleted file mode 100644 index a6bb56c6..00000000 --- a/src/prompts/architecture-planner.ts +++ /dev/null @@ -1,32 +0,0 @@ -/** - * Architecture Planner Prompt - * - * Designs software component architecture for the task. - * - * Placeholders: {TASK}, {RESEARCH_CONTEXT} - */ - -export const ARCHITECTURE_PLANNER_PROMPT = `You are an Architecture Planner specializing in software component design. - -## YOUR TASK -Design the architecture for implementing this task: - -## TASK DESCRIPTION -{TASK} - -{RESEARCH_CONTEXT} - -## INSTRUCTIONS -1. Identify all modules/components needed -2. Define interfaces between components -3. Specify data structures and types -4. Note configuration and setup requirements -5. Consider separation of concerns - -## OUTPUT FORMAT -Return ONLY a JSON array: -[ - {"category": "module|interface|type|config|infrastructure", "content": "component description", "rationale": "why needed"} -] - -Generate 10-20 items. Think about the complete system architecture.`; diff --git a/src/prompts/execution-optimizer.ts b/src/prompts/execution-optimizer.ts deleted file mode 100644 index 951b5bd2..00000000 --- a/src/prompts/execution-optimizer.ts +++ /dev/null @@ -1,134 +0,0 @@ -/** - * Execution Optimizer Prompt - * - * Optimizes the plan for Claude Code execution with parallel groups, - * agent types, model recommendations, and token estimates. - * - * Placeholders: {TASK}, {PLAN} - */ - -export const EXECUTION_OPTIMIZER_PROMPT = `You are a Claude Code Execution Optimizer. Your job is to analyze an implementation plan and optimize it for efficient execution using Claude Code's agent system. - -## ORIGINAL TASK -{TASK} - -## CURRENT PLAN -{PLAN} - -## YOUR MISSION -Analyze and enhance this plan for optimal Claude Code execution: - -### 1. PARALLEL EXECUTION GROUPS -Identify tasks that can run simultaneously in separate agents: -- Tasks with NO dependencies between them -- Tasks that modify DIFFERENT files -- Tasks that read-only operations (exploration, analysis) -- Assign a parallelGroup ID (e.g., "parallel-1", "parallel-2") to related tasks - -### 2. AGENT TYPE RECOMMENDATIONS -For each task, recommend the optimal Claude Code agent type: -- "explore": For codebase exploration, finding files, understanding patterns -- "implement": For writing new code, features, modifications -- "test": For writing and running tests -- "review": For code review, security analysis, best practices -- "general": For mixed or unclear tasks - -### 3. FRESH CONTEXT RECOMMENDATIONS -Mark tasks that benefit from a fresh context (new conversation): -- After large file modifications (>500 lines changed) -- When switching between unrelated features -- After test failures that need fresh analysis -- When accumulated context might cause confusion - -### 4. MODEL RECOMMENDATIONS -Suggest the optimal model for each task: -- "opus": Complex architecture, critical decisions, security review -- "sonnet": Standard implementation, most coding tasks -- "haiku": Quick exploration, simple searches, routine checks - -### 5. FILE SCOPE ANALYSIS -For each task, identify: -- inputFiles: Files the task will need to READ -- outputFiles: Files the task will CREATE or MODIFY - -### 6. TOKEN ESTIMATION -Estimate token usage for each task: -- Small (exploration, simple changes): 5000-15000 -- Medium (feature implementation): 15000-50000 -- Large (complex features, refactoring): 50000-100000 - -## OUTPUT FORMAT -Return ONLY a JSON object: -{ - "optimizedPlan": [ - { - "id": "P0-001", - "content": "Explore existing auth patterns in codebase", - "priority": "P0", - "tddPhase": "setup", - "verificationCriteria": "Documented auth patterns with file locations", - "dependencies": [], - "parallelGroup": "parallel-1", - "agentType": "explore", - "recommendedModel": "haiku", - "requiresFreshContext": false, - "estimatedTokens": 8000, - "inputFiles": ["src/auth/**/*.ts", "src/middleware/*.ts"], - "outputFiles": [], - "executionNotes": "Quick exploration, can run alongside P0-002" - }, - { - "id": "P0-002", - "content": "Explore test patterns and fixtures", - "priority": "P0", - "tddPhase": "setup", - "verificationCriteria": "Understood test setup and conventions", - "dependencies": [], - "parallelGroup": "parallel-1", - "agentType": "explore", - "recommendedModel": "haiku", - "requiresFreshContext": false, - "estimatedTokens": 6000, - "inputFiles": ["test/**/*.test.ts", "test/fixtures/**/*"], - "outputFiles": [], - "executionNotes": "Parallel with P0-001, different file scope" - } - ], - "parallelGroups": [ - { - "id": "parallel-1", - "tasks": ["P0-001", "P0-002"], - "rationale": "Independent exploration tasks with no file overlap", - "estimatedDuration": "2-3 minutes", - "totalTokens": 14000 - } - ], - "executionStrategy": { - "totalParallelGroups": 3, - "sequentialBlockers": ["P0-005 blocks all P1 tasks"], - "freshContextPoints": ["After P0-005 (large refactor)", "After P1-003 (test failures)"], - "estimatedTotalTokens": 150000, - "estimatedAgentSpawns": 8, - "criticalPath": ["P0-001", "P0-003", "P0-005", "P1-001"], - "optimizationNotes": [ - "Group 1 saves ~3 min by parallelizing exploration", - "Use haiku for 4 exploration tasks to reduce cost", - "Fresh context after auth refactor prevents confusion" - ] - } -} - -CRITICAL REQUIREMENTS: -1. Every task MUST have parallelGroup, agentType, recommendedModel -2. Parallel groups MUST NOT have overlapping outputFiles -3. Tasks in same parallelGroup MUST NOT depend on each other -4. Preserve all existing task fields (id, content, priority, etc.) -5. Add executionNotes explaining WHY this optimization - -PARALLELIZATION GUIDELINES (BE CONSERVATIVE): -- Only parallelize tasks when you are CERTAIN they have no file conflicts -- Prefer sequential execution for complex or risky tasks -- Limit parallel groups to 2-3 tasks maximum per group -- When in doubt, keep tasks sequential - correctness over speed -- Focus parallelization on exploration/read-only tasks, not implementations -- Never parallelize tasks that might share state or side effects`; diff --git a/src/prompts/final-review.ts b/src/prompts/final-review.ts deleted file mode 100644 index 3047686e..00000000 --- a/src/prompts/final-review.ts +++ /dev/null @@ -1,100 +0,0 @@ -/** - * Final Review Expert Prompt - * - * Provides holistic analysis of the complete implementation plan - * with scoring and improvement suggestions. - * - * Placeholders: {TASK}, {PLAN} - */ - -export const FINAL_REVIEW_PROMPT = `You are a Final Review Expert providing a holistic analysis of an implementation plan. - -## ORIGINAL TASK -{TASK} - -## COMPLETE PLAN -{PLAN} - -## YOUR MISSION -Review the ENTIRE plan from a high-level perspective. You have the bird's eye view. - -### 1. LOGICAL FLOW ANALYSIS -Check if the plan makes logical sense: -- Does the order of tasks make sense? -- Are there circular dependencies or impossible orderings? -- Is there a clear progression from setup → implementation → testing → review? -- Are foundation tasks (types, configs, setup) done before dependent tasks? - -### 2. COMPLETENESS CHECK -Verify nothing is missing: -- Every implementation has a corresponding test? -- Every test has clear verification criteria? -- Error handling and edge cases are covered? -- Setup and teardown steps are included? -- Documentation tasks if needed? - -### 3. COHERENCE VALIDATION -Ensure the plan is internally consistent: -- Do task descriptions match their dependencies? -- Are file references consistent across tasks? -- Do parallel groups actually make sense together? -- Are priority levels justified? - -### 4. FEASIBILITY ASSESSMENT -Is this plan actually achievable? -- Are any tasks too vague to execute? -- Are there unrealistic expectations? -- Are there hidden complexities not addressed? -- Is the scope creep under control? - -### 5. SUGGESTED IMPROVEMENTS -Provide actionable fixes: -- Tasks to add if missing -- Tasks to split if too large -- Tasks to merge if redundant -- Order changes if needed -- Clarifications needed - -## OUTPUT FORMAT -Return ONLY a JSON object: -{ - "overallAssessment": "ready|needs-revision|major-issues", - "logicScore": 0.85, - "completenessScore": 0.90, - "coherenceScore": 0.88, - "feasibilityScore": 0.82, - "overallScore": 0.86, - "summary": "Brief 2-3 sentence summary of the plan quality", - "logicIssues": [ - {"severity": "warning|error", "issue": "Description", "affectedTasks": ["P0-001"], "suggestion": "How to fix"} - ], - "missingTasks": [ - {"content": "Add database migration script", "reason": "Schema changes require migration", "insertAfter": "P0-002", "priority": "P0"} - ], - "tasksToSplit": [ - {"taskId": "P1-005", "reason": "Too complex", "splitInto": ["Implement auth logic", "Add session management"]} - ], - "tasksToMerge": [ - {"taskIds": ["P2-001", "P2-002"], "reason": "Redundant", "mergedContent": "Combined task description"} - ], - "orderChanges": [ - {"taskId": "P0-003", "currentPosition": 3, "suggestedPosition": 1, "reason": "Should run earlier"} - ], - "clarificationsNeeded": [ - {"taskId": "P1-002", "issue": "Unclear which API endpoint", "question": "Is this REST or GraphQL?"} - ], - "finalRecommendations": [ - "Start with P0 tasks in sequence for stable foundation", - "Consider adding integration tests after P1-004", - "Review security implications of auth changes" - ] -} - -SCORING GUIDELINES: -- 0.9+: Excellent, ready to execute -- 0.8-0.9: Good, minor tweaks recommended -- 0.7-0.8: Acceptable, some issues to address -- 0.6-0.7: Needs revision before execution -- <0.6: Major issues, significant rework needed - -Be thorough but constructive. The goal is to catch issues before execution, not to criticize.`; diff --git a/src/prompts/index.ts b/src/prompts/index.ts index 74789771..3919206d 100644 --- a/src/prompts/index.ts +++ b/src/prompts/index.ts @@ -6,11 +6,5 @@ */ export { RESEARCH_AGENT_PROMPT } from './research-agent.js'; -export { REQUIREMENTS_ANALYST_PROMPT } from './requirements-analyst.js'; -export { ARCHITECTURE_PLANNER_PROMPT } from './architecture-planner.js'; -export { TESTING_SPECIALIST_PROMPT } from './testing-specialist.js'; -export { RISK_ANALYST_PROMPT } from './risk-analyst.js'; +export { PLANNER_PROMPT } from './planner.js'; export { CODE_REVIEWER_PROMPT } from './code-reviewer.js'; -export { VERIFICATION_PROMPT } from './verification.js'; -export { EXECUTION_OPTIMIZER_PROMPT } from './execution-optimizer.js'; -export { FINAL_REVIEW_PROMPT } from './final-review.js'; diff --git a/src/prompts/planner.ts b/src/prompts/planner.ts new file mode 100644 index 00000000..3ad2aaf1 --- /dev/null +++ b/src/prompts/planner.ts @@ -0,0 +1,83 @@ +/** + * Planner Prompt - Single agent for TDD plan generation + * + * Combines what was previously 5 separate agents: + * - Requirements Analyst (redundant) + * - Architecture Planner (redundant) + * - Testing Specialist (kept - TDD focus) + * - Risk Analyst (redundant) + * - Verification Expert (kept - structure) + * + * Placeholders: {TASK}, {RESEARCH_CONTEXT} + */ + +export const PLANNER_PROMPT = `You are a TDD Plan Generator. Create a complete implementation plan with test-first approach. + +## TASK DESCRIPTION +{TASK} + +{RESEARCH_CONTEXT} + +## YOUR MISSION +Generate a complete TDD implementation plan with: +1. Tests BEFORE implementations (red-green-refactor) +2. Review tasks AFTER implementations +3. Clear priorities (P0=blocking, P1=required, P2=polish) +4. Dependencies between tasks + +## TDD CYCLE +For each feature: +1. Write failing test first +2. Implement to make test pass +3. Review implementation + +## PRIORITY GUIDELINES +- P0: Foundation, types, project setup, blocking dependencies +- P1: Core features, main implementation, error handling +- P2: Polish, optimization, documentation + +## OUTPUT FORMAT +Return ONLY a JSON object: +{ + "items": [ + { + "id": "P0-001", + "content": "Write failing test for user authentication endpoint", + "priority": "P0", + "tddPhase": "test", + "verificationCriteria": "Test file exists, test fails with 'not implemented'", + "testCommand": "npm test -- --grep='auth'", + "dependencies": [] + }, + { + "id": "P0-002", + "content": "Implement user authentication handler", + "priority": "P0", + "tddPhase": "impl", + "verificationCriteria": "npm test -- --grep='auth' passes", + "pairedWith": "P0-001", + "dependencies": ["P0-001"] + }, + { + "id": "P0-003", + "content": "Review auth implementation for security", + "priority": "P0", + "tddPhase": "review", + "verificationCriteria": "No security issues, follows best practices", + "reviewChecklist": ["Input validation", "XSS prevention", "Error handling"], + "pairedWith": "P0-002", + "dependencies": ["P0-002"] + } + ], + "gaps": ["any missing requirements noted"], + "warnings": ["any concerns or risks identified"] +} + +CRITICAL REQUIREMENTS: +1. Every implementation MUST have a paired test task that comes BEFORE it +2. Every implementation MUST have a review task that comes AFTER it +3. Use sequential IDs: P0-001, P0-002, P1-001, etc. +4. verificationCriteria must be SPECIFIC and observable +5. Dependencies must form a valid DAG (no cycles) + +Generate 15-40 items covering the complete implementation.`; diff --git a/src/prompts/requirements-analyst.ts b/src/prompts/requirements-analyst.ts deleted file mode 100644 index 41a4f564..00000000 --- a/src/prompts/requirements-analyst.ts +++ /dev/null @@ -1,31 +0,0 @@ -/** - * Requirements Analyst Prompt - * - * Extracts explicit and implicit requirements from task descriptions. - * - * Placeholders: {TASK}, {RESEARCH_CONTEXT} - */ - -export const REQUIREMENTS_ANALYST_PROMPT = `You are a Requirements Analyst specializing in extracting all requirements from task descriptions. - -## YOUR TASK -Analyze the following task and extract ALL requirements (explicit and implicit): - -## TASK DESCRIPTION -{TASK} - -{RESEARCH_CONTEXT} - -## INSTRUCTIONS -1. Identify explicit requirements (directly stated) -2. Infer implicit requirements (unstated but necessary) -3. Note any assumptions that should be validated -4. Consider non-functional requirements (performance, security, usability) - -## OUTPUT FORMAT -Return ONLY a JSON array: -[ - {"category": "functional|non-functional|constraint|assumption", "content": "requirement description", "rationale": "why this is needed"} -] - -Generate 8-15 items. Be thorough - missing requirements cause project failures.`; diff --git a/src/prompts/risk-analyst.ts b/src/prompts/risk-analyst.ts deleted file mode 100644 index 2d454dc9..00000000 --- a/src/prompts/risk-analyst.ts +++ /dev/null @@ -1,32 +0,0 @@ -/** - * Risk Analyst Prompt - * - * Identifies potential issues, edge cases, and blockers. - * - * Placeholders: {TASK}, {RESEARCH_CONTEXT} - */ - -export const RISK_ANALYST_PROMPT = `You are a Risk Analyst identifying potential issues and blockers. - -## YOUR TASK -Identify risks and edge cases for this task: - -## TASK DESCRIPTION -{TASK} - -{RESEARCH_CONTEXT} - -## INSTRUCTIONS -1. Identify potential failure points -2. Note edge cases that could cause bugs -3. Consider security vulnerabilities -4. Flag performance concerns -5. Identify dependencies that could block progress - -## OUTPUT FORMAT -Return ONLY a JSON array: -[ - {"category": "failure|edge-case|security|performance|dependency", "content": "risk description", "rationale": "mitigation approach"} -] - -Generate 8-15 items. Being proactive about risks prevents surprises.`; diff --git a/src/prompts/testing-specialist.ts b/src/prompts/testing-specialist.ts deleted file mode 100644 index 9b6c80a1..00000000 --- a/src/prompts/testing-specialist.ts +++ /dev/null @@ -1,76 +0,0 @@ -/** - * Testing Specialist Prompt - * - * Designs comprehensive, realistic test coverage with TDD approach. - * - * Placeholders: {TASK}, {RESEARCH_CONTEXT} - */ - -export const TESTING_SPECIALIST_PROMPT = `You are a TDD Specialist designing a comprehensive, REALISTIC test strategy. - -## YOUR TASK -Design detailed, executable test coverage for this task: - -## TASK DESCRIPTION -{TASK} - -{RESEARCH_CONTEXT} - -## INSTRUCTIONS -Create REALISTIC tests that would actually run in a real codebase: - -### 1. Unit Tests (test individual functions/methods in isolation) -- Mock external dependencies (databases, APIs, file system) -- Test pure logic with specific input/output examples -- Include exact assertion values, not placeholders - -### 2. Integration Tests (test component interactions) -- Test API endpoints with realistic request/response bodies -- Test database operations with actual schema -- Test service-to-service communication - -### 3. Edge Cases & Boundary Tests -- Empty inputs, null values, undefined -- Maximum/minimum values, overflow conditions -- Unicode, special characters, injection attempts -- Concurrent access, race conditions - -### 4. Error Scenario Tests -- Network failures, timeouts, connection refused -- Invalid input validation with specific error messages -- Authorization failures, permission denied -- Resource not found, conflict states - -### 5. Performance & Load Tests (where applicable) -- Response time thresholds -- Memory usage limits -- Concurrent user handling - -## REALISTIC TEST EXAMPLE -BAD: "Test user login" (too vague) -GOOD: "Test POST /api/auth/login with valid email 'test@example.com' and password 'ValidPass123!' returns 200 with JWT token containing userId and exp claims, sets httpOnly cookie 'session'" - -## OUTPUT FORMAT -Return ONLY a JSON array: -[ - { - "category": "unit|integration|edge-case|error|e2e|performance", - "content": "Test POST /api/users with email 'new@test.com' creates user and returns 201 with {id, email, createdAt}", - "rationale": "Validates user creation happy path with all required response fields", - "verificationCriteria": "Response status 201, body contains id (uuid), email matches input, createdAt is valid ISO timestamp", - "testCommand": "npm test -- --grep='POST /api/users creates user'", - "testSetup": "Clear users table, seed with test data", - "testTeardown": "Delete created test user", - "pairedImpl": "Implement POST /api/users endpoint with validation and database insert", - "mockDependencies": ["database connection", "email service"], - "assertionDetails": ["status === 201", "body.id matches UUID regex", "body.email === 'new@test.com'"] - } -] - -CRITICAL REQUIREMENTS: -- verificationCriteria: SPECIFIC observable outcomes with exact values -- testCommand: Actual runnable command (npm test, pytest, vitest, etc.) -- pairedImpl: The exact implementation step this test validates -- assertionDetails: List of specific assertions to make - -Generate 15-30 detailed test items. Tests MUST be specific enough to implement directly.`; diff --git a/src/prompts/verification.ts b/src/prompts/verification.ts deleted file mode 100644 index f87fb400..00000000 --- a/src/prompts/verification.ts +++ /dev/null @@ -1,96 +0,0 @@ -/** - * Verification Expert Prompt - * - * Reviews and enhances the synthesized plan with priorities, - * verification criteria, and TDD pairing. - * - * Placeholders: {TASK}, {PLAN} - */ - -export const VERIFICATION_PROMPT = `You are a Plan Verification Expert reviewing an implementation plan for completeness and quality. - -## ORIGINAL TASK -{TASK} - -## SYNTHESIZED PLAN (from multiple analysis subagents) -{PLAN} - -## YOUR MISSION -Review and enhance this plan: -1. Assign priorities (P0=critical/blocking, P1=required, P2=enhancement) -2. Add verification criteria to EVERY task (how to know it's done) -3. Pair test tasks with implementation tasks (TDD cycle) -4. Add dependencies where one task blocks another -5. Identify gaps and calculate quality score - -## PRIORITY GUIDELINES -- P0: Foundation tasks, type definitions, project setup, blocking dependencies -- P1: Core implementation, tests, main features, error handling -- P2: Polish, optimization, documentation, nice-to-have features - -## TDD + REVIEW CYCLE RULES -The complete cycle is: test → impl → review -- Every implementation task should have a corresponding test task AND review task -- Test task comes BEFORE its paired implementation task -- Review task comes AFTER the implementation it reviews -- Use "pairedWith" to link test ↔ implementation ↔ review -- Verification criteria should reference test results where applicable - -## REVIEW TASK REQUIREMENTS -After EVERY implementation task, add a review task that checks: -- Best practices for the language/framework -- Security vulnerabilities (OWASP top 10) -- Performance concerns -- Error handling completeness -- Code quality (DRY, SOLID, readability) - -## OUTPUT FORMAT -Return ONLY a JSON object: -{ - "validatedPlan": [ - { - "id": "P0-001", - "content": "Write failing test for user authentication", - "priority": "P0", - "tddPhase": "test", - "verificationCriteria": "Test file exists, test fails with 'not implemented'", - "testCommand": "npm test -- --grep='auth'", - "pairedWith": "P0-002", - "dependencies": [], - "complexity": "low" - }, - { - "id": "P0-002", - "content": "Implement user authentication handler", - "priority": "P0", - "tddPhase": "impl", - "verificationCriteria": "npm test -- --grep='auth' passes", - "pairedWith": "P0-001", - "dependencies": ["P0-001"], - "complexity": "medium" - }, - { - "id": "P0-003", - "content": "Review auth implementation for security and best practices", - "priority": "P0", - "tddPhase": "review", - "verificationCriteria": "No security issues found, follows TypeScript best practices", - "reviewChecklist": ["Input validation", "XSS prevention", "Session security", "Error handling"], - "pairedWith": "P0-002", - "dependencies": ["P0-002"], - "complexity": "low" - } - ], - "gaps": ["missing requirement 1", "missing test coverage for X"], - "warnings": ["consider Y before Z", "potential issue with..."], - "qualityScore": 0.85 -} - -CRITICAL REQUIREMENTS: -1. EVERY task MUST have verificationCriteria (how to verify completion) -2. Implementation tasks MUST have a paired test task AND a review task -3. Review tasks MUST have a reviewChecklist with specific items to check -4. Dependencies must form a valid DAG (no cycles) -5. Use sequential IDs: P0-001, P0-002, P0-003, P1-001, etc. - -Be critical but constructive. A thorough review catches issues that tests miss.`; diff --git a/src/types.ts b/src/types.ts index 2bcb9609..1cb98d72 100644 --- a/src/types.ts +++ b/src/types.ts @@ -1255,37 +1255,6 @@ export interface ImageDetectedEvent { size: number; } -// ========== Execution Bridge Re-exports ========== +// ========== Plan Orchestrator Re-exports ========== -export type { - ExecutionStatus, - ExecutionProgress, - TaskAssignment as ExecutionTaskAssignment, - PlanItem, - ExecutionHistoryEntry, -} from './execution-bridge.js'; - -export type { - ModelTier, - AgentType, - ExecutionMode, - ModelConfig, - ModelSelection, - ExecutionModeSelection, - TaskCharacteristics, -} from './model-selector.js'; - -export type { - GroupTaskStatus, - ExecutionGroupStatus, - GroupTask, - ExecutionGroup, - ExecutionSchedule, -} from './group-scheduler.js'; - -export type { - ContextRefreshMethod, - ContextRefreshStatus, - ContextRefreshRequest, - ContextRefreshResult, -} from './context-manager.js'; +export type { PlanItem } from './plan-orchestrator.js'; diff --git a/src/web/server.ts b/src/web/server.ts index 8298f611..26dbcb1c 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -34,13 +34,6 @@ import { v4 as uuidv4 } from 'uuid'; import { createRequire } from 'node:module'; import { RunSummaryTracker } from '../run-summary.js'; import { PlanOrchestrator, type DetailedPlanResult } from '../plan-orchestrator.js'; -import { - ExecutionBridge, - getExecutionBridge, - type PlanItem as ExecutionPlanItem, - type AgentSpawner, -} from '../execution-bridge.js'; -import { type ModelConfig } from '../model-selector.js'; // Load version from package.json const require = createRequire(import.meta.url); @@ -314,8 +307,6 @@ export class WebServer extends EventEmitter { private pendingRespawnStarts: Map = new Map(); // Active plan orchestrators (for cancellation via API) private activePlanOrchestrators: Map = new Map(); - // Execution bridge for parallel task execution - private executionBridge: ExecutionBridge; // Grace period before starting restored respawn controllers (2 minutes) private static readonly RESPAWN_RESTORE_GRACE_PERIOD_MS = 2 * 60 * 1000; @@ -346,10 +337,6 @@ export class WebServer extends EventEmitter { this.broadcast('screen:statsUpdated', screens); }); - // Initialize execution bridge with model config from settings - this.executionBridge = getExecutionBridge(this.loadModelConfig()); - this.setupExecutionBridgeListeners(); - // Set up subagent watcher listeners this.setupSubagentWatcherListeners(); @@ -3045,148 +3032,6 @@ NOW: Generate the implementation plan for the task above. Think step by step.`; return this.getSystemStats(); }); - // ========== Execution Bridge Endpoints ========== - - // Get execution status - this.app.get('/api/execution/status', async () => { - return { - success: true, - data: { - status: this.executionBridge.status, - progress: this.executionBridge.getProgress(), - }, - }; - }); - - // Get execution schedule (the current plan) - this.app.get('/api/execution/schedule', async () => { - const schedule = this.executionBridge['_scheduler'].schedule; - return { success: true, data: schedule }; - }); - - // Load a plan for execution - // Accepts either ExecutionPlanItem format (id, title, description) or - // PlanOrchestrator PlanItem format (id, content) and converts as needed - this.app.post('/api/execution/load', async (req) => { - const { items, workingDir } = req.body as { - items: Array<{ - id?: string; - title?: string; - description?: string; - content?: string; - parallelGroup?: number; - agentType?: string; - recommendedModel?: string; - requiresFreshContext?: boolean; - estimatedTokens?: number; - inputFiles?: string[]; - outputFiles?: string[]; - dependencies?: string[]; - }>; - workingDir?: string; - }; - if (!items || !Array.isArray(items)) { - return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'items array is required'); - } - try { - if (workingDir) { - this.executionBridge.setWorkingDir(workingDir); - } - // Convert to ExecutionPlanItem format - const execItems: ExecutionPlanItem[] = items.map((item, idx) => ({ - id: item.id || `task-${idx}`, - title: item.title || item.content?.slice(0, 50) || `Task ${idx + 1}`, - description: item.description || item.content || '', - parallelGroup: item.parallelGroup, - agentType: item.agentType, - recommendedModel: item.recommendedModel, - requiresFreshContext: item.requiresFreshContext, - estimatedTokens: item.estimatedTokens, - inputFiles: item.inputFiles, - outputFiles: item.outputFiles, - dependencies: item.dependencies, - })); - const schedule = this.executionBridge.loadPlan(execItems); - return { success: true, data: schedule }; - } catch (err) { - return createErrorResponse(ApiErrorCode.OPERATION_FAILED, getErrorMessage(err)); - } - }); - - // Start execution - this.app.post('/api/execution/start', async () => { - try { - await this.executionBridge.start(); - return { success: true }; - } catch (err) { - return createErrorResponse(ApiErrorCode.OPERATION_FAILED, getErrorMessage(err)); - } - }); - - // Pause execution - this.app.post('/api/execution/pause', async () => { - this.executionBridge.pause(); - return { success: true }; - }); - - // Resume execution - this.app.post('/api/execution/resume', async () => { - this.executionBridge.resume(); - return { success: true }; - }); - - // Cancel execution - this.app.post('/api/execution/cancel', async (req) => { - const { reason } = (req.body as { reason?: string }) || {}; - await this.executionBridge.cancel(reason || 'Cancelled via API'); - return { success: true }; - }); - - // Reset execution bridge - this.app.post('/api/execution/reset', async () => { - this.executionBridge.reset(); - return { success: true }; - }); - - // Get execution history - this.app.get('/api/execution/history', async () => { - return { success: true, data: this.executionBridge.getHistory() }; - }); - - // Get model configuration - this.app.get('/api/execution/model-config', async () => { - return { success: true, data: this.executionBridge.getModelConfig() }; - }); - - // Update model configuration - this.app.put('/api/execution/model-config', async (req) => { - const config = req.body as Partial; - try { - this.executionBridge.updateModelConfig(config); - const fullConfig = this.executionBridge.getModelConfig(); - this.saveModelConfig(fullConfig); - this.broadcast('execution:modelConfigUpdated', fullConfig); - return { success: true, data: fullConfig }; - } catch (err) { - return createErrorResponse(ApiErrorCode.OPERATION_FAILED, getErrorMessage(err)); - } - }); - - // Mark task complete (external signal) - this.app.post('/api/execution/tasks/:taskId/complete', async (req) => { - const { taskId } = req.params as { taskId: string }; - this.executionBridge.markTaskComplete(taskId); - return { success: true }; - }); - - // Mark task failed (external signal) - this.app.post('/api/execution/tasks/:taskId/failed', async (req) => { - const { taskId } = req.params as { taskId: string }; - const { error } = (req.body as { error?: string }) || {}; - this.executionBridge.markTaskFailed(taskId, error || 'Failed via API'); - return { success: true }; - }); - // ========== Subagent Monitoring (Claude Code Background Agents) ========== // List all known subagents @@ -3910,193 +3755,6 @@ NOW: Generate the implementation plan for the task above. Think step by step.`; }); } - /** - * Load model configuration from settings file. - */ - private loadModelConfig(): Partial { - const settingsPath = join(homedir(), '.claudeman', 'settings.json'); - try { - if (existsSync(settingsPath)) { - const content = readFileSync(settingsPath, 'utf-8'); - const settings = JSON.parse(content); - return settings.modelConfig || {}; - } - } catch (err) { - console.error('Failed to load model config:', getErrorMessage(err)); - } - return {}; - } - - /** - * Save model configuration to settings file. - */ - private saveModelConfig(config: ModelConfig): void { - const settingsPath = join(homedir(), '.claudeman', 'settings.json'); - try { - const dir = dirname(settingsPath); - if (!existsSync(dir)) { - mkdirSync(dir, { recursive: true }); - } - let settings: Record = {}; - if (existsSync(settingsPath)) { - settings = JSON.parse(readFileSync(settingsPath, 'utf-8')); - } - settings.modelConfig = config; - writeFileSync(settingsPath, JSON.stringify(settings, null, 2)); - } catch (err) { - console.error('Failed to save model config:', getErrorMessage(err)); - } - } - - /** - * Set up event listeners for execution bridge. - * Broadcasts execution progress to SSE clients. - */ - private setupExecutionBridgeListeners(): void { - // Create agent spawner interface - const spawner: AgentSpawner = { - spawnAgentWithModel: async (taskId, workingDir, prompt, _model, _options) => { - // Note: model and options are reserved for future model selection integration - const globalNice = this.getGlobalNiceConfig(); - const session = new Session({ - workingDir, - screenManager: this.screenManager, - useScreen: true, - mode: 'claude', - name: `exec:${taskId}`, - niceConfig: globalNice, - }); - - this.sessions.set(session.id, session); - this.store.incrementSessionsCreated(); - this.setupSessionListeners(session); - - await session.startInteractive(); - this.broadcast('session:created', session.toDetailedState()); - this.broadcast('session:interactive', { id: session.id }); - this.persistSessionState(session); - - // Configure ralph tracker for completion detection - session.ralphTracker.enable(); - - // Set up completion listener for execution bridge - session.on('ralphCompletionDetected', () => { - this.executionBridge.markTaskComplete(taskId); - }); - - // Send the prompt - setTimeout(() => { - session.writeViaScreen(prompt + '\r'); - }, 2000); - - return { sessionId: session.id }; - }, - useTaskTool: async (_sessionId, _taskId, _prompt, _model) => { - // Task tool mode - not yet implemented, falls back to session mode - throw new Error('Task tool mode not yet implemented'); - }, - isTaskComplete: (taskId) => { - // Check if task is marked complete in scheduler - const schedule = this.executionBridge['_scheduler'].schedule; - if (!schedule) return false; - for (const group of schedule.groups) { - const task = group.tasks.find(t => t.id === taskId); - if (task) return task.status === 'completed'; - } - return false; - }, - getTaskResult: (taskId) => { - const schedule = this.executionBridge['_scheduler'].schedule; - if (!schedule) return null; - for (const group of schedule.groups) { - const task = group.tasks.find(t => t.id === taskId); - if (task) { - if (task.status === 'completed') { - return { success: true }; - } else if (task.status === 'failed') { - return { success: false, error: task.error }; - } - } - } - return null; - }, - }; - - this.executionBridge.setAgentSpawner(spawner); - - // Set up session writer for context management - this.executionBridge.setSessionWriter({ - writeToSession: (sessionId, text) => { - const session = this.sessions.get(sessionId); - if (session) { - session.writeViaScreen(text); - } - }, - createSession: async (workingDir, name) => { - const globalNice = this.getGlobalNiceConfig(); - const session = new Session({ - workingDir, - screenManager: this.screenManager, - useScreen: true, - mode: 'claude', - name: name || 'exec-context', - niceConfig: globalNice, - }); - this.sessions.set(session.id, session); - this.store.incrementSessionsCreated(); - this.setupSessionListeners(session); - await session.startInteractive(); - this.broadcast('session:created', session.toDetailedState()); - this.persistSessionState(session); - return { sessionId: session.id }; - }, - }); - - // Forward execution bridge events as SSE broadcasts - this.executionBridge.on('planLoaded', (schedule) => { - this.broadcast('execution:planLoaded', { schedule }); - }); - this.executionBridge.on('started', () => { - this.broadcast('execution:started', {}); - }); - this.executionBridge.on('paused', () => { - this.broadcast('execution:paused', {}); - }); - this.executionBridge.on('resumed', () => { - this.broadcast('execution:resumed', {}); - }); - this.executionBridge.on('completed', (data) => { - this.broadcast('execution:completed', data); - }); - this.executionBridge.on('cancelled', (reason) => { - this.broadcast('execution:cancelled', { reason }); - }); - this.executionBridge.on('groupStarted', (data) => { - this.broadcast('execution:groupStarted', data); - }); - this.executionBridge.on('groupCompleted', (data) => { - this.broadcast('execution:groupCompleted', data); - }); - this.executionBridge.on('taskAssigned', (data) => { - this.broadcast('execution:taskAssigned', data); - }); - this.executionBridge.on('taskCompleted', (data) => { - this.broadcast('execution:taskCompleted', data); - }); - this.executionBridge.on('taskFailed', (data) => { - this.broadcast('execution:taskFailed', data); - }); - this.executionBridge.on('freshContext', (data) => { - this.broadcast('execution:freshContext', data); - }); - this.executionBridge.on('modelSelected', (data) => { - this.broadcast('execution:modelSelected', data); - }); - this.executionBridge.on('progress', (progress) => { - this.broadcast('execution:progress', progress); - }); - } - private setupTimedRespawn(sessionId: string, durationMinutes: number): void { // Clear existing timer if any const existing = this.respawnTimers.get(sessionId); diff --git a/test/execution-bridge.test.ts b/test/execution-bridge.test.ts deleted file mode 100644 index 4a8a2116..00000000 --- a/test/execution-bridge.test.ts +++ /dev/null @@ -1,270 +0,0 @@ -/** - * @fileoverview Tests for the Execution Bridge system. - * - * Tests the core functionality of: - * - ModelSelector - * - GroupScheduler - * - ExecutionBridge - * - * Uses mocks to avoid spawning real Claude sessions. - * - * Port: none (unit tests, no server) - */ - -import { describe, it, expect, beforeEach, afterEach } from 'vitest'; -import { - ModelSelector, - resetModelSelector, - createDefaultModelConfig, - type ModelConfig, - type AgentType, -} from '../src/model-selector.js'; -import { - GroupScheduler, - resetGroupScheduler, - type GroupTask, -} from '../src/group-scheduler.js'; -import { - ExecutionBridge, - resetExecutionBridge, - type PlanItem, -} from '../src/execution-bridge.js'; - -describe('ModelSelector', () => { - let selector: ModelSelector; - - beforeEach(() => { - resetModelSelector(); - selector = new ModelSelector(); - }); - - afterEach(() => { - resetModelSelector(); - }); - - it('should use default model when no overrides', () => { - const selection = selector.selectModel('task-1', { agentType: 'explore' }); - expect(selection.model).toBe('sonnet'); - expect(selection.usedUserDefault).toBe(true); - }); - - it('should respect user default model', () => { - selector.updateConfig({ defaultModel: 'opus' }); - const selection = selector.selectModel('task-1', { agentType: 'explore' }); - expect(selection.model).toBe('opus'); - expect(selection.usedUserDefault).toBe(true); - }); - - it('should use agent type override when set', () => { - selector.updateConfig({ - agentTypeOverrides: { explore: 'haiku' }, - }); - const selection = selector.selectModel('task-1', { agentType: 'explore' }); - expect(selection.model).toBe('haiku'); - expect(selection.usedUserDefault).toBe(false); - }); - - it('should include optimizer recommendation when different', () => { - selector.updateConfig({ defaultModel: 'opus' }); - const selection = selector.selectModel('task-1', { - agentType: 'explore', - recommendedModel: 'haiku', - }); - expect(selection.model).toBe('opus'); - expect(selection.optimizerRecommendation).toBe('haiku'); - }); - - it('should select session mode for high token tasks', () => { - const mode = selector.selectExecutionMode({ estimatedTokens: 60000 }); - expect(mode.mode).toBe('session'); - }); - - it('should select task-tool mode for low token explore tasks', () => { - const mode = selector.selectExecutionMode({ - estimatedTokens: 10000, - agentType: 'explore', - }); - expect(mode.mode).toBe('task-tool'); - }); - - it('should return correct cost multipliers', () => { - expect(selector.getModelCostMultiplier('opus')).toBe(5.0); - expect(selector.getModelCostMultiplier('sonnet')).toBe(1.0); - expect(selector.getModelCostMultiplier('haiku')).toBe(0.04); - }); -}); - -describe('GroupScheduler', () => { - let scheduler: GroupScheduler; - - beforeEach(() => { - resetGroupScheduler(); - scheduler = new GroupScheduler(); - }); - - afterEach(() => { - resetGroupScheduler(); - }); - - it('should build schedule from plan items', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1', parallelGroup: 0 }, - { id: 't2', title: 'Task 2', description: 'Desc 2', parallelGroup: 0 }, - { id: 't3', title: 'Task 3', description: 'Desc 3', parallelGroup: 1 }, - ]; - - const schedule = scheduler.buildSchedule(items); - - expect(schedule.groups).toHaveLength(2); - expect(schedule.groups[0].tasks).toHaveLength(2); - expect(schedule.groups[1].tasks).toHaveLength(1); - expect(schedule.totalTasks).toBe(3); - }); - - it('should order groups by number', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1', parallelGroup: 2 }, - { id: 't2', title: 'Task 2', description: 'Desc 2', parallelGroup: 0 }, - { id: 't3', title: 'Task 3', description: 'Desc 3', parallelGroup: 1 }, - ]; - - const schedule = scheduler.buildSchedule(items); - - expect(schedule.groups[0].groupNumber).toBe(0); - expect(schedule.groups[1].groupNumber).toBe(1); - expect(schedule.groups[2].groupNumber).toBe(2); - }); - - it('should track group dependencies', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1', parallelGroup: 0 }, - { id: 't2', title: 'Task 2', description: 'Desc 2', parallelGroup: 1, dependencies: ['t1'] }, - ]; - - const schedule = scheduler.buildSchedule(items); - - expect(schedule.groups[1].dependsOnGroups).toContain(0); - }); - - it('should return first group as ready initially', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1', parallelGroup: 0 }, - { id: 't2', title: 'Task 2', description: 'Desc 2', parallelGroup: 1 }, - ]; - - scheduler.buildSchedule(items); - const nextGroup = scheduler.getNextReadyGroup(); - - expect(nextGroup).not.toBeNull(); - expect(nextGroup!.groupNumber).toBe(0); - }); - - it('should update task status', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1', parallelGroup: 0 }, - ]; - - scheduler.buildSchedule(items); - scheduler.startGroup(0); - scheduler.updateTaskStatus('t1', 'completed'); - - const stats = scheduler.getStats(); - expect(stats.completedTasks).toBe(1); - }); - - it('should determine execution mode based on task characteristics', () => { - const items: PlanItem[] = [ - { - id: 't1', - title: 'High token task', - description: 'Desc', - parallelGroup: 0, - estimatedTokens: 60000, - }, - ]; - - const schedule = scheduler.buildSchedule(items); - expect(schedule.groups[0].executionMode).toBe('session'); - }); -}); - -describe('ExecutionBridge', () => { - let bridge: ExecutionBridge; - - beforeEach(() => { - resetExecutionBridge(); - resetGroupScheduler(); - resetModelSelector(); - bridge = new ExecutionBridge(); - }); - - afterEach(() => { - bridge.reset(); - resetExecutionBridge(); - resetGroupScheduler(); - resetModelSelector(); - }); - - it('should have idle status initially', () => { - expect(bridge.status).toBe('idle'); - }); - - it('should load a plan', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1', parallelGroup: 0 }, - { id: 't2', title: 'Task 2', description: 'Desc 2', parallelGroup: 1 }, - ]; - - const schedule = bridge.loadPlan(items); - - expect(schedule).not.toBeNull(); - expect(schedule.totalTasks).toBe(2); - expect(bridge.status).toBe('idle'); - }); - - it('should report progress', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1' }, - ]; - - bridge.loadPlan(items); - const progress = bridge.getProgress(); - - expect(progress.totalTasks).toBe(1); - expect(progress.completedTasks).toBe(0); - expect(progress.status).toBe('idle'); - }); - - it('should update model config', () => { - bridge.updateModelConfig({ defaultModel: 'opus' }); - const config = bridge.getModelConfig(); - expect(config.defaultModel).toBe('opus'); - }); - - it('should reset properly', () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1' }, - ]; - - bridge.loadPlan(items); - bridge.reset(); - - expect(bridge.status).toBe('idle'); - expect(bridge.getProgress().totalTasks).toBe(0); - }); - - it('should throw when starting without a spawner', async () => { - const items: PlanItem[] = [ - { id: 't1', title: 'Task 1', description: 'Desc 1' }, - ]; - - bridge.loadPlan(items); - - await expect(bridge.start()).rejects.toThrow('No agent spawner configured'); - }); - - it('should track execution history', () => { - const history = bridge.getHistory(); - expect(Array.isArray(history)).toBe(true); - }); -});