diff --git a/CLAUDE.md b/CLAUDE.md index a2238568..7fbfc343 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -17,7 +17,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co Claudeman is a Claude Code session manager with a web interface and autonomous Ralph Loop. It spawns Claude CLI processes via PTY, streams output in real-time via SSE, and supports scheduled/timed runs. -**Version**: 0.1345 +**Version**: 0.1346 **Tech Stack**: TypeScript (ES2022/NodeNext, strict mode), Node.js, Fastify, Server-Sent Events, node-pty diff --git a/package.json b/package.json index 94b85e81..9474d385 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "claudeman", - "version": "0.1345", + "version": "0.1346", "description": "The missing control plane for Claude Code - run 20 autonomous agents with real-time monitoring and session persistence", "type": "module", "main": "dist/index.js", diff --git a/src/ai-idle-checker.ts b/src/ai-idle-checker.ts index 67e3ef87..6a316616 100644 --- a/src/ai-idle-checker.ts +++ b/src/ai-idle-checker.ts @@ -268,13 +268,16 @@ export class AiIdleChecker extends EventEmitter { this.log('Cancelling AI check'); this.checkCancelled = true; - // Resolve the pending promise before cleanup + // Clear poll/timeout timers first to prevent race condition where + // the poll timer fires between setting checkCancelled and cleanup + this.cleanupCheck(); + + // Resolve the pending promise after cleanup if (this.checkResolve) { this.checkResolve({ verdict: 'ERROR', reasoning: 'Cancelled', durationMs: Date.now() - this.checkStartTime }); this.checkResolve = null; } - this.cleanupCheck(); this._status = 'ready'; } diff --git a/src/ai-plan-checker.ts b/src/ai-plan-checker.ts index e4671c6d..3ee33732 100644 --- a/src/ai-plan-checker.ts +++ b/src/ai-plan-checker.ts @@ -152,6 +152,7 @@ export class AiPlanChecker extends EventEmitter { // Active check state private checkScreenName: string | null = null; private checkTempFile: string | null = null; + private checkPromptFile: string | null = null; private checkPollTimer: NodeJS.Timeout | null = null; private checkTimeoutTimer: NodeJS.Timeout | null = null; private checkStartTime: number = 0; @@ -276,13 +277,16 @@ export class AiPlanChecker extends EventEmitter { this.log('Cancelling AI plan check'); this.checkCancelled = true; - // Resolve the pending promise before cleanup + // Clear poll/timeout timers first to prevent race condition where + // the poll timer fires between setting checkCancelled and cleanup + this.cleanupCheck(); + + // Resolve the pending promise after cleanup if (this.checkResolve) { this.checkResolve({ verdict: 'ERROR', reasoning: 'Cancelled', durationMs: Date.now() - this.checkStartTime }); this.checkResolve = null; } - this.cleanupCheck(); this._status = 'ready'; } @@ -329,21 +333,25 @@ export class AiPlanChecker extends EventEmitter { // Build the prompt const prompt = AI_PLAN_CHECK_PROMPT.replace('{TERMINAL_BUFFER}', trimmed); - // Generate temp file and screen name + // Generate temp files and screen name const shortId = this.sessionId.slice(0, 8); const timestamp = Date.now(); this.checkTempFile = join(tmpdir(), `claudeman-plancheck-${shortId}-${timestamp}.txt`); + this.checkPromptFile = join(tmpdir(), `claudeman-plancheck-prompt-${shortId}-${timestamp}.txt`); this.checkScreenName = `claudeman-plancheck-${shortId}`; - // Ensure temp file exists (empty) so we can poll it + // Ensure output temp file exists (empty) so we can poll it writeFileSync(this.checkTempFile, ''); - // Build the command - escape the prompt for shell - const escapedPrompt = prompt.replace(/'/g, "'\\''"); + // Write prompt to file to avoid E2BIG error (argument list too long) + // The prompt can be 8KB+ which exceeds shell argument limits + writeFileSync(this.checkPromptFile, prompt); + + // Build the command - read prompt from file via stdin to avoid argument size limits const modelArg = `--model ${this.config.model}`; const augmentedPath = getAugmentedPath(); - const claudeCmd = `claude -p ${modelArg} --output-format text '${escapedPrompt}'`; - const fullCmd = `export PATH="${augmentedPath}"; ${claudeCmd} > "${this.checkTempFile}" 2>&1; echo "${DONE_MARKER}" >> "${this.checkTempFile}"`; + const claudeCmd = `cat "${this.checkPromptFile}" | claude -p ${modelArg} --output-format text`; + const fullCmd = `export PATH="${augmentedPath}"; ${claudeCmd} > "${this.checkTempFile}" 2>&1; echo "${DONE_MARKER}" >> "${this.checkTempFile}"; rm -f "${this.checkPromptFile}"`; // Spawn screen try { @@ -447,7 +455,7 @@ export class AiPlanChecker extends EventEmitter { this.checkScreenName = null; } - // Delete temp file + // Delete temp files if (this.checkTempFile) { try { if (existsSync(this.checkTempFile)) { @@ -458,6 +466,17 @@ export class AiPlanChecker extends EventEmitter { } this.checkTempFile = null; } + + if (this.checkPromptFile) { + try { + if (existsSync(this.checkPromptFile)) { + unlinkSync(this.checkPromptFile); + } + } catch { + // Best effort cleanup + } + this.checkPromptFile = null; + } } private handleError(errorMsg: string): void { diff --git a/test/respawn-analysis.md b/test/respawn-analysis.md new file mode 100644 index 00000000..32d152b2 --- /dev/null +++ b/test/respawn-analysis.md @@ -0,0 +1,295 @@ +# Respawn Controller Test Analysis + +**Date**: 2026-01-25 +**Analyzed Files**: +- `/home/arkon/default/claudeman/src/respawn-controller.ts` +- `/home/arkon/default/claudeman/src/ai-idle-checker.ts` +- `/home/arkon/default/claudeman/src/ai-plan-checker.ts` +- `/home/arkon/default/claudeman/test/respawn-controller.test.ts` + +--- + +## 1. Test Execution Results + +### Summary +- **Total Tests**: 91 +- **Passed**: 91 +- **Failed**: 0 +- **Duration**: ~10 seconds + +### Type Check Results +- **Type Errors**: 0 + +All tests pass and the codebase is type-safe. + +--- + +## 2. Code Coverage Analysis + +### Coverage Summary + +| File | Statements | Branches | Functions | Lines | +|------|------------|----------|-----------|-------| +| respawn-controller.ts | 62.51% | 59.85% | 74.48% | 62.37% | +| ai-idle-checker.ts | 68.91% | 46.57% | 75% | 69.31% | +| ai-plan-checker.ts | 57.52% | 37.68% | 58.33% | 57.69% | + +### Uncovered Code in respawn-controller.ts + +The following areas lack test coverage: + +1. **Lines 2077-2151**: `sendClear()`, `sendInit()`, `completeCycle()` + - These functions execute during the full respawn cycle + - Tests trigger cycles but don't mock session responses to complete them + +2. **Line 2162**: `checkIdleAndMaybeStart()` + - Called when resuming from pause while already idle + - Only tested superficially with `resume()` call + +3. **Lines 2100-2103**: Clear fallback timer path when `sendInit` is false + - Edge case where `/clear` completes via fallback but init is disabled + +--- + +## 3. Potential Bugs and Issues + +### Issue 1: E2BIG Error in ai-plan-checker.ts (HIGH) + +**File**: `/home/arkon/default/claudeman/src/ai-plan-checker.ts` +**Lines**: 341-346 +**Severity**: HIGH + +**Description**: Unlike `ai-idle-checker.ts`, the plan checker passes the prompt directly as a shell argument rather than via a temp file. This can cause `E2BIG` (argument list too long) errors when the terminal buffer is large. + +**Code**: +```typescript +// ai-plan-checker.ts - PROBLEMATIC +const escapedPrompt = prompt.replace(/'/g, "'\\''"); +const claudeCmd = `claude -p ${modelArg} --output-format text '${escapedPrompt}'`; + +// ai-idle-checker.ts - CORRECT +writeFileSync(this.checkPromptFile, prompt); +const claudeCmd = `cat "${this.checkPromptFile}" | claude -p ${modelArg} --output-format text`; +``` + +**Impact**: With `maxContextChars: 8000`, the prompt can easily exceed shell argument limits (~128KB on most systems), causing the AI plan check to fail with a cryptic error. + +**Suggested Fix**: Modify `ai-plan-checker.ts` to use the same temp file approach as `ai-idle-checker.ts`: +1. Add `checkPromptFile` instance variable +2. Write prompt to temp file +3. Use `cat | claude -p` pattern +4. Clean up temp file in `cleanupCheck()` + +--- + +### Issue 2: Missing Prompt File Cleanup on Timeout (MEDIUM) + +**File**: `/home/arkon/default/claudeman/src/ai-idle-checker.ts` +**Lines**: 392-398 +**Severity**: MEDIUM + +**Description**: When the AI check times out, the `cleanupCheck()` function is called via `finally`. However, the prompt file deletion in `cleanupCheck()` may fail silently if the file is still being read by the `cat` command in the spawned screen. + +**Impact**: Temp files may accumulate in `/tmp` if checks frequently timeout. + +**Suggested Fix**: Add a small delay before file deletion or use a unique timestamp-based naming scheme that guarantees no collisions (already partially implemented but the shell command `rm -f` in the fullCmd also handles this). + +--- + +### Issue 3: Race Condition in cancel() with Promise Resolution (MEDIUM) + +**File**: `/home/arkon/default/claudeman/src/ai-idle-checker.ts` +**Lines**: 265-279 +**Severity**: MEDIUM + +**Description**: The `cancel()` method resolves the pending promise and then calls `cleanupCheck()`. However, the interval poll timer might fire between these two operations and attempt to resolve an already-resolved promise. + +**Code**: +```typescript +cancel(): void { + if (this._status !== 'checking') return; + this.checkCancelled = true; + // Race: poll timer might fire here + if (this.checkResolve) { + this.checkResolve({ verdict: 'ERROR', ... }); + this.checkResolve = null; + } + this.cleanupCheck(); // This clears the poll timer + this._status = 'ready'; +} +``` + +**Impact**: Could theoretically cause a double-resolve, though the guard `if (this.checkCancelled)` in the poll handler mitigates this. + +**Suggested Fix**: Set `checkCancelled = true` and call `cleanupCheck()` (which clears the poll timer) before resolving the promise. + +--- + +### Issue 4: Stale Screen Name Collision (LOW) + +**File**: `/home/arkon/default/claudeman/src/ai-idle-checker.ts`, `ai-plan-checker.ts` +**Lines**: 326-329 (idle), 333-336 (plan) +**Severity**: LOW + +**Description**: Screen names are generated using only `sessionId.slice(0, 8)`, without a unique suffix. While there's a `screen -X quit` to kill leftover screens, two rapid checks could conflict. + +**Code**: +```typescript +this.checkScreenName = `claudeman-aicheck-${shortId}`; +``` + +**Impact**: Very unlikely in practice since checks have cooldowns, but theoretically possible. + +**Suggested Fix**: Already partially mitigated by the `timestamp` in temp file names. Could add timestamp to screen name too: +```typescript +this.checkScreenName = `claudeman-aicheck-${shortId}-${timestamp}`; +``` + +--- + +### Issue 5: DetectionStatus Calculation During ai_checking State (LOW) + +**File**: `/home/arkon/default/claudeman/src/respawn-controller.ts` +**Lines**: 759-762 +**Severity**: LOW + +**Description**: The `outputSilent` calculation uses `config.completionConfirmMs`, but when in `ai_checking` state, the output silence threshold should conceptually be the AI check timeout, not the completion confirm time. + +**Impact**: UI display may show incorrect "silence" status during AI check. + +--- + +## 4. Coverage Gaps + +### Functions Not Fully Tested + +1. **`sendClear()`** - Never reaches execution in tests because the cycle is cut short +2. **`sendInit()`** - Same as above +3. **`completeCycle()`** - Tests don't allow cycles to complete fully +4. **`checkClearComplete()`** - Requires mocking prompt detection after /clear +5. **`checkInitComplete()`** - Requires mocking init completion flow +6. **`sendKickstart()`** - Not tested at all +7. **`checkKickstartComplete()`** - Not tested +8. **`startMonitoringInit()`** - Not tested +9. **`checkMonitoringInitIdle()`** - Not tested + +### Branches Not Tested + +1. `/clear` fallback timer completing (line 2095-2105) +2. `sendInit: false` branch in various locations +3. Kickstart prompt flow when `/init` doesn't trigger work +4. AI checker returning `WORKING` verdict (only test with timeout) +5. AI plan checker `PLAN_MODE` verdict (real check, not just pre-filter) + +### Edge Cases Not Tested + +1. Multiple rapid AI checks before cooldown +2. AI checker disabled mid-check +3. Session terminal buffer being `undefined` on start +4. Buffer trimming behavior at MAX_RESPAWN_BUFFER_SIZE +5. `signalElicitation()` called multiple times + +--- + +## 5. Timing and Race Condition Analysis + +### Potential Timing Issues + +1. **Pre-filter timer vs AI check**: If pre-filter fires while an AI check is already running, it correctly skips (line 1613). + +2. **Completion confirm during AI check**: Output during AI check cancels it (line 1173-1188), which is correct. + +3. **Auto-accept during respawn cycle**: Correctly guards against non-watching state (line 1741). + +4. **Step confirm timer**: Uses same `completionConfirmMs` as idle confirm, which may be too short for complex operations. + +### Timer Cleanup + +All timers appear to be properly cleaned up in `clearTimers()`: +- `idleTimer` +- `stepTimer` +- `clearFallbackTimer` +- `completionConfirmTimer` +- `stepConfirmTimer` +- `autoAcceptTimer` +- `preFilterTimer` +- `noOutputTimer` +- `detectionUpdateTimer` (interval) + +--- + +## 6. Recommendations (Prioritized) + +### Critical (Should Fix) + +1. **Fix E2BIG vulnerability in ai-plan-checker.ts** + - Use temp file for prompt like ai-idle-checker.ts does + - Add `checkPromptFile` instance variable + - Update `cleanupCheck()` to delete prompt file + +### High Priority + +2. **Add integration tests for full respawn cycle** + - Mock session to simulate prompt detection after each step + - Test: update -> clear -> init -> back to watching + - Test: kickstart prompt when init doesn't trigger work + +3. **Fix cancel() race condition in AI checkers** + - Clear poll timer before resolving promise + - Or use a mutex/flag check in poll handler + +### Medium Priority + +4. **Add tests for AI checker WORKING verdict** + - Mock the screen process to return WORKING + - Verify cooldown is started + - Verify controller returns to watching + +5. **Add tests for plan checker PLAN_MODE verdict** + - Mock the screen process to return PLAN_MODE + - Verify Enter is sent + - Verify state transitions + +6. **Add timestamp to screen session names** + - Prevents potential collisions during rapid operations + +### Low Priority + +7. **Improve coverage of edge cases** + - Test `sendInit: false` and `sendClear: false` combinations + - Test buffer trimming behavior + - Test pause/resume with activity detection + +8. **Consider reducing step confirm timeout** + - Currently uses `completionConfirmMs` (10s default) + - May want separate config for step confirmation + +--- + +## 7. MockSession Limitations + +The current `MockSession` class in tests is simplified and doesn't accurately simulate: + +1. **Delayed responses**: Real sessions have processing time between input and output +2. **State persistence**: Real sessions maintain state between commands +3. **Screen integration**: Real sessions use GNU screen for persistence +4. **Token tracking**: Real sessions parse and track token usage + +Consider adding a more sophisticated mock that can: +- Queue delayed responses +- Simulate multi-step workflows +- Return realistic completion messages with timing + +--- + +## 8. Conclusion + +The respawn controller tests are comprehensive for basic functionality but lack coverage for: +- Full respawn cycle completion +- AI checker verdicts (IDLE/WORKING/PLAN_MODE) +- Kickstart functionality +- Edge cases and error handling + +The main bug found is in `ai-plan-checker.ts` which can fail with E2BIG errors on large terminal buffers. This should be fixed by using the temp file approach already implemented in `ai-idle-checker.ts`. + +All 91 tests pass, and there are no type errors, indicating a stable codebase but with room for deeper integration testing. diff --git a/test/respawn-scenarios.md b/test/respawn-scenarios.md new file mode 100644 index 00000000..9272641f --- /dev/null +++ b/test/respawn-scenarios.md @@ -0,0 +1,1460 @@ +# Respawn Controller Test Scenarios + +This document identifies edge cases and scenarios not currently covered by the existing test suite in `test/respawn-controller.test.ts`. These scenarios are designed to find gaps in test coverage and ensure robust behavior. + +--- + +## Table of Contents + +1. [State Transition Scenarios](#state-transition-scenarios) +2. [AI Idle Checker Integration Scenarios](#ai-idle-checker-integration-scenarios) +3. [AI Plan Checker Integration Scenarios](#ai-plan-checker-integration-scenarios) +4. [Timeout and Cooldown Scenarios](#timeout-and-cooldown-scenarios) +5. [Error Recovery Scenarios](#error-recovery-scenarios) +6. [Concurrent Operation Scenarios](#concurrent-operation-scenarios) +7. [Configuration Change Scenarios](#configuration-change-scenarios) +8. [Buffer and Memory Scenarios](#buffer-and-memory-scenarios) +9. [Edge Cases in Pattern Detection](#edge-cases-in-pattern-detection) +10. [Event Emission and Listener Scenarios](#event-emission-and-listener-scenarios) + +--- + +## State Transition Scenarios + +### STS-001: Full Cycle with All Steps Enabled + +**Description**: Complete respawn cycle traversing all states when sendClear, sendInit, and kickstartPrompt are all enabled. + +**Initial State**: `watching` with all config options enabled + +**Actions**: +1. Start controller with `sendClear: true`, `sendInit: true`, `kickstartPrompt: 'continue'` +2. Simulate completion message +3. Wait for confirmation +4. Simulate completion for each step without triggering work after /init + +**Expected Behavior**: +- States visited in order: watching -> confirming_idle -> ai_checking -> sending_update -> waiting_update -> sending_clear -> waiting_clear -> sending_init -> waiting_init -> monitoring_init -> sending_kickstart -> waiting_kickstart -> watching +- All stepSent/stepCompleted events emitted +- Cycle count increments + +**Priority**: HIGH + +--- + +### STS-002: Cycle Skipping Clear Step + +**Description**: Respawn cycle when sendClear is disabled. + +**Initial State**: `watching` with `sendClear: false`, `sendInit: true` + +**Actions**: +1. Trigger idle detection +2. Complete update step + +**Expected Behavior**: +- Should skip directly from waiting_update to sending_init +- No 'clear' step events emitted + +**Priority**: MEDIUM + +--- + +### STS-003: Cycle Skipping Init Step + +**Description**: Respawn cycle when sendInit is disabled. + +**Initial State**: `watching` with `sendClear: true`, `sendInit: false` + +**Actions**: +1. Trigger idle detection +2. Complete update step +3. Complete clear step + +**Expected Behavior**: +- Should complete cycle after clear +- No 'init' step events emitted +- Should not enter monitoring_init state + +**Priority**: MEDIUM + +--- + +### STS-004: Cycle with Neither Clear nor Init + +**Description**: Respawn cycle when both sendClear and sendInit are disabled. + +**Initial State**: `watching` with `sendClear: false`, `sendInit: false` + +**Actions**: +1. Trigger idle detection +2. Complete update step + +**Expected Behavior**: +- Cycle completes immediately after update +- Only update step events emitted + +**Priority**: MEDIUM + +--- + +### STS-005: Work Triggered During monitoring_init + +**Description**: When /init actually triggers Claude to start working. + +**Initial State**: `monitoring_init` with kickstartPrompt configured + +**Actions**: +1. Reach monitoring_init state +2. Simulate working patterns (e.g., "Thinking...") + +**Expected Behavior**: +- Should emit stepCompleted for 'init' +- Should NOT enter sending_kickstart +- Should complete cycle and return to watching +- Log message: "/init triggered work, skipping kickstart" + +**Priority**: HIGH + +--- + +### STS-006: State Transition During Step Timer + +**Description**: Stopping controller while step delay timer is pending. + +**Initial State**: `sending_update` (during interStepDelayMs wait) + +**Actions**: +1. Start controller and trigger idle +2. While in sending_update (during delay), call stop() + +**Expected Behavior**: +- Step timer should be cleared +- No stepSent event emitted +- State transitions to stopped cleanly + +**Priority**: MEDIUM + +--- + +### STS-007: Clear Fallback Timer Trigger + +**Description**: /clear step completes via fallback timer when no prompt detected. + +**Initial State**: `waiting_clear` + +**Actions**: +1. Send /clear command +2. Do NOT simulate prompt detection +3. Wait for CLEAR_FALLBACK_TIMEOUT_MS (10s) + +**Expected Behavior**: +- Fallback timer fires +- Logs "clear fallback: proceeding to /init" +- Proceeds to sendInit (or completeCycle if sendInit false) +- stepCompleted event emitted for 'clear' + +**Priority**: MEDIUM + +--- + +### STS-008: Clear Fallback Timer Cancelled by Prompt + +**Description**: Clear fallback timer is cancelled when prompt is detected. + +**Initial State**: `waiting_clear` + +**Actions**: +1. Send /clear command +2. Simulate prompt detection before fallback timeout + +**Expected Behavior**: +- Fallback timer cancelled with reason "prompt detected" +- timerCancelled event emitted for 'clear-fallback' +- Normal flow continues + +**Priority**: MEDIUM + +--- + +### STS-009: Multiple Rapid State Transitions + +**Description**: Rapid cycling through multiple states without settling. + +**Initial State**: `watching` + +**Actions**: +1. Simulate completion message +2. Before confirmation completes, simulate working +3. Immediately after, simulate completion again +4. Repeat several times rapidly + +**Expected Behavior**: +- Controller handles transitions gracefully +- No duplicate state entries +- Timers properly cleaned up +- No memory leaks from orphaned timers + +**Priority**: HIGH + +--- + +### STS-010: Resume from Non-Watching State + +**Description**: Calling resume() when in a state other than watching. + +**Initial State**: `sending_update` or any non-watching state + +**Actions**: +1. Pause controller +2. Call resume() + +**Expected Behavior**: +- Should remain in current state +- Should not call checkIdleAndMaybeStart +- Only watching state triggers idle check on resume + +**Priority**: LOW + +--- + +## AI Idle Checker Integration Scenarios + +### AIC-001: AI Checker Returns IDLE Verdict + +**Description**: AI idle check completes successfully with IDLE verdict. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check via completion message + silence +2. Mock AI checker to return IDLE + +**Expected Behavior**: +- aiCheckCompleted event emitted with IDLE verdict +- onIdleConfirmed called +- Respawn cycle begins +- No cooldown started + +**Priority**: HIGH + +--- + +### AIC-002: AI Checker Returns WORKING Verdict + +**Description**: AI idle check completes with WORKING verdict. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock AI checker to return WORKING + +**Expected Behavior**: +- aiCheckCompleted event emitted +- State returns to watching +- Cooldown started (aiIdleCheckCooldownMs) +- aiCheckCooldown event emitted +- No respawn cycle started + +**Priority**: HIGH + +--- + +### AIC-003: AI Checker Returns ERROR Verdict + +**Description**: AI idle check fails with ERROR verdict. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock AI checker to return ERROR + +**Expected Behavior**: +- aiCheckFailed event emitted +- State returns to watching +- consecutiveErrors incremented +- Pre-filter and no-output timers restarted + +**Priority**: HIGH + +--- + +### AIC-004: AI Checker Throws Exception + +**Description**: AI idle check throws an exception during execution. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock AI checker to throw Error + +**Expected Behavior**: +- Exception caught +- aiCheckFailed event emitted +- State returns to watching +- Controller remains stable + +**Priority**: HIGH + +--- + +### AIC-005: AI Checker Cancelled Mid-Check + +**Description**: AI check cancelled by new working patterns. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Before completion, simulate working pattern + +**Expected Behavior**: +- AI check cancelled +- Log: "Working patterns detected during AI check, cancelling" +- State returns to watching +- AI check result ignored when it arrives + +**Priority**: HIGH + +--- + +### AIC-006: AI Checker Cancelled by Substantial Output + +**Description**: AI check cancelled when substantial (>2 chars) output arrives. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Simulate output longer than 2 characters (after ANSI stripping) + +**Expected Behavior**: +- AI check cancelled +- Log includes "Substantial output during AI check" +- State returns to watching + +**Priority**: MEDIUM + +--- + +### AIC-007: AI Checker on Cooldown - Check Skipped + +**Description**: AI check attempt while on cooldown from previous WORKING verdict. + +**Initial State**: `watching` with AI checker on cooldown + +**Actions**: +1. Get WORKING verdict (starts cooldown) +2. Immediately trigger another pre-filter pass + +**Expected Behavior**: +- AI check not started +- Log: "AI check on cooldown (Xs remaining), waiting..." +- Controller stays in watching state + +**Priority**: HIGH + +--- + +### AIC-008: AI Checker Cooldown Expires + +**Description**: Controller behavior when AI checker cooldown expires. + +**Initial State**: `watching` with AI checker on cooldown + +**Actions**: +1. Wait for cooldown to expire +2. Check that pre-filter timer is restarted + +**Expected Behavior**: +- aiCheckCooldown event emitted (false, null) +- Pre-filter timer restarted +- Next idle signal can trigger AI check + +**Priority**: MEDIUM + +--- + +### AIC-009: AI Checker Disabled After Max Consecutive Errors + +**Description**: AI checker auto-disables after too many errors. + +**Initial State**: `watching` with AI check enabled + +**Actions**: +1. Cause maxConsecutiveErrors (3) failures +2. Attempt another idle detection + +**Expected Behavior**: +- AI checker status becomes 'disabled' +- disabled event emitted with reason +- Falls back to noOutputTimeoutMs for detection +- Log: "AI check unavailable (disabled)" + +**Priority**: HIGH + +--- + +### AIC-010: AI Checker Re-enabled via Config Update + +**Description**: Re-enabling AI checker after it was disabled. + +**Initial State**: AI checker disabled + +**Actions**: +1. Call updateConfig({ aiIdleCheckEnabled: true }) + +**Expected Behavior**: +- AI checker status becomes 'ready' +- disabledReason cleared +- Next idle detection can use AI check + +**Priority**: MEDIUM + +--- + +### AIC-011: AI Check Triggered via No-Output Fallback + +**Description**: AI check triggered by noOutputTimeoutMs when no output at all. + +**Initial State**: `watching` with no output received + +**Actions**: +1. Start controller +2. Wait for noOutputTimeoutMs without any terminal output + +**Expected Behavior**: +- AI check triggered via no-output fallback path +- aiCheckStarted event emitted +- Log: "No-output fallback: Xs silence" + +**Priority**: MEDIUM + +--- + +### AIC-012: AI Check via Pre-Filter Path (No Completion Message) + +**Description**: AI check triggered by pre-filter timer without completion message. + +**Initial State**: `watching` with output received but no completion message + +**Actions**: +1. Receive some output (sets lastOutputTime) +2. Wait for completionConfirmMs of silence +3. Wait for working patterns to be absent for 3s +4. Wait for tokens to be stable + +**Expected Behavior**: +- Pre-filter passes +- AI check started +- Works even without completion message detection + +**Priority**: MEDIUM + +--- + +### AIC-013: State Change During AI Check - Result Ignored + +**Description**: AI check result arrives after state has changed. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Stop controller before result arrives +3. AI check completes + +**Expected Behavior**: +- Result ignored +- Log: "AI check result ignored (state is now stopped)" +- No state transition attempted + +**Priority**: MEDIUM + +--- + +## AI Plan Checker Integration Scenarios + +### APC-001: Plan Check Returns PLAN_MODE + +**Description**: AI plan check confirms plan mode, triggers auto-accept. + +**Initial State**: `watching` with plan mode UI in buffer + +**Actions**: +1. Simulate plan mode output (numbered list + selector) +2. Wait for autoAcceptDelayMs +3. Mock plan checker to return PLAN_MODE + +**Expected Behavior**: +- planCheckCompleted event emitted +- Enter sent via writeViaScreen +- autoAcceptSent event emitted +- hasReceivedOutput reset to false + +**Priority**: HIGH + +--- + +### APC-002: Plan Check Returns NOT_PLAN_MODE + +**Description**: AI plan check determines not plan mode. + +**Initial State**: `watching` with ambiguous output + +**Actions**: +1. Simulate output that passes pre-filter +2. Mock plan checker to return NOT_PLAN_MODE + +**Expected Behavior**: +- planCheckCompleted event emitted +- No Enter sent +- Cooldown started (aiPlanCheckCooldownMs) + +**Priority**: HIGH + +--- + +### APC-003: Plan Check Already Checking + +**Description**: Plan check attempt while already checking. + +**Initial State**: Plan checker status = 'checking' + +**Actions**: +1. Start a plan check +2. Trigger another auto-accept attempt before first completes + +**Expected Behavior**: +- Second check skipped +- Log: "plan check already in progress" +- Only one check runs + +**Priority**: MEDIUM + +--- + +### APC-004: Plan Check Cancelled by New Output + +**Description**: New output during plan check cancels it. + +**Initial State**: Plan check running + +**Actions**: +1. Start plan check +2. Simulate new terminal output + +**Expected Behavior**: +- Plan check cancelled +- Log: "New output during plan check, cancelling (stale)" +- Result discarded + +**Priority**: HIGH + +--- + +### APC-005: Plan Check Result Stale (Output During Check) + +**Description**: Plan check completes but output arrived during check. + +**Initial State**: Plan check running + +**Actions**: +1. Start plan check +2. Record planCheckStartTime +3. Simulate output (updates lastOutputTime) +4. Plan check returns PLAN_MODE + +**Expected Behavior**: +- Result discarded (lastOutputTime > planCheckStartTime) +- Log: "Result discarded (output arrived during check)" +- No Enter sent + +**Priority**: HIGH + +--- + +### APC-006: Plan Check Result When State Changed + +**Description**: Plan check returns PLAN_MODE but state is no longer watching. + +**Initial State**: Plan check running + +**Actions**: +1. Start plan check +2. Trigger respawn cycle (state changes to sending_update) +3. Plan check returns PLAN_MODE + +**Expected Behavior**: +- Result not acted upon +- Log includes "state is X, not sending Enter" +- No auto-accept + +**Priority**: MEDIUM + +--- + +### APC-007: Pre-Filter Blocks - No Numbered Options + +**Description**: Pre-filter rejects output without numbered options. + +**Initial State**: `watching` + +**Actions**: +1. Simulate output without numbered list pattern +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter fails (PLAN_MODE_OPTION_PATTERN not matched) +- Log: "pre-filter did not match plan mode patterns" +- No plan check started + +**Priority**: MEDIUM + +--- + +### APC-008: Pre-Filter Blocks - No Selector Arrow + +**Description**: Pre-filter rejects output without selector indicator. + +**Initial State**: `watching` + +**Actions**: +1. Simulate numbered list without selector (no ">" or arrow) +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter fails (PLAN_MODE_SELECTOR_PATTERN not matched) +- No plan check started + +**Priority**: MEDIUM + +--- + +### APC-009: Pre-Filter Blocks - Working Pattern After Selector + +**Description**: Pre-filter rejects when working pattern appears after selector. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "1. Yes\n> 1.\nThinking..." +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter fails (working pattern after selector) +- No plan check started + +**Priority**: MEDIUM + +--- + +### APC-010: Pre-Filter Passes - Working Pattern Before Selector + +**Description**: Pre-filter allows working patterns if they appear before the selector. + +**Initial State**: `watching` with AI plan check disabled + +**Actions**: +1. Simulate: "Thinking...\nDone.\n> 1. Yes\n 2. No" +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter passes (working pattern is before selector) +- Auto-accept triggered (since AI plan check disabled) + +**Priority**: MEDIUM + +--- + +### APC-011: Plan Check on Cooldown + +**Description**: Plan check skipped when on cooldown. + +**Initial State**: Plan checker on cooldown + +**Actions**: +1. Get NOT_PLAN_MODE verdict (starts cooldown) +2. Simulate plan mode output +3. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Plan check skipped +- Log: "plan checker on cooldown (Xs remaining)" +- No auto-accept + +**Priority**: MEDIUM + +--- + +## Timeout and Cooldown Scenarios + +### TC-001: Step Confirmation Timer Interrupted by Working + +**Description**: Step confirmation timer cancelled by working patterns. + +**Initial State**: `waiting_update` with step confirm timer running + +**Actions**: +1. Simulate completion message in waiting_update +2. Before confirmation timer fires, simulate working pattern + +**Expected Behavior**: +- Step confirm timer cancelled +- Log: "Step confirmation cancelled (working detected)" +- Remains in waiting_update, waiting for real completion + +**Priority**: HIGH + +--- + +### TC-002: Completion Confirm Timer Interrupted by Output + +**Description**: Confirmation timer restart when output arrives during confirmation. + +**Initial State**: `confirming_idle` with timer running + +**Actions**: +1. Detect completion message +2. Timer fires but lastOutputTime changed + +**Expected Behavior**: +- Timer restarts instead of confirming +- Log: "Output during confirmation, resetting" +- State remains confirming_idle + +**Priority**: MEDIUM + +--- + +### TC-003: Multiple Cooldowns Active Simultaneously + +**Description**: Both AI idle checker and plan checker on cooldown. + +**Initial State**: Both checkers on cooldown + +**Actions**: +1. Trigger WORKING verdict on idle checker +2. Trigger NOT_PLAN_MODE on plan checker +3. Try to trigger both checks + +**Expected Behavior**: +- Both checks skipped +- Controller falls back appropriately +- Cooldowns expire independently + +**Priority**: MEDIUM + +--- + +### TC-004: Zero-Duration Timers + +**Description**: Configuration with zero-duration timeouts. + +**Initial State**: `watching` with `completionConfirmMs: 0`, `interStepDelayMs: 0` + +**Actions**: +1. Trigger completion message +2. Complete full cycle + +**Expected Behavior**: +- Timers fire immediately (or within next event loop) +- Cycle completes without hanging +- No errors from zero-duration setTimeout + +**Priority**: LOW + +--- + +### TC-005: Very Long Timeout Values + +**Description**: Configuration with extremely long timeouts. + +**Initial State**: `watching` with `noOutputTimeoutMs: 3600000` (1 hour) + +**Actions**: +1. Start controller +2. Stop before timeout + +**Expected Behavior**: +- Timer properly cleared on stop +- No dangling timers +- Memory not leaked + +**Priority**: LOW + +--- + +### TC-006: Timer Tracking Accuracy + +**Description**: Active timer info reflects actual remaining time. + +**Initial State**: Timer running + +**Actions**: +1. Start a tracked timer (e.g., completion-confirm) +2. Call getActiveTimers() at various intervals +3. Check remainingMs decreases appropriately + +**Expected Behavior**: +- remainingMs decreases over time +- Never negative +- Removed from list after firing + +**Priority**: LOW + +--- + +## Error Recovery Scenarios + +### ER-001: Session Event Handler Throws + +**Description**: Exception thrown in terminal data handler. + +**Initial State**: `watching` + +**Actions**: +1. Cause handleTerminalData to throw (via malformed input) +2. Continue sending normal output + +**Expected Behavior**: +- Controller should not crash +- Should handle exception gracefully +- Should continue processing subsequent events + +**Priority**: HIGH + +--- + +### ER-002: writeViaScreen Fails + +**Description**: Writing to session fails during step send. + +**Initial State**: `sending_update` + +**Actions**: +1. Mock writeViaScreen to return false or throw +2. Let step delay timer fire + +**Expected Behavior**: +- Error should be caught +- Controller should emit error event +- Should not hang in sending state indefinitely + +**Priority**: MEDIUM + +--- + +### ER-003: Recovery After Partial Cycle Failure + +**Description**: Controller recovers after failing mid-cycle. + +**Initial State**: `waiting_clear` when error occurs + +**Actions**: +1. Simulate error during waiting_clear +2. Controller returns to watching or stopped +3. Try to start new cycle + +**Expected Behavior**: +- Clean state for new cycle +- No leftover timers or state +- New cycle works normally + +**Priority**: MEDIUM + +--- + +### ER-004: AI Checker Process Spawn Failure + +**Description**: Screen process fails to spawn for AI check. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock screen spawn to fail + +**Expected Behavior**: +- Error caught +- aiCheckFailed event emitted +- consecutiveErrors incremented +- Falls back gracefully + +**Priority**: MEDIUM + +--- + +### ER-005: Temp File Cleanup Failure + +**Description**: Temp file deletion fails in AI checker cleanup. + +**Initial State**: AI check completing + +**Actions**: +1. Complete AI check +2. Mock unlink to throw + +**Expected Behavior**: +- Exception caught (best effort cleanup) +- Check still completes +- No crash + +**Priority**: LOW + +--- + +## Concurrent Operation Scenarios + +### CO-001: Start Called While Stopping + +**Description**: Calling start() immediately after stop(). + +**Initial State**: Transitioning from running to stopped + +**Actions**: +1. Call stop() +2. Immediately call start() + +**Expected Behavior**: +- Clean restart +- No duplicate listeners +- Single watching state + +**Priority**: MEDIUM + +--- + +### CO-002: Multiple Terminal Events in Same Tick + +**Description**: Multiple terminal events processed synchronously. + +**Initial State**: `watching` + +**Actions**: +1. Emit multiple 'terminal' events without yielding +2. Check state consistency + +**Expected Behavior**: +- All events processed in order +- State remains consistent +- No race conditions in timer management + +**Priority**: HIGH + +--- + +### CO-003: Config Update During AI Check + +**Description**: Updating AI config while check is in progress. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Call updateConfig({ aiIdleCheckEnabled: false }) +3. AI check completes + +**Expected Behavior**: +- Pending check completes +- Future checks use new config +- No crash + +**Priority**: LOW + +--- + +### CO-004: Pause During AI Check + +**Description**: Pausing controller while AI check is running. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Call pause() +3. AI check completes + +**Expected Behavior**: +- Timers cleared by pause +- AI check result may arrive but timers won't fire +- Resume can restart monitoring + +**Priority**: LOW + +--- + +### CO-005: Buffer Append During Trim + +**Description**: New data arrives while buffer is being trimmed. + +**Initial State**: Buffer near MAX_RESPAWN_BUFFER_SIZE + +**Actions**: +1. Fill buffer to trigger trim +2. Simultaneously append more data + +**Expected Behavior**: +- Buffer stays within limits +- No data corruption +- Append operation completes + +**Priority**: LOW + +--- + +## Configuration Change Scenarios + +### CC-001: Disable Respawn While Running + +**Description**: Setting enabled to false while controller is running. + +**Initial State**: `watching` with cycle in progress + +**Actions**: +1. Call updateConfig({ enabled: false }) +2. Check controller behavior + +**Expected Behavior**: +- Config updated (for reference) +- Controller continues current operation +- On restart, start() will be no-op + +**Priority**: LOW + +--- + +### CC-002: Change updatePrompt During Cycle + +**Description**: Changing update prompt while cycle is in progress. + +**Initial State**: `waiting_update` + +**Actions**: +1. Call updateConfig({ updatePrompt: 'new prompt' }) +2. Start next cycle + +**Expected Behavior**: +- Current cycle uses old prompt (already sent) +- Next cycle uses new prompt + +**Priority**: LOW + +--- + +### CC-003: Toggle sendClear Mid-Cycle + +**Description**: Changing sendClear while in waiting_update. + +**Initial State**: `waiting_update` with sendClear: true + +**Actions**: +1. Call updateConfig({ sendClear: false }) +2. Update step completes + +**Expected Behavior**: +- Should use updated config value +- Skip clear and proceed appropriately + +**Priority**: MEDIUM + +--- + +### CC-004: Change Timeout Values During Wait + +**Description**: Modifying timeout values while timer is running. + +**Initial State**: Timer running with completionConfirmMs: 10000 + +**Actions**: +1. Call updateConfig({ completionConfirmMs: 1000 }) +2. Check when timer fires + +**Expected Behavior**: +- Current timer uses old value +- Next timer uses new value +- No crash or undefined behavior + +**Priority**: LOW + +--- + +### CC-005: Update kickstartPrompt to Undefined + +**Description**: Removing kickstart prompt during monitoring_init. + +**Initial State**: `monitoring_init` with kickstartPrompt set + +**Actions**: +1. Call updateConfig({ kickstartPrompt: undefined }) +2. /init doesn't trigger work + +**Expected Behavior**: +- Should check config at decision time +- May skip kickstart or use stale value +- (Behavior should be defined) + +**Priority**: LOW + +--- + +## Buffer and Memory Scenarios + +### BM-001: Buffer Trim at Exact Boundary + +**Description**: Buffer exactly at MAX_RESPAWN_BUFFER_SIZE. + +**Initial State**: Buffer at exactly 1MB + +**Actions**: +1. Fill buffer to exactly 1MB +2. Append 1 more character + +**Expected Behavior**: +- Trim triggered +- Buffer reduced to RESPAWN_BUFFER_TRIM_SIZE (512KB) +- Most recent data preserved + +**Priority**: LOW + +--- + +### BM-002: Continuous High-Volume Output + +**Description**: Sustained high-volume terminal output. + +**Initial State**: `watching` + +**Actions**: +1. Send 1MB of data per second for 10 seconds +2. Check memory usage and behavior + +**Expected Behavior**: +- Buffer stays bounded +- No memory growth over time +- Detection still functions + +**Priority**: MEDIUM + +--- + +### BM-003: Buffer Clear During Pattern Match + +**Description**: Buffer cleared while pattern detection is occurring. + +**Initial State**: Processing terminal data + +**Actions**: +1. Process data that triggers completeCycle +2. completeCycle clears buffer +3. More data arrives in same handler + +**Expected Behavior**: +- New data appended to fresh buffer +- No stale data patterns matched + +**Priority**: LOW + +--- + +### BM-004: Action Log Growth + +**Description**: Action log entries accumulating over time. + +**Initial State**: Many cycles completed + +**Actions**: +1. Run many cycles (>20) +2. Check recentActions length + +**Expected Behavior**: +- Limited to 20 entries max +- Oldest entries discarded +- Memory stable + +**Priority**: LOW + +--- + +## Edge Cases in Pattern Detection + +### PD-001: Completion Message Without "Worked" + +**Description**: Time duration pattern without "Worked" prefix. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "Waiting for 5s before retry" +2. Check if completion detected + +**Expected Behavior**: +- NOT detected as completion (requires "Worked" prefix) +- No false positive + +**Priority**: HIGH + +--- + +### PD-002: Nested Working Patterns in Text + +**Description**: Working pattern words in non-working context. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "The 'Thinking' file was created" +2. Check workingDetected status + +**Expected Behavior**: +- Currently: Would detect as working (false positive) +- Expected: Should not detect (pattern in quotes) +- (This is a known limitation) + +**Priority**: LOW (known limitation) + +--- + +### PD-003: Unicode Prompt Variants + +**Description**: Various Unicode prompt characters. + +**Initial State**: `watching` + +**Actions**: +1. Test: Regular '>' vs Unicode '>' vs fullwidth '>' +2. Check promptDetected + +**Expected Behavior**: +- Standard patterns detected +- Variant characters may not be detected +- (Document which are supported) + +**Priority**: LOW + +--- + +### PD-004: ANSI Codes Splitting Patterns + +**Description**: ANSI escape codes inserted within patterns. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "Wor\x1b[32mked\x1b[0m for 2m 46s" +2. Check completion detection + +**Expected Behavior**: +- Pattern may not match (ANSI in middle of "Worked") +- (This is edge case behavior) + +**Priority**: LOW + +--- + +### PD-005: Token Count at Boundary Values + +**Description**: Token counts with k/M suffixes. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "999.9k tokens" -> "1.0M tokens" +2. Check lastTokenCount value + +**Expected Behavior**: +- 999.9k = 999900 +- 1.0M = 1000000 +- Token change detected + +**Priority**: LOW + +--- + +### PD-006: Empty String Token Pattern + +**Description**: Token pattern with malformed numbers. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "tokens", " tokens", ".k tokens" +2. Check extractTokenCount returns + +**Expected Behavior**: +- Returns null for invalid patterns +- No crash + +**Priority**: LOW + +--- + +### PD-007: Multiple Completion Messages Same Data + +**Description**: Multiple "Worked for Xm Xs" in single terminal chunk. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "Worked for 1s... Worked for 2m 30s" +2. Check behavior + +**Expected Behavior**: +- First match triggers detection +- completionMessageTime set +- Single confirmation timer started + +**Priority**: LOW + +--- + +### PD-008: Spinner Character in Normal Text + +**Description**: Spinner Unicode character in regular output. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "The sequence is: ⠋ ⠙ ⠹" +2. Check workingDetected + +**Expected Behavior**: +- Currently: Detects as working +- This is expected behavior (conservative) + +**Priority**: LOW (expected behavior) + +--- + +## Event Emission and Listener Scenarios + +### EE-001: No Listeners Registered + +**Description**: Events emitted with no listeners. + +**Initial State**: Controller with no event listeners + +**Actions**: +1. Start controller +2. Run through cycle + +**Expected Behavior**: +- No errors +- Events still emitted (just not handled) +- Controller functions normally + +**Priority**: LOW + +--- + +### EE-002: Listener Throws Exception + +**Description**: Event listener throws during event handling. + +**Initial State**: Listener registered that throws + +**Actions**: +1. Register listener: on('stateChanged', () => { throw Error }) +2. Cause state change + +**Expected Behavior**: +- Exception propagates (EventEmitter default) +- (May want to catch in production) + +**Priority**: MEDIUM + +--- + +### EE-003: Listener Removes Itself + +**Description**: Listener that removes itself during event handling. + +**Initial State**: Self-removing listener registered + +**Actions**: +1. Register one-time listener +2. Cause event + +**Expected Behavior**: +- Listener called once +- Properly removed +- No memory leak + +**Priority**: LOW + +--- + +### EE-004: DetectionUpdate Interval Cleanup + +**Description**: Detection update interval properly cleaned up. + +**Initial State**: Controller running with detection updates + +**Actions**: +1. Start controller (starts 500ms interval) +2. Stop controller +3. Check interval cleared + +**Expected Behavior**: +- detectionUpdateTimer cleared on stop +- No interval continuing after stop +- No memory leak + +**Priority**: MEDIUM + +--- + +### EE-005: Timer Events During Stop + +**Description**: Timer fires during stop() execution. + +**Initial State**: Multiple timers running + +**Actions**: +1. Call stop() +2. Timer fires during cleanup + +**Expected Behavior**: +- Timer callback checks state +- No action taken if stopped +- Clean shutdown + +**Priority**: LOW + +--- + +## Summary + +### Priority Distribution + +| Priority | Count | Description | +|----------|-------|-------------| +| HIGH | 18 | Critical functionality and common paths | +| MEDIUM | 24 | Important edge cases and integrations | +| LOW | 22 | Rare edge cases and nice-to-have coverage | + +### Coverage Gaps Identified + +1. **Full cycle with all steps enabled** - No test covers the complete path through all states +2. **Clear fallback timer** - Not tested (10s timeout when no prompt detected) +3. **AI checker consecutive errors leading to disable** - Not integration tested +4. **Plan checker result discarding** - Stale result handling not fully tested +5. **Step confirmation timer** - The completionConfirmMs wait after each step not tested +6. **Working pattern detection during waiting states** - Limited coverage +7. **Buffer edge cases** - Trim behavior not tested +8. **Timer tracking for UI** - getActiveTimers() accuracy not verified +9. **Event listener error handling** - Not tested +10. **Elicitation flag lifecycle** - Partial coverage + +### Recommended Test Implementation Order + +1. STS-001 (Full cycle) - Establishes complete flow understanding +2. AIC-001, AIC-002, AIC-003 - Core AI checker verdicts +3. APC-001, APC-004, APC-005 - Core plan checker scenarios +4. STS-005 (Work during monitoring_init) - Important flow branch +5. TC-001 (Step confirm interrupted) - Timer reliability +6. ER-001 (Handler throws) - Robustness +7. CO-002 (Multiple events same tick) - Concurrency safety diff --git a/test/respawn-test-plan.md b/test/respawn-test-plan.md new file mode 100644 index 00000000..3bf76f69 --- /dev/null +++ b/test/respawn-test-plan.md @@ -0,0 +1,482 @@ +# Respawn Controller Test Plan + +This document describes the testing environment, architecture, and strategies for testing the RespawnController and related AI checker components. + +## Table of Contents + +1. [Test Environment Architecture](#test-environment-architecture) +2. [Component Interaction Diagram](#component-interaction-diagram) +3. [Mock Components](#mock-components) +4. [Port Allocation](#port-allocation) +5. [Test Isolation Strategy](#test-isolation-strategy) +6. [Cleanup Procedures](#cleanup-procedures) +7. [Test Categories](#test-categories) +8. [Usage Examples](#usage-examples) + +--- + +## Test Environment Architecture + +The RespawnController testing environment uses a layered approach to isolate components and enable deterministic testing without spawning real Claude CLI processes. + +``` ++------------------------------------------+ +| Test Runner (Vitest) | ++------------------------------------------+ + | | + v v ++------------------+ +------------------+ +| Unit Tests | | Integration Tests| +| (No real I/O) | | (Uses ports) | ++------------------+ +------------------+ + | | + v v ++------------------------------------------+ +| Test Utilities Layer | +| - MockSession | +| - MockAiIdleChecker | +| - MockAiPlanChecker | +| - TimeController | +| - State/Event Recorders | ++------------------------------------------+ + | + v ++------------------------------------------+ +| Real Components Under Test | +| - RespawnController | +| - RespawnConfig types | +| - State machine logic | ++------------------------------------------+ +``` + +### Key Principles + +1. **No Real Claude CLI**: Tests use MockAiIdleChecker and MockAiPlanChecker instead of spawning real Claude processes +2. **No Real Screens**: MockSession simulates terminal I/O without GNU screen +3. **Deterministic Timing**: Tests can use real timers (short timeouts) or fake timers for precise control +4. **Isolated State**: Each test gets fresh instances with no shared state + +--- + +## Component Interaction Diagram + +``` + RespawnController + | + +-----------------+-----------------+ + | | | + v v v + Session AiIdleChecker AiPlanChecker + (MockSession) (MockAiIdleChecker) (MockAiPlanChecker) + | | | + v | | + Terminal Events AI Check AI Check + (simulateXxx) Verdicts Verdicts + | | | + +--------+--------+---------+-------+ + | + v + State Machine + (watching -> ... -> stopped) + | + v + Event Emission + (stateChanged, stepSent, etc.) +``` + +### Data Flow + +1. **Terminal Output Flow**: + - `MockSession.simulateTerminalOutput(data)` -> `session.emit('terminal', data)` + - RespawnController receives terminal event in `handleTerminalData()` + - Pattern detection (completion, working, prompt) + - Timer management (start/reset/cancel) + +2. **AI Check Flow**: + - Pre-filter conditions met -> `tryStartAiCheck()` + - MockAiIdleChecker returns queued or default verdict + - Controller processes verdict (IDLE -> start cycle, WORKING -> cooldown) + +3. **State Transition Flow**: + - Internal state changes via `setState()` + - Events emitted for external monitoring + - Timers started/cancelled based on state + +--- + +## Mock Components + +### MockSession + +Enhanced session mock that simulates Claude Code terminal behavior. + +**Key Methods**: +- `simulateTerminalOutput(data)` - Raw terminal output +- `simulateCompletionMessage(duration?)` - "Worked for Xm Xs" pattern +- `simulateWorking(text?)` - Spinner/activity indicators +- `simulatePlanModePrompt()` - Numbered selection menu +- `simulateElicitationDialog()` - AskUserQuestion prompt +- `writeBuffer` - Captures all writes for assertions + +**Usage**: +```typescript +const session = createMockSession(); +session.simulateCompletionMessage(); +await waitForState(controller, 'confirming_idle'); +``` + +### MockAiIdleChecker + +Mock for AI idle detection that returns configurable verdicts. + +**Key Features**: +- Queue-based result system (FIFO) +- Default verdict when queue empty +- Cooldown simulation +- Error/disabled state simulation + +**Usage**: +```typescript +const checker = new MockAiIdleChecker('session-id'); +checker.setNextIdle('Task completed'); +// or +checker.queueResults( + { verdict: 'WORKING', reasoning: 'Still active', durationMs: 100 }, + { verdict: 'IDLE', reasoning: 'Now idle', durationMs: 150 } +); +``` + +### MockAiPlanChecker + +Mock for AI plan mode detection. + +**Key Features**: +- Same queue/default pattern as MockAiIdleChecker +- PLAN_MODE/NOT_PLAN_MODE verdicts +- Cooldown after NOT_PLAN_MODE + +**Usage**: +```typescript +const checker = new MockAiPlanChecker('session-id'); +checker.setNextPlanMode('Approval prompt detected'); +``` + +### TimeController + +Wrapper around Vitest's fake timers for deterministic timing tests. + +**Usage**: +```typescript +const time = createTimeController(); +// In test: +await time.advanceBy(1000); // Advance 1 second +await time.runAllTimers(); // Run all pending +// In afterEach: +time.useRealTimers(); +``` + +--- + +## Port Allocation + +Integration tests that spawn web servers use unique ports to avoid conflicts. + +| Port | Test File | Notes | +|-------|------------------------------|----------------------------------| +| 3099 | quick-start.test.ts | Basic startup tests | +| 3102 | session.test.ts | Session lifecycle tests | +| 3105 | scheduled-runs.test.ts | Scheduled task tests | +| 3107 | sse-events.test.ts | Server-Sent Events tests | +| 3110 | edge-cases.test.ts | Edge case handling | +| 3115 | integration-flows.test.ts | End-to-end flows | +| 3120 | session-cleanup.test.ts | Session cleanup tests | +| 3125 | ralph-integration.test.ts | Ralph loop integration | +| 3127 | *Available* | Next integration test | +| 3128 | *Available* | Reserved for respawn integration| +| 3129+ | *Available* | Future tests | + +### Port Usage Guidelines + +1. **Unit Tests**: No port needed (MockSession, no real server) +2. **Integration Tests**: Pick next available port (3127+) +3. **Parallel Safety**: `fileParallelism: false` ensures sequential execution + +--- + +## Test Isolation Strategy + +### Per-Test Isolation + +1. **Fresh Instances**: Each test creates new MockSession, MockAiIdleChecker, RespawnController +2. **No Shared State**: No global variables between tests +3. **Timer Cleanup**: Real timers have short timeouts; fake timers reset between tests + +### Per-File Isolation + +1. **beforeEach**: Create fresh instances +2. **afterEach**: + - Call `controller.stop()` to clear timers + - Reset time controller if using fake timers + - Clear mock queues + +### Cross-File Isolation + +1. **Sequential Execution**: `fileParallelism: false` in vitest.config.ts +2. **Screen Session Limits**: Max 10 concurrent (enforced by setup.ts) +3. **Orphan Cleanup**: `afterAll` cleans up any leaked screens + +--- + +## Cleanup Procedures + +### During Test Execution + +```typescript +afterEach(() => { + controller.stop(); // Clear all timers + session.removeAllListeners(); // Remove event handlers + mockChecker.reset(); // Clear queued results +}); +``` + +### After Test Suite + +The global `test/setup.ts` handles: + +1. **Screen Cleanup**: Kills any orphaned `claudeman-*` screens created during tests +2. **Process Cleanup**: Kills any Claude processes spawned by tests +3. **Pre-existing Protection**: Never kills screens that existed before tests started + +### Emergency Cleanup + +```typescript +import { forceCleanupAllTestResources } from './setup.js'; + +// Call if tests fail catastrophically +forceCleanupAllTestResources(); +``` + +--- + +## Test Categories + +### 1. Unit Tests (MockSession-based) + +Test the RespawnController state machine without real I/O. + +**File**: `test/respawn-controller.test.ts` + +**What to Test**: +- State transitions (watching -> confirming_idle -> ai_checking -> ...) +- Timer behavior (completion confirm, no-output fallback) +- Pattern detection (completion message, working patterns) +- Configuration handling +- Event emission + +**Example**: +```typescript +it('should transition to ai_checking when pre-filter met', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, { + ...FAST_TEST_CONFIG, + aiIdleCheckEnabled: true, + }); + + controller.start(); + session.simulateCompletionMessage(); + await new Promise(r => setTimeout(r, 100)); + + expect(controller.state).toBe('ai_checking'); +}); +``` + +### 2. AI Checker Unit Tests + +Test MockAiIdleChecker and MockAiPlanChecker behavior. + +**File**: `test/ai-idle-checker.test.ts`, `test/ai-plan-checker.test.ts` + +**What to Test**: +- Verdict queuing +- Cooldown behavior +- Error handling +- Disabled state + +### 3. State Machine Tests + +Comprehensive state transition testing. + +**File**: `test/respawn-controller.test.ts` (State Machine section) + +**What to Test**: +- All state transitions in the diagram +- Skip paths (sendClear: false, sendInit: false) +- Kickstart path +- Interruption handling (working patterns during transitions) + +### 4. Integration Tests (if needed) + +Full stack tests with real server but mocked Claude CLI. + +**Port**: 3128 (reserved) + +**What to Test**: +- API endpoints for respawn control +- SSE event broadcasting +- State persistence + +--- + +## Usage Examples + +### Basic Test Setup + +```typescript +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { RespawnController } from '../src/respawn-controller.js'; +import { + createMockSession, + MockAiIdleChecker, + FAST_TEST_CONFIG, + createStateTracker, +} from './respawn-test-utils.js'; + +describe('RespawnController Example', () => { + let session: MockSession; + let controller: RespawnController; + let stateTracker: ReturnType; + + beforeEach(() => { + session = createMockSession(); + controller = new RespawnController(session, FAST_TEST_CONFIG); + stateTracker = createStateTracker(); + controller.on('stateChanged', stateTracker.record); + }); + + afterEach(() => { + controller.stop(); + }); + + it('should start in watching state', () => { + controller.start(); + expect(controller.state).toBe('watching'); + }); +}); +``` + +### Testing with Mock AI Checker + +```typescript +// Note: Currently the RespawnController creates its own AI checkers internally. +// To inject mocks, you would need to modify the controller to accept them, +// or test the mock checkers independently and test the controller with +// aiIdleCheckEnabled: false. + +describe('RespawnController with AI Check Disabled', () => { + it('should fall back to direct idle on completion', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, { + ...FAST_TEST_CONFIG, + aiIdleCheckEnabled: false, // Falls back to timer-based + }); + + let cycleStarted = false; + controller.on('respawnCycleStarted', () => { cycleStarted = true; }); + + controller.start(); + session.simulateCompletionMessage(); + + await new Promise(r => setTimeout(r, 200)); + + expect(cycleStarted).toBe(true); + }); +}); +``` + +### Testing State Transitions + +```typescript +describe('State Transitions', () => { + it('should track full respawn cycle', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, { + ...FAST_TEST_CONFIG, + sendClear: true, + sendInit: true, + }); + const tracker = createStateTracker(); + controller.on('stateChanged', tracker.record); + + controller.start(); + session.simulateCompletionMessage(); + + // Wait for full cycle + await new Promise(r => setTimeout(r, 500)); + + expect(tracker.hasVisited('watching')).toBe(true); + expect(tracker.hasVisited('confirming_idle')).toBe(true); + expect(tracker.hasVisited('sending_update')).toBe(true); + expect(tracker.hasVisited('waiting_update')).toBe(true); + }); +}); +``` + +### Testing Event Emission + +```typescript +describe('Event Emission', () => { + it('should emit all lifecycle events', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, FAST_TEST_CONFIG); + const recorder = createEventRecorder(); + + controller.on('stateChanged', recorder.handler('stateChanged')); + controller.on('respawnCycleStarted', recorder.handler('respawnCycleStarted')); + controller.on('stepSent', recorder.handler('stepSent')); + + controller.start(); + session.simulateCompletionMessage(); + + await new Promise(r => setTimeout(r, 200)); + + expect(recorder.hasEvent('respawnCycleStarted')).toBe(true); + expect(recorder.hasEvent('stepSent')).toBe(true); + }); +}); +``` + +--- + +## Future Considerations + +### Dependency Injection for AI Checkers + +To enable true mock injection, consider modifying RespawnController to accept optional AI checker instances in the constructor: + +```typescript +constructor( + session: Session, + config: Partial = {}, + aiChecker?: AiIdleChecker, + planChecker?: AiPlanChecker +) { + // Use provided checkers or create defaults + this.aiChecker = aiChecker || new AiIdleChecker(session.id, { ... }); + this.planChecker = planChecker || new AiPlanChecker(session.id, { ... }); +} +``` + +This would allow tests to inject MockAiIdleChecker/MockAiPlanChecker directly. + +### Real AI Checker Tests + +For testing the real AiIdleChecker and AiPlanChecker (which spawn Claude CLI), create a separate test file with: +- Longer timeouts +- Skip conditions for CI without Claude CLI +- Actual screen session usage + +```typescript +describe.skipIf(!process.env.CLAUDE_CLI_AVAILABLE)('Real AI Checkers', () => { + // Tests that spawn real Claude CLI +}); +``` diff --git a/test/respawn-test-utils.ts b/test/respawn-test-utils.ts new file mode 100644 index 00000000..29659ba6 --- /dev/null +++ b/test/respawn-test-utils.ts @@ -0,0 +1,1045 @@ +/** + * @fileoverview Test Utilities for RespawnController Tests + * + * Provides mocks, helpers, and utilities for testing the RespawnController + * and its related AI checker components without spawning real Claude CLI processes. + * + * ## Contents + * + * - MockSession: Enhanced mock session for testing + * - MockAiIdleChecker: Mock for AI idle checker with configurable verdicts + * - MockAiPlanChecker: Mock for AI plan mode checker with configurable verdicts + * - State transition helpers + * - Time manipulation utilities + * - Factory functions for pre-configured controllers + * + * @module test/respawn-test-utils + */ + +import { EventEmitter } from 'node:events'; +import { vi } from 'vitest'; +import type { Session } from '../src/session.js'; +import type { RespawnConfig, RespawnState, DetectionStatus } from '../src/respawn-controller.js'; +import type { + AiCheckResult, + AiCheckState, + AiCheckStatus, + AiCheckVerdict, + AiIdleCheckConfig, +} from '../src/ai-idle-checker.js'; +import type { + AiPlanCheckResult, + AiPlanCheckState, + AiPlanCheckStatus, + AiPlanCheckVerdict, + AiPlanCheckConfig, +} from '../src/ai-plan-checker.js'; + +// ========== Time Manipulation Utilities ========== + +/** + * Time controller for testing timeout-based behavior. + * Uses vitest's fake timers for deterministic time control. + */ +export interface TimeController { + /** Advance time by the specified milliseconds */ + advanceBy(ms: number): Promise; + /** Advance to the next timer */ + advanceToNextTimer(): Promise; + /** Run all pending timers */ + runAllTimers(): Promise; + /** Get the current fake time */ + now(): number; + /** Reset to real timers */ + useRealTimers(): void; +} + +/** + * Create a time controller for testing timeout-based behavior. + * Call this in beforeEach and cleanup with controller.useRealTimers() in afterEach. + */ +export function createTimeController(): TimeController { + vi.useFakeTimers(); + + return { + async advanceBy(ms: number): Promise { + await vi.advanceTimersByTimeAsync(ms); + }, + async advanceToNextTimer(): Promise { + await vi.runOnlyPendingTimersAsync(); + }, + async runAllTimers(): Promise { + await vi.runAllTimersAsync(); + }, + now(): number { + return Date.now(); + }, + useRealTimers(): void { + vi.useRealTimers(); + }, + }; +} + +// ========== MockSession ========== + +/** + * Enhanced mock session for testing RespawnController. + * Extends the existing MockSession pattern with additional utilities. + */ +export class MockSession extends EventEmitter { + id: string; + workingDir: string = '/tmp/test-workdir'; + status: 'idle' | 'working' = 'idle'; + writeBuffer: string[] = []; + terminalBuffer: string = ''; + + private _screenName: string | null = null; + + constructor(id: string = 'mock-session-id') { + super(); + this.id = id; + this._screenName = `claudeman-test-${id.slice(0, 8)}`; + } + + /** Direct PTY write (used by session.write()) */ + write(data: string): void { + this.writeBuffer.push(data); + } + + /** Write via screen (used by respawn controller) */ + writeViaScreen(data: string): boolean { + this.writeBuffer.push(data); + return true; + } + + /** Get the last written data */ + get lastWrite(): string | undefined { + return this.writeBuffer[this.writeBuffer.length - 1]; + } + + /** Clear the write buffer */ + clearWriteBuffer(): void { + this.writeBuffer = []; + } + + /** Check if a specific command was written */ + hasWritten(pattern: string | RegExp): boolean { + return this.writeBuffer.some(data => + typeof pattern === 'string' ? data.includes(pattern) : pattern.test(data) + ); + } + + // ========== Terminal Output Simulation ========== + + /** Simulate raw terminal output */ + simulateTerminalOutput(data: string): void { + this.terminalBuffer += data; + this.emit('terminal', data); + } + + /** Simulate prompt appearing (legacy fallback signal) */ + simulatePrompt(): void { + this.simulateTerminalOutput('\u276f '); + this.status = 'idle'; + this.emit('idle'); + } + + /** Simulate ready state with definitive indicator (legacy) */ + simulateReady(): void { + this.simulateTerminalOutput('\u21b5 send'); + this.status = 'idle'; + this.emit('idle'); + } + + /** + * Simulate completion message (primary idle detection in Claude Code 2024+). + * This triggers the multi-layer detection flow. + */ + simulateCompletionMessage(duration: string = '2m 46s'): void { + this.simulateTerminalOutput(`\u273b Worked for ${duration}`); + this.status = 'idle'; + } + + /** Simulate working state with spinner */ + simulateWorking(text: string = 'Thinking'): void { + this.simulateTerminalOutput(`${text}... \u280b`); + this.status = 'working'; + this.emit('working'); + } + + /** Simulate /clear completion */ + simulateClearComplete(): void { + this.simulateTerminalOutput('conversation cleared'); + setTimeout(() => this.simulateCompletionMessage(), 50); + } + + /** Simulate /init completion */ + simulateInitComplete(): void { + this.simulateTerminalOutput('Analyzing CLAUDE.md...'); + setTimeout(() => this.simulateCompletionMessage(), 100); + } + + /** + * Simulate plan mode approval prompt. + * This triggers auto-accept detection. + */ + simulatePlanModePrompt(): void { + this.simulateTerminalOutput( + 'Would you like to proceed with this plan?\n' + + '\u276f 1. Yes\n' + + ' 2. No\n' + + ' 3. Type your own\n' + ); + } + + /** + * Simulate elicitation dialog (AskUserQuestion). + * This should block auto-accept. + */ + simulateElicitationDialog(): void { + this.simulateTerminalOutput( + 'What would you like to name the new file?\n' + + '> ' + ); + } + + /** Simulate token count display */ + simulateTokenCount(tokens: number | string): void { + const formatted = typeof tokens === 'number' + ? tokens >= 1000 ? `${(tokens / 1000).toFixed(1)}k` : String(tokens) + : tokens; + this.simulateTerminalOutput(`${formatted} tokens used`); + } + + /** Simulate ANSI escape codes */ + simulateAnsiOutput(text: string, color: 'green' | 'red' | 'blue' = 'green'): void { + const codes: Record = { + green: '\x1b[32m', + red: '\x1b[31m', + blue: '\x1b[34m', + }; + this.simulateTerminalOutput(`${codes[color]}${text}\x1b[0m`); + } + + /** Clear terminal buffer */ + clearTerminalBuffer(): void { + this.terminalBuffer = ''; + } + + // ========== Session Lifecycle ========== + + /** Simulate session closing */ + close(): void { + this.emit('exit', 0); + this.removeAllListeners(); + } + + /** Get screen name (for screen-based operations) */ + get screenName(): string | null { + return this._screenName; + } +} + +// ========== MockAiIdleChecker ========== + +/** + * Configuration for MockAiIdleChecker behavior. + */ +export interface MockAiIdleCheckerOptions { + /** Default verdict to return */ + defaultVerdict?: AiCheckVerdict; + /** Default reasoning to include */ + defaultReasoning?: string; + /** Simulated check duration in ms */ + checkDurationMs?: number; + /** Whether to start in disabled state */ + startDisabled?: boolean; + /** Reason if starting disabled */ + disabledReason?: string; +} + +/** + * Mock AI idle checker for testing without spawning real Claude CLI. + * Allows configuring verdicts and simulating various states. + */ +export class MockAiIdleChecker extends EventEmitter { + private config: AiIdleCheckConfig; + private _status: AiCheckStatus = 'ready'; + private _lastVerdict: AiCheckVerdict | null = null; + private _lastReasoning: string | null = null; + private _lastCheckDurationMs: number | null = null; + private _cooldownEndsAt: number | null = null; + private _consecutiveErrors: number = 0; + private _totalChecks: number = 0; + private _disabledReason: string | null = null; + + // Queued results for sequential calls + private queuedResults: AiCheckResult[] = []; + private defaultOptions: MockAiIdleCheckerOptions; + private cooldownTimer: NodeJS.Timeout | null = null; + + constructor( + public sessionId: string, + config: Partial = {}, + options: MockAiIdleCheckerOptions = {} + ) { + super(); + this.config = { + enabled: true, + model: 'claude-opus-4-5-20251101', + maxContextChars: 16000, + checkTimeoutMs: 90000, + cooldownMs: 180000, + errorCooldownMs: 60000, + maxConsecutiveErrors: 3, + ...config, + }; + this.defaultOptions = { + defaultVerdict: 'IDLE', + defaultReasoning: 'Mock verdict', + checkDurationMs: 100, + startDisabled: false, + ...options, + }; + + if (this.defaultOptions.startDisabled) { + this._status = 'disabled'; + this._disabledReason = this.defaultOptions.disabledReason || 'Disabled for testing'; + } + } + + get status(): AiCheckStatus { + return this._status; + } + + getState(): AiCheckState { + return { + status: this._status, + lastVerdict: this._lastVerdict, + lastReasoning: this._lastReasoning, + lastCheckDurationMs: this._lastCheckDurationMs, + cooldownEndsAt: this._cooldownEndsAt, + consecutiveErrors: this._consecutiveErrors, + totalChecks: this._totalChecks, + disabledReason: this._disabledReason, + }; + } + + isOnCooldown(): boolean { + return this._cooldownEndsAt !== null && Date.now() < this._cooldownEndsAt; + } + + getCooldownRemainingMs(): number { + if (this._cooldownEndsAt === null) return 0; + return Math.max(0, this._cooldownEndsAt - Date.now()); + } + + /** + * Queue a result to be returned on the next check() call. + * Results are consumed in FIFO order. + */ + queueResult(result: AiCheckResult): void { + this.queuedResults.push(result); + } + + /** + * Queue multiple results for sequential check() calls. + */ + queueResults(...results: AiCheckResult[]): void { + this.queuedResults.push(...results); + } + + /** + * Set the next check to return IDLE verdict. + */ + setNextIdle(reasoning: string = 'Mock: IDLE'): void { + this.queueResult({ + verdict: 'IDLE', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return WORKING verdict. + */ + setNextWorking(reasoning: string = 'Mock: WORKING'): void { + this.queueResult({ + verdict: 'WORKING', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return ERROR verdict. + */ + setNextError(reasoning: string = 'Mock: ERROR'): void { + this.queueResult({ + verdict: 'ERROR', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + async check(_terminalBuffer: string): Promise { + if (this._status === 'disabled') { + return { verdict: 'ERROR', reasoning: `Disabled: ${this._disabledReason}`, durationMs: 0 }; + } + + if (this.isOnCooldown()) { + return { verdict: 'ERROR', reasoning: 'On cooldown', durationMs: 0 }; + } + + if (this._status === 'checking') { + return { verdict: 'ERROR', reasoning: 'Already checking', durationMs: 0 }; + } + + this._status = 'checking'; + this._totalChecks++; + this.emit('checkStarted'); + + // Simulate async check + await new Promise(resolve => setTimeout(resolve, 10)); + + // Get result from queue or use default + const result = this.queuedResults.shift() || { + verdict: this.defaultOptions.defaultVerdict!, + reasoning: this.defaultOptions.defaultReasoning!, + durationMs: this.defaultOptions.checkDurationMs!, + }; + + this._lastVerdict = result.verdict; + this._lastReasoning = result.reasoning; + this._lastCheckDurationMs = result.durationMs; + + if (result.verdict === 'IDLE') { + this._consecutiveErrors = 0; + this._status = 'ready'; + } else if (result.verdict === 'WORKING') { + this._consecutiveErrors = 0; + this.startCooldown(this.config.cooldownMs); + } else { + this._consecutiveErrors++; + if (this._consecutiveErrors >= this.config.maxConsecutiveErrors) { + this._status = 'disabled'; + this._disabledReason = `${this.config.maxConsecutiveErrors} consecutive errors`; + this.emit('disabled', this._disabledReason); + } else { + this.startCooldown(this.config.errorCooldownMs); + } + } + + this.emit('checkCompleted', result); + return result; + } + + cancel(): void { + if (this._status === 'checking') { + this._status = 'ready'; + this.emit('log', '[MockAiIdleChecker] Check cancelled'); + } + } + + reset(): void { + this.cancel(); + this.clearCooldown(); + this._lastVerdict = null; + this._lastReasoning = null; + this._lastCheckDurationMs = null; + this._consecutiveErrors = 0; + this.queuedResults = []; + this._status = this._disabledReason ? 'disabled' : 'ready'; + } + + updateConfig(config: Partial): void { + const filteredConfig = Object.fromEntries( + Object.entries(config).filter(([, v]) => v !== undefined) + ) as Partial; + this.config = { ...this.config, ...filteredConfig }; + if (config.enabled === false) { + this._disabledReason = 'Disabled by config'; + this._status = 'disabled'; + } else if (config.enabled === true && this._status === 'disabled') { + this._disabledReason = null; + this._status = 'ready'; + } + } + + getConfig(): AiIdleCheckConfig { + return { ...this.config }; + } + + // ========== Test Helpers ========== + + /** Force into cooldown state for testing */ + forceCooldown(durationMs: number): void { + this.startCooldown(durationMs); + } + + /** Force into disabled state for testing */ + forceDisabled(reason: string = 'Forced disabled for testing'): void { + this._status = 'disabled'; + this._disabledReason = reason; + this.emit('disabled', reason); + } + + /** Force ready state for testing */ + forceReady(): void { + this.clearCooldown(); + this._status = 'ready'; + this._disabledReason = null; + } + + private startCooldown(durationMs: number): void { + this.clearCooldown(); + this._cooldownEndsAt = Date.now() + durationMs; + this._status = 'cooldown'; + this.emit('cooldownStarted', this._cooldownEndsAt); + + this.cooldownTimer = setTimeout(() => { + this._cooldownEndsAt = null; + this._status = 'ready'; + this.emit('cooldownEnded'); + }, durationMs); + } + + private clearCooldown(): void { + if (this.cooldownTimer) { + clearTimeout(this.cooldownTimer); + this.cooldownTimer = null; + } + this._cooldownEndsAt = null; + if (this._status === 'cooldown') { + this._status = 'ready'; + } + } +} + +// ========== MockAiPlanChecker ========== + +/** + * Configuration for MockAiPlanChecker behavior. + */ +export interface MockAiPlanCheckerOptions { + /** Default verdict to return */ + defaultVerdict?: AiPlanCheckVerdict; + /** Default reasoning to include */ + defaultReasoning?: string; + /** Simulated check duration in ms */ + checkDurationMs?: number; + /** Whether to start in disabled state */ + startDisabled?: boolean; + /** Reason if starting disabled */ + disabledReason?: string; +} + +/** + * Mock AI plan checker for testing without spawning real Claude CLI. + * Allows configuring verdicts and simulating various states. + */ +export class MockAiPlanChecker extends EventEmitter { + private config: AiPlanCheckConfig; + private _status: AiPlanCheckStatus = 'ready'; + private _lastVerdict: AiPlanCheckVerdict | null = null; + private _lastReasoning: string | null = null; + private _lastCheckDurationMs: number | null = null; + private _cooldownEndsAt: number | null = null; + private _consecutiveErrors: number = 0; + private _totalChecks: number = 0; + private _disabledReason: string | null = null; + + // Queued results for sequential calls + private queuedResults: AiPlanCheckResult[] = []; + private defaultOptions: MockAiPlanCheckerOptions; + private cooldownTimer: NodeJS.Timeout | null = null; + + constructor( + public sessionId: string, + config: Partial = {}, + options: MockAiPlanCheckerOptions = {} + ) { + super(); + this.config = { + enabled: true, + model: 'claude-opus-4-5-20251101', + maxContextChars: 8000, + checkTimeoutMs: 60000, + cooldownMs: 30000, + errorCooldownMs: 30000, + maxConsecutiveErrors: 3, + ...config, + }; + this.defaultOptions = { + defaultVerdict: 'PLAN_MODE', + defaultReasoning: 'Mock verdict', + checkDurationMs: 100, + startDisabled: false, + ...options, + }; + + if (this.defaultOptions.startDisabled) { + this._status = 'disabled'; + this._disabledReason = this.defaultOptions.disabledReason || 'Disabled for testing'; + } + } + + get status(): AiPlanCheckStatus { + return this._status; + } + + getState(): AiPlanCheckState { + return { + status: this._status, + lastVerdict: this._lastVerdict, + lastReasoning: this._lastReasoning, + lastCheckDurationMs: this._lastCheckDurationMs, + cooldownEndsAt: this._cooldownEndsAt, + consecutiveErrors: this._consecutiveErrors, + totalChecks: this._totalChecks, + disabledReason: this._disabledReason, + }; + } + + isOnCooldown(): boolean { + return this._cooldownEndsAt !== null && Date.now() < this._cooldownEndsAt; + } + + getCooldownRemainingMs(): number { + if (this._cooldownEndsAt === null) return 0; + return Math.max(0, this._cooldownEndsAt - Date.now()); + } + + /** + * Queue a result to be returned on the next check() call. + */ + queueResult(result: AiPlanCheckResult): void { + this.queuedResults.push(result); + } + + /** + * Queue multiple results for sequential check() calls. + */ + queueResults(...results: AiPlanCheckResult[]): void { + this.queuedResults.push(...results); + } + + /** + * Set the next check to return PLAN_MODE verdict. + */ + setNextPlanMode(reasoning: string = 'Mock: PLAN_MODE'): void { + this.queueResult({ + verdict: 'PLAN_MODE', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return NOT_PLAN_MODE verdict. + */ + setNextNotPlanMode(reasoning: string = 'Mock: NOT_PLAN_MODE'): void { + this.queueResult({ + verdict: 'NOT_PLAN_MODE', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return ERROR verdict. + */ + setNextError(reasoning: string = 'Mock: ERROR'): void { + this.queueResult({ + verdict: 'ERROR', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + async check(_terminalBuffer: string): Promise { + if (this._status === 'disabled') { + return { verdict: 'ERROR', reasoning: `Disabled: ${this._disabledReason}`, durationMs: 0 }; + } + + if (this.isOnCooldown()) { + return { verdict: 'ERROR', reasoning: 'On cooldown', durationMs: 0 }; + } + + if (this._status === 'checking') { + return { verdict: 'ERROR', reasoning: 'Already checking', durationMs: 0 }; + } + + this._status = 'checking'; + this._totalChecks++; + this.emit('checkStarted'); + + // Simulate async check + await new Promise(resolve => setTimeout(resolve, 10)); + + // Get result from queue or use default + const result = this.queuedResults.shift() || { + verdict: this.defaultOptions.defaultVerdict!, + reasoning: this.defaultOptions.defaultReasoning!, + durationMs: this.defaultOptions.checkDurationMs!, + }; + + this._lastVerdict = result.verdict; + this._lastReasoning = result.reasoning; + this._lastCheckDurationMs = result.durationMs; + + if (result.verdict === 'PLAN_MODE') { + this._consecutiveErrors = 0; + this._status = 'ready'; + } else if (result.verdict === 'NOT_PLAN_MODE') { + this._consecutiveErrors = 0; + this.startCooldown(this.config.cooldownMs); + } else { + this._consecutiveErrors++; + if (this._consecutiveErrors >= this.config.maxConsecutiveErrors) { + this._status = 'disabled'; + this._disabledReason = `${this.config.maxConsecutiveErrors} consecutive errors`; + this.emit('disabled', this._disabledReason); + } else { + this.startCooldown(this.config.errorCooldownMs); + } + } + + this.emit('checkCompleted', result); + return result; + } + + cancel(): void { + if (this._status === 'checking') { + this._status = 'ready'; + this.emit('log', '[MockAiPlanChecker] Check cancelled'); + } + } + + reset(): void { + this.cancel(); + this.clearCooldown(); + this._lastVerdict = null; + this._lastReasoning = null; + this._lastCheckDurationMs = null; + this._consecutiveErrors = 0; + this.queuedResults = []; + this._status = this._disabledReason ? 'disabled' : 'ready'; + } + + updateConfig(config: Partial): void { + const filteredConfig = Object.fromEntries( + Object.entries(config).filter(([, v]) => v !== undefined) + ) as Partial; + this.config = { ...this.config, ...filteredConfig }; + if (config.enabled === false) { + this._disabledReason = 'Disabled by config'; + this._status = 'disabled'; + } else if (config.enabled === true && this._status === 'disabled') { + this._disabledReason = null; + this._status = 'ready'; + } + } + + getConfig(): AiPlanCheckConfig { + return { ...this.config }; + } + + // ========== Test Helpers ========== + + /** Force into cooldown state for testing */ + forceCooldown(durationMs: number): void { + this.startCooldown(durationMs); + } + + /** Force into disabled state for testing */ + forceDisabled(reason: string = 'Forced disabled for testing'): void { + this._status = 'disabled'; + this._disabledReason = reason; + this.emit('disabled', reason); + } + + /** Force ready state for testing */ + forceReady(): void { + this.clearCooldown(); + this._status = 'ready'; + this._disabledReason = null; + } + + private startCooldown(durationMs: number): void { + this.clearCooldown(); + this._cooldownEndsAt = Date.now() + durationMs; + this._status = 'cooldown'; + this.emit('cooldownStarted', this._cooldownEndsAt); + + this.cooldownTimer = setTimeout(() => { + this._cooldownEndsAt = null; + this._status = 'ready'; + this.emit('cooldownEnded'); + }, durationMs); + } + + private clearCooldown(): void { + if (this.cooldownTimer) { + clearTimeout(this.cooldownTimer); + this.cooldownTimer = null; + } + this._cooldownEndsAt = null; + if (this._status === 'cooldown') { + this._status = 'ready'; + } + } +} + +// ========== State Transition Helpers ========== + +/** + * State transition event for tracking. + */ +export interface StateTransition { + from: RespawnState; + to: RespawnState; + timestamp: number; +} + +/** + * Create a state tracker that records all state transitions. + * Attach to a RespawnController via controller.on('stateChanged', tracker.record). + */ +export function createStateTracker() { + const transitions: StateTransition[] = []; + + return { + /** Record a state transition (use as event handler) */ + record(to: RespawnState, from: RespawnState): void { + transitions.push({ from, to, timestamp: Date.now() }); + }, + + /** Get all recorded transitions */ + getTransitions(): StateTransition[] { + return [...transitions]; + }, + + /** Get only the state values (not timestamps) */ + getStates(): RespawnState[] { + return transitions.map(t => t.to); + }, + + /** Check if a specific state was visited */ + hasVisited(state: RespawnState): boolean { + return transitions.some(t => t.to === state); + }, + + /** Check if a specific transition occurred */ + hasTransition(from: RespawnState, to: RespawnState): boolean { + return transitions.some(t => t.from === from && t.to === to); + }, + + /** Get the most recent state */ + getCurrentState(): RespawnState | undefined { + return transitions.length > 0 ? transitions[transitions.length - 1].to : undefined; + }, + + /** Clear all recorded transitions */ + clear(): void { + transitions.length = 0; + }, + }; +} + +/** + * Create an event recorder for tracking all events from a RespawnController. + */ +export function createEventRecorder() { + const events: Array<{ type: string; args: unknown[]; timestamp: number }> = []; + + return { + /** Create a handler for a specific event type */ + handler(type: string): (...args: unknown[]) => void { + return (...args: unknown[]) => { + events.push({ type, args, timestamp: Date.now() }); + }; + }, + + /** Get all recorded events */ + getEvents(): Array<{ type: string; args: unknown[]; timestamp: number }> { + return [...events]; + }, + + /** Get events of a specific type */ + getEventsOfType(type: string): Array<{ type: string; args: unknown[]; timestamp: number }> { + return events.filter(e => e.type === type); + }, + + /** Check if an event type was emitted */ + hasEvent(type: string): boolean { + return events.some(e => e.type === type); + }, + + /** Count events of a specific type */ + countEvents(type: string): number { + return events.filter(e => e.type === type).length; + }, + + /** Clear all recorded events */ + clear(): void { + events.length = 0; + }, + }; +} + +// ========== Factory Functions ========== + +/** + * Default fast test configuration. + * Uses short timeouts for faster test execution. + */ +export const FAST_TEST_CONFIG: Partial = { + idleTimeoutMs: 50, + interStepDelayMs: 20, + completionConfirmMs: 50, + noOutputTimeoutMs: 300, + autoAcceptDelayMs: 100, + aiIdleCheckEnabled: false, + aiPlanCheckEnabled: false, +}; + +/** + * Configuration with AI checks enabled but short timeouts. + */ +export const AI_ENABLED_TEST_CONFIG: Partial = { + ...FAST_TEST_CONFIG, + aiIdleCheckEnabled: true, + aiIdleCheckTimeoutMs: 500, + aiIdleCheckCooldownMs: 200, + aiPlanCheckEnabled: true, + aiPlanCheckTimeoutMs: 500, + aiPlanCheckCooldownMs: 200, +}; + +/** + * Create a mock session typed as Session for use with RespawnController. + */ +export function createMockSession(id?: string): MockSession & Session { + return new MockSession(id) as MockSession & Session; +} + +// ========== Test Assertion Helpers ========== + +/** + * Wait for a specific state to be reached. + * Useful for async state transition testing. + */ +export async function waitForState( + controller: { state: RespawnState; on: (event: string, handler: (state: RespawnState) => void) => void }, + targetState: RespawnState, + timeoutMs: number = 1000 +): Promise { + if (controller.state === targetState) return; + + return new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error(`Timeout waiting for state ${targetState}, current: ${controller.state}`)); + }, timeoutMs); + + const handler = (state: RespawnState) => { + if (state === targetState) { + clearTimeout(timeout); + resolve(); + } + }; + + controller.on('stateChanged', handler); + }); +} + +/** + * Wait for a specific event to be emitted. + */ +export async function waitForEvent( + emitter: EventEmitter, + eventName: string, + timeoutMs: number = 1000 +): Promise { + return new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error(`Timeout waiting for event ${eventName}`)); + }, timeoutMs); + + emitter.once(eventName, (...args: unknown[]) => { + clearTimeout(timeout); + resolve(args[0] as T); + }); + }); +} + +/** + * Create a deferred promise for controlled async testing. + */ +export function createDeferred(): { + promise: Promise; + resolve: (value: T) => void; + reject: (error: Error) => void; +} { + let resolve!: (value: T) => void; + let reject!: (error: Error) => void; + const promise = new Promise((res, rej) => { + resolve = res; + reject = rej; + }); + return { promise, resolve, reject }; +} + +// ========== Terminal Output Generators ========== + +/** + * Generate realistic terminal output for testing. + */ +export const terminalOutputs = { + /** Standard completion message */ + completion(duration: string = '2m 46s'): string { + return `\n\u273b Worked for ${duration}\n 123.4k tokens used\n`; + }, + + /** Working spinner output */ + working(activity: string = 'Thinking'): string { + return `${activity}... \u280b`; + }, + + /** Plan mode prompt */ + planMode(question: string = 'Would you like to proceed?'): string { + return [ + `\n${question}\n`, + '\u276f 1. Yes\n', + ' 2. No\n', + ' 3. Type your own\n', + ].join(''); + }, + + /** Prompt character */ + prompt(): string { + return '\n\u276f '; + }, + + /** Token count display */ + tokens(count: number): string { + const formatted = count >= 1000000 + ? `${(count / 1000000).toFixed(1)}M` + : count >= 1000 + ? `${(count / 1000).toFixed(1)}k` + : String(count); + return ` ${formatted} tokens\n`; + }, + + /** Large output for buffer testing */ + largeOutput(sizeKb: number = 100): string { + const baseText = 'Lorem ipsum dolor sit amet. '.repeat(100); + const repetitions = Math.ceil((sizeKb * 1024) / baseText.length); + return baseText.repeat(repetitions).slice(0, sizeKb * 1024); + }, + + /** ANSI colored output */ + ansiColored(text: string): string { + return `\x1b[32m${text}\x1b[0m`; + }, +};