From eb98caa812b60d59feccc534786b56681bc91c6e Mon Sep 17 00:00:00 2001 From: arkon Date: Sun, 25 Jan 2026 07:49:04 +0100 Subject: [PATCH] fix: use temp file for plan checker prompt to avoid E2BIG + fix cancel race condition The ai-plan-checker.ts was passing the prompt directly as a shell argument, which can cause E2BIG errors when the terminal buffer is large (8KB+). This fix applies the same temp file approach already used in ai-idle-checker.ts: - Write prompt to a temp file instead of passing as shell argument - Pipe the file to claude via stdin: `cat prompt.txt | claude -p ...` - Clean up prompt file after check completes Also fixes a race condition in the cancel() method of both AI checkers where the poll timer could fire between setting checkCancelled and clearing timers. Now timers are cleared before resolving the promise to prevent this race. Includes test utilities and analysis documents for the respawn controller created by other agents: - test/respawn-test-utils.ts - MockSession, MockAiIdleChecker utilities - test/respawn-analysis.md - Code analysis and issue identification - test/respawn-scenarios.md - Test scenario documentation - test/respawn-test-plan.md - Testing architecture documentation Co-Authored-By: Claude Opus 4.5 --- CLAUDE.md | 2 +- package.json | 2 +- src/ai-idle-checker.ts | 7 +- src/ai-plan-checker.ts | 37 +- test/respawn-analysis.md | 295 ++++++++ test/respawn-scenarios.md | 1460 ++++++++++++++++++++++++++++++++++++ test/respawn-test-plan.md | 482 ++++++++++++ test/respawn-test-utils.ts | 1045 ++++++++++++++++++++++++++ 8 files changed, 3317 insertions(+), 13 deletions(-) create mode 100644 test/respawn-analysis.md create mode 100644 test/respawn-scenarios.md create mode 100644 test/respawn-test-plan.md create mode 100644 test/respawn-test-utils.ts diff --git a/CLAUDE.md b/CLAUDE.md index a2238568..7fbfc343 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -17,7 +17,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co Claudeman is a Claude Code session manager with a web interface and autonomous Ralph Loop. It spawns Claude CLI processes via PTY, streams output in real-time via SSE, and supports scheduled/timed runs. -**Version**: 0.1345 +**Version**: 0.1346 **Tech Stack**: TypeScript (ES2022/NodeNext, strict mode), Node.js, Fastify, Server-Sent Events, node-pty diff --git a/package.json b/package.json index 94b85e81..9474d385 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "claudeman", - "version": "0.1345", + "version": "0.1346", "description": "The missing control plane for Claude Code - run 20 autonomous agents with real-time monitoring and session persistence", "type": "module", "main": "dist/index.js", diff --git a/src/ai-idle-checker.ts b/src/ai-idle-checker.ts index 67e3ef87..6a316616 100644 --- a/src/ai-idle-checker.ts +++ b/src/ai-idle-checker.ts @@ -268,13 +268,16 @@ export class AiIdleChecker extends EventEmitter { this.log('Cancelling AI check'); this.checkCancelled = true; - // Resolve the pending promise before cleanup + // Clear poll/timeout timers first to prevent race condition where + // the poll timer fires between setting checkCancelled and cleanup + this.cleanupCheck(); + + // Resolve the pending promise after cleanup if (this.checkResolve) { this.checkResolve({ verdict: 'ERROR', reasoning: 'Cancelled', durationMs: Date.now() - this.checkStartTime }); this.checkResolve = null; } - this.cleanupCheck(); this._status = 'ready'; } diff --git a/src/ai-plan-checker.ts b/src/ai-plan-checker.ts index e4671c6d..3ee33732 100644 --- a/src/ai-plan-checker.ts +++ b/src/ai-plan-checker.ts @@ -152,6 +152,7 @@ export class AiPlanChecker extends EventEmitter { // Active check state private checkScreenName: string | null = null; private checkTempFile: string | null = null; + private checkPromptFile: string | null = null; private checkPollTimer: NodeJS.Timeout | null = null; private checkTimeoutTimer: NodeJS.Timeout | null = null; private checkStartTime: number = 0; @@ -276,13 +277,16 @@ export class AiPlanChecker extends EventEmitter { this.log('Cancelling AI plan check'); this.checkCancelled = true; - // Resolve the pending promise before cleanup + // Clear poll/timeout timers first to prevent race condition where + // the poll timer fires between setting checkCancelled and cleanup + this.cleanupCheck(); + + // Resolve the pending promise after cleanup if (this.checkResolve) { this.checkResolve({ verdict: 'ERROR', reasoning: 'Cancelled', durationMs: Date.now() - this.checkStartTime }); this.checkResolve = null; } - this.cleanupCheck(); this._status = 'ready'; } @@ -329,21 +333,25 @@ export class AiPlanChecker extends EventEmitter { // Build the prompt const prompt = AI_PLAN_CHECK_PROMPT.replace('{TERMINAL_BUFFER}', trimmed); - // Generate temp file and screen name + // Generate temp files and screen name const shortId = this.sessionId.slice(0, 8); const timestamp = Date.now(); this.checkTempFile = join(tmpdir(), `claudeman-plancheck-${shortId}-${timestamp}.txt`); + this.checkPromptFile = join(tmpdir(), `claudeman-plancheck-prompt-${shortId}-${timestamp}.txt`); this.checkScreenName = `claudeman-plancheck-${shortId}`; - // Ensure temp file exists (empty) so we can poll it + // Ensure output temp file exists (empty) so we can poll it writeFileSync(this.checkTempFile, ''); - // Build the command - escape the prompt for shell - const escapedPrompt = prompt.replace(/'/g, "'\\''"); + // Write prompt to file to avoid E2BIG error (argument list too long) + // The prompt can be 8KB+ which exceeds shell argument limits + writeFileSync(this.checkPromptFile, prompt); + + // Build the command - read prompt from file via stdin to avoid argument size limits const modelArg = `--model ${this.config.model}`; const augmentedPath = getAugmentedPath(); - const claudeCmd = `claude -p ${modelArg} --output-format text '${escapedPrompt}'`; - const fullCmd = `export PATH="${augmentedPath}"; ${claudeCmd} > "${this.checkTempFile}" 2>&1; echo "${DONE_MARKER}" >> "${this.checkTempFile}"`; + const claudeCmd = `cat "${this.checkPromptFile}" | claude -p ${modelArg} --output-format text`; + const fullCmd = `export PATH="${augmentedPath}"; ${claudeCmd} > "${this.checkTempFile}" 2>&1; echo "${DONE_MARKER}" >> "${this.checkTempFile}"; rm -f "${this.checkPromptFile}"`; // Spawn screen try { @@ -447,7 +455,7 @@ export class AiPlanChecker extends EventEmitter { this.checkScreenName = null; } - // Delete temp file + // Delete temp files if (this.checkTempFile) { try { if (existsSync(this.checkTempFile)) { @@ -458,6 +466,17 @@ export class AiPlanChecker extends EventEmitter { } this.checkTempFile = null; } + + if (this.checkPromptFile) { + try { + if (existsSync(this.checkPromptFile)) { + unlinkSync(this.checkPromptFile); + } + } catch { + // Best effort cleanup + } + this.checkPromptFile = null; + } } private handleError(errorMsg: string): void { diff --git a/test/respawn-analysis.md b/test/respawn-analysis.md new file mode 100644 index 00000000..32d152b2 --- /dev/null +++ b/test/respawn-analysis.md @@ -0,0 +1,295 @@ +# Respawn Controller Test Analysis + +**Date**: 2026-01-25 +**Analyzed Files**: +- `/home/arkon/default/claudeman/src/respawn-controller.ts` +- `/home/arkon/default/claudeman/src/ai-idle-checker.ts` +- `/home/arkon/default/claudeman/src/ai-plan-checker.ts` +- `/home/arkon/default/claudeman/test/respawn-controller.test.ts` + +--- + +## 1. Test Execution Results + +### Summary +- **Total Tests**: 91 +- **Passed**: 91 +- **Failed**: 0 +- **Duration**: ~10 seconds + +### Type Check Results +- **Type Errors**: 0 + +All tests pass and the codebase is type-safe. + +--- + +## 2. Code Coverage Analysis + +### Coverage Summary + +| File | Statements | Branches | Functions | Lines | +|------|------------|----------|-----------|-------| +| respawn-controller.ts | 62.51% | 59.85% | 74.48% | 62.37% | +| ai-idle-checker.ts | 68.91% | 46.57% | 75% | 69.31% | +| ai-plan-checker.ts | 57.52% | 37.68% | 58.33% | 57.69% | + +### Uncovered Code in respawn-controller.ts + +The following areas lack test coverage: + +1. **Lines 2077-2151**: `sendClear()`, `sendInit()`, `completeCycle()` + - These functions execute during the full respawn cycle + - Tests trigger cycles but don't mock session responses to complete them + +2. **Line 2162**: `checkIdleAndMaybeStart()` + - Called when resuming from pause while already idle + - Only tested superficially with `resume()` call + +3. **Lines 2100-2103**: Clear fallback timer path when `sendInit` is false + - Edge case where `/clear` completes via fallback but init is disabled + +--- + +## 3. Potential Bugs and Issues + +### Issue 1: E2BIG Error in ai-plan-checker.ts (HIGH) + +**File**: `/home/arkon/default/claudeman/src/ai-plan-checker.ts` +**Lines**: 341-346 +**Severity**: HIGH + +**Description**: Unlike `ai-idle-checker.ts`, the plan checker passes the prompt directly as a shell argument rather than via a temp file. This can cause `E2BIG` (argument list too long) errors when the terminal buffer is large. + +**Code**: +```typescript +// ai-plan-checker.ts - PROBLEMATIC +const escapedPrompt = prompt.replace(/'/g, "'\\''"); +const claudeCmd = `claude -p ${modelArg} --output-format text '${escapedPrompt}'`; + +// ai-idle-checker.ts - CORRECT +writeFileSync(this.checkPromptFile, prompt); +const claudeCmd = `cat "${this.checkPromptFile}" | claude -p ${modelArg} --output-format text`; +``` + +**Impact**: With `maxContextChars: 8000`, the prompt can easily exceed shell argument limits (~128KB on most systems), causing the AI plan check to fail with a cryptic error. + +**Suggested Fix**: Modify `ai-plan-checker.ts` to use the same temp file approach as `ai-idle-checker.ts`: +1. Add `checkPromptFile` instance variable +2. Write prompt to temp file +3. Use `cat | claude -p` pattern +4. Clean up temp file in `cleanupCheck()` + +--- + +### Issue 2: Missing Prompt File Cleanup on Timeout (MEDIUM) + +**File**: `/home/arkon/default/claudeman/src/ai-idle-checker.ts` +**Lines**: 392-398 +**Severity**: MEDIUM + +**Description**: When the AI check times out, the `cleanupCheck()` function is called via `finally`. However, the prompt file deletion in `cleanupCheck()` may fail silently if the file is still being read by the `cat` command in the spawned screen. + +**Impact**: Temp files may accumulate in `/tmp` if checks frequently timeout. + +**Suggested Fix**: Add a small delay before file deletion or use a unique timestamp-based naming scheme that guarantees no collisions (already partially implemented but the shell command `rm -f` in the fullCmd also handles this). + +--- + +### Issue 3: Race Condition in cancel() with Promise Resolution (MEDIUM) + +**File**: `/home/arkon/default/claudeman/src/ai-idle-checker.ts` +**Lines**: 265-279 +**Severity**: MEDIUM + +**Description**: The `cancel()` method resolves the pending promise and then calls `cleanupCheck()`. However, the interval poll timer might fire between these two operations and attempt to resolve an already-resolved promise. + +**Code**: +```typescript +cancel(): void { + if (this._status !== 'checking') return; + this.checkCancelled = true; + // Race: poll timer might fire here + if (this.checkResolve) { + this.checkResolve({ verdict: 'ERROR', ... }); + this.checkResolve = null; + } + this.cleanupCheck(); // This clears the poll timer + this._status = 'ready'; +} +``` + +**Impact**: Could theoretically cause a double-resolve, though the guard `if (this.checkCancelled)` in the poll handler mitigates this. + +**Suggested Fix**: Set `checkCancelled = true` and call `cleanupCheck()` (which clears the poll timer) before resolving the promise. + +--- + +### Issue 4: Stale Screen Name Collision (LOW) + +**File**: `/home/arkon/default/claudeman/src/ai-idle-checker.ts`, `ai-plan-checker.ts` +**Lines**: 326-329 (idle), 333-336 (plan) +**Severity**: LOW + +**Description**: Screen names are generated using only `sessionId.slice(0, 8)`, without a unique suffix. While there's a `screen -X quit` to kill leftover screens, two rapid checks could conflict. + +**Code**: +```typescript +this.checkScreenName = `claudeman-aicheck-${shortId}`; +``` + +**Impact**: Very unlikely in practice since checks have cooldowns, but theoretically possible. + +**Suggested Fix**: Already partially mitigated by the `timestamp` in temp file names. Could add timestamp to screen name too: +```typescript +this.checkScreenName = `claudeman-aicheck-${shortId}-${timestamp}`; +``` + +--- + +### Issue 5: DetectionStatus Calculation During ai_checking State (LOW) + +**File**: `/home/arkon/default/claudeman/src/respawn-controller.ts` +**Lines**: 759-762 +**Severity**: LOW + +**Description**: The `outputSilent` calculation uses `config.completionConfirmMs`, but when in `ai_checking` state, the output silence threshold should conceptually be the AI check timeout, not the completion confirm time. + +**Impact**: UI display may show incorrect "silence" status during AI check. + +--- + +## 4. Coverage Gaps + +### Functions Not Fully Tested + +1. **`sendClear()`** - Never reaches execution in tests because the cycle is cut short +2. **`sendInit()`** - Same as above +3. **`completeCycle()`** - Tests don't allow cycles to complete fully +4. **`checkClearComplete()`** - Requires mocking prompt detection after /clear +5. **`checkInitComplete()`** - Requires mocking init completion flow +6. **`sendKickstart()`** - Not tested at all +7. **`checkKickstartComplete()`** - Not tested +8. **`startMonitoringInit()`** - Not tested +9. **`checkMonitoringInitIdle()`** - Not tested + +### Branches Not Tested + +1. `/clear` fallback timer completing (line 2095-2105) +2. `sendInit: false` branch in various locations +3. Kickstart prompt flow when `/init` doesn't trigger work +4. AI checker returning `WORKING` verdict (only test with timeout) +5. AI plan checker `PLAN_MODE` verdict (real check, not just pre-filter) + +### Edge Cases Not Tested + +1. Multiple rapid AI checks before cooldown +2. AI checker disabled mid-check +3. Session terminal buffer being `undefined` on start +4. Buffer trimming behavior at MAX_RESPAWN_BUFFER_SIZE +5. `signalElicitation()` called multiple times + +--- + +## 5. Timing and Race Condition Analysis + +### Potential Timing Issues + +1. **Pre-filter timer vs AI check**: If pre-filter fires while an AI check is already running, it correctly skips (line 1613). + +2. **Completion confirm during AI check**: Output during AI check cancels it (line 1173-1188), which is correct. + +3. **Auto-accept during respawn cycle**: Correctly guards against non-watching state (line 1741). + +4. **Step confirm timer**: Uses same `completionConfirmMs` as idle confirm, which may be too short for complex operations. + +### Timer Cleanup + +All timers appear to be properly cleaned up in `clearTimers()`: +- `idleTimer` +- `stepTimer` +- `clearFallbackTimer` +- `completionConfirmTimer` +- `stepConfirmTimer` +- `autoAcceptTimer` +- `preFilterTimer` +- `noOutputTimer` +- `detectionUpdateTimer` (interval) + +--- + +## 6. Recommendations (Prioritized) + +### Critical (Should Fix) + +1. **Fix E2BIG vulnerability in ai-plan-checker.ts** + - Use temp file for prompt like ai-idle-checker.ts does + - Add `checkPromptFile` instance variable + - Update `cleanupCheck()` to delete prompt file + +### High Priority + +2. **Add integration tests for full respawn cycle** + - Mock session to simulate prompt detection after each step + - Test: update -> clear -> init -> back to watching + - Test: kickstart prompt when init doesn't trigger work + +3. **Fix cancel() race condition in AI checkers** + - Clear poll timer before resolving promise + - Or use a mutex/flag check in poll handler + +### Medium Priority + +4. **Add tests for AI checker WORKING verdict** + - Mock the screen process to return WORKING + - Verify cooldown is started + - Verify controller returns to watching + +5. **Add tests for plan checker PLAN_MODE verdict** + - Mock the screen process to return PLAN_MODE + - Verify Enter is sent + - Verify state transitions + +6. **Add timestamp to screen session names** + - Prevents potential collisions during rapid operations + +### Low Priority + +7. **Improve coverage of edge cases** + - Test `sendInit: false` and `sendClear: false` combinations + - Test buffer trimming behavior + - Test pause/resume with activity detection + +8. **Consider reducing step confirm timeout** + - Currently uses `completionConfirmMs` (10s default) + - May want separate config for step confirmation + +--- + +## 7. MockSession Limitations + +The current `MockSession` class in tests is simplified and doesn't accurately simulate: + +1. **Delayed responses**: Real sessions have processing time between input and output +2. **State persistence**: Real sessions maintain state between commands +3. **Screen integration**: Real sessions use GNU screen for persistence +4. **Token tracking**: Real sessions parse and track token usage + +Consider adding a more sophisticated mock that can: +- Queue delayed responses +- Simulate multi-step workflows +- Return realistic completion messages with timing + +--- + +## 8. Conclusion + +The respawn controller tests are comprehensive for basic functionality but lack coverage for: +- Full respawn cycle completion +- AI checker verdicts (IDLE/WORKING/PLAN_MODE) +- Kickstart functionality +- Edge cases and error handling + +The main bug found is in `ai-plan-checker.ts` which can fail with E2BIG errors on large terminal buffers. This should be fixed by using the temp file approach already implemented in `ai-idle-checker.ts`. + +All 91 tests pass, and there are no type errors, indicating a stable codebase but with room for deeper integration testing. diff --git a/test/respawn-scenarios.md b/test/respawn-scenarios.md new file mode 100644 index 00000000..9272641f --- /dev/null +++ b/test/respawn-scenarios.md @@ -0,0 +1,1460 @@ +# Respawn Controller Test Scenarios + +This document identifies edge cases and scenarios not currently covered by the existing test suite in `test/respawn-controller.test.ts`. These scenarios are designed to find gaps in test coverage and ensure robust behavior. + +--- + +## Table of Contents + +1. [State Transition Scenarios](#state-transition-scenarios) +2. [AI Idle Checker Integration Scenarios](#ai-idle-checker-integration-scenarios) +3. [AI Plan Checker Integration Scenarios](#ai-plan-checker-integration-scenarios) +4. [Timeout and Cooldown Scenarios](#timeout-and-cooldown-scenarios) +5. [Error Recovery Scenarios](#error-recovery-scenarios) +6. [Concurrent Operation Scenarios](#concurrent-operation-scenarios) +7. [Configuration Change Scenarios](#configuration-change-scenarios) +8. [Buffer and Memory Scenarios](#buffer-and-memory-scenarios) +9. [Edge Cases in Pattern Detection](#edge-cases-in-pattern-detection) +10. [Event Emission and Listener Scenarios](#event-emission-and-listener-scenarios) + +--- + +## State Transition Scenarios + +### STS-001: Full Cycle with All Steps Enabled + +**Description**: Complete respawn cycle traversing all states when sendClear, sendInit, and kickstartPrompt are all enabled. + +**Initial State**: `watching` with all config options enabled + +**Actions**: +1. Start controller with `sendClear: true`, `sendInit: true`, `kickstartPrompt: 'continue'` +2. Simulate completion message +3. Wait for confirmation +4. Simulate completion for each step without triggering work after /init + +**Expected Behavior**: +- States visited in order: watching -> confirming_idle -> ai_checking -> sending_update -> waiting_update -> sending_clear -> waiting_clear -> sending_init -> waiting_init -> monitoring_init -> sending_kickstart -> waiting_kickstart -> watching +- All stepSent/stepCompleted events emitted +- Cycle count increments + +**Priority**: HIGH + +--- + +### STS-002: Cycle Skipping Clear Step + +**Description**: Respawn cycle when sendClear is disabled. + +**Initial State**: `watching` with `sendClear: false`, `sendInit: true` + +**Actions**: +1. Trigger idle detection +2. Complete update step + +**Expected Behavior**: +- Should skip directly from waiting_update to sending_init +- No 'clear' step events emitted + +**Priority**: MEDIUM + +--- + +### STS-003: Cycle Skipping Init Step + +**Description**: Respawn cycle when sendInit is disabled. + +**Initial State**: `watching` with `sendClear: true`, `sendInit: false` + +**Actions**: +1. Trigger idle detection +2. Complete update step +3. Complete clear step + +**Expected Behavior**: +- Should complete cycle after clear +- No 'init' step events emitted +- Should not enter monitoring_init state + +**Priority**: MEDIUM + +--- + +### STS-004: Cycle with Neither Clear nor Init + +**Description**: Respawn cycle when both sendClear and sendInit are disabled. + +**Initial State**: `watching` with `sendClear: false`, `sendInit: false` + +**Actions**: +1. Trigger idle detection +2. Complete update step + +**Expected Behavior**: +- Cycle completes immediately after update +- Only update step events emitted + +**Priority**: MEDIUM + +--- + +### STS-005: Work Triggered During monitoring_init + +**Description**: When /init actually triggers Claude to start working. + +**Initial State**: `monitoring_init` with kickstartPrompt configured + +**Actions**: +1. Reach monitoring_init state +2. Simulate working patterns (e.g., "Thinking...") + +**Expected Behavior**: +- Should emit stepCompleted for 'init' +- Should NOT enter sending_kickstart +- Should complete cycle and return to watching +- Log message: "/init triggered work, skipping kickstart" + +**Priority**: HIGH + +--- + +### STS-006: State Transition During Step Timer + +**Description**: Stopping controller while step delay timer is pending. + +**Initial State**: `sending_update` (during interStepDelayMs wait) + +**Actions**: +1. Start controller and trigger idle +2. While in sending_update (during delay), call stop() + +**Expected Behavior**: +- Step timer should be cleared +- No stepSent event emitted +- State transitions to stopped cleanly + +**Priority**: MEDIUM + +--- + +### STS-007: Clear Fallback Timer Trigger + +**Description**: /clear step completes via fallback timer when no prompt detected. + +**Initial State**: `waiting_clear` + +**Actions**: +1. Send /clear command +2. Do NOT simulate prompt detection +3. Wait for CLEAR_FALLBACK_TIMEOUT_MS (10s) + +**Expected Behavior**: +- Fallback timer fires +- Logs "clear fallback: proceeding to /init" +- Proceeds to sendInit (or completeCycle if sendInit false) +- stepCompleted event emitted for 'clear' + +**Priority**: MEDIUM + +--- + +### STS-008: Clear Fallback Timer Cancelled by Prompt + +**Description**: Clear fallback timer is cancelled when prompt is detected. + +**Initial State**: `waiting_clear` + +**Actions**: +1. Send /clear command +2. Simulate prompt detection before fallback timeout + +**Expected Behavior**: +- Fallback timer cancelled with reason "prompt detected" +- timerCancelled event emitted for 'clear-fallback' +- Normal flow continues + +**Priority**: MEDIUM + +--- + +### STS-009: Multiple Rapid State Transitions + +**Description**: Rapid cycling through multiple states without settling. + +**Initial State**: `watching` + +**Actions**: +1. Simulate completion message +2. Before confirmation completes, simulate working +3. Immediately after, simulate completion again +4. Repeat several times rapidly + +**Expected Behavior**: +- Controller handles transitions gracefully +- No duplicate state entries +- Timers properly cleaned up +- No memory leaks from orphaned timers + +**Priority**: HIGH + +--- + +### STS-010: Resume from Non-Watching State + +**Description**: Calling resume() when in a state other than watching. + +**Initial State**: `sending_update` or any non-watching state + +**Actions**: +1. Pause controller +2. Call resume() + +**Expected Behavior**: +- Should remain in current state +- Should not call checkIdleAndMaybeStart +- Only watching state triggers idle check on resume + +**Priority**: LOW + +--- + +## AI Idle Checker Integration Scenarios + +### AIC-001: AI Checker Returns IDLE Verdict + +**Description**: AI idle check completes successfully with IDLE verdict. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check via completion message + silence +2. Mock AI checker to return IDLE + +**Expected Behavior**: +- aiCheckCompleted event emitted with IDLE verdict +- onIdleConfirmed called +- Respawn cycle begins +- No cooldown started + +**Priority**: HIGH + +--- + +### AIC-002: AI Checker Returns WORKING Verdict + +**Description**: AI idle check completes with WORKING verdict. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock AI checker to return WORKING + +**Expected Behavior**: +- aiCheckCompleted event emitted +- State returns to watching +- Cooldown started (aiIdleCheckCooldownMs) +- aiCheckCooldown event emitted +- No respawn cycle started + +**Priority**: HIGH + +--- + +### AIC-003: AI Checker Returns ERROR Verdict + +**Description**: AI idle check fails with ERROR verdict. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock AI checker to return ERROR + +**Expected Behavior**: +- aiCheckFailed event emitted +- State returns to watching +- consecutiveErrors incremented +- Pre-filter and no-output timers restarted + +**Priority**: HIGH + +--- + +### AIC-004: AI Checker Throws Exception + +**Description**: AI idle check throws an exception during execution. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock AI checker to throw Error + +**Expected Behavior**: +- Exception caught +- aiCheckFailed event emitted +- State returns to watching +- Controller remains stable + +**Priority**: HIGH + +--- + +### AIC-005: AI Checker Cancelled Mid-Check + +**Description**: AI check cancelled by new working patterns. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Before completion, simulate working pattern + +**Expected Behavior**: +- AI check cancelled +- Log: "Working patterns detected during AI check, cancelling" +- State returns to watching +- AI check result ignored when it arrives + +**Priority**: HIGH + +--- + +### AIC-006: AI Checker Cancelled by Substantial Output + +**Description**: AI check cancelled when substantial (>2 chars) output arrives. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Simulate output longer than 2 characters (after ANSI stripping) + +**Expected Behavior**: +- AI check cancelled +- Log includes "Substantial output during AI check" +- State returns to watching + +**Priority**: MEDIUM + +--- + +### AIC-007: AI Checker on Cooldown - Check Skipped + +**Description**: AI check attempt while on cooldown from previous WORKING verdict. + +**Initial State**: `watching` with AI checker on cooldown + +**Actions**: +1. Get WORKING verdict (starts cooldown) +2. Immediately trigger another pre-filter pass + +**Expected Behavior**: +- AI check not started +- Log: "AI check on cooldown (Xs remaining), waiting..." +- Controller stays in watching state + +**Priority**: HIGH + +--- + +### AIC-008: AI Checker Cooldown Expires + +**Description**: Controller behavior when AI checker cooldown expires. + +**Initial State**: `watching` with AI checker on cooldown + +**Actions**: +1. Wait for cooldown to expire +2. Check that pre-filter timer is restarted + +**Expected Behavior**: +- aiCheckCooldown event emitted (false, null) +- Pre-filter timer restarted +- Next idle signal can trigger AI check + +**Priority**: MEDIUM + +--- + +### AIC-009: AI Checker Disabled After Max Consecutive Errors + +**Description**: AI checker auto-disables after too many errors. + +**Initial State**: `watching` with AI check enabled + +**Actions**: +1. Cause maxConsecutiveErrors (3) failures +2. Attempt another idle detection + +**Expected Behavior**: +- AI checker status becomes 'disabled' +- disabled event emitted with reason +- Falls back to noOutputTimeoutMs for detection +- Log: "AI check unavailable (disabled)" + +**Priority**: HIGH + +--- + +### AIC-010: AI Checker Re-enabled via Config Update + +**Description**: Re-enabling AI checker after it was disabled. + +**Initial State**: AI checker disabled + +**Actions**: +1. Call updateConfig({ aiIdleCheckEnabled: true }) + +**Expected Behavior**: +- AI checker status becomes 'ready' +- disabledReason cleared +- Next idle detection can use AI check + +**Priority**: MEDIUM + +--- + +### AIC-011: AI Check Triggered via No-Output Fallback + +**Description**: AI check triggered by noOutputTimeoutMs when no output at all. + +**Initial State**: `watching` with no output received + +**Actions**: +1. Start controller +2. Wait for noOutputTimeoutMs without any terminal output + +**Expected Behavior**: +- AI check triggered via no-output fallback path +- aiCheckStarted event emitted +- Log: "No-output fallback: Xs silence" + +**Priority**: MEDIUM + +--- + +### AIC-012: AI Check via Pre-Filter Path (No Completion Message) + +**Description**: AI check triggered by pre-filter timer without completion message. + +**Initial State**: `watching` with output received but no completion message + +**Actions**: +1. Receive some output (sets lastOutputTime) +2. Wait for completionConfirmMs of silence +3. Wait for working patterns to be absent for 3s +4. Wait for tokens to be stable + +**Expected Behavior**: +- Pre-filter passes +- AI check started +- Works even without completion message detection + +**Priority**: MEDIUM + +--- + +### AIC-013: State Change During AI Check - Result Ignored + +**Description**: AI check result arrives after state has changed. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Stop controller before result arrives +3. AI check completes + +**Expected Behavior**: +- Result ignored +- Log: "AI check result ignored (state is now stopped)" +- No state transition attempted + +**Priority**: MEDIUM + +--- + +## AI Plan Checker Integration Scenarios + +### APC-001: Plan Check Returns PLAN_MODE + +**Description**: AI plan check confirms plan mode, triggers auto-accept. + +**Initial State**: `watching` with plan mode UI in buffer + +**Actions**: +1. Simulate plan mode output (numbered list + selector) +2. Wait for autoAcceptDelayMs +3. Mock plan checker to return PLAN_MODE + +**Expected Behavior**: +- planCheckCompleted event emitted +- Enter sent via writeViaScreen +- autoAcceptSent event emitted +- hasReceivedOutput reset to false + +**Priority**: HIGH + +--- + +### APC-002: Plan Check Returns NOT_PLAN_MODE + +**Description**: AI plan check determines not plan mode. + +**Initial State**: `watching` with ambiguous output + +**Actions**: +1. Simulate output that passes pre-filter +2. Mock plan checker to return NOT_PLAN_MODE + +**Expected Behavior**: +- planCheckCompleted event emitted +- No Enter sent +- Cooldown started (aiPlanCheckCooldownMs) + +**Priority**: HIGH + +--- + +### APC-003: Plan Check Already Checking + +**Description**: Plan check attempt while already checking. + +**Initial State**: Plan checker status = 'checking' + +**Actions**: +1. Start a plan check +2. Trigger another auto-accept attempt before first completes + +**Expected Behavior**: +- Second check skipped +- Log: "plan check already in progress" +- Only one check runs + +**Priority**: MEDIUM + +--- + +### APC-004: Plan Check Cancelled by New Output + +**Description**: New output during plan check cancels it. + +**Initial State**: Plan check running + +**Actions**: +1. Start plan check +2. Simulate new terminal output + +**Expected Behavior**: +- Plan check cancelled +- Log: "New output during plan check, cancelling (stale)" +- Result discarded + +**Priority**: HIGH + +--- + +### APC-005: Plan Check Result Stale (Output During Check) + +**Description**: Plan check completes but output arrived during check. + +**Initial State**: Plan check running + +**Actions**: +1. Start plan check +2. Record planCheckStartTime +3. Simulate output (updates lastOutputTime) +4. Plan check returns PLAN_MODE + +**Expected Behavior**: +- Result discarded (lastOutputTime > planCheckStartTime) +- Log: "Result discarded (output arrived during check)" +- No Enter sent + +**Priority**: HIGH + +--- + +### APC-006: Plan Check Result When State Changed + +**Description**: Plan check returns PLAN_MODE but state is no longer watching. + +**Initial State**: Plan check running + +**Actions**: +1. Start plan check +2. Trigger respawn cycle (state changes to sending_update) +3. Plan check returns PLAN_MODE + +**Expected Behavior**: +- Result not acted upon +- Log includes "state is X, not sending Enter" +- No auto-accept + +**Priority**: MEDIUM + +--- + +### APC-007: Pre-Filter Blocks - No Numbered Options + +**Description**: Pre-filter rejects output without numbered options. + +**Initial State**: `watching` + +**Actions**: +1. Simulate output without numbered list pattern +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter fails (PLAN_MODE_OPTION_PATTERN not matched) +- Log: "pre-filter did not match plan mode patterns" +- No plan check started + +**Priority**: MEDIUM + +--- + +### APC-008: Pre-Filter Blocks - No Selector Arrow + +**Description**: Pre-filter rejects output without selector indicator. + +**Initial State**: `watching` + +**Actions**: +1. Simulate numbered list without selector (no ">" or arrow) +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter fails (PLAN_MODE_SELECTOR_PATTERN not matched) +- No plan check started + +**Priority**: MEDIUM + +--- + +### APC-009: Pre-Filter Blocks - Working Pattern After Selector + +**Description**: Pre-filter rejects when working pattern appears after selector. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "1. Yes\n> 1.\nThinking..." +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter fails (working pattern after selector) +- No plan check started + +**Priority**: MEDIUM + +--- + +### APC-010: Pre-Filter Passes - Working Pattern Before Selector + +**Description**: Pre-filter allows working patterns if they appear before the selector. + +**Initial State**: `watching` with AI plan check disabled + +**Actions**: +1. Simulate: "Thinking...\nDone.\n> 1. Yes\n 2. No" +2. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Pre-filter passes (working pattern is before selector) +- Auto-accept triggered (since AI plan check disabled) + +**Priority**: MEDIUM + +--- + +### APC-011: Plan Check on Cooldown + +**Description**: Plan check skipped when on cooldown. + +**Initial State**: Plan checker on cooldown + +**Actions**: +1. Get NOT_PLAN_MODE verdict (starts cooldown) +2. Simulate plan mode output +3. Wait for autoAcceptDelayMs + +**Expected Behavior**: +- Plan check skipped +- Log: "plan checker on cooldown (Xs remaining)" +- No auto-accept + +**Priority**: MEDIUM + +--- + +## Timeout and Cooldown Scenarios + +### TC-001: Step Confirmation Timer Interrupted by Working + +**Description**: Step confirmation timer cancelled by working patterns. + +**Initial State**: `waiting_update` with step confirm timer running + +**Actions**: +1. Simulate completion message in waiting_update +2. Before confirmation timer fires, simulate working pattern + +**Expected Behavior**: +- Step confirm timer cancelled +- Log: "Step confirmation cancelled (working detected)" +- Remains in waiting_update, waiting for real completion + +**Priority**: HIGH + +--- + +### TC-002: Completion Confirm Timer Interrupted by Output + +**Description**: Confirmation timer restart when output arrives during confirmation. + +**Initial State**: `confirming_idle` with timer running + +**Actions**: +1. Detect completion message +2. Timer fires but lastOutputTime changed + +**Expected Behavior**: +- Timer restarts instead of confirming +- Log: "Output during confirmation, resetting" +- State remains confirming_idle + +**Priority**: MEDIUM + +--- + +### TC-003: Multiple Cooldowns Active Simultaneously + +**Description**: Both AI idle checker and plan checker on cooldown. + +**Initial State**: Both checkers on cooldown + +**Actions**: +1. Trigger WORKING verdict on idle checker +2. Trigger NOT_PLAN_MODE on plan checker +3. Try to trigger both checks + +**Expected Behavior**: +- Both checks skipped +- Controller falls back appropriately +- Cooldowns expire independently + +**Priority**: MEDIUM + +--- + +### TC-004: Zero-Duration Timers + +**Description**: Configuration with zero-duration timeouts. + +**Initial State**: `watching` with `completionConfirmMs: 0`, `interStepDelayMs: 0` + +**Actions**: +1. Trigger completion message +2. Complete full cycle + +**Expected Behavior**: +- Timers fire immediately (or within next event loop) +- Cycle completes without hanging +- No errors from zero-duration setTimeout + +**Priority**: LOW + +--- + +### TC-005: Very Long Timeout Values + +**Description**: Configuration with extremely long timeouts. + +**Initial State**: `watching` with `noOutputTimeoutMs: 3600000` (1 hour) + +**Actions**: +1. Start controller +2. Stop before timeout + +**Expected Behavior**: +- Timer properly cleared on stop +- No dangling timers +- Memory not leaked + +**Priority**: LOW + +--- + +### TC-006: Timer Tracking Accuracy + +**Description**: Active timer info reflects actual remaining time. + +**Initial State**: Timer running + +**Actions**: +1. Start a tracked timer (e.g., completion-confirm) +2. Call getActiveTimers() at various intervals +3. Check remainingMs decreases appropriately + +**Expected Behavior**: +- remainingMs decreases over time +- Never negative +- Removed from list after firing + +**Priority**: LOW + +--- + +## Error Recovery Scenarios + +### ER-001: Session Event Handler Throws + +**Description**: Exception thrown in terminal data handler. + +**Initial State**: `watching` + +**Actions**: +1. Cause handleTerminalData to throw (via malformed input) +2. Continue sending normal output + +**Expected Behavior**: +- Controller should not crash +- Should handle exception gracefully +- Should continue processing subsequent events + +**Priority**: HIGH + +--- + +### ER-002: writeViaScreen Fails + +**Description**: Writing to session fails during step send. + +**Initial State**: `sending_update` + +**Actions**: +1. Mock writeViaScreen to return false or throw +2. Let step delay timer fire + +**Expected Behavior**: +- Error should be caught +- Controller should emit error event +- Should not hang in sending state indefinitely + +**Priority**: MEDIUM + +--- + +### ER-003: Recovery After Partial Cycle Failure + +**Description**: Controller recovers after failing mid-cycle. + +**Initial State**: `waiting_clear` when error occurs + +**Actions**: +1. Simulate error during waiting_clear +2. Controller returns to watching or stopped +3. Try to start new cycle + +**Expected Behavior**: +- Clean state for new cycle +- No leftover timers or state +- New cycle works normally + +**Priority**: MEDIUM + +--- + +### ER-004: AI Checker Process Spawn Failure + +**Description**: Screen process fails to spawn for AI check. + +**Initial State**: `ai_checking` + +**Actions**: +1. Trigger AI check +2. Mock screen spawn to fail + +**Expected Behavior**: +- Error caught +- aiCheckFailed event emitted +- consecutiveErrors incremented +- Falls back gracefully + +**Priority**: MEDIUM + +--- + +### ER-005: Temp File Cleanup Failure + +**Description**: Temp file deletion fails in AI checker cleanup. + +**Initial State**: AI check completing + +**Actions**: +1. Complete AI check +2. Mock unlink to throw + +**Expected Behavior**: +- Exception caught (best effort cleanup) +- Check still completes +- No crash + +**Priority**: LOW + +--- + +## Concurrent Operation Scenarios + +### CO-001: Start Called While Stopping + +**Description**: Calling start() immediately after stop(). + +**Initial State**: Transitioning from running to stopped + +**Actions**: +1. Call stop() +2. Immediately call start() + +**Expected Behavior**: +- Clean restart +- No duplicate listeners +- Single watching state + +**Priority**: MEDIUM + +--- + +### CO-002: Multiple Terminal Events in Same Tick + +**Description**: Multiple terminal events processed synchronously. + +**Initial State**: `watching` + +**Actions**: +1. Emit multiple 'terminal' events without yielding +2. Check state consistency + +**Expected Behavior**: +- All events processed in order +- State remains consistent +- No race conditions in timer management + +**Priority**: HIGH + +--- + +### CO-003: Config Update During AI Check + +**Description**: Updating AI config while check is in progress. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Call updateConfig({ aiIdleCheckEnabled: false }) +3. AI check completes + +**Expected Behavior**: +- Pending check completes +- Future checks use new config +- No crash + +**Priority**: LOW + +--- + +### CO-004: Pause During AI Check + +**Description**: Pausing controller while AI check is running. + +**Initial State**: `ai_checking` + +**Actions**: +1. Start AI check +2. Call pause() +3. AI check completes + +**Expected Behavior**: +- Timers cleared by pause +- AI check result may arrive but timers won't fire +- Resume can restart monitoring + +**Priority**: LOW + +--- + +### CO-005: Buffer Append During Trim + +**Description**: New data arrives while buffer is being trimmed. + +**Initial State**: Buffer near MAX_RESPAWN_BUFFER_SIZE + +**Actions**: +1. Fill buffer to trigger trim +2. Simultaneously append more data + +**Expected Behavior**: +- Buffer stays within limits +- No data corruption +- Append operation completes + +**Priority**: LOW + +--- + +## Configuration Change Scenarios + +### CC-001: Disable Respawn While Running + +**Description**: Setting enabled to false while controller is running. + +**Initial State**: `watching` with cycle in progress + +**Actions**: +1. Call updateConfig({ enabled: false }) +2. Check controller behavior + +**Expected Behavior**: +- Config updated (for reference) +- Controller continues current operation +- On restart, start() will be no-op + +**Priority**: LOW + +--- + +### CC-002: Change updatePrompt During Cycle + +**Description**: Changing update prompt while cycle is in progress. + +**Initial State**: `waiting_update` + +**Actions**: +1. Call updateConfig({ updatePrompt: 'new prompt' }) +2. Start next cycle + +**Expected Behavior**: +- Current cycle uses old prompt (already sent) +- Next cycle uses new prompt + +**Priority**: LOW + +--- + +### CC-003: Toggle sendClear Mid-Cycle + +**Description**: Changing sendClear while in waiting_update. + +**Initial State**: `waiting_update` with sendClear: true + +**Actions**: +1. Call updateConfig({ sendClear: false }) +2. Update step completes + +**Expected Behavior**: +- Should use updated config value +- Skip clear and proceed appropriately + +**Priority**: MEDIUM + +--- + +### CC-004: Change Timeout Values During Wait + +**Description**: Modifying timeout values while timer is running. + +**Initial State**: Timer running with completionConfirmMs: 10000 + +**Actions**: +1. Call updateConfig({ completionConfirmMs: 1000 }) +2. Check when timer fires + +**Expected Behavior**: +- Current timer uses old value +- Next timer uses new value +- No crash or undefined behavior + +**Priority**: LOW + +--- + +### CC-005: Update kickstartPrompt to Undefined + +**Description**: Removing kickstart prompt during monitoring_init. + +**Initial State**: `monitoring_init` with kickstartPrompt set + +**Actions**: +1. Call updateConfig({ kickstartPrompt: undefined }) +2. /init doesn't trigger work + +**Expected Behavior**: +- Should check config at decision time +- May skip kickstart or use stale value +- (Behavior should be defined) + +**Priority**: LOW + +--- + +## Buffer and Memory Scenarios + +### BM-001: Buffer Trim at Exact Boundary + +**Description**: Buffer exactly at MAX_RESPAWN_BUFFER_SIZE. + +**Initial State**: Buffer at exactly 1MB + +**Actions**: +1. Fill buffer to exactly 1MB +2. Append 1 more character + +**Expected Behavior**: +- Trim triggered +- Buffer reduced to RESPAWN_BUFFER_TRIM_SIZE (512KB) +- Most recent data preserved + +**Priority**: LOW + +--- + +### BM-002: Continuous High-Volume Output + +**Description**: Sustained high-volume terminal output. + +**Initial State**: `watching` + +**Actions**: +1. Send 1MB of data per second for 10 seconds +2. Check memory usage and behavior + +**Expected Behavior**: +- Buffer stays bounded +- No memory growth over time +- Detection still functions + +**Priority**: MEDIUM + +--- + +### BM-003: Buffer Clear During Pattern Match + +**Description**: Buffer cleared while pattern detection is occurring. + +**Initial State**: Processing terminal data + +**Actions**: +1. Process data that triggers completeCycle +2. completeCycle clears buffer +3. More data arrives in same handler + +**Expected Behavior**: +- New data appended to fresh buffer +- No stale data patterns matched + +**Priority**: LOW + +--- + +### BM-004: Action Log Growth + +**Description**: Action log entries accumulating over time. + +**Initial State**: Many cycles completed + +**Actions**: +1. Run many cycles (>20) +2. Check recentActions length + +**Expected Behavior**: +- Limited to 20 entries max +- Oldest entries discarded +- Memory stable + +**Priority**: LOW + +--- + +## Edge Cases in Pattern Detection + +### PD-001: Completion Message Without "Worked" + +**Description**: Time duration pattern without "Worked" prefix. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "Waiting for 5s before retry" +2. Check if completion detected + +**Expected Behavior**: +- NOT detected as completion (requires "Worked" prefix) +- No false positive + +**Priority**: HIGH + +--- + +### PD-002: Nested Working Patterns in Text + +**Description**: Working pattern words in non-working context. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "The 'Thinking' file was created" +2. Check workingDetected status + +**Expected Behavior**: +- Currently: Would detect as working (false positive) +- Expected: Should not detect (pattern in quotes) +- (This is a known limitation) + +**Priority**: LOW (known limitation) + +--- + +### PD-003: Unicode Prompt Variants + +**Description**: Various Unicode prompt characters. + +**Initial State**: `watching` + +**Actions**: +1. Test: Regular '>' vs Unicode '>' vs fullwidth '>' +2. Check promptDetected + +**Expected Behavior**: +- Standard patterns detected +- Variant characters may not be detected +- (Document which are supported) + +**Priority**: LOW + +--- + +### PD-004: ANSI Codes Splitting Patterns + +**Description**: ANSI escape codes inserted within patterns. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "Wor\x1b[32mked\x1b[0m for 2m 46s" +2. Check completion detection + +**Expected Behavior**: +- Pattern may not match (ANSI in middle of "Worked") +- (This is edge case behavior) + +**Priority**: LOW + +--- + +### PD-005: Token Count at Boundary Values + +**Description**: Token counts with k/M suffixes. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "999.9k tokens" -> "1.0M tokens" +2. Check lastTokenCount value + +**Expected Behavior**: +- 999.9k = 999900 +- 1.0M = 1000000 +- Token change detected + +**Priority**: LOW + +--- + +### PD-006: Empty String Token Pattern + +**Description**: Token pattern with malformed numbers. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "tokens", " tokens", ".k tokens" +2. Check extractTokenCount returns + +**Expected Behavior**: +- Returns null for invalid patterns +- No crash + +**Priority**: LOW + +--- + +### PD-007: Multiple Completion Messages Same Data + +**Description**: Multiple "Worked for Xm Xs" in single terminal chunk. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "Worked for 1s... Worked for 2m 30s" +2. Check behavior + +**Expected Behavior**: +- First match triggers detection +- completionMessageTime set +- Single confirmation timer started + +**Priority**: LOW + +--- + +### PD-008: Spinner Character in Normal Text + +**Description**: Spinner Unicode character in regular output. + +**Initial State**: `watching` + +**Actions**: +1. Simulate: "The sequence is: ⠋ ⠙ ⠹" +2. Check workingDetected + +**Expected Behavior**: +- Currently: Detects as working +- This is expected behavior (conservative) + +**Priority**: LOW (expected behavior) + +--- + +## Event Emission and Listener Scenarios + +### EE-001: No Listeners Registered + +**Description**: Events emitted with no listeners. + +**Initial State**: Controller with no event listeners + +**Actions**: +1. Start controller +2. Run through cycle + +**Expected Behavior**: +- No errors +- Events still emitted (just not handled) +- Controller functions normally + +**Priority**: LOW + +--- + +### EE-002: Listener Throws Exception + +**Description**: Event listener throws during event handling. + +**Initial State**: Listener registered that throws + +**Actions**: +1. Register listener: on('stateChanged', () => { throw Error }) +2. Cause state change + +**Expected Behavior**: +- Exception propagates (EventEmitter default) +- (May want to catch in production) + +**Priority**: MEDIUM + +--- + +### EE-003: Listener Removes Itself + +**Description**: Listener that removes itself during event handling. + +**Initial State**: Self-removing listener registered + +**Actions**: +1. Register one-time listener +2. Cause event + +**Expected Behavior**: +- Listener called once +- Properly removed +- No memory leak + +**Priority**: LOW + +--- + +### EE-004: DetectionUpdate Interval Cleanup + +**Description**: Detection update interval properly cleaned up. + +**Initial State**: Controller running with detection updates + +**Actions**: +1. Start controller (starts 500ms interval) +2. Stop controller +3. Check interval cleared + +**Expected Behavior**: +- detectionUpdateTimer cleared on stop +- No interval continuing after stop +- No memory leak + +**Priority**: MEDIUM + +--- + +### EE-005: Timer Events During Stop + +**Description**: Timer fires during stop() execution. + +**Initial State**: Multiple timers running + +**Actions**: +1. Call stop() +2. Timer fires during cleanup + +**Expected Behavior**: +- Timer callback checks state +- No action taken if stopped +- Clean shutdown + +**Priority**: LOW + +--- + +## Summary + +### Priority Distribution + +| Priority | Count | Description | +|----------|-------|-------------| +| HIGH | 18 | Critical functionality and common paths | +| MEDIUM | 24 | Important edge cases and integrations | +| LOW | 22 | Rare edge cases and nice-to-have coverage | + +### Coverage Gaps Identified + +1. **Full cycle with all steps enabled** - No test covers the complete path through all states +2. **Clear fallback timer** - Not tested (10s timeout when no prompt detected) +3. **AI checker consecutive errors leading to disable** - Not integration tested +4. **Plan checker result discarding** - Stale result handling not fully tested +5. **Step confirmation timer** - The completionConfirmMs wait after each step not tested +6. **Working pattern detection during waiting states** - Limited coverage +7. **Buffer edge cases** - Trim behavior not tested +8. **Timer tracking for UI** - getActiveTimers() accuracy not verified +9. **Event listener error handling** - Not tested +10. **Elicitation flag lifecycle** - Partial coverage + +### Recommended Test Implementation Order + +1. STS-001 (Full cycle) - Establishes complete flow understanding +2. AIC-001, AIC-002, AIC-003 - Core AI checker verdicts +3. APC-001, APC-004, APC-005 - Core plan checker scenarios +4. STS-005 (Work during monitoring_init) - Important flow branch +5. TC-001 (Step confirm interrupted) - Timer reliability +6. ER-001 (Handler throws) - Robustness +7. CO-002 (Multiple events same tick) - Concurrency safety diff --git a/test/respawn-test-plan.md b/test/respawn-test-plan.md new file mode 100644 index 00000000..3bf76f69 --- /dev/null +++ b/test/respawn-test-plan.md @@ -0,0 +1,482 @@ +# Respawn Controller Test Plan + +This document describes the testing environment, architecture, and strategies for testing the RespawnController and related AI checker components. + +## Table of Contents + +1. [Test Environment Architecture](#test-environment-architecture) +2. [Component Interaction Diagram](#component-interaction-diagram) +3. [Mock Components](#mock-components) +4. [Port Allocation](#port-allocation) +5. [Test Isolation Strategy](#test-isolation-strategy) +6. [Cleanup Procedures](#cleanup-procedures) +7. [Test Categories](#test-categories) +8. [Usage Examples](#usage-examples) + +--- + +## Test Environment Architecture + +The RespawnController testing environment uses a layered approach to isolate components and enable deterministic testing without spawning real Claude CLI processes. + +``` ++------------------------------------------+ +| Test Runner (Vitest) | ++------------------------------------------+ + | | + v v ++------------------+ +------------------+ +| Unit Tests | | Integration Tests| +| (No real I/O) | | (Uses ports) | ++------------------+ +------------------+ + | | + v v ++------------------------------------------+ +| Test Utilities Layer | +| - MockSession | +| - MockAiIdleChecker | +| - MockAiPlanChecker | +| - TimeController | +| - State/Event Recorders | ++------------------------------------------+ + | + v ++------------------------------------------+ +| Real Components Under Test | +| - RespawnController | +| - RespawnConfig types | +| - State machine logic | ++------------------------------------------+ +``` + +### Key Principles + +1. **No Real Claude CLI**: Tests use MockAiIdleChecker and MockAiPlanChecker instead of spawning real Claude processes +2. **No Real Screens**: MockSession simulates terminal I/O without GNU screen +3. **Deterministic Timing**: Tests can use real timers (short timeouts) or fake timers for precise control +4. **Isolated State**: Each test gets fresh instances with no shared state + +--- + +## Component Interaction Diagram + +``` + RespawnController + | + +-----------------+-----------------+ + | | | + v v v + Session AiIdleChecker AiPlanChecker + (MockSession) (MockAiIdleChecker) (MockAiPlanChecker) + | | | + v | | + Terminal Events AI Check AI Check + (simulateXxx) Verdicts Verdicts + | | | + +--------+--------+---------+-------+ + | + v + State Machine + (watching -> ... -> stopped) + | + v + Event Emission + (stateChanged, stepSent, etc.) +``` + +### Data Flow + +1. **Terminal Output Flow**: + - `MockSession.simulateTerminalOutput(data)` -> `session.emit('terminal', data)` + - RespawnController receives terminal event in `handleTerminalData()` + - Pattern detection (completion, working, prompt) + - Timer management (start/reset/cancel) + +2. **AI Check Flow**: + - Pre-filter conditions met -> `tryStartAiCheck()` + - MockAiIdleChecker returns queued or default verdict + - Controller processes verdict (IDLE -> start cycle, WORKING -> cooldown) + +3. **State Transition Flow**: + - Internal state changes via `setState()` + - Events emitted for external monitoring + - Timers started/cancelled based on state + +--- + +## Mock Components + +### MockSession + +Enhanced session mock that simulates Claude Code terminal behavior. + +**Key Methods**: +- `simulateTerminalOutput(data)` - Raw terminal output +- `simulateCompletionMessage(duration?)` - "Worked for Xm Xs" pattern +- `simulateWorking(text?)` - Spinner/activity indicators +- `simulatePlanModePrompt()` - Numbered selection menu +- `simulateElicitationDialog()` - AskUserQuestion prompt +- `writeBuffer` - Captures all writes for assertions + +**Usage**: +```typescript +const session = createMockSession(); +session.simulateCompletionMessage(); +await waitForState(controller, 'confirming_idle'); +``` + +### MockAiIdleChecker + +Mock for AI idle detection that returns configurable verdicts. + +**Key Features**: +- Queue-based result system (FIFO) +- Default verdict when queue empty +- Cooldown simulation +- Error/disabled state simulation + +**Usage**: +```typescript +const checker = new MockAiIdleChecker('session-id'); +checker.setNextIdle('Task completed'); +// or +checker.queueResults( + { verdict: 'WORKING', reasoning: 'Still active', durationMs: 100 }, + { verdict: 'IDLE', reasoning: 'Now idle', durationMs: 150 } +); +``` + +### MockAiPlanChecker + +Mock for AI plan mode detection. + +**Key Features**: +- Same queue/default pattern as MockAiIdleChecker +- PLAN_MODE/NOT_PLAN_MODE verdicts +- Cooldown after NOT_PLAN_MODE + +**Usage**: +```typescript +const checker = new MockAiPlanChecker('session-id'); +checker.setNextPlanMode('Approval prompt detected'); +``` + +### TimeController + +Wrapper around Vitest's fake timers for deterministic timing tests. + +**Usage**: +```typescript +const time = createTimeController(); +// In test: +await time.advanceBy(1000); // Advance 1 second +await time.runAllTimers(); // Run all pending +// In afterEach: +time.useRealTimers(); +``` + +--- + +## Port Allocation + +Integration tests that spawn web servers use unique ports to avoid conflicts. + +| Port | Test File | Notes | +|-------|------------------------------|----------------------------------| +| 3099 | quick-start.test.ts | Basic startup tests | +| 3102 | session.test.ts | Session lifecycle tests | +| 3105 | scheduled-runs.test.ts | Scheduled task tests | +| 3107 | sse-events.test.ts | Server-Sent Events tests | +| 3110 | edge-cases.test.ts | Edge case handling | +| 3115 | integration-flows.test.ts | End-to-end flows | +| 3120 | session-cleanup.test.ts | Session cleanup tests | +| 3125 | ralph-integration.test.ts | Ralph loop integration | +| 3127 | *Available* | Next integration test | +| 3128 | *Available* | Reserved for respawn integration| +| 3129+ | *Available* | Future tests | + +### Port Usage Guidelines + +1. **Unit Tests**: No port needed (MockSession, no real server) +2. **Integration Tests**: Pick next available port (3127+) +3. **Parallel Safety**: `fileParallelism: false` ensures sequential execution + +--- + +## Test Isolation Strategy + +### Per-Test Isolation + +1. **Fresh Instances**: Each test creates new MockSession, MockAiIdleChecker, RespawnController +2. **No Shared State**: No global variables between tests +3. **Timer Cleanup**: Real timers have short timeouts; fake timers reset between tests + +### Per-File Isolation + +1. **beforeEach**: Create fresh instances +2. **afterEach**: + - Call `controller.stop()` to clear timers + - Reset time controller if using fake timers + - Clear mock queues + +### Cross-File Isolation + +1. **Sequential Execution**: `fileParallelism: false` in vitest.config.ts +2. **Screen Session Limits**: Max 10 concurrent (enforced by setup.ts) +3. **Orphan Cleanup**: `afterAll` cleans up any leaked screens + +--- + +## Cleanup Procedures + +### During Test Execution + +```typescript +afterEach(() => { + controller.stop(); // Clear all timers + session.removeAllListeners(); // Remove event handlers + mockChecker.reset(); // Clear queued results +}); +``` + +### After Test Suite + +The global `test/setup.ts` handles: + +1. **Screen Cleanup**: Kills any orphaned `claudeman-*` screens created during tests +2. **Process Cleanup**: Kills any Claude processes spawned by tests +3. **Pre-existing Protection**: Never kills screens that existed before tests started + +### Emergency Cleanup + +```typescript +import { forceCleanupAllTestResources } from './setup.js'; + +// Call if tests fail catastrophically +forceCleanupAllTestResources(); +``` + +--- + +## Test Categories + +### 1. Unit Tests (MockSession-based) + +Test the RespawnController state machine without real I/O. + +**File**: `test/respawn-controller.test.ts` + +**What to Test**: +- State transitions (watching -> confirming_idle -> ai_checking -> ...) +- Timer behavior (completion confirm, no-output fallback) +- Pattern detection (completion message, working patterns) +- Configuration handling +- Event emission + +**Example**: +```typescript +it('should transition to ai_checking when pre-filter met', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, { + ...FAST_TEST_CONFIG, + aiIdleCheckEnabled: true, + }); + + controller.start(); + session.simulateCompletionMessage(); + await new Promise(r => setTimeout(r, 100)); + + expect(controller.state).toBe('ai_checking'); +}); +``` + +### 2. AI Checker Unit Tests + +Test MockAiIdleChecker and MockAiPlanChecker behavior. + +**File**: `test/ai-idle-checker.test.ts`, `test/ai-plan-checker.test.ts` + +**What to Test**: +- Verdict queuing +- Cooldown behavior +- Error handling +- Disabled state + +### 3. State Machine Tests + +Comprehensive state transition testing. + +**File**: `test/respawn-controller.test.ts` (State Machine section) + +**What to Test**: +- All state transitions in the diagram +- Skip paths (sendClear: false, sendInit: false) +- Kickstart path +- Interruption handling (working patterns during transitions) + +### 4. Integration Tests (if needed) + +Full stack tests with real server but mocked Claude CLI. + +**Port**: 3128 (reserved) + +**What to Test**: +- API endpoints for respawn control +- SSE event broadcasting +- State persistence + +--- + +## Usage Examples + +### Basic Test Setup + +```typescript +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { RespawnController } from '../src/respawn-controller.js'; +import { + createMockSession, + MockAiIdleChecker, + FAST_TEST_CONFIG, + createStateTracker, +} from './respawn-test-utils.js'; + +describe('RespawnController Example', () => { + let session: MockSession; + let controller: RespawnController; + let stateTracker: ReturnType; + + beforeEach(() => { + session = createMockSession(); + controller = new RespawnController(session, FAST_TEST_CONFIG); + stateTracker = createStateTracker(); + controller.on('stateChanged', stateTracker.record); + }); + + afterEach(() => { + controller.stop(); + }); + + it('should start in watching state', () => { + controller.start(); + expect(controller.state).toBe('watching'); + }); +}); +``` + +### Testing with Mock AI Checker + +```typescript +// Note: Currently the RespawnController creates its own AI checkers internally. +// To inject mocks, you would need to modify the controller to accept them, +// or test the mock checkers independently and test the controller with +// aiIdleCheckEnabled: false. + +describe('RespawnController with AI Check Disabled', () => { + it('should fall back to direct idle on completion', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, { + ...FAST_TEST_CONFIG, + aiIdleCheckEnabled: false, // Falls back to timer-based + }); + + let cycleStarted = false; + controller.on('respawnCycleStarted', () => { cycleStarted = true; }); + + controller.start(); + session.simulateCompletionMessage(); + + await new Promise(r => setTimeout(r, 200)); + + expect(cycleStarted).toBe(true); + }); +}); +``` + +### Testing State Transitions + +```typescript +describe('State Transitions', () => { + it('should track full respawn cycle', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, { + ...FAST_TEST_CONFIG, + sendClear: true, + sendInit: true, + }); + const tracker = createStateTracker(); + controller.on('stateChanged', tracker.record); + + controller.start(); + session.simulateCompletionMessage(); + + // Wait for full cycle + await new Promise(r => setTimeout(r, 500)); + + expect(tracker.hasVisited('watching')).toBe(true); + expect(tracker.hasVisited('confirming_idle')).toBe(true); + expect(tracker.hasVisited('sending_update')).toBe(true); + expect(tracker.hasVisited('waiting_update')).toBe(true); + }); +}); +``` + +### Testing Event Emission + +```typescript +describe('Event Emission', () => { + it('should emit all lifecycle events', async () => { + const session = createMockSession(); + const controller = new RespawnController(session, FAST_TEST_CONFIG); + const recorder = createEventRecorder(); + + controller.on('stateChanged', recorder.handler('stateChanged')); + controller.on('respawnCycleStarted', recorder.handler('respawnCycleStarted')); + controller.on('stepSent', recorder.handler('stepSent')); + + controller.start(); + session.simulateCompletionMessage(); + + await new Promise(r => setTimeout(r, 200)); + + expect(recorder.hasEvent('respawnCycleStarted')).toBe(true); + expect(recorder.hasEvent('stepSent')).toBe(true); + }); +}); +``` + +--- + +## Future Considerations + +### Dependency Injection for AI Checkers + +To enable true mock injection, consider modifying RespawnController to accept optional AI checker instances in the constructor: + +```typescript +constructor( + session: Session, + config: Partial = {}, + aiChecker?: AiIdleChecker, + planChecker?: AiPlanChecker +) { + // Use provided checkers or create defaults + this.aiChecker = aiChecker || new AiIdleChecker(session.id, { ... }); + this.planChecker = planChecker || new AiPlanChecker(session.id, { ... }); +} +``` + +This would allow tests to inject MockAiIdleChecker/MockAiPlanChecker directly. + +### Real AI Checker Tests + +For testing the real AiIdleChecker and AiPlanChecker (which spawn Claude CLI), create a separate test file with: +- Longer timeouts +- Skip conditions for CI without Claude CLI +- Actual screen session usage + +```typescript +describe.skipIf(!process.env.CLAUDE_CLI_AVAILABLE)('Real AI Checkers', () => { + // Tests that spawn real Claude CLI +}); +``` diff --git a/test/respawn-test-utils.ts b/test/respawn-test-utils.ts new file mode 100644 index 00000000..29659ba6 --- /dev/null +++ b/test/respawn-test-utils.ts @@ -0,0 +1,1045 @@ +/** + * @fileoverview Test Utilities for RespawnController Tests + * + * Provides mocks, helpers, and utilities for testing the RespawnController + * and its related AI checker components without spawning real Claude CLI processes. + * + * ## Contents + * + * - MockSession: Enhanced mock session for testing + * - MockAiIdleChecker: Mock for AI idle checker with configurable verdicts + * - MockAiPlanChecker: Mock for AI plan mode checker with configurable verdicts + * - State transition helpers + * - Time manipulation utilities + * - Factory functions for pre-configured controllers + * + * @module test/respawn-test-utils + */ + +import { EventEmitter } from 'node:events'; +import { vi } from 'vitest'; +import type { Session } from '../src/session.js'; +import type { RespawnConfig, RespawnState, DetectionStatus } from '../src/respawn-controller.js'; +import type { + AiCheckResult, + AiCheckState, + AiCheckStatus, + AiCheckVerdict, + AiIdleCheckConfig, +} from '../src/ai-idle-checker.js'; +import type { + AiPlanCheckResult, + AiPlanCheckState, + AiPlanCheckStatus, + AiPlanCheckVerdict, + AiPlanCheckConfig, +} from '../src/ai-plan-checker.js'; + +// ========== Time Manipulation Utilities ========== + +/** + * Time controller for testing timeout-based behavior. + * Uses vitest's fake timers for deterministic time control. + */ +export interface TimeController { + /** Advance time by the specified milliseconds */ + advanceBy(ms: number): Promise; + /** Advance to the next timer */ + advanceToNextTimer(): Promise; + /** Run all pending timers */ + runAllTimers(): Promise; + /** Get the current fake time */ + now(): number; + /** Reset to real timers */ + useRealTimers(): void; +} + +/** + * Create a time controller for testing timeout-based behavior. + * Call this in beforeEach and cleanup with controller.useRealTimers() in afterEach. + */ +export function createTimeController(): TimeController { + vi.useFakeTimers(); + + return { + async advanceBy(ms: number): Promise { + await vi.advanceTimersByTimeAsync(ms); + }, + async advanceToNextTimer(): Promise { + await vi.runOnlyPendingTimersAsync(); + }, + async runAllTimers(): Promise { + await vi.runAllTimersAsync(); + }, + now(): number { + return Date.now(); + }, + useRealTimers(): void { + vi.useRealTimers(); + }, + }; +} + +// ========== MockSession ========== + +/** + * Enhanced mock session for testing RespawnController. + * Extends the existing MockSession pattern with additional utilities. + */ +export class MockSession extends EventEmitter { + id: string; + workingDir: string = '/tmp/test-workdir'; + status: 'idle' | 'working' = 'idle'; + writeBuffer: string[] = []; + terminalBuffer: string = ''; + + private _screenName: string | null = null; + + constructor(id: string = 'mock-session-id') { + super(); + this.id = id; + this._screenName = `claudeman-test-${id.slice(0, 8)}`; + } + + /** Direct PTY write (used by session.write()) */ + write(data: string): void { + this.writeBuffer.push(data); + } + + /** Write via screen (used by respawn controller) */ + writeViaScreen(data: string): boolean { + this.writeBuffer.push(data); + return true; + } + + /** Get the last written data */ + get lastWrite(): string | undefined { + return this.writeBuffer[this.writeBuffer.length - 1]; + } + + /** Clear the write buffer */ + clearWriteBuffer(): void { + this.writeBuffer = []; + } + + /** Check if a specific command was written */ + hasWritten(pattern: string | RegExp): boolean { + return this.writeBuffer.some(data => + typeof pattern === 'string' ? data.includes(pattern) : pattern.test(data) + ); + } + + // ========== Terminal Output Simulation ========== + + /** Simulate raw terminal output */ + simulateTerminalOutput(data: string): void { + this.terminalBuffer += data; + this.emit('terminal', data); + } + + /** Simulate prompt appearing (legacy fallback signal) */ + simulatePrompt(): void { + this.simulateTerminalOutput('\u276f '); + this.status = 'idle'; + this.emit('idle'); + } + + /** Simulate ready state with definitive indicator (legacy) */ + simulateReady(): void { + this.simulateTerminalOutput('\u21b5 send'); + this.status = 'idle'; + this.emit('idle'); + } + + /** + * Simulate completion message (primary idle detection in Claude Code 2024+). + * This triggers the multi-layer detection flow. + */ + simulateCompletionMessage(duration: string = '2m 46s'): void { + this.simulateTerminalOutput(`\u273b Worked for ${duration}`); + this.status = 'idle'; + } + + /** Simulate working state with spinner */ + simulateWorking(text: string = 'Thinking'): void { + this.simulateTerminalOutput(`${text}... \u280b`); + this.status = 'working'; + this.emit('working'); + } + + /** Simulate /clear completion */ + simulateClearComplete(): void { + this.simulateTerminalOutput('conversation cleared'); + setTimeout(() => this.simulateCompletionMessage(), 50); + } + + /** Simulate /init completion */ + simulateInitComplete(): void { + this.simulateTerminalOutput('Analyzing CLAUDE.md...'); + setTimeout(() => this.simulateCompletionMessage(), 100); + } + + /** + * Simulate plan mode approval prompt. + * This triggers auto-accept detection. + */ + simulatePlanModePrompt(): void { + this.simulateTerminalOutput( + 'Would you like to proceed with this plan?\n' + + '\u276f 1. Yes\n' + + ' 2. No\n' + + ' 3. Type your own\n' + ); + } + + /** + * Simulate elicitation dialog (AskUserQuestion). + * This should block auto-accept. + */ + simulateElicitationDialog(): void { + this.simulateTerminalOutput( + 'What would you like to name the new file?\n' + + '> ' + ); + } + + /** Simulate token count display */ + simulateTokenCount(tokens: number | string): void { + const formatted = typeof tokens === 'number' + ? tokens >= 1000 ? `${(tokens / 1000).toFixed(1)}k` : String(tokens) + : tokens; + this.simulateTerminalOutput(`${formatted} tokens used`); + } + + /** Simulate ANSI escape codes */ + simulateAnsiOutput(text: string, color: 'green' | 'red' | 'blue' = 'green'): void { + const codes: Record = { + green: '\x1b[32m', + red: '\x1b[31m', + blue: '\x1b[34m', + }; + this.simulateTerminalOutput(`${codes[color]}${text}\x1b[0m`); + } + + /** Clear terminal buffer */ + clearTerminalBuffer(): void { + this.terminalBuffer = ''; + } + + // ========== Session Lifecycle ========== + + /** Simulate session closing */ + close(): void { + this.emit('exit', 0); + this.removeAllListeners(); + } + + /** Get screen name (for screen-based operations) */ + get screenName(): string | null { + return this._screenName; + } +} + +// ========== MockAiIdleChecker ========== + +/** + * Configuration for MockAiIdleChecker behavior. + */ +export interface MockAiIdleCheckerOptions { + /** Default verdict to return */ + defaultVerdict?: AiCheckVerdict; + /** Default reasoning to include */ + defaultReasoning?: string; + /** Simulated check duration in ms */ + checkDurationMs?: number; + /** Whether to start in disabled state */ + startDisabled?: boolean; + /** Reason if starting disabled */ + disabledReason?: string; +} + +/** + * Mock AI idle checker for testing without spawning real Claude CLI. + * Allows configuring verdicts and simulating various states. + */ +export class MockAiIdleChecker extends EventEmitter { + private config: AiIdleCheckConfig; + private _status: AiCheckStatus = 'ready'; + private _lastVerdict: AiCheckVerdict | null = null; + private _lastReasoning: string | null = null; + private _lastCheckDurationMs: number | null = null; + private _cooldownEndsAt: number | null = null; + private _consecutiveErrors: number = 0; + private _totalChecks: number = 0; + private _disabledReason: string | null = null; + + // Queued results for sequential calls + private queuedResults: AiCheckResult[] = []; + private defaultOptions: MockAiIdleCheckerOptions; + private cooldownTimer: NodeJS.Timeout | null = null; + + constructor( + public sessionId: string, + config: Partial = {}, + options: MockAiIdleCheckerOptions = {} + ) { + super(); + this.config = { + enabled: true, + model: 'claude-opus-4-5-20251101', + maxContextChars: 16000, + checkTimeoutMs: 90000, + cooldownMs: 180000, + errorCooldownMs: 60000, + maxConsecutiveErrors: 3, + ...config, + }; + this.defaultOptions = { + defaultVerdict: 'IDLE', + defaultReasoning: 'Mock verdict', + checkDurationMs: 100, + startDisabled: false, + ...options, + }; + + if (this.defaultOptions.startDisabled) { + this._status = 'disabled'; + this._disabledReason = this.defaultOptions.disabledReason || 'Disabled for testing'; + } + } + + get status(): AiCheckStatus { + return this._status; + } + + getState(): AiCheckState { + return { + status: this._status, + lastVerdict: this._lastVerdict, + lastReasoning: this._lastReasoning, + lastCheckDurationMs: this._lastCheckDurationMs, + cooldownEndsAt: this._cooldownEndsAt, + consecutiveErrors: this._consecutiveErrors, + totalChecks: this._totalChecks, + disabledReason: this._disabledReason, + }; + } + + isOnCooldown(): boolean { + return this._cooldownEndsAt !== null && Date.now() < this._cooldownEndsAt; + } + + getCooldownRemainingMs(): number { + if (this._cooldownEndsAt === null) return 0; + return Math.max(0, this._cooldownEndsAt - Date.now()); + } + + /** + * Queue a result to be returned on the next check() call. + * Results are consumed in FIFO order. + */ + queueResult(result: AiCheckResult): void { + this.queuedResults.push(result); + } + + /** + * Queue multiple results for sequential check() calls. + */ + queueResults(...results: AiCheckResult[]): void { + this.queuedResults.push(...results); + } + + /** + * Set the next check to return IDLE verdict. + */ + setNextIdle(reasoning: string = 'Mock: IDLE'): void { + this.queueResult({ + verdict: 'IDLE', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return WORKING verdict. + */ + setNextWorking(reasoning: string = 'Mock: WORKING'): void { + this.queueResult({ + verdict: 'WORKING', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return ERROR verdict. + */ + setNextError(reasoning: string = 'Mock: ERROR'): void { + this.queueResult({ + verdict: 'ERROR', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + async check(_terminalBuffer: string): Promise { + if (this._status === 'disabled') { + return { verdict: 'ERROR', reasoning: `Disabled: ${this._disabledReason}`, durationMs: 0 }; + } + + if (this.isOnCooldown()) { + return { verdict: 'ERROR', reasoning: 'On cooldown', durationMs: 0 }; + } + + if (this._status === 'checking') { + return { verdict: 'ERROR', reasoning: 'Already checking', durationMs: 0 }; + } + + this._status = 'checking'; + this._totalChecks++; + this.emit('checkStarted'); + + // Simulate async check + await new Promise(resolve => setTimeout(resolve, 10)); + + // Get result from queue or use default + const result = this.queuedResults.shift() || { + verdict: this.defaultOptions.defaultVerdict!, + reasoning: this.defaultOptions.defaultReasoning!, + durationMs: this.defaultOptions.checkDurationMs!, + }; + + this._lastVerdict = result.verdict; + this._lastReasoning = result.reasoning; + this._lastCheckDurationMs = result.durationMs; + + if (result.verdict === 'IDLE') { + this._consecutiveErrors = 0; + this._status = 'ready'; + } else if (result.verdict === 'WORKING') { + this._consecutiveErrors = 0; + this.startCooldown(this.config.cooldownMs); + } else { + this._consecutiveErrors++; + if (this._consecutiveErrors >= this.config.maxConsecutiveErrors) { + this._status = 'disabled'; + this._disabledReason = `${this.config.maxConsecutiveErrors} consecutive errors`; + this.emit('disabled', this._disabledReason); + } else { + this.startCooldown(this.config.errorCooldownMs); + } + } + + this.emit('checkCompleted', result); + return result; + } + + cancel(): void { + if (this._status === 'checking') { + this._status = 'ready'; + this.emit('log', '[MockAiIdleChecker] Check cancelled'); + } + } + + reset(): void { + this.cancel(); + this.clearCooldown(); + this._lastVerdict = null; + this._lastReasoning = null; + this._lastCheckDurationMs = null; + this._consecutiveErrors = 0; + this.queuedResults = []; + this._status = this._disabledReason ? 'disabled' : 'ready'; + } + + updateConfig(config: Partial): void { + const filteredConfig = Object.fromEntries( + Object.entries(config).filter(([, v]) => v !== undefined) + ) as Partial; + this.config = { ...this.config, ...filteredConfig }; + if (config.enabled === false) { + this._disabledReason = 'Disabled by config'; + this._status = 'disabled'; + } else if (config.enabled === true && this._status === 'disabled') { + this._disabledReason = null; + this._status = 'ready'; + } + } + + getConfig(): AiIdleCheckConfig { + return { ...this.config }; + } + + // ========== Test Helpers ========== + + /** Force into cooldown state for testing */ + forceCooldown(durationMs: number): void { + this.startCooldown(durationMs); + } + + /** Force into disabled state for testing */ + forceDisabled(reason: string = 'Forced disabled for testing'): void { + this._status = 'disabled'; + this._disabledReason = reason; + this.emit('disabled', reason); + } + + /** Force ready state for testing */ + forceReady(): void { + this.clearCooldown(); + this._status = 'ready'; + this._disabledReason = null; + } + + private startCooldown(durationMs: number): void { + this.clearCooldown(); + this._cooldownEndsAt = Date.now() + durationMs; + this._status = 'cooldown'; + this.emit('cooldownStarted', this._cooldownEndsAt); + + this.cooldownTimer = setTimeout(() => { + this._cooldownEndsAt = null; + this._status = 'ready'; + this.emit('cooldownEnded'); + }, durationMs); + } + + private clearCooldown(): void { + if (this.cooldownTimer) { + clearTimeout(this.cooldownTimer); + this.cooldownTimer = null; + } + this._cooldownEndsAt = null; + if (this._status === 'cooldown') { + this._status = 'ready'; + } + } +} + +// ========== MockAiPlanChecker ========== + +/** + * Configuration for MockAiPlanChecker behavior. + */ +export interface MockAiPlanCheckerOptions { + /** Default verdict to return */ + defaultVerdict?: AiPlanCheckVerdict; + /** Default reasoning to include */ + defaultReasoning?: string; + /** Simulated check duration in ms */ + checkDurationMs?: number; + /** Whether to start in disabled state */ + startDisabled?: boolean; + /** Reason if starting disabled */ + disabledReason?: string; +} + +/** + * Mock AI plan checker for testing without spawning real Claude CLI. + * Allows configuring verdicts and simulating various states. + */ +export class MockAiPlanChecker extends EventEmitter { + private config: AiPlanCheckConfig; + private _status: AiPlanCheckStatus = 'ready'; + private _lastVerdict: AiPlanCheckVerdict | null = null; + private _lastReasoning: string | null = null; + private _lastCheckDurationMs: number | null = null; + private _cooldownEndsAt: number | null = null; + private _consecutiveErrors: number = 0; + private _totalChecks: number = 0; + private _disabledReason: string | null = null; + + // Queued results for sequential calls + private queuedResults: AiPlanCheckResult[] = []; + private defaultOptions: MockAiPlanCheckerOptions; + private cooldownTimer: NodeJS.Timeout | null = null; + + constructor( + public sessionId: string, + config: Partial = {}, + options: MockAiPlanCheckerOptions = {} + ) { + super(); + this.config = { + enabled: true, + model: 'claude-opus-4-5-20251101', + maxContextChars: 8000, + checkTimeoutMs: 60000, + cooldownMs: 30000, + errorCooldownMs: 30000, + maxConsecutiveErrors: 3, + ...config, + }; + this.defaultOptions = { + defaultVerdict: 'PLAN_MODE', + defaultReasoning: 'Mock verdict', + checkDurationMs: 100, + startDisabled: false, + ...options, + }; + + if (this.defaultOptions.startDisabled) { + this._status = 'disabled'; + this._disabledReason = this.defaultOptions.disabledReason || 'Disabled for testing'; + } + } + + get status(): AiPlanCheckStatus { + return this._status; + } + + getState(): AiPlanCheckState { + return { + status: this._status, + lastVerdict: this._lastVerdict, + lastReasoning: this._lastReasoning, + lastCheckDurationMs: this._lastCheckDurationMs, + cooldownEndsAt: this._cooldownEndsAt, + consecutiveErrors: this._consecutiveErrors, + totalChecks: this._totalChecks, + disabledReason: this._disabledReason, + }; + } + + isOnCooldown(): boolean { + return this._cooldownEndsAt !== null && Date.now() < this._cooldownEndsAt; + } + + getCooldownRemainingMs(): number { + if (this._cooldownEndsAt === null) return 0; + return Math.max(0, this._cooldownEndsAt - Date.now()); + } + + /** + * Queue a result to be returned on the next check() call. + */ + queueResult(result: AiPlanCheckResult): void { + this.queuedResults.push(result); + } + + /** + * Queue multiple results for sequential check() calls. + */ + queueResults(...results: AiPlanCheckResult[]): void { + this.queuedResults.push(...results); + } + + /** + * Set the next check to return PLAN_MODE verdict. + */ + setNextPlanMode(reasoning: string = 'Mock: PLAN_MODE'): void { + this.queueResult({ + verdict: 'PLAN_MODE', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return NOT_PLAN_MODE verdict. + */ + setNextNotPlanMode(reasoning: string = 'Mock: NOT_PLAN_MODE'): void { + this.queueResult({ + verdict: 'NOT_PLAN_MODE', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + /** + * Set the next check to return ERROR verdict. + */ + setNextError(reasoning: string = 'Mock: ERROR'): void { + this.queueResult({ + verdict: 'ERROR', + reasoning, + durationMs: this.defaultOptions.checkDurationMs || 100, + }); + } + + async check(_terminalBuffer: string): Promise { + if (this._status === 'disabled') { + return { verdict: 'ERROR', reasoning: `Disabled: ${this._disabledReason}`, durationMs: 0 }; + } + + if (this.isOnCooldown()) { + return { verdict: 'ERROR', reasoning: 'On cooldown', durationMs: 0 }; + } + + if (this._status === 'checking') { + return { verdict: 'ERROR', reasoning: 'Already checking', durationMs: 0 }; + } + + this._status = 'checking'; + this._totalChecks++; + this.emit('checkStarted'); + + // Simulate async check + await new Promise(resolve => setTimeout(resolve, 10)); + + // Get result from queue or use default + const result = this.queuedResults.shift() || { + verdict: this.defaultOptions.defaultVerdict!, + reasoning: this.defaultOptions.defaultReasoning!, + durationMs: this.defaultOptions.checkDurationMs!, + }; + + this._lastVerdict = result.verdict; + this._lastReasoning = result.reasoning; + this._lastCheckDurationMs = result.durationMs; + + if (result.verdict === 'PLAN_MODE') { + this._consecutiveErrors = 0; + this._status = 'ready'; + } else if (result.verdict === 'NOT_PLAN_MODE') { + this._consecutiveErrors = 0; + this.startCooldown(this.config.cooldownMs); + } else { + this._consecutiveErrors++; + if (this._consecutiveErrors >= this.config.maxConsecutiveErrors) { + this._status = 'disabled'; + this._disabledReason = `${this.config.maxConsecutiveErrors} consecutive errors`; + this.emit('disabled', this._disabledReason); + } else { + this.startCooldown(this.config.errorCooldownMs); + } + } + + this.emit('checkCompleted', result); + return result; + } + + cancel(): void { + if (this._status === 'checking') { + this._status = 'ready'; + this.emit('log', '[MockAiPlanChecker] Check cancelled'); + } + } + + reset(): void { + this.cancel(); + this.clearCooldown(); + this._lastVerdict = null; + this._lastReasoning = null; + this._lastCheckDurationMs = null; + this._consecutiveErrors = 0; + this.queuedResults = []; + this._status = this._disabledReason ? 'disabled' : 'ready'; + } + + updateConfig(config: Partial): void { + const filteredConfig = Object.fromEntries( + Object.entries(config).filter(([, v]) => v !== undefined) + ) as Partial; + this.config = { ...this.config, ...filteredConfig }; + if (config.enabled === false) { + this._disabledReason = 'Disabled by config'; + this._status = 'disabled'; + } else if (config.enabled === true && this._status === 'disabled') { + this._disabledReason = null; + this._status = 'ready'; + } + } + + getConfig(): AiPlanCheckConfig { + return { ...this.config }; + } + + // ========== Test Helpers ========== + + /** Force into cooldown state for testing */ + forceCooldown(durationMs: number): void { + this.startCooldown(durationMs); + } + + /** Force into disabled state for testing */ + forceDisabled(reason: string = 'Forced disabled for testing'): void { + this._status = 'disabled'; + this._disabledReason = reason; + this.emit('disabled', reason); + } + + /** Force ready state for testing */ + forceReady(): void { + this.clearCooldown(); + this._status = 'ready'; + this._disabledReason = null; + } + + private startCooldown(durationMs: number): void { + this.clearCooldown(); + this._cooldownEndsAt = Date.now() + durationMs; + this._status = 'cooldown'; + this.emit('cooldownStarted', this._cooldownEndsAt); + + this.cooldownTimer = setTimeout(() => { + this._cooldownEndsAt = null; + this._status = 'ready'; + this.emit('cooldownEnded'); + }, durationMs); + } + + private clearCooldown(): void { + if (this.cooldownTimer) { + clearTimeout(this.cooldownTimer); + this.cooldownTimer = null; + } + this._cooldownEndsAt = null; + if (this._status === 'cooldown') { + this._status = 'ready'; + } + } +} + +// ========== State Transition Helpers ========== + +/** + * State transition event for tracking. + */ +export interface StateTransition { + from: RespawnState; + to: RespawnState; + timestamp: number; +} + +/** + * Create a state tracker that records all state transitions. + * Attach to a RespawnController via controller.on('stateChanged', tracker.record). + */ +export function createStateTracker() { + const transitions: StateTransition[] = []; + + return { + /** Record a state transition (use as event handler) */ + record(to: RespawnState, from: RespawnState): void { + transitions.push({ from, to, timestamp: Date.now() }); + }, + + /** Get all recorded transitions */ + getTransitions(): StateTransition[] { + return [...transitions]; + }, + + /** Get only the state values (not timestamps) */ + getStates(): RespawnState[] { + return transitions.map(t => t.to); + }, + + /** Check if a specific state was visited */ + hasVisited(state: RespawnState): boolean { + return transitions.some(t => t.to === state); + }, + + /** Check if a specific transition occurred */ + hasTransition(from: RespawnState, to: RespawnState): boolean { + return transitions.some(t => t.from === from && t.to === to); + }, + + /** Get the most recent state */ + getCurrentState(): RespawnState | undefined { + return transitions.length > 0 ? transitions[transitions.length - 1].to : undefined; + }, + + /** Clear all recorded transitions */ + clear(): void { + transitions.length = 0; + }, + }; +} + +/** + * Create an event recorder for tracking all events from a RespawnController. + */ +export function createEventRecorder() { + const events: Array<{ type: string; args: unknown[]; timestamp: number }> = []; + + return { + /** Create a handler for a specific event type */ + handler(type: string): (...args: unknown[]) => void { + return (...args: unknown[]) => { + events.push({ type, args, timestamp: Date.now() }); + }; + }, + + /** Get all recorded events */ + getEvents(): Array<{ type: string; args: unknown[]; timestamp: number }> { + return [...events]; + }, + + /** Get events of a specific type */ + getEventsOfType(type: string): Array<{ type: string; args: unknown[]; timestamp: number }> { + return events.filter(e => e.type === type); + }, + + /** Check if an event type was emitted */ + hasEvent(type: string): boolean { + return events.some(e => e.type === type); + }, + + /** Count events of a specific type */ + countEvents(type: string): number { + return events.filter(e => e.type === type).length; + }, + + /** Clear all recorded events */ + clear(): void { + events.length = 0; + }, + }; +} + +// ========== Factory Functions ========== + +/** + * Default fast test configuration. + * Uses short timeouts for faster test execution. + */ +export const FAST_TEST_CONFIG: Partial = { + idleTimeoutMs: 50, + interStepDelayMs: 20, + completionConfirmMs: 50, + noOutputTimeoutMs: 300, + autoAcceptDelayMs: 100, + aiIdleCheckEnabled: false, + aiPlanCheckEnabled: false, +}; + +/** + * Configuration with AI checks enabled but short timeouts. + */ +export const AI_ENABLED_TEST_CONFIG: Partial = { + ...FAST_TEST_CONFIG, + aiIdleCheckEnabled: true, + aiIdleCheckTimeoutMs: 500, + aiIdleCheckCooldownMs: 200, + aiPlanCheckEnabled: true, + aiPlanCheckTimeoutMs: 500, + aiPlanCheckCooldownMs: 200, +}; + +/** + * Create a mock session typed as Session for use with RespawnController. + */ +export function createMockSession(id?: string): MockSession & Session { + return new MockSession(id) as MockSession & Session; +} + +// ========== Test Assertion Helpers ========== + +/** + * Wait for a specific state to be reached. + * Useful for async state transition testing. + */ +export async function waitForState( + controller: { state: RespawnState; on: (event: string, handler: (state: RespawnState) => void) => void }, + targetState: RespawnState, + timeoutMs: number = 1000 +): Promise { + if (controller.state === targetState) return; + + return new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error(`Timeout waiting for state ${targetState}, current: ${controller.state}`)); + }, timeoutMs); + + const handler = (state: RespawnState) => { + if (state === targetState) { + clearTimeout(timeout); + resolve(); + } + }; + + controller.on('stateChanged', handler); + }); +} + +/** + * Wait for a specific event to be emitted. + */ +export async function waitForEvent( + emitter: EventEmitter, + eventName: string, + timeoutMs: number = 1000 +): Promise { + return new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error(`Timeout waiting for event ${eventName}`)); + }, timeoutMs); + + emitter.once(eventName, (...args: unknown[]) => { + clearTimeout(timeout); + resolve(args[0] as T); + }); + }); +} + +/** + * Create a deferred promise for controlled async testing. + */ +export function createDeferred(): { + promise: Promise; + resolve: (value: T) => void; + reject: (error: Error) => void; +} { + let resolve!: (value: T) => void; + let reject!: (error: Error) => void; + const promise = new Promise((res, rej) => { + resolve = res; + reject = rej; + }); + return { promise, resolve, reject }; +} + +// ========== Terminal Output Generators ========== + +/** + * Generate realistic terminal output for testing. + */ +export const terminalOutputs = { + /** Standard completion message */ + completion(duration: string = '2m 46s'): string { + return `\n\u273b Worked for ${duration}\n 123.4k tokens used\n`; + }, + + /** Working spinner output */ + working(activity: string = 'Thinking'): string { + return `${activity}... \u280b`; + }, + + /** Plan mode prompt */ + planMode(question: string = 'Would you like to proceed?'): string { + return [ + `\n${question}\n`, + '\u276f 1. Yes\n', + ' 2. No\n', + ' 3. Type your own\n', + ].join(''); + }, + + /** Prompt character */ + prompt(): string { + return '\n\u276f '; + }, + + /** Token count display */ + tokens(count: number): string { + const formatted = count >= 1000000 + ? `${(count / 1000000).toFixed(1)}M` + : count >= 1000 + ? `${(count / 1000).toFixed(1)}k` + : String(count); + return ` ${formatted} tokens\n`; + }, + + /** Large output for buffer testing */ + largeOutput(sizeKb: number = 100): string { + const baseText = 'Lorem ipsum dolor sit amet. '.repeat(100); + const repetitions = Math.ceil((sizeKb * 1024) / baseText.length); + return baseText.repeat(repetitions).slice(0, sizeKb * 1024); + }, + + /** ANSI colored output */ + ansiColored(text: string): string { + return `\x1b[32m${text}\x1b[0m`; + }, +};