diff --git a/@fix_plan.md b/@fix_plan.md new file mode 100644 index 00000000..6234c582 --- /dev/null +++ b/@fix_plan.md @@ -0,0 +1,54 @@ +# Ralph Loop Inception - Improvement Plan + +This plan improves the Ralph Loop system to make it more reliable for 24+ hour autonomous runs. + +## Phase 1: Stuck-State Detection & Recovery (P0 - Critical) + +- [x] P0-001: Add stuck-state detection to RespawnController - detect when the same state persists for too long without progress +- [x] P0-002: Add iteration stall detection to RalphTracker - detect when iteration count stops incrementing despite respawn cycles +- [x] P0-003: Add automatic recovery action when stuck detected - escalate from soft reset to hard reset (implemented in handleStuckStateRecovery) +- [x] P0-004: Add stuck-state metrics to DetectionStatus for UI visibility (added stuckState to DetectionStatus) + +## Phase 2: Enhanced Idle Detection (P0 - Critical) + +- [x] P0-005: Add confidence decay over time - if no definitive signal for extended period, gradually lower confidence threshold +- [x] P0-006: Add Session.isWorking integration check before AI idle check - skip expensive AI call if session reports working +- [x] P0-007: Add RALPH_STATUS block integration with respawn controller - use EXIT_SIGNAL for more reliable completion detection (already implemented) + +## Phase 3: Promise Detection Improvements (P1 - High) + +- [x] P1-001: Add fuzzy matching for completion phrases - handle minor variations like whitespace or case +- [x] P1-002: Add promise phrase validation - warn if phrase is too common (likely false positives) +- [x] P1-003: Add multi-phrase support - allow multiple valid completion phrases for complex workflows + +## Phase 4: Error Recovery & Resilience (P1 - High) + +- [x] P1-004: Add circuit breaker reset on successful iteration - prevent permanent disabled state +- [x] P1-005: Add exponential backoff for AI check failures instead of immediate disable +- [x] P1-006: Add session health check before respawn cycle - skip if session is in error state + +## Phase 5: Todo Tracking Improvements (P1 - High) + +- [x] P1-007: Add todo deduplication by content similarity - prevent duplicate todos from repeated output +- [x] P1-008: Add todo priority inference from keywords - automatically set priority based on content +- [x] P1-009: Add todo progress estimation - estimate completion based on historical patterns + +## Phase 6: Respawn Cycle Optimization (P2 - Medium) + +- [x] P2-001: Add adaptive timing based on session behavior - adjust timeouts based on observed patterns (uses rolling 75th percentile of idle detection times) +- [x] P2-002: Add skip-clear optimization - skip /clear if context usage is low (below 30% by default) +- [ ] P2-003: Add smart kickstart prompt generation - use context to generate relevant kickstart prompts (deferred - requires AI generation) + +## Phase 7: Monitoring & Observability (P2 - Medium) + +- [x] P2-004: Add respawn cycle metrics - track success rate, average duration, failure reasons (RespawnCycleMetrics, RespawnAggregateMetrics types) +- [x] P2-005: Add Ralph Loop health score - aggregate metric for loop reliability (calculateHealthScore() method with 5 component scores) +- [ ] P2-006: Add automated anomaly detection - alert on unusual patterns (deferred - requires statistical analysis) + +## Completion Criteria + +All P0 and P1 tasks must be completed. P2 tasks are nice-to-have. +Tests must pass after each change. +Documentation must be updated. + +When ALL P0 and P1 tasks are complete, output: RALPH_INCEPTION_COMPLETE diff --git a/CLAUDE.md b/CLAUDE.md index 409e25bf..d16361b7 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -16,7 +16,7 @@ When user says "COM": 1. Increment version in BOTH `package.json` AND `CLAUDE.md` 2. Run: `git add -A && git commit -m "chore: bump version to X.XXXX" && git push && npm run build && systemctl --user restart claudeman-web` -**Version**: 0.1437 (must match `package.json`) +**Version**: 0.1438 (must match `package.json`) ## Project Overview @@ -35,10 +35,14 @@ Claudeman is a Claude Code session manager with web interface and autonomous Ral **Default port**: `3000` (web UI at `http://localhost:3000`) ```bash +# Setup +npm install # Install dependencies + # Development npx tsx src/index.ts web # Dev server (RECOMMENDED) npx tsx src/index.ts web --https # With TLS (only needed for remote access) npm run typecheck # Type check +tsc --noEmit --watch # Continuous type checking # Testing npx vitest run # All tests @@ -46,6 +50,7 @@ npx vitest run test/.test.ts # Single file npx vitest run -t "pattern" # Tests matching name npm run test:coverage # With coverage report npm run test:e2e # Browser E2E (requires: npx playwright install chromium) +npm run test:e2e:quick # Quick E2E (just quick-start workflow) # Production npm run build @@ -73,20 +78,35 @@ journalctl --user -u claudeman-web -f | `src/subagent-watcher.ts` | Monitors Claude Code's Task tool (background agents) | | `src/run-summary.ts` | Timeline events for "what happened while away" | | `src/ai-idle-checker.ts` | AI-powered idle detection with `ai-checker-base.ts` | +| `src/bash-tool-parser.ts` | Parses Claude's bash tool invocations from output | +| `src/transcript-watcher.ts` | Watches Claude's transcript files for changes | +| `src/hooks-config.ts` | Manages `.claude/settings.local.json` hook configuration | +| `src/image-watcher.ts` | Watches for image file creation (screenshots, etc.) | | `src/plan-orchestrator.ts` | Multi-agent plan generation with research and planning phases | | `src/prompts/*.ts` | Agent prompts (research-agent, code-reviewer, planner) | | `src/web/server.ts` | Fastify REST API + SSE at `/api/events` | | `src/web/public/app.js` | Frontend: xterm.js, tab management, subagent windows | | `src/types.ts` | All TypeScript interfaces | +### Config Files (`src/config/`) + +| File | Purpose | +|------|---------| +| `buffer-limits.ts` | Terminal/text buffer size limits | +| `map-limits.ts` | Global limits for Maps, sessions, watchers | + ### Utility Files (`src/utils/`) | File | Purpose | |------|---------| +| `index.ts` | Re-exports all utilities (standard import point) | | `lru-map.ts` | LRU eviction Map for bounded caches | | `stale-expiration-map.ts` | TTL-based Map with lazy expiration | | `cleanup-manager.ts` | Centralized resource disposal | | `buffer-accumulator.ts` | Chunk accumulator with size limits | +| `string-similarity.ts` | String matching utilities (fuzzy matching) | +| `token-validation.ts` | Token count parsing and validation | +| `regex-patterns.ts` | Shared regex patterns for parsing | ### Data Flow @@ -103,15 +123,17 @@ journalctl --user -u claudeman-web -f **Token tracking**: Interactive mode parses status line ("123.4k tokens"), estimates 60/40 input/output split. -**Memory leak prevention**: Frontend runs long; clear all Maps/timers on SSE reconnect in `handleInit()`. Backend clears `_recentTaskDescriptions` in Session.stop(), nulls promise callbacks on error, and removes watcher listeners on shutdown. +**Hook events**: Claude Code hooks trigger notifications via `/api/hook-event`. Key events: `permission_prompt` (tool approval needed), `elicitation_dialog` (Claude asking question), `idle_prompt` (waiting for input), `stop` (response complete). See `src/hooks-config.ts`. ## Adding Features -- **API endpoint**: Types in `types.ts`, route in `server.ts:buildServer()`, use `createErrorResponse()` +- **API endpoint**: Types in `types.ts`, route in `server.ts:buildServer()`, use `createErrorResponse()`. Validate request bodies with Zod schemas. - **SSE event**: Emit via `broadcast()`, handle in `app.js:handleSSEEvent()` - **Session setting**: Add to `SessionState` in `types.ts`, include in `session.toState()`, call `persistSessionState()` - **New test**: Pick unique port (see below), add port comment to test file header +**Validation**: Uses Zod v4 for request validation. Define schemas near route handlers and use `.parse()` or `.safeParse()`. + ## State Files | File | Purpose | @@ -120,13 +142,41 @@ journalctl --user -u claudeman-web -f | `~/.claudeman/screens.json` | Screen metadata for recovery | | `~/.claudeman/settings.json` | User preferences | +## Default Settings + +UI defaults are optimized for minimal distraction. Set in `src/web/public/app.js` (using `??` operator). + +**Display Settings** (default values): +| Setting | Default | Description | +|---------|---------|-------------| +| `showFontControls` | `false` | Font size controls in header | +| `showSystemStats` | `true` | CPU/memory stats in header | +| `showTokenCount` | `true` | Token counter in header | +| `showCost` | `false` | Cost display | +| `showMonitor` | `true` | Monitor panel | +| `showProjectInsights` | `false` | Project insights panel | +| `showFileBrowser` | `false` | File browser panel | +| `showSubagents` | `false` | Subagent windows panel | + +**Tracking Settings**: +| Setting | Default | Description | +|---------|---------|-------------| +| `ralphTrackerEnabled` | `false` | Ralph/Todo loop tracking | +| `subagentTrackingEnabled` | `true` | Background agent monitoring | +| `subagentActiveTabOnly` | `true` | Show subagents only for active session | +| `imageWatcherEnabled` | `false` | Watch for image file creation | + +**Notification Defaults**: Browser notifications enabled, audio alerts disabled. Critical events (permission prompts, questions) notify by default; info events (respawn cycles, token milestones) are silent. + +To change defaults, edit the `??` fallback values in `openAppSettings()` and `apply*Visibility()` functions. + ## Testing **Port allocation**: E2E tests use centralized ports in `test/e2e/e2e.config.ts`. Unit/integration tests pick unique ports manually. Search `const PORT =` or `TEST_PORT` in test files to find used ports before adding new tests. **E2E tests**: Use Playwright. Run `npx playwright install chromium` first. See `test/e2e/fixtures/` for helpers. E2E config (`test/e2e/e2e.config.ts`) provides ports (3183-3190), timeouts, and helpers. -**Test config**: Vitest runs with `globals: true` (no imports needed for `describe`/`it`/`expect`) and `fileParallelism: false` (files run sequentially to respect screen limits). Unit test timeout is 30s, teardown timeout is 60s. E2E tests have longer timeouts defined in `test/e2e/e2e.config.ts` (90s test, 30s session creation). +**Test config**: Vitest runs with `globals: true` (no imports needed for `describe`/`it`/`expect`/`vi`) and `fileParallelism: false` (files run sequentially to respect screen limits). Unit test timeout is 30s, teardown timeout is 60s. E2E tests have longer timeouts defined in `test/e2e/e2e.config.ts` (90s test, 30s session creation). **Test safety**: `test/setup.ts` provides: - Screen concurrency limiter (max 10) @@ -183,7 +233,7 @@ Use `LRUMap` for bounded caches with eviction, `StaleExpirationMap` for TTL-base | **Ralph Loop guide** | `docs/ralph-wiggum-guide.md` | | **Claude Code hooks** | `docs/claude-code-hooks-reference.md` | | **Browser/E2E testing** | `docs/browser-testing-guide.md` | -| **API routes** | `src/web/server.ts:buildServer()` or README.md | +| **API routes** | `src/web/server.ts:buildServer()` or README.md (full endpoint tables) | | **SSE events** | Search `broadcast(` in `server.ts` | | **CLI commands** | `claudeman --help` | | **Frontend patterns** | `src/web/public/app.js` (subagent windows, notifications) | @@ -193,6 +243,7 @@ Use `LRUMap` for bounded caches with eviction, `StaleExpirationMap` for TTL-base | **Test utilities** | `test/respawn-test-utils.ts` | | **Memory leak patterns** | `test/memory-leak-prevention.test.ts` | | **Keyboard shortcuts** | README.md or App Settings in web UI | +| **Mobile/SSH access** | README.md (Claudeman Screens / `sc` command) | | **Plan orchestrator** | `src/plan-orchestrator.ts` file header | | **Agent prompts** | `src/prompts/` directory | @@ -201,7 +252,7 @@ Use `LRUMap` for bounded caches with eviction, `StaleExpirationMap` for TTL-base | Script | Purpose | |--------|---------| | `scripts/screen-manager.sh` | Safe screen management (use instead of direct kill commands) | -| `scripts/screen-chooser.sh` | Mobile-friendly screen session picker for Termius/iPhone | +| `scripts/screen-chooser.sh` | Claudeman Screens - mobile-friendly session picker (`sc` alias, see README for usage) | | `scripts/monitor-respawn.sh` | Monitor respawn state machine in real-time | | `scripts/postinstall.js` | npm postinstall hook for setup | @@ -209,19 +260,9 @@ Use `LRUMap` for bounded caches with eviction, `StaleExpirationMap` for TTL-base The TUI (Terminal UI) has been removed in favor of the web interface. Files in `src/tui/` are excluded from compilation via `tsconfig.json`. -## Recent Memory Leak Fixes (2026-01-30) +## Memory Leak Prevention -All P0 memory leak issues have been fixed in commit `e3e0d22`: - -### Backend Fixes -- **Session._recentTaskDescriptions**: Now cleared in `stop()` and `clearBuffers()` -- **Session promise callbacks**: Nulled after rejection in `runPrompt()` catch block -- **Watcher listeners**: SubagentWatcher and ImageWatcher listeners stored and removed on server shutdown - -### Frontend Fixes -- **Plan file windows**: Drag/resize handlers stored on elements and cleaned up via `closePlanFileWindow()` -- **Plan file manager**: Drag handler stored and cleaned up via `closePlanFileManager()` -- **cleanupAllFloatingWindows()**: Now cleans up plan file windows +Frontend runs long (24+ hour sessions); all Maps/timers must be cleaned up. ### Cleanup Patterns When adding new event listeners or timers: @@ -229,8 +270,8 @@ When adding new event listeners or timers: 2. Add cleanup to appropriate `stop()` or `cleanup*()` method 3. For singleton watchers, store refs in class properties and remove in server `stop()` -### Verification Tests -Memory leak prevention patterns are tested in `test/memory-leak-prevention.test.ts`. Run with: -```bash -npx vitest run test/memory-leak-prevention.test.ts -``` +**Backend**: Clear Maps in `stop()`, null promise callbacks on error, remove watcher listeners on shutdown. + +**Frontend**: Store drag/resize handlers on elements, clean up in `close*()` functions. SSE reconnect calls `handleInit()` which resets state. + +Run `npx vitest run test/memory-leak-prevention.test.ts` to verify patterns. diff --git a/README.md b/README.md index 12cdca84..16552792 100644 --- a/README.md +++ b/README.md @@ -282,6 +282,37 @@ claudeman web --- +## Mobile Access (Termius/SSH) + +**Claudeman Screens** (`sc`) is a mobile-friendly screen session chooser, optimized for Termius on iPhone. + +```bash +sc # Interactive chooser +sc 2 # Quick attach to session 2 +sc -l # List sessions +sc -h # Help +``` + +**Features:** +- Single-digit selection (1-9) for fast thumb typing +- Color-coded status indicators (attached/detached/respawn) +- Token count display +- Session names from Claudeman state +- Pagination for many sessions +- Auto-refresh every 60 seconds + +**Indicators:** +| Symbol | Meaning | +|--------|---------| +| `*` / `●` | Attached (someone connected) | +| `-` / `○` | Detached (available) | +| `R` | Respawn enabled | +| `45k` | Token count | + +**Tip:** Detach from a screen with `Ctrl+A D` + +--- + ## API ### Sessions diff --git a/install.sh b/install.sh index b969fb01..229fa5cf 100755 --- a/install.sh +++ b/install.sh @@ -588,6 +588,32 @@ add_to_path() { success "Added to $profile - restart your shell or run: source $profile" } +setup_sc_alias() { + local profile + profile=$(detect_shell_profile) + + # Check if alias already exists + if [[ -f "$profile" ]] && grep -qE "^alias sc=" "$profile" 2>/dev/null; then + info "Alias 'sc' already configured in $profile" + return 0 + fi + + local shell_name + shell_name="$(basename "${SHELL:-/bin/bash}")" + + if [[ "$shell_name" == "fish" ]]; then + echo "" >> "$profile" + echo "# Claudeman Screens shortcut" >> "$profile" + echo "alias sc='screen-chooser'" >> "$profile" + else + echo "" >> "$profile" + echo "# Claudeman Screens shortcut" >> "$profile" + echo "alias sc='screen-chooser'" >> "$profile" + fi + + info "Added 'sc' alias for screen-chooser" +} + # ============================================================================ # Screen Configuration # ============================================================================ @@ -876,6 +902,14 @@ main() { ln -sf "$INSTALL_DIR/dist/index.js" "$symlink_dir/claudeman" info "Created symlink: $symlink_dir/claudeman" + # Install screen-chooser as 'screen-chooser' command + if [[ -f "$INSTALL_DIR/scripts/screen-chooser.sh" ]]; then + ln -sf "$INSTALL_DIR/scripts/screen-chooser.sh" "$symlink_dir/screen-chooser" + info "Created symlink: $symlink_dir/screen-chooser" + # Add 'sc' alias for quick access + setup_sc_alias + fi + # Add ~/.local/bin to PATH if not already there if [[ ":$PATH:" != *":$symlink_dir:"* ]]; then add_to_path "$symlink_dir" @@ -919,6 +953,12 @@ main() { echo -e " ${CYAN}# Open in browser${NC}" echo -e " http://localhost:3000" echo "" + echo -e " ${BOLD}Mobile Access (Termius/SSH):${NC}" + echo "" + echo -e " ${CYAN}sc${NC} # Interactive screen session chooser" + echo -e " ${CYAN}sc 2${NC} # Quick attach to session 2" + echo -e " ${CYAN}sc -h${NC} # Help" + echo "" if [[ "$os" == "linux" ]] && [[ -f "$HOME/.config/systemd/user/claudeman-web.service" ]]; then echo -e " ${BOLD}Systemd Service:${NC}" diff --git a/package.json b/package.json index eb68964a..15e974f8 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "claudeman", - "version": "0.1437", + "version": "0.1438", "description": "The missing control plane for Claude Code - run 20 autonomous agents with real-time monitoring and session persistence", "type": "module", "main": "dist/index.js", diff --git a/scripts/screen-chooser.sh b/scripts/screen-chooser.sh index e05f47a1..149964df 100755 --- a/scripts/screen-chooser.sh +++ b/scripts/screen-chooser.sh @@ -1,7 +1,7 @@ #!/bin/bash # ============================================================================ -# Screen Chooser for iPhone/Termius -# Optimized for iPhone 17 Pro (portrait ~45 chars, landscape ~95 chars) +# Claudeman Screens - Mobile-friendly Screen Session Chooser +# Optimized for iPhone/Termius (portrait ~45 chars, landscape ~95 chars) # ============================================================================ # # Design principles: @@ -12,12 +12,12 @@ # - Minimal keystrokes to attach # # Usage: -# ./screen-chooser.sh # Interactive chooser -# ./screen-chooser.sh 1 # Quick attach to session 1 -# ./screen-chooser.sh -l # List only (no interactive) -# ./screen-chooser.sh -h # Help +# screen-chooser # Interactive chooser +# screen-chooser 1 # Quick attach to session 1 +# screen-chooser -l # List only (non-interactive) +# screen-chooser -h # Help # -# Alias: alias sc='path/to/screen-chooser.sh' +# Alias (added by installer): alias sc='screen-chooser' # Then: sc (interactive) # sc 2 (attach session 2) # @@ -343,7 +343,7 @@ clear_screen() { # Print header print_header() { local count=${#SCREEN_PIDS[@]} - echo -e "${B}${CYAN}${ICON_SCREEN} Screens${R} ${D}($count)${R}" + echo -e "${B}${CYAN}Claudeman Screens${R} ${D}($count)${R}" echo -e "${D}$(printf '%.0s─' {1..32})${R}" } @@ -423,7 +423,7 @@ print_footer() { # Print no screens message print_no_screens() { clear_screen - echo -e "${B}${CYAN}${ICON_SCREEN} Screens${R}" + echo -e "${B}${CYAN}Claudeman Screens${R}" echo -e "${D}$(printf '%.0s─' {1..32})${R}" echo "" echo -e " ${YELLOW}No screen sessions found${R}" @@ -631,7 +631,7 @@ quick_attach() { show_help() { cat << 'EOF' -Screen Chooser for iPhone/Termius +Claudeman Screens - Mobile-friendly Screen Session Chooser USAGE: sc Interactive chooser @@ -653,9 +653,9 @@ INDICATORS: 45k Token count TIPS: - - Alias: alias sc='path/to/screen-chooser.sh' - - Detach: Ctrl+A D + - Detach from screen: Ctrl+A D - Session names from Claudeman state + - Optimized for Termius/iPhone EOF } diff --git a/src/ai-checker-base.ts b/src/ai-checker-base.ts index ec2d53c1..a4db9153 100644 --- a/src/ai-checker-base.ts +++ b/src/ai-checker-base.ts @@ -508,7 +508,15 @@ export abstract class AiCheckerBase< if (this.consecutiveErrors >= this.config.maxConsecutiveErrors) { this.disable(`${this.config.maxConsecutiveErrors} consecutive errors: ${errorMsg}`); } else { - this.startCooldown(this.config.errorCooldownMs); + // P1-005: Exponential backoff for errors + // Base cooldown * 2^(consecutiveErrors-1), capped at 5 minutes + const backoffMultiplier = Math.pow(2, this.consecutiveErrors - 1); + const backoffCooldownMs = Math.min( + this.config.errorCooldownMs * backoffMultiplier, + 5 * 60 * 1000 // Max 5 minutes + ); + this.log(`Exponential backoff: ${Math.round(backoffCooldownMs / 1000)}s (error #${this.consecutiveErrors})`); + this.startCooldown(backoffCooldownMs); } } diff --git a/src/ai-idle-checker.ts b/src/ai-idle-checker.ts index f335f02e..629050b1 100644 --- a/src/ai-idle-checker.ts +++ b/src/ai-idle-checker.ts @@ -70,33 +70,66 @@ const DEFAULT_AI_CHECK_CONFIG: AiIdleCheckConfig = { /** Pattern to match IDLE or WORKING as the first word of output */ const VERDICT_PATTERN = /^\s*(IDLE|WORKING)\b/i; -/** The prompt sent to the AI checker */ -const AI_CHECK_PROMPT = `Analyze this terminal output from a running Claude Code session. Determine if the session is IDLE (done working, waiting for new input) or WORKING (still actively processing). +/** + * The prompt sent to the AI idle checker. + * + * P1-005: Enhanced with more specific working pattern examples and clearer structure. + */ +const AI_CHECK_PROMPT = `You are analyzing terminal output from a Claude Code CLI session. Determine if Claude has FINISHED working (IDLE) or is STILL WORKING (WORKING). -IMPORTANT: When in doubt, answer WORKING. Brief pauses between tool executions do NOT mean the session is idle. Claude may be processing or about to output more. +CRITICAL RULE: When in doubt, ALWAYS answer WORKING. False positives (saying IDLE when Claude is working) cause session interruptions. It's safer to wait longer than to interrupt active work. -IDLE indicators (need MULTIPLE of these to confirm idle): -- Completion summary shown (e.g., "✻ Worked for 2m 46s", "Worked for 5s") -- Prompt character visible at the end (❯ or similar) -- Cost summary displayed (e.g., "$0.12 spent") -- Clear end of output with no pending work +## IDLE Indicators (need AT LEAST 2 of these together) -WORKING indicators (ANY of these means WORKING): -- Spinner characters (⠋ ⠙ ⠹ ⠸ ⠼ ⠴ ⠦ ⠧ ⠇ ⠏ or similar) -- Activity text: Thinking, Writing, Reading, Running, Searching, Editing, Creating, Deleting, Analyzing, Executing, Synthesizing, Compiling, Building, Processing, Loading, Generating, Testing, Checking, Validating -- Tool execution in progress (commands being run) -- Truncated or partial lines at the end -- File operations in progress -- Output that appears mid-stream or incomplete -- No completion summary visible yet +1. **Completion Summary** - The most reliable signal: + - "✻ Worked for Xm Ys" (e.g., "✻ Worked for 2m 46s") + - "Worked for Xs" (e.g., "Worked for 5s") + - Cost summary: "$X.XX spent" or "X tokens used" -Terminal output (most recent at bottom): +2. **Input Prompt Visible**: + - The ❯ prompt character at the very end + - Empty line after completion summary + - Waiting cursor position + +3. **Task Completion Language**: + - "All done", "Finished", "Completed successfully" + - Explicit "waiting for input" or similar + +## WORKING Indicators (ANY ONE of these = answer WORKING) + +### Active Processing Indicators: +- **Spinners**: ⠋ ⠙ ⠹ ⠸ ⠼ ⠴ ⠦ ⠧ ⠇ ⠏ (Braille), ◐ ◓ ◑ ◒ (quarter), ⣾ ⣽ ⣻ ⢿ ⡿ ⣟ ⣯ ⣷ +- **Activity Words**: Thinking, Writing, Reading, Running, Searching, Editing, Creating, Deleting, Analyzing, Executing, Synthesizing, Compiling, Building, Processing, Loading, Generating, Testing, Checking, Validating, Brewing, Formatting, Linting, Installing, Fetching, Downloading + +### Tool Execution in Progress: +- Bash commands with no result shown yet +- "Running: npm test", "Executing command..." +- File read/write operations incomplete +- Progress bars or percentage indicators +- Test suite running (dots appearing, "Test Suites: X passed") + +### Output Structure Issues: +- Truncated lines without completion +- JSON/code blocks not closed +- Multi-line output clearly incomplete +- "..." indicating more to come +- Output ending mid-sentence or mid-word + +### Claude Planning/Thinking: +- "Let me...", "I'll...", "Now I need to..." +- TodoWrite updates without completion +- Plan mode approval prompts (numbered options) + +## Terminal Output to Analyze --- {TERMINAL_BUFFER} --- -Answer with EXACTLY one word on the first line: IDLE or WORKING -If uncertain, answer WORKING. Then briefly explain why.`; +## Your Response +First line: EXACTLY "IDLE" or "WORKING" (nothing else) +Second line onwards: Brief explanation of your reasoning. + +Remember: When uncertain, answer WORKING.`; // ========== AiIdleChecker Class ========== diff --git a/src/ralph-tracker.ts b/src/ralph-tracker.ts index 083aaca0..9ac85f04 100644 --- a/src/ralph-tracker.ts +++ b/src/ralph-tracker.ts @@ -27,10 +27,17 @@ import { RalphTestsStatus, RalphWorkType, CircuitBreakerStatus, + RalphTodoProgress, + CompletionConfidence, createInitialRalphTrackerState, createInitialCircuitBreakerStatus, } from './types.js'; -import { ANSI_ESCAPE_PATTERN_SIMPLE } from './utils/index.js'; +import { + ANSI_ESCAPE_PATTERN_SIMPLE, + fuzzyPhraseMatch, + todoContentHash, + stringSimilarity, +} from './utils/index.js'; // ========== Enhanced Plan Task Interface ========== @@ -119,6 +126,14 @@ const TODO_EXPIRY_MS = 60 * 60 * 1000; */ const CLEANUP_THROTTLE_MS = 30 * 1000; +/** + * Similarity threshold for todo deduplication. + * Todos with similarity >= this value are considered duplicates. + * Range: 0.0 (no similarity) to 1.0 (identical) + * Default: 0.85 (85% similar) + */ +const TODO_SIMILARITY_THRESHOLD = 0.85; + /** * Debounce interval for event emissions (milliseconds). * Prevents UI jitter from rapid consecutive updates. @@ -132,6 +147,24 @@ const EVENT_DEBOUNCE_MS = 50; */ const MAX_COMPLETION_PHRASE_ENTRIES = 50; +/** + * Common/generic completion phrases that may cause false positives. + * These phrases are likely to appear in Claude's natural output, + * making them unreliable as completion signals. + * + * P1-002: Configurable false positive prevention + */ +const COMMON_COMPLETION_PHRASES = new Set([ + 'DONE', 'COMPLETE', 'FINISHED', 'OK', 'YES', 'TRUE', 'SUCCESS', + 'READY', 'COMPLETED', 'PASSED', 'END', 'STOP', 'EXIT', +]); + +/** + * Minimum recommended phrase length for completion detection. + * Shorter phrases are more likely to cause false positives. + */ +const MIN_RECOMMENDED_PHRASE_LENGTH = 6; + /** * Maximum line buffer size to prevent unbounded growth from long lines. */ @@ -150,8 +183,20 @@ const MAX_LINE_BUFFER_SIZE = 64 * 1024; * - Numbers: TASK_123 * - Underscores: ALL_TASKS_DONE * - Hyphens: TESTS-PASS, TIME-COMPLETE + * + * Now also tolerates: + * - Whitespace/newlines inside tags: COMPLETE + * - Case variations in tag names: , */ -const PROMISE_PATTERN = /([^<]+)<\/promise>/; +const PROMISE_PATTERN = /\s*([^<]+?)\s*<\/promise>/i; + +/** + * Pattern for detecting partial/incomplete promise tags at end of buffer. + * Used for cross-chunk promise detection when tags are split across PTY writes. + * Captures: + * - Group 1: Partial opening tag content after (may be incomplete) + */ +const PROMISE_PARTIAL_PATTERN = /\s*([^<]*)$/i; // ---------- Todo Item Patterns ---------- // Claude Code outputs todos in multiple formats; we detect all of them @@ -390,6 +435,16 @@ export interface RalphTrackerEvents { circuitBreakerUpdate: (status: CircuitBreakerStatus) => void; /** Emitted when dual-condition exit gate is met (completion indicators >= 2 AND EXIT_SIGNAL: true) */ exitGateMet: (data: { completionIndicators: number; exitSignal: boolean }) => void; + /** Emitted when iteration count hasn't changed for an extended period (stall warning) */ + iterationStallWarning: (data: { iteration: number; stallDurationMs: number }) => void; + /** Emitted when iteration count hasn't changed for critical period (stall critical) */ + iterationStallCritical: (data: { iteration: number; stallDurationMs: number }) => void; + /** Emitted when a common/risky completion phrase is detected (P1-002) */ + phraseValidationWarning: (data: { + phrase: string; + reason: 'common' | 'short' | 'numeric'; + suggestedPhrase: string; + }) => void; } /** @@ -467,6 +522,16 @@ export class RalphTracker extends EventEmitter { /** Maps task numbers from "✔ Task #N" format to their content for status updates */ private _taskNumberToContent: Map = new Map(); + /** + * Buffer for partial promise tags split across PTY chunks. + * Holds content after '' when closing tag hasn't arrived yet. + * Max 256 chars to prevent unbounded growth from malformed tags. + */ + private _partialPromiseBuffer: string = ''; + + /** Maximum size of partial promise buffer */ + private static readonly MAX_PARTIAL_PROMISE_SIZE = 256; + // ========== RALPH_STATUS Block State ========== /** Circuit breaker state tracking */ @@ -535,6 +600,43 @@ export class RalphTracker extends EventEmitter { /** Last checkpoint iteration */ private _lastCheckpointIteration: number = 0; + // ========== Iteration Stall Detection ========== + + /** Timestamp when iteration count last changed */ + private _lastIterationChangeTime: number = 0; + + /** Last observed iteration count for stall detection */ + private _lastObservedIteration: number = 0; + + /** Timer for iteration stall detection */ + private _iterationStallTimer: NodeJS.Timeout | null = null; + + /** Iteration stall warning threshold (ms) - default 10 minutes */ + private _iterationStallWarningMs: number = 10 * 60 * 1000; + + /** Iteration stall critical threshold (ms) - default 20 minutes */ + private _iterationStallCriticalMs: number = 20 * 60 * 1000; + + /** Whether stall warning has been emitted */ + private _iterationStallWarned: boolean = false; + + /** Alternate completion phrases (P1-003: multi-phrase support) */ + private _alternateCompletionPhrases: string[] = []; + + // ========== P1-009: Progress Estimation ========== + + /** History of todo completion times (ms) for averaging */ + private _completionTimes: number[] = []; + + /** Maximum number of completion times to track */ + private static readonly MAX_COMPLETION_TIMES = 50; + + /** Timestamp when todos started being tracked for this session */ + private _todosStartedAt: number = 0; + + /** Map of todo ID to timestamp when it started (for duration tracking) */ + private _todoStartTimes: Map = new Map(); + /** * Creates a new RalphTracker instance. * Starts in disabled state until Ralph patterns are detected. @@ -543,6 +645,60 @@ export class RalphTracker extends EventEmitter { super(); this._loopState = createInitialRalphTrackerState(); this._circuitBreaker = createInitialCircuitBreakerStatus(); + this._lastIterationChangeTime = Date.now(); + } + + /** + * Add an alternate completion phrase (P1-003: multi-phrase support). + * Multiple phrases can trigger completion (useful for complex workflows). + * @param phrase - Additional phrase that can trigger completion + */ + addAlternateCompletionPhrase(phrase: string): void { + if (!this._alternateCompletionPhrases.includes(phrase)) { + this._alternateCompletionPhrases.push(phrase); + this._loopState.alternateCompletionPhrases = [...this._alternateCompletionPhrases]; + this.emit('loopUpdate', this.loopState); + } + } + + /** + * Remove an alternate completion phrase. + * @param phrase - Phrase to remove + */ + removeAlternateCompletionPhrase(phrase: string): void { + const index = this._alternateCompletionPhrases.indexOf(phrase); + if (index !== -1) { + this._alternateCompletionPhrases.splice(index, 1); + this._loopState.alternateCompletionPhrases = [...this._alternateCompletionPhrases]; + this.emit('loopUpdate', this.loopState); + } + } + + /** + * Check if a phrase matches any valid completion phrase (primary or alternate). + * @param phrase - Phrase to check + * @returns True if phrase matches any valid completion phrase + */ + isValidCompletionPhrase(phrase: string): boolean { + return this.findMatchingCompletionPhrase(phrase) !== null; + } + + /** + * Find which completion phrase (primary or alternate) matches the given phrase. + * @param phrase - Phrase to check + * @returns The matched canonical phrase, or null if no match + */ + private findMatchingCompletionPhrase(phrase: string): string | null { + const primary = this._loopState.completionPhrase; + if (primary && this.isFuzzyPhraseMatch(phrase, primary)) { + return primary; + } + for (const alt of this._alternateCompletionPhrases) { + if (this.isFuzzyPhraseMatch(phrase, alt)) { + return alt; + } + } + return null; } /** @@ -738,6 +894,7 @@ export class RalphTracker extends EventEmitter { this._completionPhraseCount.clear(); this._taskNumberToContent.clear(); this._lineBuffer = ''; + this._partialPromiseBuffer = ''; // Reset RALPH_STATUS block state this._statusBlockBuffer = []; this._inStatusBlock = false; @@ -868,14 +1025,208 @@ export class RalphTracker extends EventEmitter { * Get a copy of the current loop state. * @returns Shallow copy of loop state (safe to modify) */ + // ========== Iteration Stall Detection Methods ========== + + /** + * Start iteration stall detection timer. + * Should be called when the loop becomes active. + */ + startIterationStallDetection(): void { + this.stopIterationStallDetection(); + this._lastIterationChangeTime = Date.now(); + this._iterationStallWarned = false; + + // Check every minute + this._iterationStallTimer = setInterval(() => { + this.checkIterationStall(); + }, 60 * 1000); + } + + /** + * Stop iteration stall detection timer. + */ + stopIterationStallDetection(): void { + if (this._iterationStallTimer) { + clearInterval(this._iterationStallTimer); + this._iterationStallTimer = null; + } + } + + /** + * Check for iteration stall and emit appropriate events. + */ + private checkIterationStall(): void { + if (!this._loopState.active) return; + + const stallDurationMs = Date.now() - this._lastIterationChangeTime; + + // Critical stall (longer duration) + if (stallDurationMs >= this._iterationStallCriticalMs) { + this.emit('iterationStallCritical', { + iteration: this._loopState.cycleCount, + stallDurationMs, + }); + return; + } + + // Warning stall + if (stallDurationMs >= this._iterationStallWarningMs && !this._iterationStallWarned) { + this._iterationStallWarned = true; + this.emit('iterationStallWarning', { + iteration: this._loopState.cycleCount, + stallDurationMs, + }); + } + } + + /** + * Get iteration stall metrics for monitoring. + */ + getIterationStallMetrics(): { + lastIterationChangeTime: number; + stallDurationMs: number; + warningThresholdMs: number; + criticalThresholdMs: number; + isWarned: boolean; + currentIteration: number; + } { + return { + lastIterationChangeTime: this._lastIterationChangeTime, + stallDurationMs: Date.now() - this._lastIterationChangeTime, + warningThresholdMs: this._iterationStallWarningMs, + criticalThresholdMs: this._iterationStallCriticalMs, + isWarned: this._iterationStallWarned, + currentIteration: this._loopState.cycleCount, + }; + } + + /** + * Configure iteration stall thresholds. + * @param warningMs - Warning threshold in milliseconds + * @param criticalMs - Critical threshold in milliseconds + */ + configureIterationStallThresholds(warningMs: number, criticalMs: number): void { + this._iterationStallWarningMs = warningMs; + this._iterationStallCriticalMs = criticalMs; + } + get loopState(): RalphTrackerState { return { ...this._loopState, planVersion: this._planVersion, planHistoryLength: this._planHistory.length, + completionConfidence: this._lastCompletionConfidence, }; } + /** Last calculated completion confidence */ + private _lastCompletionConfidence: CompletionConfidence | undefined; + + /** Confidence threshold for triggering completion (0-100) */ + private static readonly COMPLETION_CONFIDENCE_THRESHOLD = 70; + + /** + * Calculate confidence score for a potential completion signal. + * + * Scoring weights: + * - Promise tag with proper format: +30 + * - Matches expected phrase: +25 + * - All todos complete: +20 + * - EXIT_SIGNAL: true: +15 + * - Multiple completion indicators (>=2): +10 + * - Context appropriate (not in prompt/explanation): +10 + * - Loop was explicitly active: +10 + * + * @param phrase - The detected phrase to evaluate + * @param context - Optional surrounding context for the phrase + * @returns CompletionConfidence assessment + */ + calculateCompletionConfidence(phrase: string, context?: string): CompletionConfidence { + let score = 0; + const signals = { + hasPromiseTag: false, + matchesExpected: false, + allTodosComplete: false, + hasExitSignal: false, + multipleIndicators: false, + contextAppropriate: true, // Default to true, deduct if inappropriate + }; + + // Check for promise tag format (adds 30 points) + if (context && PROMISE_PATTERN.test(context)) { + signals.hasPromiseTag = true; + score += 30; + } + + // Check if phrase matches expected completion phrase (adds 25 points) + const expectedPhrase = this._loopState.completionPhrase; + if (expectedPhrase) { + const matchedPhrase = this.findMatchingCompletionPhrase(phrase); + if (matchedPhrase) { + signals.matchesExpected = true; + score += 25; + } + } + + // Check if all todos are complete (adds 20 points) + const todoArray = Array.from(this._todos.values()); + if (todoArray.length > 0 && todoArray.every(t => t.status === 'completed')) { + signals.allTodosComplete = true; + score += 20; + } + + // Check for EXIT_SIGNAL from RALPH_STATUS block (adds 15 points) + if (this._lastStatusBlock?.exitSignal === true) { + signals.hasExitSignal = true; + score += 15; + } + + // Check for multiple completion indicators (adds 10 points) + if (this._completionIndicators >= 2) { + signals.multipleIndicators = true; + score += 10; + } + + // Check context appropriateness (deduct if inappropriate) + if (context) { + const lowerContext = context.toLowerCase(); + // Deduct points if phrase appears in prompt-like context + if (lowerContext.includes('output:') || + lowerContext.includes('completion phrase') || + lowerContext.includes('output exactly') || + lowerContext.includes('when done')) { + signals.contextAppropriate = false; + score -= 20; + } else { + score += 10; + } + } + + // Bonus for active loop state (adds 10 points) + if (this._loopState.active) { + score += 10; + } + + // Bonus for 2nd+ occurrence (adds 15 points) + const count = this._completionPhraseCount.get(phrase) || 0; + if (count >= 2) { + score += 15; + } + + // Clamp score to 0-100 + score = Math.max(0, Math.min(100, score)); + + const confidence: CompletionConfidence = { + score, + isConfident: score >= RalphTracker.COMPLETION_CONFIDENCE_THRESHOLD, + signals, + calculatedAt: Date.now(), + }; + + this._lastCompletionConfidence = confidence; + return confidence; + } + /** * Get all tracked todo items as an array. * @returns Array of todo items (copy, safe to modify) @@ -1163,13 +1514,42 @@ export class RalphTracker extends EventEmitter { /** * Check for multi-line patterns that might span line boundaries. * Completion phrases can be split across PTY chunks. + * + * Handles cross-chunk promise tags by: + * 1. Checking combined buffer + new data for complete tags + * 2. Detecting partial tags at end of chunk and buffering + * 3. Clearing buffer when complete tag found or buffer gets stale + * * @param data - The full data chunk (may contain multiple lines) */ private checkMultiLinePatterns(data: string): void { - // Completion phrase can span lines, so check the whole chunk - const promiseMatch = data.match(PROMISE_PATTERN); + // If we have a partial promise buffer, prepend it to the new data + const combinedData = this._partialPromiseBuffer + data; + + // Try to find a complete promise tag in combined data + const promiseMatch = combinedData.match(PROMISE_PATTERN); if (promiseMatch) { - this.handleCompletionPhrase(promiseMatch[1]); + // Found complete tag - extract phrase and clear buffer + const phrase = promiseMatch[1].trim(); + this._partialPromiseBuffer = ''; + this.handleCompletionPhrase(phrase); + return; + } + + // Check for partial promise tag at end of combined data + const partialMatch = combinedData.match(PROMISE_PARTIAL_PATTERN); + if (partialMatch) { + // Buffer the partial content (with size limit) + const partialContent = partialMatch[0]; + if (partialContent.length <= RalphTracker.MAX_PARTIAL_PROMISE_SIZE) { + this._partialPromiseBuffer = partialContent; + } else { + // Partial is too long, likely malformed - discard + this._partialPromiseBuffer = ''; + } + } else { + // No partial tag - clear buffer + this._partialPromiseBuffer = ''; } } @@ -1267,9 +1647,10 @@ export class RalphTracker extends EventEmitter { /** * Handle a detected completion phrase * - * Uses occurrence-based detection to distinguish prompt from actual completion: + * Uses occurrence-based detection combined with confidence scoring + * to distinguish prompt from actual completion: * - 1st occurrence: Store as expected phrase (likely in prompt) - * - 2nd occurrence: Emit completionDetected (actual completion) + * - 2nd occurrence OR high confidence: Emit completionDetected (actual completion) * - If loop already active: Emit immediately (explicit loop start) */ private handleCompletionPhrase(phrase: string): void { @@ -1297,9 +1678,42 @@ export class RalphTracker extends EventEmitter { if (!this._loopState.completionPhrase) { this._loopState.completionPhrase = phrase; this._loopState.lastActivity = Date.now(); + + // P1-002: Validate phrase and emit warning if risky + this.validateCompletionPhrase(phrase); + this.emit('loopUpdate', this.loopState); } + // Check for fuzzy match with primary phrase or any alternate phrase (P1-003) + // This handles minor variations like whitespace, case, underscores vs hyphens + const matchedPhrase = this.findMatchingCompletionPhrase(phrase); + + if (matchedPhrase) { + // Use the matched phrase (canonical) for tracking + const canonicalCount = (this._completionPhraseCount.get(matchedPhrase) || 0); + // If this is a match of an expected phrase, treat as if we saw it + if (canonicalCount >= 1 || this._loopState.active) { + // Mark as completion + this._loopState.active = false; + this._loopState.lastActivity = Date.now(); + // Mark all todos as complete + let updated = false; + for (const todo of this._todos.values()) { + if (todo.status !== 'completed') { + todo.status = 'completed'; + updated = true; + } + } + if (updated) { + this.emit('todoUpdate', this.todos); + } + this.emit('completionDetected', matchedPhrase); + this.emit('loopUpdate', this.loopState); + return; + } + } + // Emit completion if loop is active OR this is 2nd+ occurrence if (this._loopState.active || count >= 2) { // Mark all todos as complete when completion phrase is detected @@ -1321,6 +1735,76 @@ export class RalphTracker extends EventEmitter { } } + /** + * Check if two phrases match with fuzzy tolerance. + * Handles variations in: + * - Case (COMPLETE vs Complete) + * - Whitespace (TASK_DONE vs TASK DONE) + * - Separators (TASK_DONE vs TASK-DONE) + * - Minor typos with Levenshtein distance (COMPLET vs COMPLETE) + * + * @param phrase1 - First phrase to compare + * @param phrase2 - Second phrase to compare + * @param maxDistance - Maximum edit distance for fuzzy match (default: 2) + * @returns True if phrases are fuzzy-equal + */ + private isFuzzyPhraseMatch(phrase1: string, phrase2: string, maxDistance = 2): boolean { + return fuzzyPhraseMatch(phrase1, phrase2, maxDistance); + } + + /** + * Validate a completion phrase and emit warnings if it's risky. + * + * P1-002: Configurable false positive prevention + * + * Checks for: + * - Common/generic phrases (DONE, COMPLETE, etc.) + * - Short phrases (< MIN_RECOMMENDED_PHRASE_LENGTH) + * - Numeric-only phrases + * + * @param phrase - The completion phrase to validate + * @fires phraseValidationWarning - When a risky phrase is detected + */ + private validateCompletionPhrase(phrase: string): void { + const normalized = phrase.toUpperCase().replace(/[\s_\-\.]+/g, ''); + + // Generate a suggested unique phrase + const uniqueSuffix = Date.now().toString(36).slice(-4).toUpperCase(); + const suggestedPhrase = `${phrase}_${uniqueSuffix}`; + + // Check for common phrases + if (COMMON_COMPLETION_PHRASES.has(normalized)) { + console.warn(`[RalphTracker] Warning: Completion phrase "${phrase}" is very common and may cause false positives. Consider using: "${suggestedPhrase}"`); + this.emit('phraseValidationWarning', { + phrase, + reason: 'common', + suggestedPhrase, + }); + return; + } + + // Check for short phrases + if (normalized.length < MIN_RECOMMENDED_PHRASE_LENGTH) { + console.warn(`[RalphTracker] Warning: Completion phrase "${phrase}" is too short (${normalized.length} chars). Consider using: "${suggestedPhrase}"`); + this.emit('phraseValidationWarning', { + phrase, + reason: 'short', + suggestedPhrase, + }); + return; + } + + // Check for numeric-only phrases + if (/^\d+$/.test(normalized)) { + console.warn(`[RalphTracker] Warning: Completion phrase "${phrase}" is numeric-only and may cause false positives. Consider using: "${suggestedPhrase}"`); + this.emit('phraseValidationWarning', { + phrase, + reason: 'numeric', + suggestedPhrase, + }); + } + } + /** * Activate the loop if not already active. * @@ -1386,6 +1870,29 @@ export class RalphTracker extends EventEmitter { if (!isNaN(currentIter)) { this.activateLoopIfNeeded(); + // Track iteration changes for stall detection + if (currentIter !== this._lastObservedIteration) { + this._lastIterationChangeTime = Date.now(); + this._lastObservedIteration = currentIter; + this._iterationStallWarned = false; // Reset warning on iteration change + + // P1-004: Reset circuit breaker on successful iteration progress + // If we're making progress, the loop is healthy + if (this._circuitBreaker.state === 'HALF_OPEN' || + this._circuitBreaker.consecutiveNoProgress > 0 || + this._circuitBreaker.consecutiveSameError > 0 || + this._circuitBreaker.consecutiveTestsFailure > 0) { + this._circuitBreaker.consecutiveNoProgress = 0; + this._circuitBreaker.consecutiveSameError = 0; + this._circuitBreaker.lastProgressIteration = currentIter; + if (this._circuitBreaker.state === 'HALF_OPEN') { + this._circuitBreaker.state = 'CLOSED'; + this._circuitBreaker.reason = 'Iteration progress detected'; + this._circuitBreaker.reasonCode = 'progress_detected'; + this.emit('circuitBreakerUpdate', { ...this._circuitBreaker }); + } + } + } this._loopState.cycleCount = currentIter; if (maxIter !== null && !isNaN(maxIter)) { this._loopState.maxIterations = maxIter; @@ -1601,8 +2108,12 @@ export class RalphTracker extends EventEmitter { /** * Parse priority from todo content. - * Looks for patterns like: "P0:", "P1:", "P2:", "(P0)", "(P1)", "(P2)", - * "Critical:", "Blocker:", "High Priority:", etc. + * P1-008: Enhanced keyword-based priority inference. + * + * Priority levels: + * - P0 (Critical): Explicit P0, "critical", "blocker", "urgent", "security", "crash", "broken" + * - P1 (High): Explicit P1, "important", "high priority", "bug", "fix", "error", "fail" + * - P2 (Medium): Explicit P2, "nice to have", "low priority", "refactor", "cleanup", "improve" * * @param content - Todo content text * @returns Parsed priority level or null @@ -1610,15 +2121,71 @@ export class RalphTracker extends EventEmitter { private parsePriority(content: string): RalphTodoPriority { const upper = content.toUpperCase(); - // Direct P0/P1/P2 patterns - if (/\bP0\b|^\(P0\)|:?\s*P0\s*:|\bCRITICAL\b|\bBLOCKER\b/.test(upper)) { - return 'P0'; + // P0 patterns - Critical issues + const p0Patterns = [ + /\bP0\b|\(P0\)|:?\s*P0\s*:/, // Explicit P0 + /\bCRITICAL\b/, // Critical keyword + /\bBLOCKER\b/, // Blocker + /\bURGENT\b/, // Urgent + /\bSECURITY\b/, // Security issues + /\bCRASH(?:ES|ING)?\b/, // Crash, crashes, crashing + /\bBROKEN\b/, // Broken + /\bDATA\s*LOSS\b/, // Data loss + /\bPRODUCTION\s*(?:DOWN|ISSUE|BUG)\b/, // Production issues + /\bHOTFIX\b/, // Hotfix + /\bSEVERITY\s*1\b/, // Severity 1 + ]; + + // P1 patterns - High priority issues + const p1Patterns = [ + /\bP1\b|\(P1\)|:?\s*P1\s*:/, // Explicit P1 + /\bHIGH\s*PRIORITY\b/, // High priority + /\bIMPORTANT\b/, // Important + /\bBUG\b/, // Bug + /\bFIX\b/, // Fix (as task type) + /\bERROR\b/, // Error + /\bFAIL(?:S|ED|ING|URE)?\b/, // Fail variants + /\bREGRESSION\b/, // Regression + /\bMUST\s*(?:HAVE|FIX|DO)\b/, // Must have/fix/do + /\bSEVERITY\s*2\b/, // Severity 2 + /\bREQUIRED\b/, // Required + ]; + + // P2 patterns - Lower priority + const p2Patterns = [ + /\bP2\b|\(P2\)|:?\s*P2\s*:/, // Explicit P2 + /\bNICE\s*TO\s*HAVE\b/, // Nice to have + /\bLOW\s*PRIORITY\b/, // Low priority + /\bREFACTOR\b/, // Refactor + /\bCLEANUP\b/, // Cleanup + /\bIMPROVE(?:MENT)?\b/, // Improve/Improvement + /\bOPTIMIZ(?:E|ATION)\b/, // Optimize/Optimization + /\bCONSIDER\b/, // Consider + /\bWOULD\s*BE\s*NICE\b/, // Would be nice + /\bENHANCE(?:MENT)?\b/, // Enhance/Enhancement + /\bTECH(?:NICAL)?\s*DEBT\b/, // Tech debt + /\bDOCUMENT(?:ATION)?\b/, // Documentation + ]; + + // Check P0 first (highest priority wins) + for (const pattern of p0Patterns) { + if (pattern.test(upper)) { + return 'P0'; + } } - if (/\bP1\b|^\(P1\)|:?\s*P1\s*:|\bHIGH\s*PRIORITY\b/.test(upper)) { - return 'P1'; + + // Check P1 + for (const pattern of p1Patterns) { + if (pattern.test(upper)) { + return 'P1'; + } } - if (/\bP2\b|^\(P2\)|:?\s*P2\s*:|\bNICE\s*TO\s*HAVE\b|\bLOW\s*PRIORITY\b/.test(upper)) { - return 'P2'; + + // Check P2 + for (const pattern of p2Patterns) { + if (pattern.test(upper)) { + return 'P2'; + } } return null; @@ -1652,17 +2219,70 @@ export class RalphTracker extends EventEmitter { // Parse priority from content const priority = this.parsePriority(cleanContent); + // P1-009: Estimate complexity for duration tracking + const estimatedComplexity = this.estimateComplexity(cleanContent); + // Generate a stable ID from normalized content const id = this.generateTodoId(cleanContent); const existing = this._todos.get(id); if (existing) { - // Update existing todo + // P1-009: Track status transitions for progress estimation + const wasCompleted = existing.status === 'completed'; + const isNowCompleted = status === 'completed'; + const wasInProgress = existing.status === 'in_progress'; + const isNowInProgress = status === 'in_progress'; + + // Update existing todo (exact match by ID) existing.status = status; existing.detectedAt = Date.now(); // Update priority if parsed (don't overwrite with null) if (priority) existing.priority = priority; + // Update complexity estimate if not already set + if (!existing.estimatedComplexity) { + existing.estimatedComplexity = estimatedComplexity; + } + + // P1-009: Track completion time + if (!wasCompleted && isNowCompleted) { + this.recordTodoCompletion(id); + } + // P1-009: Start tracking when status changes to in_progress + if (!wasInProgress && isNowInProgress) { + this.startTrackingTodo(id); + } } else { + // P1-007: Check for similar existing todo (deduplication) + const similar = this.findSimilarTodo(cleanContent); + if (similar) { + // P1-009: Track status transitions on similar todo + const wasCompleted = similar.status === 'completed'; + const isNowCompleted = status === 'completed'; + const wasInProgress = similar.status === 'in_progress'; + const isNowInProgress = status === 'in_progress'; + + // Update similar todo instead of creating duplicate + similar.status = status; + similar.detectedAt = Date.now(); + // Update priority if new content has priority and existing doesn't + if (priority && !similar.priority) { + similar.priority = priority; + } + // Keep the longer/more descriptive content + if (cleanContent.length > similar.content.length) { + similar.content = cleanContent; + } + + // P1-009: Track completion time + if (!wasCompleted && isNowCompleted) { + this.recordTodoCompletion(similar.id); + } + if (!wasInProgress && isNowInProgress) { + this.startTrackingTodo(similar.id); + } + return; + } + // Add new todo if (this._todos.size >= MAX_TODO_ITEMS) { // Remove oldest todo to make room @@ -1672,13 +2292,22 @@ export class RalphTracker extends EventEmitter { } } + const estimatedDurationMs = this.getEstimatedDuration(estimatedComplexity); + this._todos.set(id, { id, content: cleanContent, status, detectedAt: Date.now(), priority, + estimatedComplexity, + estimatedDurationMs, }); + + // P1-009: Start tracking if already in_progress + if (status === 'in_progress') { + this.startTrackingTodo(id); + } } } @@ -1705,6 +2334,327 @@ export class RalphTracker extends EventEmitter { .toLowerCase(); } + /** + * Calculate similarity between two strings. + * + * P1-007: Uses a hybrid approach combining: + * 1. Levenshtein-based similarity for edit-distance tolerance + * 2. Bigram (Dice coefficient) for reordering tolerance + * Returns the maximum of both methods. + * + * @param str1 - First string (will be normalized) + * @param str2 - Second string (will be normalized) + * @returns Similarity score from 0.0 (no similarity) to 1.0 (identical) + */ + private calculateSimilarity(str1: string, str2: string): number { + const norm1 = this.normalizeTodoContent(str1); + const norm2 = this.normalizeTodoContent(str2); + + // Identical after normalization + if (norm1 === norm2) return 1.0; + + // If either is empty, no similarity + if (!norm1 || !norm2) return 0.0; + + // Method 1: Levenshtein-based similarity (good for typos/minor edits) + const levenshteinSim = stringSimilarity(norm1, norm2); + + // Method 2: Bigram/Dice similarity (good for word reordering) + const bigramSim = this.calculateBigramSimilarity(norm1, norm2); + + // Return the higher of the two scores + return Math.max(levenshteinSim, bigramSim); + } + + /** + * Calculate bigram (Dice coefficient) similarity. + * Good for detecting near-duplicates with word reordering. + * + * @param norm1 - First normalized string + * @param norm2 - Second normalized string + * @returns Similarity score from 0.0 to 1.0 + */ + private calculateBigramSimilarity(norm1: string, norm2: string): number { + // Short strings: use simple character overlap + if (norm1.length < 3 || norm2.length < 3) { + const shorter = norm1.length <= norm2.length ? norm1 : norm2; + const longer = norm1.length > norm2.length ? norm1 : norm2; + return longer.includes(shorter) ? 0.9 : 0.0; + } + + // Extract bigrams (pairs of consecutive characters) + const getBigrams = (s: string): Set => { + const bigrams = new Set(); + for (let i = 0; i < s.length - 1; i++) { + bigrams.add(s.substring(i, i + 2)); + } + return bigrams; + }; + + const bigrams1 = getBigrams(norm1); + const bigrams2 = getBigrams(norm2); + + // Count intersection + let intersection = 0; + for (const bigram of bigrams1) { + if (bigrams2.has(bigram)) { + intersection++; + } + } + + // Dice coefficient: 2 * intersection / (total bigrams) + const totalBigrams = bigrams1.size + bigrams2.size; + if (totalBigrams === 0) return 0.0; + + return (2 * intersection) / totalBigrams; + } + + /** + * Find an existing todo that is similar to the given content. + * Returns the most similar todo if similarity >= threshold. + * + * Deduplication is intentionally conservative: + * - Short strings (< 30 chars): require 95% similarity (nearly identical) + * - Medium strings (30-60 chars): require 90% similarity + * - Longer strings: use default 85% threshold + * + * This prevents over-aggressive deduplication of brief, numbered items + * like "Task 1", "Task 2" while still catching true duplicates. + * + * @param content - New todo content to check against existing todos + * @returns Similar todo item if found, undefined otherwise + */ + private findSimilarTodo(content: string): RalphTodoItem | undefined { + const normalized = this.normalizeTodoContent(content); + + // Determine appropriate threshold based on string length + // Shorter strings need higher threshold to avoid false positives + let threshold: number; + if (normalized.length < 30) { + threshold = 0.95; // Very strict for short strings + } else if (normalized.length < 60) { + threshold = 0.90; // Strict for medium strings + } else { + threshold = TODO_SIMILARITY_THRESHOLD; // 0.85 for longer strings + } + + let bestMatch: RalphTodoItem | undefined; + let bestSimilarity = 0; + + for (const todo of this._todos.values()) { + const similarity = this.calculateSimilarity(content, todo.content); + if (similarity >= threshold && similarity > bestSimilarity) { + bestSimilarity = similarity; + bestMatch = todo; + } + } + + return bestMatch; + } + + // ========== P1-009: Progress Estimation Methods ========== + + /** + * Estimate complexity of a todo based on content keywords. + * Used for duration estimation. + * + * @param content - Todo content text + * @returns Complexity category + */ + private estimateComplexity(content: string): 'trivial' | 'simple' | 'moderate' | 'complex' { + const lower = content.toLowerCase(); + + // Trivial: Simple fixes, typos, documentation + const trivialPatterns = [ + /\btypo\b/, + /\bspelling\b/, + /\bcomment\b/, + /\bupdate\s+(?:version|readme)\b/, + /\brename\b/, + /\bformat(?:ting)?\b/, + ]; + + // Complex: Architecture, refactoring, security, testing + const complexPatterns = [ + /\barchitect(?:ure)?\b/, + /\brefactor\b/, + /\brewrite\b/, + /\bsecurity\b/, + /\bmigrat(?:e|ion)\b/, + /\btest(?:s|ing)?\b/, + /\bintegrat(?:e|ion)\b/, + /\bperformance\b/, + /\boptimiz(?:e|ation)\b/, + /\bmultiple\s+files?\b/, + ]; + + // Moderate: Bugs, features, enhancements + const moderatePatterns = [ + /\bbug\b/, + /\bfeature\b/, + /\benhance(?:ment)?\b/, + /\bimplement\b/, + /\badd\b/, + /\bfix\b/, + ]; + + for (const pattern of complexPatterns) { + if (pattern.test(lower)) return 'complex'; + } + + for (const trivialPattern of trivialPatterns) { + if (trivialPattern.test(lower)) return 'trivial'; + } + + for (const moderatePattern of moderatePatterns) { + if (moderatePattern.test(lower)) return 'moderate'; + } + + return 'simple'; + } + + /** + * Get estimated duration for a complexity level (ms). + * Based on historical patterns from similar tasks. + * + * @param complexity - Complexity category + * @returns Estimated duration in milliseconds + */ + private getEstimatedDuration(complexity: 'trivial' | 'simple' | 'moderate' | 'complex'): number { + // If we have historical data, use average adjusted by complexity + const avgTime = this.getAverageCompletionTime(); + if (avgTime !== null) { + const multipliers = { + trivial: 0.25, + simple: 0.5, + moderate: 1.0, + complex: 2.0, + }; + return Math.round(avgTime * multipliers[complexity]); + } + + // Default estimates (in ms) based on typical task durations + const defaults = { + trivial: 1 * 60 * 1000, // 1 minute + simple: 3 * 60 * 1000, // 3 minutes + moderate: 10 * 60 * 1000, // 10 minutes + complex: 30 * 60 * 1000, // 30 minutes + }; + return defaults[complexity]; + } + + /** + * Get average completion time from historical data. + * @returns Average time in ms, or null if no data + */ + private getAverageCompletionTime(): number | null { + if (this._completionTimes.length === 0) return null; + const sum = this._completionTimes.reduce((a, b) => a + b, 0); + return Math.round(sum / this._completionTimes.length); + } + + /** + * Record a todo completion for progress tracking. + * @param todoId - ID of the completed todo + */ + private recordTodoCompletion(todoId: string): void { + const startTime = this._todoStartTimes.get(todoId); + if (startTime) { + const duration = Date.now() - startTime; + this._completionTimes.push(duration); + + // Keep only recent completion times + while (this._completionTimes.length > RalphTracker.MAX_COMPLETION_TIMES) { + this._completionTimes.shift(); + } + + this._todoStartTimes.delete(todoId); + } + } + + /** + * Start tracking a todo for duration estimation. + * @param todoId - ID of the todo being started + */ + private startTrackingTodo(todoId: string): void { + if (!this._todoStartTimes.has(todoId)) { + this._todoStartTimes.set(todoId, Date.now()); + } + + // Initialize session tracking if needed + if (this._todosStartedAt === 0) { + this._todosStartedAt = Date.now(); + } + } + + /** + * Get progress estimation for the todo list. + * P1-009: Provides completion percentage, estimated remaining time, + * and projected completion timestamp. + * + * @returns Progress estimation object + */ + public getTodoProgress(): RalphTodoProgress { + const todos = Array.from(this._todos.values()); + const total = todos.length; + const completed = todos.filter(t => t.status === 'completed').length; + const inProgress = todos.filter(t => t.status === 'in_progress').length; + const pending = todos.filter(t => t.status === 'pending').length; + + const percentComplete = total > 0 ? Math.round((completed / total) * 100) : 0; + + // Calculate estimated remaining time + let estimatedRemainingMs: number | null = null; + let avgCompletionTimeMs: number | null = null; + let projectedCompletionAt: number | null = null; + + avgCompletionTimeMs = this.getAverageCompletionTime(); + + if (total > 0 && completed > 0) { + // Method 1: Use historical average if available + if (avgCompletionTimeMs !== null) { + const remaining = total - completed; + estimatedRemainingMs = remaining * avgCompletionTimeMs; + } else { + // Method 2: Calculate based on elapsed time and progress + const elapsed = Date.now() - this._todosStartedAt; + if (elapsed > 0 && completed > 0) { + const timePerTodo = elapsed / completed; + avgCompletionTimeMs = Math.round(timePerTodo); + const remaining = total - completed; + estimatedRemainingMs = Math.round(remaining * timePerTodo); + } + } + + // Calculate projected completion timestamp + if (estimatedRemainingMs !== null) { + projectedCompletionAt = Date.now() + estimatedRemainingMs; + } + } else if (total > 0 && completed === 0) { + // No completions yet - use complexity-based estimates + let totalEstimate = 0; + for (const todo of todos) { + if (todo.status !== 'completed') { + const complexity = todo.estimatedComplexity || this.estimateComplexity(todo.content); + totalEstimate += this.getEstimatedDuration(complexity); + } + } + estimatedRemainingMs = totalEstimate; + projectedCompletionAt = Date.now() + totalEstimate; + } + + return { + total, + completed, + inProgress, + pending, + percentComplete, + estimatedRemainingMs, + avgCompletionTimeMs, + projectedCompletionAt, + }; + } + /** * Generate a stable ID from todo content using djb2 hash. * @@ -1714,20 +2664,21 @@ export class RalphTracker extends EventEmitter { * @param content - Todo content text * @returns Stable ID in format `todo-{hash}` (base36 encoded) */ + /** + * Generate a stable ID from todo content using content hashing. + * + * P1-007: Uses centralized todoContentHash utility for consistency + * with deduplication logic. + * + * @param content - Todo content text + * @returns Unique ID string prefixed with "todo-" + */ private generateTodoId(content: string): string { if (!content) return 'todo-empty'; - // Normalize content for consistent hashing - const normalized = this.normalizeTodoContent(content); - if (!normalized) return 'todo-empty'; - - // djb2 hash algorithm - good distribution for strings - let hash = 5381; - for (let i = 0; i < normalized.length; i++) { - hash = ((hash << 5) + hash) ^ normalized.charCodeAt(i); - hash = hash | 0; // Convert to 32-bit integer - } - return `todo-${Math.abs(hash).toString(36)}`; + // Use centralized hashing utility + const hash = todoContentHash(content); + return `todo-${hash}`; } /** @@ -1888,6 +2839,7 @@ export class RalphTracker extends EventEmitter { this._todos.clear(); this._taskNumberToContent.clear(); this._lineBuffer = ''; + this._partialPromiseBuffer = ''; this._completionPhraseCount.clear(); // Clear RALPH_STATUS block and circuit breaker state this._statusBlockBuffer = []; @@ -2000,6 +2952,8 @@ export class RalphTracker extends EventEmitter { /** * Parse buffered RALPH_STATUS block lines into structured data. * + * P1-004: Enhanced with schema validation and error recovery + * * @param lines - Array of lines between block markers * @fires statusBlockDetected - When parsing succeeds */ @@ -2007,75 +2961,129 @@ export class RalphTracker extends EventEmitter { const block: Partial = { parsedAt: Date.now(), }; + const parseErrors: string[] = []; + const unknownFields: string[] = []; for (const line of lines) { - // STATUS field - const statusMatch = line.match(RALPH_STATUS_FIELD_PATTERN); + const trimmedLine = line.trim(); + if (!trimmedLine) continue; + + // Track whether this line matched any known field + let matched = false; + + // STATUS field (required) + const statusMatch = trimmedLine.match(RALPH_STATUS_FIELD_PATTERN); if (statusMatch) { - block.status = statusMatch[1].toUpperCase() as RalphStatusValue; - continue; + const value = statusMatch[1].toUpperCase(); + if (['IN_PROGRESS', 'COMPLETE', 'BLOCKED'].includes(value)) { + block.status = value as RalphStatusValue; + } else { + parseErrors.push(`Invalid STATUS value: "${value}". Expected: IN_PROGRESS, COMPLETE, or BLOCKED`); + } + matched = true; } // TASKS_COMPLETED_THIS_LOOP field - const tasksMatch = line.match(RALPH_TASKS_COMPLETED_PATTERN); + const tasksMatch = trimmedLine.match(RALPH_TASKS_COMPLETED_PATTERN); if (tasksMatch) { - block.tasksCompletedThisLoop = parseInt(tasksMatch[1], 10); - continue; + const value = parseInt(tasksMatch[1], 10); + if (!isNaN(value) && value >= 0) { + block.tasksCompletedThisLoop = value; + } else { + parseErrors.push(`Invalid TASKS_COMPLETED_THIS_LOOP value: "${tasksMatch[1]}". Expected: non-negative integer`); + } + matched = true; } // FILES_MODIFIED field - const filesMatch = line.match(RALPH_FILES_MODIFIED_PATTERN); + const filesMatch = trimmedLine.match(RALPH_FILES_MODIFIED_PATTERN); if (filesMatch) { - block.filesModified = parseInt(filesMatch[1], 10); - continue; + const value = parseInt(filesMatch[1], 10); + if (!isNaN(value) && value >= 0) { + block.filesModified = value; + } else { + parseErrors.push(`Invalid FILES_MODIFIED value: "${filesMatch[1]}". Expected: non-negative integer`); + } + matched = true; } // TESTS_STATUS field - const testsMatch = line.match(RALPH_TESTS_STATUS_PATTERN); + const testsMatch = trimmedLine.match(RALPH_TESTS_STATUS_PATTERN); if (testsMatch) { - block.testsStatus = testsMatch[1].toUpperCase() as RalphTestsStatus; - continue; + const value = testsMatch[1].toUpperCase(); + if (['PASSING', 'FAILING', 'NOT_RUN'].includes(value)) { + block.testsStatus = value as RalphTestsStatus; + } else { + parseErrors.push(`Invalid TESTS_STATUS value: "${value}". Expected: PASSING, FAILING, or NOT_RUN`); + } + matched = true; } // WORK_TYPE field - const workMatch = line.match(RALPH_WORK_TYPE_PATTERN); + const workMatch = trimmedLine.match(RALPH_WORK_TYPE_PATTERN); if (workMatch) { - block.workType = workMatch[1].toUpperCase() as RalphWorkType; - continue; + const value = workMatch[1].toUpperCase(); + if (['IMPLEMENTATION', 'TESTING', 'DOCUMENTATION', 'REFACTORING'].includes(value)) { + block.workType = value as RalphWorkType; + } else { + parseErrors.push(`Invalid WORK_TYPE value: "${value}". Expected: IMPLEMENTATION, TESTING, DOCUMENTATION, or REFACTORING`); + } + matched = true; } // EXIT_SIGNAL field - const exitMatch = line.match(RALPH_EXIT_SIGNAL_PATTERN); + const exitMatch = trimmedLine.match(RALPH_EXIT_SIGNAL_PATTERN); if (exitMatch) { block.exitSignal = exitMatch[1].toLowerCase() === 'true'; - continue; + matched = true; } // RECOMMENDATION field - const recMatch = line.match(RALPH_RECOMMENDATION_PATTERN); + const recMatch = trimmedLine.match(RALPH_RECOMMENDATION_PATTERN); if (recMatch) { block.recommendation = recMatch[1].trim(); - continue; + matched = true; + } + + // Track unknown fields for debugging (only if looks like a field) + if (!matched && trimmedLine.includes(':')) { + const fieldName = trimmedLine.split(':')[0].trim().toUpperCase(); + if (fieldName && !['#', '//'].some(c => fieldName.startsWith(c))) { + unknownFields.push(fieldName); + } } } - // Only process if we have at least the STATUS field - if (block.status !== undefined) { - // Fill in defaults for missing fields - const fullBlock: RalphStatusBlock = { - status: block.status, - tasksCompletedThisLoop: block.tasksCompletedThisLoop ?? 0, - filesModified: block.filesModified ?? 0, - testsStatus: block.testsStatus ?? 'NOT_RUN', - workType: block.workType ?? 'IMPLEMENTATION', - exitSignal: block.exitSignal ?? false, - recommendation: block.recommendation ?? '', - parsedAt: block.parsedAt!, - }; - - this._lastStatusBlock = fullBlock; - this.handleStatusBlock(fullBlock); + // Log parse errors if any + if (parseErrors.length > 0) { + console.warn(`[RalphTracker] RALPH_STATUS parse errors:\n - ${parseErrors.join('\n - ')}`); } + + // Log unknown fields if any + if (unknownFields.length > 0) { + console.warn(`[RalphTracker] RALPH_STATUS unknown fields: ${unknownFields.join(', ')}`); + } + + // Validate required field: STATUS + if (block.status === undefined) { + console.warn('[RalphTracker] RALPH_STATUS block missing required STATUS field, skipping'); + return; + } + + // Fill in defaults for missing optional fields + const fullBlock: RalphStatusBlock = { + status: block.status, + tasksCompletedThisLoop: block.tasksCompletedThisLoop ?? 0, + filesModified: block.filesModified ?? 0, + testsStatus: block.testsStatus ?? 'NOT_RUN', + workType: block.workType ?? 'IMPLEMENTATION', + exitSignal: block.exitSignal ?? false, + recommendation: block.recommendation ?? '', + parsedAt: block.parsedAt!, + }; + + this._lastStatusBlock = fullBlock; + this.handleStatusBlock(fullBlock); } /** diff --git a/src/respawn-controller.ts b/src/respawn-controller.ts index bc66d3fe..02c6bc1d 100644 --- a/src/respawn-controller.ts +++ b/src/respawn-controller.ts @@ -47,6 +47,14 @@ import { MAX_RESPAWN_BUFFER_SIZE, TRIM_RESPAWN_BUFFER_TO as RESPAWN_BUFFER_TRIM_SIZE, } from './config/buffer-limits.js'; +import type { + RespawnCycleMetrics, + RespawnAggregateMetrics, + RalphLoopHealthScore, + TimingHistory, + CycleOutcome, + HealthStatus, +} from './types.js'; // ========== Constants ========== @@ -137,6 +145,22 @@ export interface DetectionStatus { /** Next expected action */ nextAction: string; + + /** Stuck-state detection metrics */ + stuckState: { + /** How long the controller has been in the current state (ms) */ + currentStateDurationMs: number; + /** Warning threshold (ms) */ + warningThresholdMs: number; + /** Recovery threshold (ms) */ + recoveryThresholdMs: number; + /** Number of recovery attempts made */ + recoveryAttempts: number; + /** Maximum allowed recoveries */ + maxRecoveries: number; + /** Whether a warning has been emitted for current state */ + isWarned: boolean; + }; } // ========== Type Definitions ========== @@ -332,6 +356,109 @@ export interface RespawnConfig { * @default 30000 (30 seconds) */ aiPlanCheckCooldownMs: number; + + /** + * Enable stuck-state detection. + * Detects when the controller stays in the same state for too long. + * @default true + */ + stuckStateDetectionEnabled: boolean; + + /** + * Threshold for stuck-state warning in ms. + * If state doesn't change for this duration, emits a warning. + * @default 300000 (5 minutes) + */ + stuckStateWarningMs: number; + + /** + * Threshold for stuck-state recovery in ms. + * If state doesn't change for this duration, triggers recovery action. + * @default 600000 (10 minutes) + */ + stuckStateRecoveryMs: number; + + /** + * Maximum consecutive stuck-state recoveries before giving up. + * @default 3 + */ + maxStuckRecoveries: number; + + // ========== P2-001: Adaptive Timing ========== + + /** + * Enable adaptive timing based on historical patterns. + * Adjusts completion confirm timeout dynamically. + * @default true + */ + adaptiveTimingEnabled?: boolean; + + /** + * Minimum adaptive completion confirm timeout (ms). + * @default 5000 + */ + adaptiveMinConfirmMs?: number; + + /** + * Maximum adaptive completion confirm timeout (ms). + * @default 30000 + */ + adaptiveMaxConfirmMs?: number; + + // ========== P2-002: Skip-Clear Optimization ========== + + /** + * Skip /clear when context usage is low. + * @default true + */ + skipClearWhenLowContext?: boolean; + + /** + * Context threshold percentage below which to skip /clear. + * @default 30 + */ + skipClearThresholdPercent?: number; + + // ========== P2-004: Cycle Metrics ========== + + /** + * Enable tracking of respawn cycle metrics. + * @default true + */ + trackCycleMetrics?: boolean; + + // ========== P2-001: Confidence Scoring ========== + + /** + * Minimum confidence level required to trigger idle detection. + * Below this threshold, the controller waits for more signals. + * @default 65 + */ + minIdleConfidence?: number; + + /** + * Confidence weight for completion message detection. + * @default 40 + */ + confidenceWeightCompletion?: number; + + /** + * Confidence weight for output silence. + * @default 25 + */ + confidenceWeightSilence?: number; + + /** + * Confidence weight for token stability. + * @default 20 + */ + confidenceWeightTokens?: number; + + /** + * Confidence weight for working pattern absence. + * @default 15 + */ + confidenceWeightNoWorking?: number; } /** @@ -402,6 +529,12 @@ export interface RespawnEvents { error: (error: Error) => void; /** Debug log message */ log: (message: string) => void; + /** Stuck state warning emitted */ + stuckStateWarning: (state: RespawnState, durationMs: number) => void; + /** Stuck state recovery triggered */ + stuckStateRecovery: (state: RespawnState, durationMs: number, attempt: number) => void; + /** Respawn blocked by external signal */ + respawnBlocked: (data: { reason: string; details: string }) => void; } /** Default configuration values */ @@ -426,6 +559,25 @@ const DEFAULT_CONFIG: RespawnConfig = { aiPlanCheckMaxContext: 8000, // ~2k tokens (plan mode UI is compact) aiPlanCheckTimeoutMs: 60000, // 60 seconds (thinking can be slow) aiPlanCheckCooldownMs: 30000, // 30 seconds after NOT_PLAN_MODE + stuckStateDetectionEnabled: true, // detect stuck states + stuckStateWarningMs: 300000, // 5 minutes warning threshold + stuckStateRecoveryMs: 600000, // 10 minutes recovery threshold + maxStuckRecoveries: 3, // max recovery attempts + // P2-001: Adaptive timing + adaptiveTimingEnabled: true, // Use adaptive timing based on historical patterns + adaptiveMinConfirmMs: 5000, // Minimum 5 seconds + adaptiveMaxConfirmMs: 30000, // Maximum 30 seconds + // P2-002: Skip-clear optimization + skipClearWhenLowContext: true, // Skip /clear when token count is low + skipClearThresholdPercent: 30, // Skip if below 30% of max context + // P2-004: Cycle metrics + trackCycleMetrics: true, // Track and persist cycle metrics + // P2-001: Confidence scoring + minIdleConfidence: 65, // Minimum confidence to trigger idle (0-100) + confidenceWeightCompletion: 40, // Weight for completion message + confidenceWeightSilence: 25, // Weight for output silence + confidenceWeightTokens: 20, // Weight for token stability + confidenceWeightNoWorking: 15, // Weight for working pattern absence }; /** @@ -581,6 +733,60 @@ export class RespawnController extends EventEmitter { /** Recent action log entries (for UI display, max 20) */ private recentActions: ActionLogEntry[] = []; + // ========== Stuck-State Detection State ========== + + /** Timestamp when the current state was entered */ + private stateEnteredAt: number = 0; + + /** Timer for stuck-state detection */ + private stuckStateTimer: NodeJS.Timeout | null = null; + + /** Whether a stuck-state warning has been emitted for current state */ + private stuckStateWarned: boolean = false; + + /** Number of stuck-state recovery attempts */ + private stuckRecoveryCount: number = 0; + + // ========== P2-001: Adaptive Timing State ========== + + /** Historical timing data for adaptive adjustments */ + private timingHistory: TimingHistory = { + recentIdleDetectionMs: [], + recentCycleDurationMs: [], + adaptiveCompletionConfirmMs: 10000, // Start with default + sampleCount: 0, + maxSamples: 20, // Keep last 20 samples for rolling average + lastUpdatedAt: Date.now(), + }; + + // ========== P2-004: Cycle Metrics State ========== + + /** Current cycle being tracked */ + private currentCycleMetrics: Partial | null = null; + + /** Timestamp when idle detection started for current cycle */ + private idleDetectionStartTime: number = 0; + + /** Recent cycle metrics (rolling window for aggregate calculation) */ + private recentCycleMetrics: RespawnCycleMetrics[] = []; + + /** Maximum number of cycle metrics to keep in memory */ + private static readonly MAX_CYCLE_METRICS_IN_MEMORY = 100; + + /** Aggregate metrics across all tracked cycles */ + private aggregateMetrics: RespawnAggregateMetrics = { + totalCycles: 0, + successfulCycles: 0, + stuckRecoveryCycles: 0, + blockedCycles: 0, + errorCycles: 0, + avgCycleDurationMs: 0, + avgIdleDetectionMs: 0, + p90CycleDurationMs: 0, + successRate: 100, + lastUpdatedAt: Date.now(), + }; + // ========== Multi-Layer Detection State ========== /** Layer 1: Timestamp when completion message was detected */ @@ -786,17 +992,33 @@ export class RespawnController extends EventEmitter { const tokensStable = msSinceTokenChange >= this.config.completionConfirmMs; const workingPatternsAbsent = msSinceLastWorking >= RespawnController.MIN_WORKING_PATTERN_ABSENCE_MS; - // Calculate confidence level (0-100) + // Calculate confidence level (0-100) using configurable weights + // P2-001: Configurable confidence scoring // Hook signals are definitive (100% confidence) let confidence = 0; if (this.stopHookReceived || this.idlePromptReceived) { confidence = 100; } else { - if (completionMessageDetected) confidence += 40; - if (outputSilent) confidence += 25; - if (tokensStable) confidence += 20; - if (workingPatternsAbsent) confidence += 15; + // Use configured weights (with fallback to defaults) + const weightCompletion = this.config.confidenceWeightCompletion ?? 40; + const weightSilence = this.config.confidenceWeightSilence ?? 25; + const weightTokens = this.config.confidenceWeightTokens ?? 20; + const weightNoWorking = this.config.confidenceWeightNoWorking ?? 15; + + if (completionMessageDetected) confidence += weightCompletion; + if (outputSilent) confidence += weightSilence; + if (tokensStable) confidence += weightTokens; + if (workingPatternsAbsent) confidence += weightNoWorking; + + // Confidence decay: if no output for extended time, add bonus confidence + // This helps detect stuck states where Claude is truly idle but no completion message + const extendedSilenceBonus = Math.min(20, Math.floor(msSinceLastOutput / 30000) * 5); + if (msSinceLastOutput > 30000) { + confidence += extendedSilenceBonus; + } } + // Cap at 100 + confidence = Math.min(100, confidence); // Determine status text and what we're waiting for let statusText: string; @@ -923,6 +1145,7 @@ export class RespawnController extends EventEmitter { recentActions: this.recentActions.slice(0, 10), currentPhase, nextAction, + stuckState: this.getStuckStateMetrics(), }; } @@ -962,9 +1185,19 @@ export class RespawnController extends EventEmitter { const prevState = this._state; this._state = newState; + this.stateEnteredAt = Date.now(); + this.stuckStateWarned = false; // Reset warning for new state this.log(`State: ${prevState} → ${newState}`); this.logAction('state', `${prevState} → ${newState}`); this.emit('stateChanged', newState, prevState); + + // Reset stuck recovery count on successful state transition to normal states + if (newState === 'watching' && prevState !== 'stopped') { + this.stuckRecoveryCount = 0; + } + + // Start/restart stuck-state detection timer + this.startStuckStateTimer(); } /** @@ -1035,6 +1268,9 @@ export class RespawnController extends EventEmitter { if (this.config.autoAcceptPrompts) { this.startAutoAcceptTimer(); } + + // P2-001: Initialize idle detection start time + this.idleDetectionStartTime = Date.now(); } /** @@ -1283,8 +1519,24 @@ export class RespawnController extends EventEmitter { this.log('Update completed (ready indicator)'); this.emit('stepCompleted', 'update'); + // P2-004: Record step completion + this.recordCycleStep('update'); + if (this.config.sendClear) { - this.sendClear(); + // P2-002: Check if we should skip /clear + if (this.shouldSkipClear()) { + if (this.currentCycleMetrics) { + this.currentCycleMetrics.clearSkipped = true; + } + // Skip /clear, go directly to /init or complete + if (this.config.sendInit) { + this.sendInit(); + } else { + this.completeCycle(); + } + } else { + this.sendClear(); + } } else if (this.config.sendInit) { this.sendInit(); } else { @@ -1305,6 +1557,9 @@ export class RespawnController extends EventEmitter { this.logAction('step', '/clear completed'); this.emit('stepCompleted', 'clear'); + // P2-004: Record step completion + this.recordCycleStep('clear'); + if (this.config.sendInit) { this.sendInit(); } else { @@ -1322,6 +1577,9 @@ export class RespawnController extends EventEmitter { this.clearIdleTimer(); this.log('/init completed (ready indicator)'); + // P2-004: Record step completion + this.recordCycleStep('init'); + // If kickstart prompt is configured, monitor to see if /init triggered work if (this.config.kickstartPrompt) { this.startMonitoringInit(); @@ -1408,6 +1666,10 @@ export class RespawnController extends EventEmitter { this.clearIdleTimer(); this.log('Kickstart completed (ready indicator)'); this.emit('stepCompleted', 'kickstart'); + + // P2-004: Record step completion + this.recordCycleStep('kickstart'); + this.completeCycle(); } @@ -1457,10 +1719,162 @@ export class RespawnController extends EventEmitter { clearTimeout(this.hookConfirmTimer); this.hookConfirmTimer = null; } + if (this.stuckStateTimer) { + clearTimeout(this.stuckStateTimer); + this.stuckStateTimer = null; + } // Clear all tracked timers this.activeTimers.clear(); } + // ========== Stuck-State Detection Methods ========== + + /** + * Start or restart the stuck-state detection timer. + * Emits warning after stuckStateWarningMs, triggers recovery after stuckStateRecoveryMs. + */ + private startStuckStateTimer(): void { + if (!this.config.stuckStateDetectionEnabled) return; + if (this._state === 'stopped') return; + + // Clear existing timer + if (this.stuckStateTimer) { + clearTimeout(this.stuckStateTimer); + this.stuckStateTimer = null; + } + + // Check interval for stuck state + const checkIntervalMs = Math.min(this.config.stuckStateWarningMs, 60000); // Check every minute max + + this.stuckStateTimer = setInterval(() => { + this.checkStuckState(); + }, checkIntervalMs); + } + + /** + * Check if the controller is stuck in the current state. + * Emits warnings and triggers recovery actions as appropriate. + */ + private checkStuckState(): void { + if (this._state === 'stopped') return; + + const durationMs = Date.now() - this.stateEnteredAt; + + // Check for recovery threshold (more severe) + if (durationMs >= this.config.stuckStateRecoveryMs) { + if (this.stuckRecoveryCount < this.config.maxStuckRecoveries) { + this.stuckRecoveryCount++; + this.logAction('stuck', `Recovery attempt ${this.stuckRecoveryCount}/${this.config.maxStuckRecoveries}`); + this.log(`Stuck-state recovery triggered (state: ${this._state}, duration: ${Math.round(durationMs / 1000)}s, attempt: ${this.stuckRecoveryCount})`); + this.emit('stuckStateRecovery', this._state, durationMs, this.stuckRecoveryCount); + this.handleStuckStateRecovery(); + } else { + this.logAction('stuck', `Max recoveries (${this.config.maxStuckRecoveries}) reached - manual intervention needed`); + this.log(`Stuck-state: max recoveries reached, manual intervention needed`); + } + return; + } + + // Check for warning threshold + if (durationMs >= this.config.stuckStateWarningMs && !this.stuckStateWarned) { + this.stuckStateWarned = true; + this.logAction('stuck', `Warning: in state '${this._state}' for ${Math.round(durationMs / 1000)}s`); + this.log(`Stuck-state warning: state '${this._state}' for ${Math.round(durationMs / 1000)}s without progress`); + this.emit('stuckStateWarning', this._state, durationMs); + } + } + + /** + * Handle stuck-state recovery by resetting to a known good state. + * Uses escalating recovery strategies based on current state. + */ + private handleStuckStateRecovery(): void { + const currentState = this._state; + + // P2-004: Complete current cycle metrics with stuck_recovery outcome + if (this.currentCycleMetrics) { + this.completeCycleMetrics('stuck_recovery', `Stuck in state: ${currentState}`); + } + + // Cancel any running AI checks + if (this.aiChecker.status === 'checking') { + this.aiChecker.cancel(); + } + if (this.planChecker.status === 'checking') { + this.planChecker.cancel(); + } + + // Clear all timers and reset detection state + this.clearTimers(); + this.completionMessageTime = null; + this.promptDetected = false; + this.workingDetected = false; + this.resetHookState(); + + // Escalating recovery strategies + switch (currentState) { + case 'watching': + case 'confirming_idle': + case 'ai_checking': + // For detection states, try forcing idle confirmation + this.log('Recovery: forcing idle confirmation'); + this.onIdleConfirmed(`stuck-state recovery (was ${currentState})`); + break; + + case 'waiting_update': + case 'waiting_clear': + case 'waiting_init': + case 'waiting_kickstart': + case 'monitoring_init': + // For waiting states, skip to next step or complete cycle + this.log('Recovery: skipping stuck step'); + this.completeCycle(); + break; + + case 'sending_update': + case 'sending_clear': + case 'sending_init': + case 'sending_kickstart': + // For sending states, retry the send + this.log('Recovery: returning to watching state'); + this.setState('watching'); + this.startNoOutputTimer(); + this.startPreFilterTimer(); + if (this.config.autoAcceptPrompts) { + this.startAutoAcceptTimer(); + } + break; + + default: + // Fallback: reset to watching + this.log('Recovery: fallback to watching state'); + this.setState('watching'); + this.startNoOutputTimer(); + this.startPreFilterTimer(); + } + } + + /** + * Get stuck-state metrics for UI display. + */ + getStuckStateMetrics(): { + currentStateDurationMs: number; + warningThresholdMs: number; + recoveryThresholdMs: number; + recoveryAttempts: number; + maxRecoveries: number; + isWarned: boolean; + } { + return { + currentStateDurationMs: this._state !== 'stopped' ? Date.now() - this.stateEnteredAt : 0, + warningThresholdMs: this.config.stuckStateWarningMs, + recoveryThresholdMs: this.config.stuckStateRecoveryMs, + recoveryAttempts: this.stuckRecoveryCount, + maxRecoveries: this.config.maxStuckRecoveries, + isWarned: this.stuckStateWarned, + }; + } + // ========== Timer Tracking Methods ========== /** @@ -1676,6 +2090,13 @@ export class RespawnController extends EventEmitter { * @param reason - What triggered this attempt (for logging) */ private tryStartAiCheck(reason: string): void { + // P0-006: Check Session.isWorking first to skip expensive AI call if session reports working + if (this.session.isWorking) { + this.log(`Skipping AI check - Session reports isWorking=true (reason: ${reason})`); + this.logAction('detection', 'Skipped AI check: Session is working'); + return; + } + // If AI check is disabled or errored out, fall back to direct idle confirmation if (!this.config.aiIdleCheckEnabled || this.aiChecker.status === 'disabled') { this.log(`AI check unavailable (${this.aiChecker.status}), confirming idle directly via: ${reason}`); @@ -2266,18 +2687,21 @@ export class RespawnController extends EventEmitter { // ========== RALPH_STATUS Integration ========== // Check circuit breaker status - if OPEN, pause respawn - const circuitBreaker = this.session.ralphTracker.circuitBreakerStatus; - if (circuitBreaker.state === 'OPEN') { - this.log(`Respawn blocked - Circuit breaker OPEN: ${circuitBreaker.reason}`); - this.logAction('ralph', `Circuit breaker OPEN: ${circuitBreaker.reason}`); - this.emit('respawnBlocked', { reason: 'circuit_breaker_open', details: circuitBreaker.reason }); - this.setState('watching'); - // Don't restart timers - wait for manual reset or circuit breaker resolution - return; + const ralphTracker = this.session.ralphTracker; + if (ralphTracker) { + const circuitBreaker = ralphTracker.circuitBreakerStatus; + if (circuitBreaker.state === 'OPEN') { + this.log(`Respawn blocked - Circuit breaker OPEN: ${circuitBreaker.reason}`); + this.logAction('ralph', `Circuit breaker OPEN: ${circuitBreaker.reason}`); + this.emit('respawnBlocked', { reason: 'circuit_breaker_open', details: circuitBreaker.reason }); + this.setState('watching'); + // Don't restart timers - wait for manual reset or circuit breaker resolution + return; + } } // Check RALPH_STATUS EXIT_SIGNAL - if true, loop is complete - const statusBlock = this.session.ralphTracker.lastStatusBlock; + const statusBlock = ralphTracker?.lastStatusBlock; if (statusBlock?.exitSignal) { this.log(`Respawn paused - RALPH_STATUS EXIT_SIGNAL=true`); this.logAction('ralph', `Exit signal detected: ${statusBlock.recommendation || 'Task complete'}`); @@ -2315,11 +2739,41 @@ export class RespawnController extends EventEmitter { return; } + // P1-006: Session health check before respawn cycle + // Skip if session is in error state or not running + if (this.session.status === 'error') { + this.log('Skipping respawn cycle - session is in error state'); + this.logAction('health', 'Respawn skipped: Session error state'); + this.emit('respawnBlocked', { reason: 'session_error', details: 'Session is in error state' }); + this.setState('watching'); + return; + } + + if (this.session.status === 'stopped') { + this.log('Skipping respawn cycle - session is stopped'); + this.logAction('health', 'Respawn skipped: Session stopped'); + this.emit('respawnBlocked', { reason: 'session_stopped', details: 'Session is stopped' }); + this.setState('watching'); + return; + } + + // Check if session PTY is still alive (via PID) + if (!this.session.pid) { + this.log('Skipping respawn cycle - session PTY not running (no PID)'); + this.logAction('health', 'Respawn skipped: No PTY process'); + this.emit('respawnBlocked', { reason: 'no_pty', details: 'Session PTY process not running' }); + this.setState('watching'); + return; + } + // Start the respawn cycle this.cycleCount++; this.log(`Starting respawn cycle #${this.cycleCount}`); this.emit('respawnCycleStarted', this.cycleCount); + // P2-004: Start tracking cycle metrics + this.startCycleMetrics('idle_confirmed'); + this.sendUpdateDocs(); } @@ -2340,7 +2794,7 @@ export class RespawnController extends EventEmitter { this.stepTimer = null; // Use RALPH_STATUS RECOMMENDATION if available, otherwise fall back to config - const statusBlock = this.session.ralphTracker.lastStatusBlock; + const statusBlock = this.session.ralphTracker?.lastStatusBlock; let updatePrompt = this.config.updatePrompt; if (statusBlock?.recommendation) { @@ -2440,6 +2894,9 @@ export class RespawnController extends EventEmitter { this.log(`Respawn cycle #${this.cycleCount} completed`); this.emit('respawnCycleCompleted', this.cycleCount); + // P2-004: Complete cycle metrics with success outcome + this.completeCycleMetrics('success'); + // Go back to watching state for next cycle this.setState('watching'); this.terminalBuffer.clear(); @@ -2448,6 +2905,9 @@ export class RespawnController extends EventEmitter { this.workingDetected = false; this.resetHookState(); // Clear hook signals for next cycle + // P2-001: Reset idle detection start time for next cycle + this.idleDetectionStartTime = Date.now(); + // Restart detection timers for next cycle this.startNoOutputTimer(); this.startPreFilterTimer(); @@ -2548,4 +3008,424 @@ export class RespawnController extends EventEmitter { config: this.config, }; } + + // ========== P2-001: Adaptive Timing Methods ========== + + /** + * Get the current completion confirm timeout, potentially adjusted by adaptive timing. + * Uses historical idle detection durations to calculate an optimal timeout. + * + * @returns Completion confirm timeout in milliseconds + */ + getAdaptiveCompletionConfirmMs(): number { + if (!this.config.adaptiveTimingEnabled) { + return this.config.completionConfirmMs ?? 10000; + } + + // Need at least 5 samples before adjusting + if (this.timingHistory.sampleCount < 5) { + return this.config.completionConfirmMs ?? 10000; + } + + return this.timingHistory.adaptiveCompletionConfirmMs; + } + + /** + * Record timing data from a completed cycle for adaptive adjustments. + * + * @param idleDetectionMs - Time spent detecting idle + * @param cycleDurationMs - Total cycle duration + */ + private recordTimingData(idleDetectionMs: number, cycleDurationMs: number): void { + if (!this.config.adaptiveTimingEnabled) return; + + const history = this.timingHistory; + + // Add to rolling windows + history.recentIdleDetectionMs.push(idleDetectionMs); + history.recentCycleDurationMs.push(cycleDurationMs); + + // Trim to max samples + if (history.recentIdleDetectionMs.length > history.maxSamples) { + history.recentIdleDetectionMs.shift(); + } + if (history.recentCycleDurationMs.length > history.maxSamples) { + history.recentCycleDurationMs.shift(); + } + + history.sampleCount = history.recentIdleDetectionMs.length; + history.lastUpdatedAt = Date.now(); + + // Recalculate adaptive timing + this.updateAdaptiveTiming(); + } + + /** + * Recalculate the adaptive completion confirm timeout based on historical data. + * Uses the 75th percentile of recent idle detection times as the new timeout, + * with a 20% buffer for safety. + */ + private updateAdaptiveTiming(): void { + const history = this.timingHistory; + const minMs = this.config.adaptiveMinConfirmMs ?? 5000; + const maxMs = this.config.adaptiveMaxConfirmMs ?? 30000; + + if (history.recentIdleDetectionMs.length < 5) return; + + // Sort for percentile calculation + const sorted = [...history.recentIdleDetectionMs].sort((a, b) => a - b); + + // Use 75th percentile with 20% buffer + const p75Index = Math.floor(sorted.length * 0.75); + const p75Value = sorted[p75Index]; + const withBuffer = Math.round(p75Value * 1.2); + + // Clamp to configured bounds + const clamped = Math.max(minMs, Math.min(maxMs, withBuffer)); + + history.adaptiveCompletionConfirmMs = clamped; + this.log(`Adaptive timing updated: ${clamped}ms (p75=${p75Value}ms, samples=${sorted.length})`); + } + + /** + * Get the current timing history for monitoring. + * @returns Copy of timing history + */ + getTimingHistory(): TimingHistory { + return { ...this.timingHistory }; + } + + // ========== P2-002: Skip-Clear Optimization Methods ========== + + /** + * Determine whether to skip the /clear step based on current context usage. + * Skips if token count is below the configured threshold percentage. + * + * @returns True if /clear should be skipped + */ + private shouldSkipClear(): boolean { + if (!this.config.skipClearWhenLowContext) return false; + + const thresholdPercent = this.config.skipClearThresholdPercent ?? 30; + const maxContext = 200000; // Approximate max context for Claude + + // Use the session's token count if available + const currentTokens = this.lastTokenCount; + if (currentTokens === 0) return false; // Can't determine, don't skip + + const usagePercent = (currentTokens / maxContext) * 100; + + if (usagePercent < thresholdPercent) { + this.log(`Skip-clear optimization: ${usagePercent.toFixed(1)}% < ${thresholdPercent}% threshold`); + this.logAction('optimization', `Skipping /clear (${usagePercent.toFixed(1)}% context used)`); + return true; + } + + return false; + } + + // ========== P2-004: Cycle Metrics Methods ========== + + /** + * Start tracking metrics for a new cycle. + * Called when a respawn cycle begins. + */ + private startCycleMetrics(idleReason: string): void { + if (!this.config.trackCycleMetrics) return; + + const now = Date.now(); + this.currentCycleMetrics = { + cycleId: `${this.session.id}:${this.cycleCount}`, + sessionId: this.session.id, + cycleNumber: this.cycleCount, + startedAt: now, + idleReason, + idleDetectionMs: now - this.idleDetectionStartTime, + stepsCompleted: [], + clearSkipped: false, + tokenCountAtStart: this.lastTokenCount, + completionConfirmMsUsed: this.getAdaptiveCompletionConfirmMs(), + }; + } + + /** + * Record a completed step in the current cycle. + * @param step - Name of the step (e.g., 'update', 'clear', 'init') + */ + private recordCycleStep(step: string): void { + if (!this.config.trackCycleMetrics || !this.currentCycleMetrics) return; + this.currentCycleMetrics.stepsCompleted?.push(step); + } + + /** + * Complete the current cycle metrics with outcome. + * Adds to recent metrics and updates aggregates. + * + * @param outcome - Outcome of the cycle + * @param errorMessage - Optional error message if outcome is 'error' + */ + private completeCycleMetrics(outcome: CycleOutcome, errorMessage?: string): void { + if (!this.config.trackCycleMetrics || !this.currentCycleMetrics) return; + + const now = Date.now(); + const metrics: RespawnCycleMetrics = { + ...this.currentCycleMetrics as RespawnCycleMetrics, + completedAt: now, + durationMs: now - (this.currentCycleMetrics.startedAt ?? now), + outcome, + errorMessage, + tokenCountAtEnd: this.lastTokenCount, + }; + + // Add to recent metrics + this.recentCycleMetrics.push(metrics); + if (this.recentCycleMetrics.length > RespawnController.MAX_CYCLE_METRICS_IN_MEMORY) { + this.recentCycleMetrics.shift(); + } + + // Record timing data for adaptive timing + this.recordTimingData(metrics.idleDetectionMs, metrics.durationMs); + + // Update aggregate metrics + this.updateAggregateMetrics(metrics); + + // Clear current cycle + this.currentCycleMetrics = null; + + this.log(`Cycle #${metrics.cycleNumber} metrics: ${outcome}, duration=${metrics.durationMs}ms, idle_detection=${metrics.idleDetectionMs}ms`); + } + + /** + * Update aggregate metrics with a new cycle's data. + * @param metrics - The completed cycle metrics + */ + private updateAggregateMetrics(metrics: RespawnCycleMetrics): void { + const agg = this.aggregateMetrics; + + agg.totalCycles++; + + switch (metrics.outcome) { + case 'success': + agg.successfulCycles++; + break; + case 'stuck_recovery': + agg.stuckRecoveryCycles++; + break; + case 'blocked': + agg.blockedCycles++; + break; + case 'error': + agg.errorCycles++; + break; + } + + // Recalculate averages using all recent metrics + const durations = this.recentCycleMetrics.map(m => m.durationMs); + const idleTimes = this.recentCycleMetrics.map(m => m.idleDetectionMs); + + if (durations.length > 0) { + agg.avgCycleDurationMs = Math.round(durations.reduce((a, b) => a + b, 0) / durations.length); + agg.avgIdleDetectionMs = Math.round(idleTimes.reduce((a, b) => a + b, 0) / idleTimes.length); + + // Calculate P90 + const sortedDurations = [...durations].sort((a, b) => a - b); + const p90Index = Math.floor(sortedDurations.length * 0.9); + agg.p90CycleDurationMs = sortedDurations[p90Index]; + } + + // Calculate success rate + agg.successRate = agg.totalCycles > 0 + ? Math.round((agg.successfulCycles / agg.totalCycles) * 100) + : 100; + + agg.lastUpdatedAt = Date.now(); + } + + /** + * Get aggregate metrics for monitoring. + * @returns Copy of aggregate metrics + */ + getAggregateMetrics(): RespawnAggregateMetrics { + return { ...this.aggregateMetrics }; + } + + /** + * Get recent cycle metrics for analysis. + * @param limit - Maximum number of metrics to return (default: 20) + * @returns Recent cycle metrics, newest first + */ + getRecentCycleMetrics(limit: number = 20): RespawnCycleMetrics[] { + return this.recentCycleMetrics.slice(-limit).reverse(); + } + + // ========== P2-005: Health Score Methods ========== + + /** + * Calculate a comprehensive health score for the Ralph Loop system. + * Aggregates multiple health signals into a single score (0-100). + * + * @returns Health score with component breakdown + */ + calculateHealthScore(): RalphLoopHealthScore { + const now = Date.now(); + const components = { + cycleSuccess: this.calculateCycleSuccessScore(), + circuitBreaker: this.calculateCircuitBreakerScore(), + iterationProgress: this.calculateIterationProgressScore(), + aiChecker: this.calculateAiCheckerScore(), + stuckRecovery: this.calculateStuckRecoveryScore(), + }; + + // Weighted average (cycle success is most important) + const weights = { + cycleSuccess: 0.35, + circuitBreaker: 0.20, + iterationProgress: 0.20, + aiChecker: 0.15, + stuckRecovery: 0.10, + }; + + const score = Math.round( + components.cycleSuccess * weights.cycleSuccess + + components.circuitBreaker * weights.circuitBreaker + + components.iterationProgress * weights.iterationProgress + + components.aiChecker * weights.aiChecker + + components.stuckRecovery * weights.stuckRecovery + ); + + // Determine status + let status: HealthStatus; + if (score >= 90) status = 'excellent'; + else if (score >= 70) status = 'good'; + else if (score >= 50) status = 'degraded'; + else status = 'critical'; + + // Generate recommendations + const recommendations = this.generateHealthRecommendations(components); + + // Generate summary + const summary = this.generateHealthSummary(score, status, components); + + return { + score, + status, + components, + summary, + recommendations, + calculatedAt: now, + }; + } + + /** + * Calculate score based on recent cycle success rate. + */ + private calculateCycleSuccessScore(): number { + if (this.aggregateMetrics.totalCycles === 0) return 100; // No data = assume healthy + return this.aggregateMetrics.successRate; + } + + /** + * Calculate score based on circuit breaker state. + */ + private calculateCircuitBreakerScore(): number { + const tracker = this.session.ralphTracker; + if (!tracker) return 100; + + const cb = tracker.circuitBreakerStatus; + switch (cb.state) { + case 'CLOSED': return 100; + case 'HALF_OPEN': return 50; + case 'OPEN': return 0; + default: return 100; + } + } + + /** + * Calculate score based on iteration progress. + */ + private calculateIterationProgressScore(): number { + const tracker = this.session.ralphTracker; + if (!tracker) return 100; + + const stallMetrics = tracker.getIterationStallMetrics(); + const { stallDurationMs, warningThresholdMs, criticalThresholdMs } = stallMetrics; + + if (stallDurationMs >= criticalThresholdMs) return 0; + if (stallDurationMs >= warningThresholdMs) return 30; + if (stallDurationMs >= warningThresholdMs / 2) return 70; + return 100; + } + + /** + * Calculate score based on AI checker health. + */ + private calculateAiCheckerScore(): number { + const state = this.aiChecker.getState(); + if (state.status === 'disabled') return 30; + if (state.status === 'cooldown') return 70; + if (state.consecutiveErrors > 0) return 50; + return 100; + } + + /** + * Calculate score based on stuck-state recovery count. + */ + private calculateStuckRecoveryScore(): number { + const maxRecoveries = this.config.maxStuckRecoveries ?? 3; + if (this.stuckRecoveryCount === 0) return 100; + if (this.stuckRecoveryCount >= maxRecoveries) return 0; + return Math.round(100 - (this.stuckRecoveryCount / maxRecoveries) * 100); + } + + /** + * Generate health recommendations based on component scores. + */ + private generateHealthRecommendations(components: RalphLoopHealthScore['components']): string[] { + const recommendations: string[] = []; + + if (components.cycleSuccess < 70) { + recommendations.push('Cycle success rate is low. Check for recurring errors or stuck states.'); + } + if (components.circuitBreaker < 50) { + recommendations.push('Circuit breaker is open or half-open. Review recent errors and consider manual reset.'); + } + if (components.iterationProgress < 50) { + recommendations.push('Iteration progress has stalled. Check if Claude is stuck on a task.'); + } + if (components.aiChecker < 50) { + recommendations.push('AI idle checker has errors. May need to check Claude CLI availability.'); + } + if (components.stuckRecovery < 50) { + recommendations.push('Multiple stuck-state recoveries occurred. Consider increasing timeouts.'); + } + + if (recommendations.length === 0) { + recommendations.push('System is healthy. No action needed.'); + } + + return recommendations; + } + + /** + * Generate a human-readable health summary. + */ + private generateHealthSummary( + score: number, + status: HealthStatus, + components: RalphLoopHealthScore['components'] + ): string { + const lowest = Object.entries(components).reduce((min, [key, val]) => + val < min.val ? { key, val } : min, { key: '', val: 100 }); + + if (status === 'excellent') { + return `Ralph Loop is operating excellently (${score}/100). All systems healthy.`; + } + if (status === 'good') { + return `Ralph Loop is operating well (${score}/100). Minor issues in ${lowest.key}.`; + } + if (status === 'degraded') { + return `Ralph Loop is degraded (${score}/100). Primary issue: ${lowest.key} (${lowest.val}/100).`; + } + return `Ralph Loop is in critical state (${score}/100). Immediate attention needed: ${lowest.key}.`; + } } diff --git a/src/session.ts b/src/session.ts index 9de67f44..521c57e1 100644 --- a/src/session.ts +++ b/src/session.ts @@ -727,6 +727,7 @@ export class Session extends EventEmitter { inputTokens: this._totalInputTokens, outputTokens: this._totalOutputTokens, ralphEnabled: this._ralphTracker.enabled, + ralphAutoEnableDisabled: this._ralphTracker.autoEnableDisabled || undefined, ralphCompletionPhrase: this._ralphTracker.loopState.completionPhrase || undefined, parentAgentId: this._parentAgentId || undefined, childAgentIds: this._childAgentIds.length > 0 ? this._childAgentIds : undefined, diff --git a/src/types.ts b/src/types.ts index 563d4f50..f31b37c4 100644 --- a/src/types.ts +++ b/src/types.ts @@ -166,6 +166,8 @@ export interface SessionState { respawnConfig?: RespawnConfig & { durationMinutes?: number }; /** Ralph / Todo tracker enabled */ ralphEnabled?: boolean; + /** Ralph auto-enable disabled (user explicitly turned off Ralph) */ + ralphAutoEnableDisabled?: boolean; /** Ralph completion phrase (if set) */ ralphCompletionPhrase?: string; /** Parent agent ID if this session is a spawned agent */ @@ -394,6 +396,159 @@ export interface RespawnConfig { aiPlanCheckTimeoutMs?: number; /** Cooldown after NOT_PLAN_MODE verdict in ms */ aiPlanCheckCooldownMs?: number; + + // ========== P2-001: Adaptive Timing ========== + + /** Whether to use adaptive timing based on historical patterns */ + adaptiveTimingEnabled?: boolean; + /** Minimum value for adaptive completion confirm (ms) */ + adaptiveMinConfirmMs?: number; + /** Maximum value for adaptive completion confirm (ms) */ + adaptiveMaxConfirmMs?: number; + + // ========== P2-002: Skip-Clear Optimization ========== + + /** Whether to skip /clear when context is below threshold */ + skipClearWhenLowContext?: boolean; + /** Token percentage threshold below which /clear is skipped (0-100) */ + skipClearThresholdPercent?: number; + + // ========== P2-004: Cycle Metrics ========== + + /** Whether to track and persist cycle metrics */ + trackCycleMetrics?: boolean; +} + +// ========== P2-004: Respawn Cycle Metrics ========== + +/** + * Outcome of a respawn cycle + */ +export type CycleOutcome = + | 'success' // Cycle completed normally + | 'stuck_recovery' // Stuck-state recovery triggered + | 'blocked' // Blocked by circuit breaker or exit signal + | 'error' // Error during cycle + | 'cancelled'; // Cancelled (e.g., controller stopped) + +/** + * Metrics for a single respawn cycle. + * Persisted for post-mortem analysis of long-running loops. + */ +export interface RespawnCycleMetrics { + /** Unique cycle ID (session-id:cycle-number) */ + cycleId: string; + /** Session ID this cycle belongs to */ + sessionId: string; + /** Cycle number within the session */ + cycleNumber: number; + /** Timestamp when cycle started */ + startedAt: number; + /** Timestamp when cycle completed */ + completedAt: number; + /** Total duration of cycle (ms) */ + durationMs: number; + /** What triggered idle detection */ + idleReason: string; + /** Time spent detecting idle (from start of watching to idle confirmed) */ + idleDetectionMs: number; + /** Steps completed in this cycle */ + stepsCompleted: string[]; + /** Whether /clear was skipped (P2-002) */ + clearSkipped: boolean; + /** Outcome of the cycle */ + outcome: CycleOutcome; + /** Error message if outcome is 'error' */ + errorMessage?: string; + /** Token count at start of cycle */ + tokenCountAtStart?: number; + /** Token count at end of cycle */ + tokenCountAtEnd?: number; + /** Completion confirm time used (may be adaptive) */ + completionConfirmMsUsed: number; +} + +/** + * Aggregate metrics across multiple cycles for health scoring. + */ +export interface RespawnAggregateMetrics { + /** Total cycles tracked */ + totalCycles: number; + /** Successful cycles */ + successfulCycles: number; + /** Cycles that required stuck-state recovery */ + stuckRecoveryCycles: number; + /** Blocked cycles */ + blockedCycles: number; + /** Error cycles */ + errorCycles: number; + /** Average cycle duration (ms) */ + avgCycleDurationMs: number; + /** Average idle detection time (ms) */ + avgIdleDetectionMs: number; + /** 90th percentile cycle duration (ms) */ + p90CycleDurationMs: number; + /** Success rate (0-100) */ + successRate: number; + /** Last updated timestamp */ + lastUpdatedAt: number; +} + +// ========== P2-005: Ralph Loop Health Score ========== + +/** + * Health status levels for the Ralph Loop system. + */ +export type HealthStatus = 'excellent' | 'good' | 'degraded' | 'critical'; + +/** + * Comprehensive health score for a Ralph Loop session. + * Aggregates multiple health signals into a single score. + */ +export interface RalphLoopHealthScore { + /** Overall health score (0-100) */ + score: number; + /** Health status based on score thresholds */ + status: HealthStatus; + /** Individual component scores (0-100 each) */ + components: { + /** Based on recent cycle success rate */ + cycleSuccess: number; + /** Based on circuit breaker state */ + circuitBreaker: number; + /** Based on iteration stall metrics */ + iterationProgress: number; + /** Based on AI checker error rate */ + aiChecker: number; + /** Based on stuck-state recovery count */ + stuckRecovery: number; + }; + /** Human-readable summary of health */ + summary: string; + /** Recommendations for improvement */ + recommendations: string[]; + /** Timestamp when score was calculated */ + calculatedAt: number; +} + +// ========== Timing History for Adaptive Timing ========== + +/** + * Historical timing data for adaptive adjustments. + */ +export interface TimingHistory { + /** Rolling window of recent idle detection durations (ms) */ + recentIdleDetectionMs: number[]; + /** Rolling window of recent cycle durations (ms) */ + recentCycleDurationMs: number[]; + /** Calculated adaptive completion confirm value (ms) */ + adaptiveCompletionConfirmMs: number; + /** Number of samples in rolling windows */ + sampleCount: number; + /** Maximum samples to keep */ + maxSamples: number; + /** Last updated timestamp */ + lastUpdatedAt: number; } /** @@ -829,13 +984,43 @@ export type RalphTodoStatus = 'pending' | 'in_progress' | 'completed'; /** * State of per-session Ralph / Todo tracking (detected from Claude output) */ +/** + * Confidence scoring for completion detection. + * Helps distinguish genuine completion signals from false positives. + */ +export interface CompletionConfidence { + /** Overall confidence level (0-100) */ + score: number; + /** Whether score is above threshold for triggering completion */ + isConfident: boolean; + /** Individual signal contributions */ + signals: { + /** Promise tag detected with proper formatting */ + hasPromiseTag: boolean; + /** Phrase matches expected completion phrase */ + matchesExpected: boolean; + /** All todos are marked complete */ + allTodosComplete: boolean; + /** EXIT_SIGNAL: true in RALPH_STATUS block */ + hasExitSignal: boolean; + /** Multiple completion indicators present */ + multipleIndicators: boolean; + /** Output context suggests completion (not in prompt/explanation) */ + contextAppropriate: boolean; + }; + /** Timestamp of last confidence calculation */ + calculatedAt: number; +} + export interface RalphTrackerState { /** Whether the tracker is actively monitoring (disabled by default) */ enabled: boolean; /** Whether a loop is currently active */ active: boolean; - /** Detected completion phrase */ + /** Detected completion phrase (primary) */ completionPhrase: string | null; + /** Additional valid completion phrases (P1-003: multi-phrase support) */ + alternateCompletionPhrases?: string[]; /** Timestamp when loop started */ startedAt: number | null; /** Number of cycles/iterations detected */ @@ -850,6 +1035,8 @@ export interface RalphTrackerState { planVersion?: number; /** Number of versions in history (for versioning UI) */ planHistoryLength?: number; + /** Last completion confidence assessment */ + completionConfidence?: CompletionConfidence; } /** @@ -872,6 +1059,32 @@ export interface RalphTodoItem { detectedAt: number; /** Priority level (P0=critical, P1=high, P2=normal) */ priority: RalphTodoPriority; + /** P1-009: Estimated time to complete (ms), based on historical patterns */ + estimatedDurationMs?: number; + /** P1-009: Complexity category for progress estimation */ + estimatedComplexity?: 'trivial' | 'simple' | 'moderate' | 'complex'; +} + +/** + * Progress estimation for the todo list + */ +export interface RalphTodoProgress { + /** Total number of todos */ + total: number; + /** Number completed */ + completed: number; + /** Number in progress */ + inProgress: number; + /** Number pending */ + pending: number; + /** Completion percentage (0-100) */ + percentComplete: number; + /** Estimated remaining time (ms), based on historical completion rate */ + estimatedRemainingMs: number | null; + /** Average time per todo completion (ms) */ + avgCompletionTimeMs: number | null; + /** Projected completion timestamp (epoch ms) */ + projectedCompletionAt: number | null; } /** diff --git a/src/utils/index.ts b/src/utils/index.ts index a5539464..c4a31700 100644 --- a/src/utils/index.ts +++ b/src/utils/index.ts @@ -24,3 +24,12 @@ export { validateTokenCounts, validateTokensAndCost, } from './token-validation.js'; +export { + levenshteinDistance, + stringSimilarity, + isSimilar, + isSimilarByDistance, + normalizePhrase, + fuzzyPhraseMatch, + todoContentHash, +} from './string-similarity.js'; diff --git a/src/utils/string-similarity.ts b/src/utils/string-similarity.ts new file mode 100644 index 00000000..869949df --- /dev/null +++ b/src/utils/string-similarity.ts @@ -0,0 +1,208 @@ +/** + * @fileoverview String similarity utilities for fuzzy matching. + * + * Provides Levenshtein distance and similarity scoring for: + * - Completion phrase fuzzy matching + * - Todo content deduplication + * + * @module utils/string-similarity + */ + +/** + * Calculate the Levenshtein (edit) distance between two strings. + * This is the minimum number of single-character edits (insertions, + * deletions, or substitutions) required to change one string into the other. + * + * Uses Wagner-Fischer algorithm with O(min(m,n)) space optimization. + * + * @param a - First string + * @param b - Second string + * @returns The edit distance (0 = identical) + * + * @example + * levenshteinDistance('hello', 'hello') // 0 + * levenshteinDistance('hello', 'helo') // 1 (one deletion) + * levenshteinDistance('COMPLETE', 'COMPLET') // 1 (one deletion) + */ +export function levenshteinDistance(a: string, b: string): number { + // Ensure a is the shorter string for space efficiency + if (a.length > b.length) { + [a, b] = [b, a]; + } + + const m = a.length; + const n = b.length; + + // Early exit for identical strings + if (a === b) return 0; + + // Early exit for empty strings + if (m === 0) return n; + + // Use single array (space optimization) + let prev = new Array(m + 1); + let curr = new Array(m + 1); + + // Initialize first row + for (let i = 0; i <= m; i++) { + prev[i] = i; + } + + // Fill the matrix row by row + for (let j = 1; j <= n; j++) { + curr[0] = j; + + for (let i = 1; i <= m; i++) { + const cost = a[i - 1] === b[j - 1] ? 0 : 1; + curr[i] = Math.min( + prev[i] + 1, // deletion + curr[i - 1] + 1, // insertion + prev[i - 1] + cost // substitution + ); + } + + // Swap rows + [prev, curr] = [curr, prev]; + } + + return prev[m]; +} + +/** + * Calculate similarity ratio between two strings (0 to 1). + * Uses Levenshtein distance normalized by the longer string's length. + * + * @param a - First string + * @param b - Second string + * @returns Similarity ratio (1.0 = identical, 0.0 = completely different) + * + * @example + * stringSimilarity('hello', 'hello') // 1.0 + * stringSimilarity('hello', 'helo') // 0.8 (4/5 similar) + * stringSimilarity('abc', 'xyz') // 0.0 (3 edits, length 3) + */ +export function stringSimilarity(a: string, b: string): number { + if (a === b) return 1.0; + if (a.length === 0 && b.length === 0) return 1.0; + if (a.length === 0 || b.length === 0) return 0.0; + + const distance = levenshteinDistance(a, b); + const maxLength = Math.max(a.length, b.length); + return 1 - distance / maxLength; +} + +/** + * Check if two strings are similar within a given threshold. + * + * @param a - First string + * @param b - Second string + * @param threshold - Minimum similarity ratio (default: 0.85 = 85% similar) + * @returns True if similarity >= threshold + * + * @example + * isSimilar('COMPLETE', 'COMPLET', 0.85) // true (87.5% similar) + * isSimilar('COMPLETE', 'DONE', 0.85) // false (0% similar) + */ +export function isSimilar(a: string, b: string, threshold = 0.85): boolean { + return stringSimilarity(a, b) >= threshold; +} + +/** + * Check if two strings are similar with edit distance tolerance. + * More intuitive for short strings than percentage-based threshold. + * + * @param a - First string + * @param b - Second string + * @param maxDistance - Maximum allowed edit distance (default: 2) + * @returns True if edit distance <= maxDistance + * + * @example + * isSimilarByDistance('COMPLETE', 'COMPLET', 2) // true (distance 1) + * isSimilarByDistance('COMPLETE', 'COMP', 2) // false (distance 4) + */ +export function isSimilarByDistance(a: string, b: string, maxDistance = 2): boolean { + return levenshteinDistance(a, b) <= maxDistance; +} + +/** + * Normalize a completion phrase for comparison. + * Handles variations in case, whitespace, and separators. + * + * @param phrase - Raw completion phrase + * @returns Normalized phrase (uppercase, no separators) + * + * @example + * normalizePhrase('task_done') // 'TASKDONE' + * normalizePhrase('TASK-DONE') // 'TASKDONE' + * normalizePhrase('Task Done') // 'TASKDONE' + */ +export function normalizePhrase(phrase: string): string { + return phrase + .toUpperCase() + .replace(/[\s_\-\.]+/g, '') // Remove whitespace, underscores, hyphens, dots + .trim(); +} + +/** + * Check if two completion phrases match with fuzzy tolerance. + * + * First normalizes both phrases, then checks: + * 1. Exact match after normalization + * 2. Edit distance <= maxDistance for typo tolerance + * + * @param phrase1 - First phrase to compare + * @param phrase2 - Second phrase to compare + * @param maxDistance - Maximum edit distance for fuzzy match (default: 2) + * @returns True if phrases match (exact or fuzzy) + * + * @example + * fuzzyPhraseMatch('COMPLETE', 'COMPLETE') // true (exact) + * fuzzyPhraseMatch('COMPLETE', 'COMPLET') // true (typo) + * fuzzyPhraseMatch('TASK_DONE', 'TASKDONE') // true (separator) + * fuzzyPhraseMatch('COMPLETE', 'FINISHED') // false (different word) + */ +export function fuzzyPhraseMatch( + phrase1: string, + phrase2: string, + maxDistance = 2 +): boolean { + const norm1 = normalizePhrase(phrase1); + const norm2 = normalizePhrase(phrase2); + + // Exact match after normalization + if (norm1 === norm2) return true; + + // For short phrases (< 6 chars), require exact match to avoid false positives + // e.g., "DONE" shouldn't match "DENY" + if (norm1.length < 6 || norm2.length < 6) { + return false; + } + + // Fuzzy match with edit distance + return isSimilarByDistance(norm1, norm2, maxDistance); +} + +/** + * Generate a content hash for todo deduplication. + * Normalizes content and generates a simple hash. + * + * @param content - Todo item content + * @returns Normalized hash string for comparison + */ +export function todoContentHash(content: string): string { + // Normalize: lowercase, collapse whitespace, remove punctuation + const normalized = content + .toLowerCase() + .replace(/\s+/g, ' ') + .replace(/[^\w\s]/g, '') + .trim(); + + // Simple hash using reduce (fast, good enough for deduplication) + let hash = 0; + for (let i = 0; i < normalized.length; i++) { + const char = normalized.charCodeAt(i); + hash = ((hash << 5) - hash) + char; + hash = hash & hash; // Convert to 32-bit integer + } + return hash.toString(36); +} diff --git a/src/web/public/app.js b/src/web/public/app.js index 56275474..317a75eb 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -6504,8 +6504,14 @@ class ClaudemanApp { if (total > 0) { tokensEl.style.display = ''; const tokenStr = this.formatTokens(total); - const estimatedCost = this.estimateCost(input, output); - tokensEl.textContent = `${tokenStr} tokens · $${estimatedCost.toFixed(2)}`; + const settings = this.loadAppSettingsFromStorage(); + const showCost = settings.showCost ?? false; + if (showCost) { + const estimatedCost = this.estimateCost(input, output); + tokensEl.textContent = `${tokenStr} tokens · $${estimatedCost.toFixed(2)}`; + } else { + tokensEl.textContent = `${tokenStr} tokens`; + } } else { tokensEl.style.display = 'none'; } @@ -6865,10 +6871,14 @@ class ClaudemanApp { const estimatedCost = this.estimateCost(totalInput, totalOutput); const tokenEl = this.$('headerTokens'); if (tokenEl) { - tokenEl.textContent = total > 0 ? `${display} tokens · $${estimatedCost.toFixed(2)}` : '0 tokens'; + const settings = this.loadAppSettingsFromStorage(); + const showCost = settings.showCost ?? false; + tokenEl.textContent = total > 0 + ? (showCost ? `${display} tokens · $${estimatedCost.toFixed(2)}` : `${display} tokens`) + : '0 tokens'; tokenEl.title = this.globalStats - ? `Lifetime: ${this.globalStats.totalSessionsCreated} sessions created\nEstimated cost based on Claude Opus pricing` - : 'Token usage across active sessions\nEstimated cost based on Claude Opus pricing'; + ? `Lifetime: ${this.globalStats.totalSessionsCreated} sessions created${showCost ? '\nEstimated cost based on Claude Opus pricing' : ''}` + : `Token usage across active sessions${showCost ? '\nEstimated cost based on Claude Opus pricing' : ''}`; } } @@ -7781,16 +7791,17 @@ class ClaudemanApp { document.getElementById('appSettingsDefaultDir').value = settings.defaultWorkingDir || ''; document.getElementById('appSettingsRalphEnabled').checked = settings.ralphTrackerEnabled ?? false; // Header visibility settings (default to true/enabled) - document.getElementById('appSettingsShowFontControls').checked = settings.showFontControls ?? true; + document.getElementById('appSettingsShowFontControls').checked = settings.showFontControls ?? false; document.getElementById('appSettingsShowSystemStats').checked = settings.showSystemStats ?? true; document.getElementById('appSettingsShowTokenCount').checked = settings.showTokenCount ?? true; + document.getElementById('appSettingsShowCost').checked = settings.showCost ?? false; document.getElementById('appSettingsShowMonitor').checked = settings.showMonitor ?? true; - document.getElementById('appSettingsShowProjectInsights').checked = settings.showProjectInsights ?? true; + document.getElementById('appSettingsShowProjectInsights').checked = settings.showProjectInsights ?? false; document.getElementById('appSettingsShowFileBrowser').checked = settings.showFileBrowser ?? false; - document.getElementById('appSettingsShowSubagents').checked = settings.showSubagents ?? true; + document.getElementById('appSettingsShowSubagents').checked = settings.showSubagents ?? false; document.getElementById('appSettingsSubagentTracking').checked = settings.subagentTrackingEnabled ?? true; - document.getElementById('appSettingsSubagentActiveTabOnly').checked = settings.subagentActiveTabOnly ?? false; - document.getElementById('appSettingsImageWatcherEnabled').checked = settings.imageWatcherEnabled ?? true; + document.getElementById('appSettingsSubagentActiveTabOnly').checked = settings.subagentActiveTabOnly ?? true; + document.getElementById('appSettingsImageWatcherEnabled').checked = settings.imageWatcherEnabled ?? false; // Claude CLI settings const claudeModeSelect = document.getElementById('appSettingsClaudeMode'); const allowedToolsRow = document.getElementById('allowedToolsRow'); @@ -7906,6 +7917,7 @@ class ClaudemanApp { showFontControls: document.getElementById('appSettingsShowFontControls').checked, showSystemStats: document.getElementById('appSettingsShowSystemStats').checked, showTokenCount: document.getElementById('appSettingsShowTokenCount').checked, + showCost: document.getElementById('appSettingsShowCost').checked, showMonitor: document.getElementById('appSettingsShowMonitor').checked, showProjectInsights: document.getElementById('appSettingsShowProjectInsights').checked, showFileBrowser: document.getElementById('appSettingsShowFileBrowser').checked, @@ -8107,7 +8119,7 @@ class ClaudemanApp { applyHeaderVisibilitySettings() { const settings = this.loadAppSettingsFromStorage(); // Default all to true (enabled) if not set - const showFontControls = settings.showFontControls ?? true; + const showFontControls = settings.showFontControls ?? false; const showSystemStats = settings.showSystemStats ?? true; const showTokenCount = settings.showTokenCount ?? true; @@ -8141,7 +8153,7 @@ class ClaudemanApp { applyMonitorVisibility() { const settings = this.loadAppSettingsFromStorage(); const showMonitor = settings.showMonitor ?? true; - const showSubagents = settings.showSubagents ?? true; + const showSubagents = settings.showSubagents ?? false; const showFileBrowser = settings.showFileBrowser ?? false; const monitorPanel = document.getElementById('monitorPanel'); @@ -10679,7 +10691,7 @@ class ClaudemanApp { */ updateSubagentWindowVisibility() { const settings = this.loadAppSettingsFromStorage(); - const activeTabOnly = settings.subagentActiveTabOnly ?? false; + const activeTabOnly = settings.subagentActiveTabOnly ?? true; for (const [agentId, windowInfo] of this.subagentWindows) { // Get parent from PERSISTENT map (THE source of truth) @@ -10722,7 +10734,7 @@ class ClaudemanApp { const existing = this.subagentWindows.get(agentId); const agent = this.subagents.get(agentId); const settings = this.loadAppSettingsFromStorage(); - const activeTabOnly = settings.subagentActiveTabOnly ?? false; + const activeTabOnly = settings.subagentActiveTabOnly ?? true; // If window is hidden (different tab) and activeTabOnly is enabled, switch to parent tab if (existing.hidden && agent?.parentSessionId && activeTabOnly) { @@ -10883,7 +10895,7 @@ class ClaudemanApp { // Check if this window should be visible based on settings // Use the PERSISTENT parent map for accurate tab-based visibility const settings = this.loadAppSettingsFromStorage(); - const activeTabOnly = settings.subagentActiveTabOnly ?? false; + const activeTabOnly = settings.subagentActiveTabOnly ?? true; let shouldHide = false; if (activeTabOnly) { const storedParent = this.subagentParentMap.get(agentId); @@ -11132,7 +11144,7 @@ class ClaudemanApp { if (windowData) { const settings = this.loadAppSettingsFromStorage(); - const activeTabOnly = settings.subagentActiveTabOnly ?? false; + const activeTabOnly = settings.subagentActiveTabOnly ?? true; // Get parent from PERSISTENT map (THE source of truth) const storedParent = this.subagentParentMap.get(agentId); @@ -11559,7 +11571,7 @@ class ClaudemanApp { // Check if panel is enabled in settings const settings = this.loadAppSettingsFromStorage(); - const showProjectInsights = settings.showProjectInsights ?? true; + const showProjectInsights = settings.showProjectInsights ?? false; if (!showProjectInsights) { panel.classList.remove('visible'); this.projectInsightsPanelVisible = false; diff --git a/src/web/public/index.html b/src/web/public/index.html index 6a48006a..b65f6896 100644 --- a/src/web/public/index.html +++ b/src/web/public/index.html @@ -731,6 +731,13 @@ +
+ Show Cost ($) + +
Panels
diff --git a/src/web/public/styles.css b/src/web/public/styles.css index 9b9b695c..495affc3 100644 --- a/src/web/public/styles.css +++ b/src/web/public/styles.css @@ -5118,22 +5118,24 @@ kbd { } .connection-line { - stroke: #00ff41; - stroke-width: 2.5; + stroke: #3b82f6; + stroke-width: 3; stroke-dasharray: 5 3; fill: none; - opacity: 0.75; - /* Dark outline for contrast, subtle glow */ - filter: drop-shadow(0 0 1px rgba(0, 0, 0, 0.7)) - drop-shadow(0 0 3px rgba(0, 255, 65, 0.5)); + opacity: 0.9; + /* Dark outline for contrast, vibrant blue glow */ + filter: drop-shadow(0 0 2px rgba(0, 0, 0, 0.8)) + drop-shadow(0 0 4px rgba(59, 130, 246, 0.8)) + drop-shadow(0 0 8px rgba(59, 130, 246, 0.5)); transition: opacity 0.2s, stroke-width 0.2s, filter 0.2s; } .connection-line:hover { - opacity: 0.9; - stroke-width: 3; - filter: drop-shadow(0 0 2px rgba(0, 0, 0, 0.8)) - drop-shadow(0 0 5px rgba(0, 255, 65, 0.7)); + opacity: 1; + stroke-width: 3.5; + filter: drop-shadow(0 0 2px rgba(0, 0, 0, 0.9)) + drop-shadow(0 0 6px rgba(59, 130, 246, 1)) + drop-shadow(0 0 12px rgba(59, 130, 246, 0.7)); } .connection-line.spawning-line { @@ -5160,24 +5162,26 @@ kbd { /* Plan subagent to regular subagent connection lines (Opus → Haiku) */ .connection-line.plan-to-subagent-line { - stroke: #00ff41; - stroke-width: 2.5; - /* Dark outline for contrast, subtle glow */ - filter: drop-shadow(0 0 1px rgba(0, 0, 0, 0.7)) - drop-shadow(0 0 3px rgba(0, 255, 65, 0.5)); + stroke: #3b82f6; + stroke-width: 3; + /* Dark outline for contrast, vibrant blue glow */ + filter: drop-shadow(0 0 2px rgba(0, 0, 0, 0.8)) + drop-shadow(0 0 4px rgba(59, 130, 246, 0.8)) + drop-shadow(0 0 8px rgba(59, 130, 246, 0.5)); stroke-dasharray: 5 3; animation: plan-subagent-pulse 1.2s ease-in-out infinite; } .connection-line.plan-to-subagent-line:hover { - stroke-width: 3; - filter: drop-shadow(0 0 2px rgba(0, 0, 0, 0.8)) - drop-shadow(0 0 5px rgba(0, 255, 65, 0.7)); + stroke-width: 3.5; + filter: drop-shadow(0 0 2px rgba(0, 0, 0, 0.9)) + drop-shadow(0 0 6px rgba(59, 130, 246, 1)) + drop-shadow(0 0 12px rgba(59, 130, 246, 0.7)); } @keyframes plan-subagent-pulse { - 0%, 100% { opacity: 0.65; } - 50% { opacity: 0.85; } + 0%, 100% { opacity: 0.8; } + 50% { opacity: 1; } } /* ========== Project Insights Panel (Bash File Viewers) ========== */ diff --git a/src/web/server.ts b/src/web/server.ts index 125b142e..2ccc9642 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -1121,8 +1121,12 @@ export class WebServer extends EventEmitter { if (enabled !== undefined) { if (enabled) { session.ralphTracker.enable(); + // Allow re-enabling on restart if user explicitly enabled + session.ralphTracker.enableAutoEnable(); } else { session.ralphTracker.disable(); + // Prevent re-enabling on restart when user explicitly disabled + session.ralphTracker.disableAutoEnable(); } // Persist Ralph enabled state this.screenManager.updateRalphEnabled(id, enabled); @@ -1369,8 +1373,8 @@ export class WebServer extends EventEmitter { } try { - // Auto-detect completion phrase from CLAUDE.md BEFORE starting (only if globally enabled) - if (this.store.getConfig().ralphEnabled) { + // Auto-detect completion phrase from CLAUDE.md BEFORE starting (only if globally enabled and not explicitly disabled by user) + if (this.store.getConfig().ralphEnabled && !session.ralphTracker.autoEnableDisabled) { autoConfigureRalph(session, session.workingDir, () => {}); if (!session.ralphTracker.enabled) { session.ralphTracker.enable(); @@ -1688,8 +1692,8 @@ export class WebServer extends EventEmitter { } try { - // Auto-detect completion phrase from CLAUDE.md BEFORE starting (only if globally enabled) - if (this.store.getConfig().ralphEnabled) { + // Auto-detect completion phrase from CLAUDE.md BEFORE starting (only if globally enabled and not explicitly disabled by user) + if (this.store.getConfig().ralphEnabled && !session.ralphTracker.autoEnableDisabled) { autoConfigureRalph(session, session.workingDir, () => {}); if (!session.ralphTracker.enabled) { session.ralphTracker.enable(); @@ -2288,6 +2292,7 @@ export class WebServer extends EventEmitter { autoConfigureRalph(session, casePath, () => {}); // no broadcast yet if (!session.ralphTracker.enabled) { session.ralphTracker.enable(); + session.ralphTracker.enableAutoEnable(); // Allow re-enabling on restart } } @@ -4561,6 +4566,13 @@ NOW: Generate the implementation plan for the task above. Think step by step.`; } } // Ralph / Todo tracker + if (savedState.ralphAutoEnableDisabled) { + session.ralphTracker.disableAutoEnable(); + console.log(`[Server] Restored Ralph auto-enable disabled for session ${session.id}`); + } else if (savedState.ralphEnabled) { + // If Ralph was enabled and not explicitly disabled, allow re-enabling on restart + session.ralphTracker.enableAutoEnable(); + } if (savedState.ralphEnabled) { session.ralphTracker.enable(); if (savedState.ralphCompletionPhrase) { @@ -4594,8 +4606,8 @@ NOW: Generate the implementation plan for the task above. Think step by step.`; } } - // Fallback: restore Ralph state from state-inner.json if not already set - if (!session.ralphTracker.enabled) { + // Fallback: restore Ralph state from state-inner.json if not already set and not explicitly disabled + if (!session.ralphTracker.enabled && !session.ralphTracker.autoEnableDisabled) { const ralphState = this.store.getRalphState(screen.sessionId); if (ralphState?.loop?.enabled) { session.ralphTracker.restoreState(ralphState.loop, ralphState.todos); diff --git a/test/ai-idle-checker.test.ts b/test/ai-idle-checker.test.ts index 53f20389..e71c5631 100644 --- a/test/ai-idle-checker.test.ts +++ b/test/ai-idle-checker.test.ts @@ -295,6 +295,12 @@ describe('AiIdleChecker', () => { }); it('should disable after maxConsecutiveErrors', async () => { + // With P1-005 exponential backoff, cooldowns increase: + // Error 1: 1000ms * 2^0 = 1000ms + // Error 2: 1000ms * 2^1 = 2000ms + // Error 3: disabled (no cooldown) + const cooldowns = [1100, 2100]; // Wait slightly longer than each cooldown + for (let i = 0; i < 3; i++) { mockedReadFileSync.mockReturnValueOnce('') .mockReturnValueOnce('garbage\n__AICHECK_DONE__'); @@ -305,7 +311,7 @@ describe('AiIdleChecker', () => { // Clear cooldown for next check (except after the last one which disables) if (i < 2) { - await vi.advanceTimersByTimeAsync(1100); + await vi.advanceTimersByTimeAsync(cooldowns[i]); } } @@ -464,13 +470,19 @@ describe('AiIdleChecker', () => { const handler = vi.fn(); checker.on('disabled', handler); + // With exponential backoff (P1-005), cooldown increases: + // Error 1: 1000ms * 2^0 = 1000ms + // Error 2: 1000ms * 2^1 = 2000ms + // Error 3: disabled (no cooldown needed) + const cooldowns = [1100, 2100]; // Wait longer than exponential backoff + for (let i = 0; i < 3; i++) { mockedReadFileSync.mockReturnValueOnce('') .mockReturnValueOnce('garbage\n__AICHECK_DONE__'); const checkPromise = checker.check('output'); await vi.advanceTimersByTimeAsync(1000); await checkPromise; - if (i < 2) await vi.advanceTimersByTimeAsync(1100); + if (i < 2) await vi.advanceTimersByTimeAsync(cooldowns[i]); } expect(handler).toHaveBeenCalledWith(expect.stringContaining('consecutive errors')); diff --git a/test/respawn-controller.test.ts b/test/respawn-controller.test.ts index 5dfa62d0..abee16ca 100644 --- a/test/respawn-controller.test.ts +++ b/test/respawn-controller.test.ts @@ -15,6 +15,8 @@ class MockSession extends EventEmitter { id = 'mock-session-id'; workingDir = '/tmp'; status = 'idle'; + pid = 12345; // Mock PID for P1-006 health check + isWorking = false; // P0-006 Session.isWorking integration writeBuffer: string[] = []; write(data: string): void {