From 83033b4299995bfea07622be99db05aec0c9963f Mon Sep 17 00:00:00 2001 From: Devvyn <22340871+opticon454@users.noreply.github.com> Date: Wed, 16 Sep 2026 15:21:09 +0800 Subject: [PATCH] fix(custom-model): show a status toast during Claude's native-boot-then-restart window Claude stays on the launch-then-restart path (see runCustomModelEntry's own comment for why), but with nothing on screen during that window, a native boot that briefly talks to the cloud model read as "the endpoint didn't apply" rather than "the switch hasn't happened yet". A sticky "Claude started - switching to ..." toast now covers the whole window from the native launch through the apply call, updated in place (never stacked) as the outcome resolves: dismissed on cancel or failure (replaced by the existing cancellation/error toast), handed off to _watchLlamaSwapLoading's own sticky toast when a model swap is in progress, or updated to the existing "Pointed at ... - restarting" message and auto-dismissed after 3s on a plain success. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG --- src/web/public/session-ui.js | 16 ++++++- test/custom-model-run-menu-ui.test.ts | 67 +++++++++++++++++++++++++++ 2 files changed, 82 insertions(+), 1 deletion(-) diff --git a/src/web/public/session-ui.js b/src/web/public/session-ui.js index aee49b4e..ccb97887 100644 --- a/src/web/public/session-ui.js +++ b/src/web/public/session-ui.js @@ -810,6 +810,13 @@ Object.assign(CodemanApp.prototype, { const sessionId = this.activeSessionId; if (!sessionId || sessionId === before) return; + // Claude just launched on the NATIVE backend and is about to be restarted onto + // the endpoint — without something saying so, that native boot (which can talk + // to Opus for a moment) reads as "the endpoint didn't apply" rather than "the + // switch hasn't happened yet". Sticky until the apply below settles one way or + // the other, or hands off to _watchLlamaSwapLoading's own sticky toast. + const switchingToast = this.showToast(`Claude started — switching to ${endpointId}…`, 'info', { duration: 0 }); + // A freshly launched CLI reports its OWN startup as 'busy' (spinner, the // workspace-trust check, whatever else it does before its first prompt) — // measured landing well before this line reliably reaches it — and the @@ -849,6 +856,7 @@ Object.assign(CodemanApp.prototype, { `for ${payload.affectedSessions.length === 1 ? 'that session' : 'those sessions'} too. Continue?` ); if (!proceed) { + switchingToast?.dismiss(); this.showToast('Kept the native backend — model switch cancelled', 'info'); return; } @@ -857,20 +865,26 @@ Object.assign(CodemanApp.prototype, { } if (!ok || !data || data.success === false) { + switchingToast?.dismiss(); const detail = data?.error ? `: ${data.error}` : res ? ` (HTTP ${res.status})` : ' (request failed)'; this.showToast(`Session started on the native backend — could not apply the custom endpoint${detail}`, 'error'); return; } - this.showToast(`Pointed at ${endpointId} — restarting the session...`, 'info'); // The apply above already succeeded — the session IS pointed at the endpoint — but // llama-swap itself may still be unloading the old model and loading this one, which // can take well over a minute. Without this, a prompt sent during that window either // hangs silently or (the bug this whole feature exists to fix) gets answered by // whatever was loaded a moment ago, reading as "it's still using the wrong model." + // Hand off to its own sticky toast rather than stacking a second one on top. if (payload?.modelSwapInProgress) { + switchingToast?.dismiss(); void this._watchLlamaSwapLoading(endpointId, modelId); + return; } + + switchingToast?.setMessage(`Pointed at ${endpointId} — restarting the session...`); + setTimeout(() => switchingToast?.dismiss(), 3000); }, /** POST /api/sessions/:id/custom-model, returning {ok, data, res} rather than throwing — diff --git a/test/custom-model-run-menu-ui.test.ts b/test/custom-model-run-menu-ui.test.ts index 2597d1a3..fedce7fe 100644 --- a/test/custom-model-run-menu-ui.test.ts +++ b/test/custom-model-run-menu-ui.test.ts @@ -398,6 +398,73 @@ describe('Custom Model Endpoint Profiles: applying a picked entry', () => { expect(toastType).toBe('error'); }); + it('shows a status toast for the native-boot-then-restart window, so it never reads as the endpoint failing to apply', async () => { + // Claude still goes through this two-step launch (see runCustomModelEntry's own + // comment for why) — without something saying so, the native boot it starts with + // (which can genuinely talk to the cloud model for a moment) reads as "the + // endpoint didn't apply" rather than "the switch hasn't happened yet". + const { app } = bootApp({}); + app.activeSessionId = 'old-session'; + app.run = async () => { + app.activeSessionId = 'new-session'; + }; + app._api = async () => ({ + ok: true, + json: async () => ({ success: true, data: { customModel: { endpointId: 'llama-box' }, restarted: true } }), + }); + const toasts: Array<{ message: string; dismissed: boolean }> = []; + const messageHistory: string[] = []; + app.showToast = (message: string) => { + const entry = { message, dismissed: false }; + toasts.push(entry); + messageHistory.push(message); + return { + dismiss: () => { + entry.dismissed = true; + }, + setMessage: (next: string) => { + entry.message = next; + messageHistory.push(next); + }, + }; + }; + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(toasts).toHaveLength(1); // updated in place, not stacked with a second toast + expect(messageHistory[0]).toContain('Claude started — switching to llama-box'); + expect(messageHistory.at(-1)).toContain('Pointed at llama-box — restarting'); + }); + + it('dismisses the status toast on a failed apply rather than leaving it stuck on "switching"', async () => { + const { app } = bootApp({}); + app.activeSessionId = 'old-session'; + app.run = async () => { + app.activeSessionId = 'new-session'; + }; + app._api = async () => ({ + ok: false, + status: 500, + json: async () => ({ success: false, error: 'boom' }), + }); + const toasts: Array<{ message: string; dismissed: boolean }> = []; + app.showToast = (message: string) => { + const entry = { message, dismissed: false }; + toasts.push(entry); + return { + dismiss: () => { + entry.dismissed = true; + }, + setMessage: () => {}, + }; + }; + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(toasts[0].dismissed).toBe(true); // the "switching..." toast, cleaned up + expect(toasts.at(-1)?.message).toContain('boom'); // the error toast, separate from it + }); + it('routes through run() itself, so the Run in-flight lock actually engages', async () => { // CLAUDE.md, Run launch synchronization: the lock exists so a double click // cannot create duplicate sessions. A hardcoded dispatch table bypassing