feat(custom-model): live countdown on the loading banner; timeout is now an error

The loading banner now shows a live countdown against its own timeout
(updated every poll, so every second by default) instead of a static
"this can take a while" — e.g. "Loading qwen3.8-27b (16.4 GB, typically
~1-3 min) on llama-swap - 47s remaining".

If the countdown reaches zero and the model still isn't ready, this is now
treated as a real failure rather than a "keep waiting" shrug:
- The banner turns into a sticky error (_showCenterStatus gains a `type`
  option - 'error' drops the spinner and adds a close button, since nothing
  is "in progress" anymore and a sticky message needs a way to dismiss it),
  naming the llama-swap server's own logs as where to look for detail.
- The session that load was for is closed automatically (closeSession) -
  requested explicitly: a console left open and pointed at a model that
  never finished loading is worse than no console at all. Both apply paths
  now thread the new session's id through to _watchLlamaSwapLoading for
  this (new required 3rd parameter, after endpointId/modelId).

_watchLlamaSwapGeneration's existing stale-call guard extends naturally to
this: a superseded call's own eventual timeout recognises it no longer owns
the banner and neither shows the error nor closes a session that may by
then belong to a different, newer launch.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
Devvyn
2026-09-16 20:22:02 +08:00
co-authored by Claude Sonnet 5
parent 55dae31530
commit 7bbe408e44
7 changed files with 181 additions and 64 deletions
+40 -18
View File
@@ -5556,37 +5556,59 @@ Object.assign(CodemanApp.prototype, {
* genuinely worth interrupting the eye for rather than living in the corner with every
* other toast (currently: a custom-model session's "switching backends" and "loading
* model" states, both of which can sit on screen for well over a minute and are easy to
* mistake for nothing happening). Non-blocking (`pointer-events: none`, no backdrop) —
* this is informational, never a gate the user has to dismiss to keep working. Only one
* is ever shown at a time (the DOM node is created once and reused), which matches every
* current caller: each hands off to the next rather than stacking.
* mistake for nothing happening). Non-blocking (`pointer-events: none` on the wrapper,
* restored only on the card) — an info banner is never a gate the user has to dismiss to
* keep working. Only one is ever shown at a time (the DOM node is created once and
* reused), which matches every current caller: each hands off to the next rather than
* stacking.
*
* `opts.type` — `'info'` (default, spinner, no close button — a caller ends it itself via
* `dismiss()`) or `'error'` (no spinner — nothing is in progress once this shows — with a
* close button, since a sticky error the user cannot dismiss would just sit there). The
* DOM is rebuilt fresh each call rather than patched, since which children exist differs
* by type; `setMessage` still only ever touches the text node afterwards.
*/
_showCenterStatus(message) {
_showCenterStatus(message, opts = {}) {
const { type = 'info' } = opts;
let el = document.getElementById('customModelCenterStatus');
if (!el) {
el = document.createElement('div');
el.id = 'customModelCenterStatus';
el.className = 'center-status-banner';
document.body.appendChild(el);
}
el.className = `center-status-banner center-status-${type}`;
el.innerHTML = '';
const dismiss = () => {
el.classList.remove('show');
setTimeout(() => {
el.hidden = true;
}, 200);
};
if (type !== 'error') {
const spinner = document.createElement('span');
spinner.className = 'center-status-spinner';
spinner.setAttribute('aria-hidden', 'true');
const text = document.createElement('span');
text.className = 'center-status-text';
el.appendChild(spinner);
el.appendChild(text);
document.body.appendChild(el);
}
const textEl = el.querySelector('.center-status-text');
if (textEl) textEl.textContent = message;
const text = document.createElement('span');
text.className = 'center-status-text';
text.textContent = message;
el.appendChild(text);
if (type === 'error') {
const closeBtn = document.createElement('button');
closeBtn.className = 'center-status-close';
closeBtn.textContent = '×';
closeBtn.setAttribute('aria-label', 'Dismiss');
closeBtn.onclick = (e) => {
e.stopPropagation();
dismiss();
};
el.appendChild(closeBtn);
}
el.hidden = false;
requestAnimationFrame(() => el.classList.add('show'));
return {
dismiss: () => {
el.classList.remove('show');
setTimeout(() => {
el.hidden = true;
}, 200);
},
dismiss,
setMessage: (next) => {
const t = el.querySelector('.center-status-text');
if (t) t.textContent = next;
+44 -16
View File
@@ -765,7 +765,7 @@ Object.assign(CodemanApp.prototype, {
const result = this._lastCustomModelLaunchResult;
this._lastCustomModelLaunchResult = undefined;
if (result?.modelSwapInProgress) {
void this._watchLlamaSwapLoading(endpointId, modelId);
void this._watchLlamaSwapLoading(endpointId, modelId, result.sessionId);
}
},
@@ -906,7 +906,7 @@ Object.assign(CodemanApp.prototype, {
// Hand off to its own sticky toast rather than stacking a second one on top.
if (payload?.modelSwapInProgress) {
switchingToast?.dismiss();
void this._watchLlamaSwapLoading(endpointId, modelId);
void this._watchLlamaSwapLoading(endpointId, modelId, sessionId);
return;
}
@@ -967,29 +967,43 @@ Object.assign(CodemanApp.prototype, {
return this._MODEL_LOAD_TIME_MATRIX.find((bracket) => sizeGB <= bracket.maxGB) ?? null;
},
/** `ms` -> `"1m 08s remaining"` / `"8s remaining"`, for the loading banner's live countdown. */
_formatRemaining(ms) {
const totalSec = Math.max(0, Math.ceil(ms / 1000));
const mins = Math.floor(totalSec / 60);
const secs = totalSec % 60;
return mins > 0 ? `${mins}m ${String(secs).padStart(2, '0')}s remaining` : `${secs}s remaining`;
},
/**
* Polls llama-swap's own `/running` (via the read-only running-status route) until
* `modelId` reports `state: 'ready'`, showing a sticky banner the whole time so a slow
* unload/reload (measured well over a minute for a large model) reads as "loading",
* never as silence or a wrong answer from whatever was loaded before. Checks immediately
* (a fast load, or a re-apply onto an already-ready model, shouldn't wait a full interval
* to say so), then every `pollIntervalMs`. Bounded at `maxWaitMs` — defaults to a rough,
* size-scaled estimate (`_estimateModelLoad`) when the model's discovered size is known,
* falling back to a flat 5 minutes when it isn't; still not ready by then gets a toast
* saying so rather than polling forever.
* `modelId` reports `state: 'ready'`, showing a sticky banner with a live countdown the
* whole time so a slow unload/reload (measured well over a minute for a large model)
* reads as "loading, N seconds left", never as silence or a wrong answer from whatever
* was loaded before. Checks immediately (a fast load, or a re-apply onto an
* already-ready model, shouldn't wait a full interval to say so), then every
* `pollIntervalMs`. Bounded at `maxWaitMs` — defaults to a rough, size-scaled estimate
* (`_estimateModelLoad`) when the model's discovered size is known, falling back to a
* flat 5 minutes when it isn't.
*
* If the countdown reaches zero with the model still not ready, this is a real failure,
* not a "keep waiting" — the banner turns into a sticky error naming the llama-swap
* server's own logs as where to look, and `sessionId` (the session this was launched
* for) is closed automatically: a console left open and pointed at a model that never
* finished loading is worse than no console at all.
*
* `_watchLlamaSwapGeneration` guards against two overlapping calls (a second launch
* started before the first one's loop finished) clobbering each other's banner:
* `_showCenterStatus` reuses one shared DOM node, so an older loop's `dismiss()`/message
* update firing after a newer one has already taken over the banner would otherwise hide
* or overwrite the WRONG one. Each call claims the counter as its own "generation" and
* checks it still owns it before touching the banner.
* or overwrite the WRONG one, or close the WRONG session. Each call claims the counter
* as its own "generation" and checks it still owns it before touching either.
*
* `pollIntervalMs`/`maxWaitMs` exist to let a test drive this in milliseconds instead of
* minutes — real callers never pass `maxWaitMs`, which is what keeps the size-scaled
* default live here rather than only in a test fixture.
*/
async _watchLlamaSwapLoading(endpointId, modelId, pollIntervalMs = 1000, maxWaitMs) {
async _watchLlamaSwapLoading(endpointId, modelId, sessionId, pollIntervalMs = 1000, maxWaitMs) {
const generation = (this._watchLlamaSwapGeneration = (this._watchLlamaSwapGeneration || 0) + 1);
const isCurrent = () => this._watchLlamaSwapGeneration === generation;
const sizeGB = await this._lookupModelSizeGB(endpointId, modelId);
@@ -999,10 +1013,11 @@ Object.assign(CodemanApp.prototype, {
const sizeSuffix = sizeGB
? ` (${sizeGB.toFixed(1)} GB${estimate ? `, typically ${estimate.label}` : ''})`
: '';
const baseMessage = `Loading ${modelId}${sizeSuffix} on ${endpointId} —`;
// Prominent and screen-centred, not a corner toast — a real llama-swap model load can
// sit on screen for well over a minute, easy to mistake for nothing happening there.
const toast = this._showCenterStatus(`Loading ${modelId}${sizeSuffix} on ${endpointId}… this can take a while`);
const deadline = Date.now() + effectiveMaxWaitMs;
const toast = this._showCenterStatus(`${baseMessage} ${this._formatRemaining(deadline - Date.now())}`);
while (Date.now() < deadline) {
const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(endpointId)}/running-status`);
if (!isCurrent()) return; // a newer launch took over the banner — this loop is done
@@ -1018,11 +1033,24 @@ Object.assign(CodemanApp.prototype, {
this.showToast(`${modelId} is ready`, 'success', { duration: 2500 });
return;
}
if (!isCurrent()) return;
toast?.setMessage(`${baseMessage} ${this._formatRemaining(deadline - Date.now())}`);
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs));
}
if (!isCurrent()) return;
toast?.dismiss();
this.showToast(`Still waiting for ${modelId} to finish loading on ${endpointId} — check the llama-swap server`, 'warning');
this._showCenterStatus(
`${modelId} did not finish loading on ${endpointId} within the expected time. ` +
`Check the llama-swap server logs for details.` +
(sessionId ? ' The session has been closed.' : ''),
{ type: 'error' }
);
if (sessionId) {
try {
await this.closeSession(sessionId);
} catch {
// closeSession already reports its own failure via toast — nothing more to do here
}
}
},
/**
+26
View File
@@ -8577,11 +8577,37 @@ kbd {
}
.center-status-text {
flex: 1;
pointer-events: auto;
white-space: pre-wrap;
word-break: break-word;
}
/* Error variant: the load didn't finish in time — nothing is "in progress" anymore (no
spinner), and since this one doesn't dismiss itself, it needs a close button the user
can actually click, so pointer-events is restored here too (see the wrapper's own
comment on why that's `none` by default). */
.center-status-error {
border-color: rgba(239, 68, 68, 0.5);
}
.center-status-close {
flex-shrink: 0;
pointer-events: auto;
background: none;
border: none;
color: inherit;
opacity: 0.6;
font-size: 1.2rem;
line-height: 1;
padding: 0 0.15rem;
cursor: pointer;
}
.center-status-close:hover {
opacity: 1;
}
.toast-success { border-color: rgba(34, 197, 94, 0.4); }
.toast-error { border-color: rgba(239, 68, 68, 0.4); }
.toast-warning { border-color: rgba(234, 179, 8, 0.4); }