mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
feat(custom-model): show real-time llama.cpp backend status in the loading banner
Answers the underlying request behind investigating llama.cpp log
access: surface what the backend is actually doing, live, on top of
the existing countdown timer during a model load.
- getLatestLlamaSwapLogLine()/pruneIdleLlamaSwapLogTails()
(custom-model-routes.ts): one persistent GET /api/events (SSE)
connection held open per endpoint, parsing logData frames and
keeping the latest source:"upstream" (backend llama-server) line —
filtering out llama-swap's own source:"proxy" request-access lines.
Idle-closed after 30s of no polling, same 20s sweep as the existing
swap-displacement check.
- running-status route now returns logLine alongside the existing
isLlamaSwap/running fields.
- Frontend: _watchLlamaSwapLoading's banner gains a second line
("llama.cpp: <line>", bootlog timestamp/level/component prefix
stripped for display) that stays on the last real thing llama.cpp
said rather than clearing to blank between polls.
⚠️ Caught and fixed before merge, not after: the first cut targeted
GET /logs (the endpoint the name suggests), shipped a working-looking
implementation with passing tests, and only failed a live check against
the real Nemesis llama-swap deployment — /logs turns out to carry ONLY
llama-swap's own proxy request-access log and never once showed a
single backend line, even seconds after a real, confirmed model swap
triggered via a direct API call. GET /api/events's logData frames
(with an explicit source field distinguishing upstream from proxy) are
the only source that actually has backend output; corrected and
re-verified live end-to-end through an actual forced swap before
writing this commit, confirmed live to hold its connection open
indefinitely (unlike /logs, which closes after a fixed ~100KB).
12 tests for the corrected /api/events parsing (SSE frame buffering
across chunk boundaries, source filtering, malformed/wrong-type frames,
connection reuse, idle pruning) plus 2 for the frontend banner
rendering. Typecheck/lint/frontend-syntax clean; full suite shows no
new regressions (14 more passing than baseline, matching the new
tests; same pre-existing Windows-environment failures).
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
5ddc028a2f
commit
2d3fc65758
@@ -626,6 +626,54 @@ describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () =>
|
||||
expect(toastCalls.at(-1)).toMatch(/ready/i);
|
||||
});
|
||||
|
||||
it('adds a second line with the real llama.cpp log line once one is available, stripped of the bootlog prefix', async () => {
|
||||
const { app } = bootApp({});
|
||||
const bannerMessages: string[] = [];
|
||||
app._showCenterStatus = (message: string) => {
|
||||
bannerMessages.push(message);
|
||||
return { dismiss: () => {}, setMessage: (next: string) => bannerMessages.push(next) };
|
||||
};
|
||||
app.showToast = () => {};
|
||||
let statusCalls = 0;
|
||||
app._apiJson = async (path: string) => {
|
||||
if (path === '/api/model-endpoints') return null; // size lookup — unrelated to this test
|
||||
statusCalls += 1;
|
||||
if (statusCalls === 1) {
|
||||
return {
|
||||
isLlamaSwap: true,
|
||||
running: [{ model: 'qwen3', state: 'starting' }],
|
||||
logLine: '0.31.428.568 I srv llama_server: model loaded',
|
||||
};
|
||||
}
|
||||
return { isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] };
|
||||
};
|
||||
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 'sess-1', 5, 200);
|
||||
|
||||
// First render (before any poll has landed) has no log line at all.
|
||||
expect(bannerMessages[0]).not.toMatch(/llama\.cpp:/);
|
||||
// Second render carries the log line, bootlog prefix (timestamp/level/component) stripped.
|
||||
const withLogLine = bannerMessages.find((m) => m.includes('llama.cpp:'));
|
||||
expect(withLogLine).toContain('llama.cpp: llama_server: model loaded');
|
||||
expect(withLogLine).not.toContain('0.31.428.568');
|
||||
expect(withLogLine).not.toContain(' I srv');
|
||||
});
|
||||
|
||||
it('shows no second line at all when the endpoint has no logLine to offer', async () => {
|
||||
const { app } = bootApp({});
|
||||
const bannerMessages: string[] = [];
|
||||
app._showCenterStatus = (message: string) => {
|
||||
bannerMessages.push(message);
|
||||
return { dismiss: () => {}, setMessage: (next: string) => bannerMessages.push(next) };
|
||||
};
|
||||
app.showToast = () => {};
|
||||
app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] });
|
||||
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 'sess-1', 5, 200);
|
||||
|
||||
expect(bannerMessages.some((m) => m.includes('llama.cpp:'))).toBe(false);
|
||||
});
|
||||
|
||||
it('gives up after the bounded wait, turns the banner into a sticky error, and closes the session', async () => {
|
||||
const { app } = bootApp({});
|
||||
const banners: Array<{ message: string; opts: unknown }> = [];
|
||||
|
||||
Reference in New Issue
Block a user