From f85419b7b76f6a493d64893ab30794ff2110ada9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Troed=20S=C3=A5ngberg?= Date: Thu, 8 Oct 2026 09:55:57 +0000 Subject: [PATCH] feat: send the session model on all stats calls --- README.md | 2 +- src/stats.ts | 20 ++++++++++++++++++++ test-backoff.ts | 23 +++++++++++++++++++++++ tui.tsx | 43 ++++++++++++++++++++++++------------------- 4 files changed, 68 insertions(+), 20 deletions(-) diff --git a/README.md b/README.md index 6f7b376..f3e12fc 100644 --- a/README.md +++ b/README.md @@ -77,7 +77,7 @@ The plugin detects each server's type automatically by probing `GET /health`: a ### Slot Polling -Every 500ms, the plugin polls each discovered server. For a llama.cpp server it calls `GET /slots?model=` (the model parameter is required by that endpoint; if the model cannot be discovered from the current route's session, the plugin skips polling). For a Strata server it calls `GET /metrics` and reads the `live` object instead. +Every 500ms, the plugin polls each discovered server, always carrying the current session's model as `?model=`. For a llama.cpp server it calls `GET /slots?model=`; for a Strata server it calls `GET /health?model=` (to identify the server) and `GET /metrics?model=` (reading the `live` object). Plain llama.cpp and Strata servers ignore the extra parameter; a routing layer that dispatches by model (e.g. one fronting several nodes) uses it to forward the call to the backend serving that model. If the model cannot be discovered from the current route's session, the plugin polls without it — direct servers still answer, while a model-dispatched router rejects the bare call and the plugin shows `n/a`. When a server returns an error (e.g. HTTP 502) or is unreachable, the plugin backs off with an exponentially increasing delay per server (1s, 2s, 4s, ...) capped at 10s, with a small random jitter to stagger multiple servers. Backoff resets as soon as the server responds successfully again. diff --git a/src/stats.ts b/src/stats.ts index 85b5252..1f36966 100644 --- a/src/stats.ts +++ b/src/stats.ts @@ -372,3 +372,23 @@ export function mapStrataLive(live: unknown, tracker: TrackerState): void { tracker.lastGeneratedTps = 0 } } + +// --- Model-scoped stats URLs ------------------------------------------------ +// Stats are requested with the session's model so a routing layer that +// dispatches by model (a LAN router in front of several nodes) can forward the +// call to the right backend. Plain llama.cpp / Strata servers ignore the extra +// query parameter. + +/** Build a stats URL, adding the model as `?model=` when one is known. */ +export function statsUrl(baseUrl: string, path: string, model?: string): string { + return model ? `${baseUrl}${path}?model=${encodeURIComponent(model)}` : `${baseUrl}${path}` +} + +/** + * Cache key for the detected server kind. One URL can front different backends + * per model (e.g. a router), so the model belongs in the key: the same conduct + * URL may be a Strata node for one model and a llama.cpp node for another. + */ +export function serverKindKey(baseUrl: string, model?: string): string { + return `${baseUrl}\u0000${model ?? ""}` +} diff --git a/test-backoff.ts b/test-backoff.ts index 632f40d..a52ce16 100644 --- a/test-backoff.ts +++ b/test-backoff.ts @@ -7,6 +7,8 @@ import { serverTypeOverride, isStrataHealth, mapStrataLive, + statsUrl, + serverKindKey, type TrackerState, } from "./src/stats.ts" @@ -388,3 +390,24 @@ test("mapStrataLive treats non-numeric rates as zero", () => { assert.equal(t.isGenerating, true) assert.equal(t.lastGeneratedTps, 0) }) + +// --- Model-scoped stats URLs ------------------------------------------------ +// The plugin polls stats with the session's model so a routing layer in front +// of the backend (one that dispatches by model) can forward the call to the +// right node. Plain llama.cpp / Strata servers ignore the extra parameter. + +test("statsUrl appends the model as a query parameter", () => { + assert.equal(statsUrl("http://h:8100", "/metrics", "m"), "http://h:8100/metrics?model=m") + assert.equal(statsUrl("http://h:8100", "/health", "qwen 3"), "http://h:8100/health?model=qwen%203") +}) + +test("statsUrl omits the parameter when there is no model", () => { + assert.equal(statsUrl("http://h:8100", "/slots"), "http://h:8100/slots") + assert.equal(statsUrl("http://h:8100", "/metrics", ""), "http://h:8100/metrics") +}) + +test("serverKindKey distinguishes models on the same server", () => { + assert.equal(serverKindKey("http://h:8100", "a"), serverKindKey("http://h:8100", "a")) + assert.notEqual(serverKindKey("http://h:8100", "a"), serverKindKey("http://h:8100", "b")) + assert.notEqual(serverKindKey("http://h:8100", "a"), serverKindKey("http://h:8100")) +}) diff --git a/tui.tsx b/tui.tsx index 092b735..9e06818 100644 --- a/tui.tsx +++ b/tui.tsx @@ -10,6 +10,8 @@ import { serverTypeOverride, isStrataHealth, mapStrataLive, + statsUrl, + serverKindKey, type ServerType, } from "./src/stats.ts" @@ -54,8 +56,7 @@ function estimateStreamTokens(delta: string) { async function fetchSlots(baseUrl: string, model?: string): Promise { try { - const url = model ? `${baseUrl}/slots?model=${encodeURIComponent(model)}` : `${baseUrl}/slots` - const resp = await fetch(url) + const resp = await fetch(statsUrl(baseUrl, "/slots", model)) if (!resp.ok) return null return await resp.json() } catch { @@ -63,9 +64,9 @@ async function fetchSlots(baseUrl: string, model?: string): Promise { +async function fetchHealth(baseUrl: string, model?: string): Promise { try { - const resp = await fetch(`${baseUrl}/health`) + const resp = await fetch(statsUrl(baseUrl, "/health", model)) if (!resp.ok) return null return await resp.json() } catch { @@ -74,10 +75,12 @@ async function fetchHealth(baseUrl: string): Promise { } // Strata reports the live rates under /metrics.live (prefill_tok_s_mean while -// reading, tok_s while generating) rather than in the minimal /slots stub. -async function fetchStrataLive(baseUrl: string): Promise { +// reading, tok_s while generating) rather than in the minimal /slots stub. The +// model is passed so a routing layer in front of the backend can forward the +// call; plain llama.cpp / Strata servers ignore the parameter. +async function fetchStrataLive(baseUrl: string, model?: string): Promise { try { - const resp = await fetch(`${baseUrl}/metrics`) + const resp = await fetch(statsUrl(baseUrl, "/metrics", model)) if (!resp.ok) return null const body = await resp.json() if (body && typeof body === "object" && "live" in body) return (body as { live: unknown }).live @@ -215,14 +218,15 @@ const tui: TuiPlugin = async (api, options) => { const serverKindByUrl = new Map() const forcedServerType = serverTypeOverride((options as Record | undefined)?.serverType) - const resolveServerKind = async (baseUrl: string): Promise => { + const resolveServerKind = async (baseUrl: string, model?: string): Promise => { if (forcedServerType) return forcedServerType - const cached = serverKindByUrl.get(baseUrl) + const key = serverKindKey(baseUrl, model) + const cached = serverKindByUrl.get(key) if (cached) return cached - const health = await fetchHealth(baseUrl) + const health = await fetchHealth(baseUrl, model) if (health != null) { const kind = isStrataHealth(health) ? "strata" : "llama" - serverKindByUrl.set(baseUrl, kind) + serverKindByUrl.set(key, kind) return kind } return "llama" @@ -271,8 +275,8 @@ const tui: TuiPlugin = async (api, options) => { continue } - if ((await resolveServerKind(baseUrl)) === "strata") { - const live = await fetchStrataLive(baseUrl) + if ((await resolveServerKind(baseUrl, model)) === "strata") { + const live = await fetchStrataLive(baseUrl, model) if (live == null) { noteServerFailure(baseUrl) continue @@ -504,14 +508,15 @@ const v2Setup = async (context: any) => { const serverKindByUrl = new Map() const forcedServerType = serverTypeOverride((context.options as Record | undefined)?.serverType) - const resolveServerKind = async (baseUrl: string): Promise => { + const resolveServerKind = async (baseUrl: string, model?: string): Promise => { if (forcedServerType) return forcedServerType - const cached = serverKindByUrl.get(baseUrl) + const key = serverKindKey(baseUrl, model) + const cached = serverKindByUrl.get(key) if (cached) return cached - const health = await fetchHealth(baseUrl) + const health = await fetchHealth(baseUrl, model) if (health != null) { const kind = isStrataHealth(health) ? "strata" : "llama" - serverKindByUrl.set(baseUrl, kind) + serverKindByUrl.set(key, kind) return kind } return "llama" @@ -565,8 +570,8 @@ const v2Setup = async (context: any) => { continue } - if ((await resolveServerKind(baseUrl)) === "strata") { - const live = await fetchStrataLive(baseUrl) + if ((await resolveServerKind(baseUrl, model)) === "strata") { + const live = await fetchStrataLive(baseUrl, model) if (live == null) { noteServerFailure(baseUrl) continue