From f127895ecdb7afcf949d71e999cfac71f7d57544 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Troed=20S=C3=A5ngberg?= Date: Thu, 8 Oct 2026 09:21:34 +0000 Subject: [PATCH] feat: Strata support (detect via /health, read /metrics.live) --- README.md | 28 +++++++++- src/stats.ts | 64 ++++++++++++++++++++++ test-backoff.ts | 101 ++++++++++++++++++++++++++++++++++- tui.tsx | 137 +++++++++++++++++++++++++++++++++++++++++------- 4 files changed, 307 insertions(+), 23 deletions(-) diff --git a/README.md b/README.md index 21bdc47..6f7b376 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ There are many plugins to show tokens per second, but the reason for this one is Another motivation was to display the data of interest, but in a non-intrusive way with no UI elements jumping around. This plugin thus only shows a single numeric value, tokens per second, with an indicator as to whether the model is currently doing prompt processing or inference (token generation). -I'm using the llama-server /slots endpoint to get the needed data, which means if you connect opencode to another provider the plugin will just display "-" since it's not getting any data to display. +I'm using the llama-server /slots endpoint to get the needed data. Strata (a llama.cpp-derived server with its own HTTP layer) is also supported: it exposes live prefill/decode rates under /metrics, and the plugin detects it automatically. If you connect opencode to an unrelated provider the plugin will just display "-" since it's not getting any data to display. Note: As explained in further detail below the data that's displayed has to be deduced from llama-server's output. Sometimes the plugin might display PP for prompt processing while in reality the model is doing TG. If additional developments are made to the llama-server output data the plugin might be able to discern between them in a better way, but I think is as good as it gets for now. @@ -63,9 +63,21 @@ If your llama-server provider is named differently, configure the endpoint expli A `baseURL`-style value ending in `/v1` is also accepted. +The plugin detects each server's type automatically by probing `GET /health`: a Strata server reports `"service": "strata"`. You can bypass detection when needed with a `serverType` option: + +```json +{ + "plugin": [ + ["@troed/oc-ls-stats@latest", { "serverType": "strata" }] + ] +} +``` + +`serverType` accepts `"llama"` or `"strata"`; omit it (or set `"auto"`) for automatic detection. + ### Slot Polling -Every 500ms, the plugin polls `GET /slots?model=` on each discovered server. The model parameter is required by the `/slots` endpoint. If the model cannot be discovered from the current route's session, the plugin skips polling. +Every 500ms, the plugin polls each discovered server. For a llama.cpp server it calls `GET /slots?model=` (the model parameter is required by that endpoint; if the model cannot be discovered from the current route's session, the plugin skips polling). For a Strata server it calls `GET /metrics` and reads the `live` object instead. When a server returns an error (e.g. HTTP 502) or is unreachable, the plugin backs off with an exponentially increasing delay per server (1s, 2s, 4s, ...) capped at 10s, with a small random jitter to stagger multiple servers. Backoff resets as soon as the server responds successfully again. @@ -106,6 +118,18 @@ During generation, the plugin calculates the instantaneous token generation rate Slot reuse is tracked via `generateSlotId` to detect when a new generation starts on a different slot. +### Strata Support + +Strata is a llama.cpp-derived server with its own Python HTTP layer. Its `/slots` route is a minimal llama-server compatibility stub (`id`, `n_ctx`, `is_processing` only) with no `next_token`, `id_task`, or `n_prompt_tokens`, so the delta-based classification above cannot produce a rate against it. + +Strata instead reports the live rates directly in `GET /metrics`, under `live`: + +- `state` — `reading` (prompt processing), `generating`, `idle`, or `unloaded` +- `prefill_tok_s_mean` — prompt processing rate while `reading` (PP) +- `tok_s` — instantaneous decode rate while `generating` (TG) + +The plugin detects Strata per server URL from `GET /health` (`service: "strata"`) and maps these fields straight onto the display, so PP/TG are shown without delta estimation. Detection is cached per URL, and servers of both kinds can be polled side by side. Like the `/slots` requests, the plugin sends no API key; if Strata is configured to require one, it cannot read the stats. + ## Limitations ### Progress Percentage diff --git a/src/stats.ts b/src/stats.ts index 443cf53..85b5252 100644 --- a/src/stats.ts +++ b/src/stats.ts @@ -308,3 +308,67 @@ export function selectSessionUrls( } return [] } + +// --- Strata adapter --------------------------------------------------------- +// Strata is a llama.cpp-derived server whose HTTP layer is Python. Its /slots +// route is a minimal compatibility stub, but /metrics.live carries the live +// rates directly: prefill_tok_s_mean while reading, tok_s while generating. + +export type ServerType = "llama" | "strata" + +/** Returns the explicit server type from the plugin option, or undefined for auto. */ +export function serverTypeOverride(value: unknown): ServerType | undefined { + if (value === "llama" || value === "strata") return value + return undefined +} + +/** True when a parsed /health body identifies the server as Strata. */ +export function isStrataHealth(health: unknown): boolean { + if (!health || typeof health !== "object") return false + return (health as Record).service === "strata" +} + +function finiteOrZero(value: unknown): number { + return typeof value === "number" && Number.isFinite(value) ? value : 0 +} + +/** + * Map a Strata `/metrics.live` object onto the shared tracker. Strata reports + * the rates itself, so no delta classification is needed: `state` says which + * phase is active and `prefill_tok_s_mean` / `tok_s` carry the rate. + */ +export function mapStrataLive(live: unknown, tracker: TrackerState): void { + const l = live && typeof live === "object" ? (live as Record) : {} + const state = typeof l.state === "string" ? l.state : "" + + if (state === "reading") { + tracker.isPrefilling = true + tracker.isGenerating = false + tracker.prefillSlotId = 0 + tracker.lastPrefillRate = finiteOrZero(l.prefill_tok_s_mean) + tracker.generateSlotId = null + tracker.generatePrevNd = 0 + tracker.generateStartAt = 0 + tracker.lastGeneratedTps = 0 + } else if (state === "generating") { + tracker.isGenerating = true + tracker.isPrefilling = false + tracker.generateSlotId = 0 + tracker.lastGeneratedTps = finiteOrZero(l.tok_s) + tracker.prefillSlotId = null + tracker.prefillCapturedTokens = null + tracker.prefillStartAt = 0 + tracker.lastPrefillRate = 0 + } else { + tracker.isPrefilling = false + tracker.isGenerating = false + tracker.prefillSlotId = null + tracker.prefillCapturedTokens = null + tracker.prefillStartAt = 0 + tracker.lastPrefillRate = 0 + tracker.generateSlotId = null + tracker.generatePrevNd = 0 + tracker.generateStartAt = 0 + tracker.lastGeneratedTps = 0 + } +} diff --git a/test-backoff.ts b/test-backoff.ts index 752a8bd..632f40d 100644 --- a/test-backoff.ts +++ b/test-backoff.ts @@ -1,6 +1,14 @@ import { test } from "node:test" import assert from "node:assert/strict" -import { backoffDelayMs, resolveServerUrls, selectSessionUrls } from "./src/stats.ts" +import { + backoffDelayMs, + resolveServerUrls, + selectSessionUrls, + serverTypeOverride, + isStrataHealth, + mapStrataLive, + type TrackerState, +} from "./src/stats.ts" test("first failure retries after the base delay", () => { assert.equal(backoffDelayMs(1), 1000) @@ -289,3 +297,94 @@ test("V2 provider settings still excludes non-candidate providers", () => { const providers = [{ id: "llama-local", name: "llama.cpp", settings: { baseURL: "http://other:1" } }] assert.deepEqual(selectSessionUrls(["http://localhost:8080"], providers, "llama-local"), []) }) + +// --- Strata adapter --------------------------------------------------------- +// Strata (a llama.cpp-derived server with a Python HTTP layer) exposes a +// minimal /slots stub and puts the live rates in /metrics.live instead. The +// plugin detects Strata per server URL and maps those rates directly, rather +// than deriving PP/TG from /slots deltas. + +function makeTracker(): TrackerState { + return { + prevNdBySlot: {}, + prevPromptTokensBySlot: {}, + isPrefilling: false, + isGenerating: false, + prefillSlotId: null, + prefillCapturedTokens: null, + prefillStartAt: 0, + lastPrefillRate: 0, + generateSlotId: null, + generatePrevNd: 0, + generateStartAt: 0, + lastGeneratedTps: 0, + } +} + +test("serverTypeOverride accepts only an explicit llama or strata", () => { + assert.equal(serverTypeOverride("strata"), "strata") + assert.equal(serverTypeOverride("llama"), "llama") + assert.equal(serverTypeOverride("auto"), undefined) + assert.equal(serverTypeOverride(undefined), undefined) + assert.equal(serverTypeOverride(42), undefined) +}) + +test("isStrataHealth detects the strata service field", () => { + assert.equal(isStrataHealth({ status: "ok", service: "strata" }), true) + assert.equal(isStrataHealth({ status: "ok" }), false) + assert.equal(isStrataHealth({ service: "llama" }), false) + assert.equal(isStrataHealth(null), false) + assert.equal(isStrataHealth("strata"), false) +}) + +test("mapStrataLive reading sets PP from prefill_tok_s_mean", () => { + const t = makeTracker() + mapStrataLive({ state: "reading", prefill_tok_s_mean: 1247.4, tok_s: null }, t) + assert.equal(t.isPrefilling, true) + assert.equal(t.isGenerating, false) + assert.equal(t.lastPrefillRate, 1247.4) +}) + +test("mapStrataLive generating sets TG from tok_s", () => { + const t = makeTracker() + mapStrataLive({ state: "generating", tok_s: 25.6, prefill_tok_s_mean: null }, t) + assert.equal(t.isGenerating, true) + assert.equal(t.isPrefilling, false) + assert.equal(t.lastGeneratedTps, 25.6) +}) + +test("mapStrataLive idle clears both phases", () => { + const t = makeTracker() + t.isGenerating = true + t.lastGeneratedTps = 42 + mapStrataLive({ state: "idle", tok_s: null, prefill_tok_s_mean: null }, t) + assert.equal(t.isPrefilling, false) + assert.equal(t.isGenerating, false) + assert.equal(t.lastGeneratedTps, 0) + assert.equal(t.lastPrefillRate, 0) +}) + +test("mapStrataLive unloaded clears both phases", () => { + const t = makeTracker() + t.isPrefilling = true + t.lastPrefillRate = 99 + mapStrataLive({ state: "unloaded" }, t) + assert.equal(t.isPrefilling, false) + assert.equal(t.isGenerating, false) +}) + +test("mapStrataLive tolerates missing live data", () => { + const t = makeTracker() + t.isGenerating = true + t.lastGeneratedTps = 10 + mapStrataLive(null, t) + assert.equal(t.isGenerating, false) + assert.equal(t.lastGeneratedTps, 0) +}) + +test("mapStrataLive treats non-numeric rates as zero", () => { + const t = makeTracker() + mapStrataLive({ state: "generating", tok_s: null }, t) + assert.equal(t.isGenerating, true) + assert.equal(t.lastGeneratedTps, 0) +}) diff --git a/tui.tsx b/tui.tsx index aedb1e7..092b735 100644 --- a/tui.tsx +++ b/tui.tsx @@ -2,7 +2,16 @@ import type { TextRenderable } from "@opentui/core" import type { TuiPlugin, TuiPluginModule } from "@opencode-ai/plugin/tui" import { onCleanup, createSignal } from "solid-js" -import { backoffDelayMs, processPoll, resolveServerUrls, selectSessionUrls } from "./src/stats.ts" +import { + backoffDelayMs, + processPoll, + resolveServerUrls, + selectSessionUrls, + serverTypeOverride, + isStrataHealth, + mapStrataLive, + type ServerType, +} from "./src/stats.ts" type StreamSample = { at: number @@ -54,6 +63,30 @@ async function fetchSlots(baseUrl: string, model?: string): Promise { + try { + const resp = await fetch(`${baseUrl}/health`) + if (!resp.ok) return null + return await resp.json() + } catch { + return null + } +} + +// Strata reports the live rates under /metrics.live (prefill_tok_s_mean while +// reading, tok_s while generating) rather than in the minimal /slots stub. +async function fetchStrataLive(baseUrl: string): Promise { + try { + const resp = await fetch(`${baseUrl}/metrics`) + if (!resp.ok) return null + const body = await resp.json() + if (body && typeof body === "object" && "live" in body) return (body as { live: unknown }).live + return null + } catch { + return null + } +} + function formatPps(value: number) { if (!Number.isFinite(value) || value <= 0) return undefined return `${Math.round(value)}` @@ -179,6 +212,33 @@ const tui: TuiPlugin = async (api, options) => { let llamaServerUrls: string[] = [] let llamaServerModel: string | undefined const backoffByUrl = new Map() + const serverKindByUrl = new Map() + const forcedServerType = serverTypeOverride((options as Record | undefined)?.serverType) + + const resolveServerKind = async (baseUrl: string): Promise => { + if (forcedServerType) return forcedServerType + const cached = serverKindByUrl.get(baseUrl) + if (cached) return cached + const health = await fetchHealth(baseUrl) + if (health != null) { + const kind = isStrataHealth(health) ? "strata" : "llama" + serverKindByUrl.set(baseUrl, kind) + return kind + } + return "llama" + } + + const noteServerFailure = (baseUrl: string) => { + const failures = (backoffByUrl.get(baseUrl)?.failures ?? 0) + 1 + const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS) + const jitter = Math.floor(Math.random() * (delay / 4)) + backoffByUrl.set(baseUrl, { + failures, + nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS), + }) + tracker.failure = "n/a" + bump() + } const pollMetrics = async () => { if (!llamaServerUrls.length) return @@ -211,17 +271,22 @@ const tui: TuiPlugin = async (api, options) => { continue } + if ((await resolveServerKind(baseUrl)) === "strata") { + const live = await fetchStrataLive(baseUrl) + if (live == null) { + noteServerFailure(baseUrl) + continue + } + backoffByUrl.delete(baseUrl) + tracker.failure = null + mapStrataLive(live, tracker) + bump() + continue + } + const slots = await fetchSlots(baseUrl, model) if (slots == null) { - const failures = (backoff?.failures ?? 0) + 1 - const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS) - const jitter = Math.floor(Math.random() * (delay / 4)) - backoffByUrl.set(baseUrl, { - failures, - nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS), - }) - tracker.failure = "n/a" - bump() + noteServerFailure(baseUrl) continue } @@ -235,7 +300,7 @@ const tui: TuiPlugin = async (api, options) => { tracker.failure = null bump() - const result = processPoll(slotList, tracker) + processPoll(slotList, tracker) bump() } } @@ -436,6 +501,33 @@ const v2Setup = async (context: any) => { let llamaServerUrls: string[] = [] let providerList: unknown[] = [] const backoffByUrl = new Map() + const serverKindByUrl = new Map() + const forcedServerType = serverTypeOverride((context.options as Record | undefined)?.serverType) + + const resolveServerKind = async (baseUrl: string): Promise => { + if (forcedServerType) return forcedServerType + const cached = serverKindByUrl.get(baseUrl) + if (cached) return cached + const health = await fetchHealth(baseUrl) + if (health != null) { + const kind = isStrataHealth(health) ? "strata" : "llama" + serverKindByUrl.set(baseUrl, kind) + return kind + } + return "llama" + } + + const noteServerFailure = (baseUrl: string) => { + const failures = (backoffByUrl.get(baseUrl)?.failures ?? 0) + 1 + const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS) + const jitter = Math.floor(Math.random() * (delay / 4)) + backoffByUrl.set(baseUrl, { + failures, + nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS), + }) + tracker.failure = "n/a" + bump() + } const currentSessionID = (): string | undefined => { const route = context.ui.router.current?.() @@ -473,17 +565,22 @@ const v2Setup = async (context: any) => { continue } + if ((await resolveServerKind(baseUrl)) === "strata") { + const live = await fetchStrataLive(baseUrl) + if (live == null) { + noteServerFailure(baseUrl) + continue + } + backoffByUrl.delete(baseUrl) + tracker.failure = null + mapStrataLive(live, tracker) + bump() + continue + } + const slots = await fetchSlots(baseUrl, model) if (slots == null) { - const failures = (backoff?.failures ?? 0) + 1 - const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS) - const jitter = Math.floor(Math.random() * (delay / 4)) - backoffByUrl.set(baseUrl, { - failures, - nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS), - }) - tracker.failure = "n/a" - bump() + noteServerFailure(baseUrl) continue }