feat: Strata support (detect via /health, read /metrics.live)

This commit is contained in:
Troed Sångberg committed 2026-10-08 09:21:34 +00:00
1 parent 55f5e52a6c
commit f127895ecd
4 files changed
+307 -23

No files matched your search

+26 -2
View File
@@ -6,7 +6,7 @@ There are many plugins to show tokens per second, but the reason for this one is
Another motivation was to display the data of interest, but in a non-intrusive way with no UI elements jumping around. This plugin thus only shows a single numeric value, tokens per second, with an indicator as to whether the model is currently doing prompt processing or inference (token generation).
I'm using the llama-server /slots endpoint to get the needed data, which means if you connect opencode to another provider the plugin will just display "-" since it's not getting any data to display.
I'm using the llama-server /slots endpoint to get the needed data. Strata (a llama.cpp-derived server with its own HTTP layer) is also supported: it exposes live prefill/decode rates under /metrics, and the plugin detects it automatically. If you connect opencode to an unrelated provider the plugin will just display "-" since it's not getting any data to display.
Note: As explained in further detail below the data that's displayed has to be deduced from llama-server's output. Sometimes the plugin might display PP for prompt processing while in reality the model is doing TG. If additional developments are made to the llama-server output data the plugin might be able to discern between them in a better way, but I think is as good as it gets for now.
@@ -63,9 +63,21 @@ If your llama-server provider is named differently, configure the endpoint expli
A `baseURL`-style value ending in `/v1` is also accepted.
The plugin detects each server's type automatically by probing `GET /health`: a Strata server reports `"service": "strata"`. You can bypass detection when needed with a `serverType` option:
```json
{
"plugin": [
["@troed/oc-ls-stats@latest", { "serverType": "strata" }]
]
}
```
`serverType` accepts `"llama"` or `"strata"`; omit it (or set `"auto"`) for automatic detection.
### Slot Polling
Every 500ms, the plugin polls `GET /slots?model=<model>` on each discovered server. The model parameter is required by the `/slots` endpoint. If the model cannot be discovered from the current route's session, the plugin skips polling.
Every 500ms, the plugin polls each discovered server. For a llama.cpp server it calls `GET /slots?model=<model>` (the model parameter is required by that endpoint; if the model cannot be discovered from the current route's session, the plugin skips polling). For a Strata server it calls `GET /metrics` and reads the `live` object instead.
When a server returns an error (e.g. HTTP 502) or is unreachable, the plugin backs off with an exponentially increasing delay per server (1s, 2s, 4s, ...) capped at 10s, with a small random jitter to stagger multiple servers. Backoff resets as soon as the server responds successfully again.
@@ -106,6 +118,18 @@ During generation, the plugin calculates the instantaneous token generation rate
Slot reuse is tracked via `generateSlotId` to detect when a new generation starts on a different slot.
### Strata Support
Strata is a llama.cpp-derived server with its own Python HTTP layer. Its `/slots` route is a minimal llama-server compatibility stub (`id`, `n_ctx`, `is_processing` only) with no `next_token`, `id_task`, or `n_prompt_tokens`, so the delta-based classification above cannot produce a rate against it.
Strata instead reports the live rates directly in `GET /metrics`, under `live`:
- `state` — `reading` (prompt processing), `generating`, `idle`, or `unloaded`
- `prefill_tok_s_mean` — prompt processing rate while `reading` (PP)
- `tok_s` — instantaneous decode rate while `generating` (TG)
The plugin detects Strata per server URL from `GET /health` (`service: "strata"`) and maps these fields straight onto the display, so PP/TG are shown without delta estimation. Detection is cached per URL, and servers of both kinds can be polled side by side. Like the `/slots` requests, the plugin sends no API key; if Strata is configured to require one, it cannot read the stats.
## Limitations
### Progress Percentage
+64
View File
@@ -308,3 +308,67 @@ export function selectSessionUrls(
}
return []
}
// --- Strata adapter ---------------------------------------------------------
// Strata is a llama.cpp-derived server whose HTTP layer is Python. Its /slots
// route is a minimal compatibility stub, but /metrics.live carries the live
// rates directly: prefill_tok_s_mean while reading, tok_s while generating.
export type ServerType = "llama" | "strata"
/** Returns the explicit server type from the plugin option, or undefined for auto. */
export function serverTypeOverride(value: unknown): ServerType | undefined {
if (value === "llama" || value === "strata") return value
return undefined
}
/** True when a parsed /health body identifies the server as Strata. */
export function isStrataHealth(health: unknown): boolean {
if (!health || typeof health !== "object") return false
return (health as Record<string, unknown>).service === "strata"
}
function finiteOrZero(value: unknown): number {
return typeof value === "number" && Number.isFinite(value) ? value : 0
}
/**
* Map a Strata `/metrics.live` object onto the shared tracker. Strata reports
* the rates itself, so no delta classification is needed: `state` says which
* phase is active and `prefill_tok_s_mean` / `tok_s` carry the rate.
*/
export function mapStrataLive(live: unknown, tracker: TrackerState): void {
const l = live && typeof live === "object" ? (live as Record<string, unknown>) : {}
const state = typeof l.state === "string" ? l.state : ""
if (state === "reading") {
tracker.isPrefilling = true
tracker.isGenerating = false
tracker.prefillSlotId = 0
tracker.lastPrefillRate = finiteOrZero(l.prefill_tok_s_mean)
tracker.generateSlotId = null
tracker.generatePrevNd = 0
tracker.generateStartAt = 0
tracker.lastGeneratedTps = 0
} else if (state === "generating") {
tracker.isGenerating = true
tracker.isPrefilling = false
tracker.generateSlotId = 0
tracker.lastGeneratedTps = finiteOrZero(l.tok_s)
tracker.prefillSlotId = null
tracker.prefillCapturedTokens = null
tracker.prefillStartAt = 0
tracker.lastPrefillRate = 0
} else {
tracker.isPrefilling = false
tracker.isGenerating = false
tracker.prefillSlotId = null
tracker.prefillCapturedTokens = null
tracker.prefillStartAt = 0
tracker.lastPrefillRate = 0
tracker.generateSlotId = null
tracker.generatePrevNd = 0
tracker.generateStartAt = 0
tracker.lastGeneratedTps = 0
}
}
+100 -1
View File
@@ -1,6 +1,14 @@
import { test } from "node:test"
import assert from "node:assert/strict"
import { backoffDelayMs, resolveServerUrls, selectSessionUrls } from "./src/stats.ts"
import {
backoffDelayMs,
resolveServerUrls,
selectSessionUrls,
serverTypeOverride,
isStrataHealth,
mapStrataLive,
type TrackerState,
} from "./src/stats.ts"
test("first failure retries after the base delay", () => {
assert.equal(backoffDelayMs(1), 1000)
@@ -289,3 +297,94 @@ test("V2 provider settings still excludes non-candidate providers", () => {
const providers = [{ id: "llama-local", name: "llama.cpp", settings: { baseURL: "http://other:1" } }]
assert.deepEqual(selectSessionUrls(["http://localhost:8080"], providers, "llama-local"), [])
})
// --- Strata adapter ---------------------------------------------------------
// Strata (a llama.cpp-derived server with a Python HTTP layer) exposes a
// minimal /slots stub and puts the live rates in /metrics.live instead. The
// plugin detects Strata per server URL and maps those rates directly, rather
// than deriving PP/TG from /slots deltas.
function makeTracker(): TrackerState {
return {
prevNdBySlot: {},
prevPromptTokensBySlot: {},
isPrefilling: false,
isGenerating: false,
prefillSlotId: null,
prefillCapturedTokens: null,
prefillStartAt: 0,
lastPrefillRate: 0,
generateSlotId: null,
generatePrevNd: 0,
generateStartAt: 0,
lastGeneratedTps: 0,
}
}
test("serverTypeOverride accepts only an explicit llama or strata", () => {
assert.equal(serverTypeOverride("strata"), "strata")
assert.equal(serverTypeOverride("llama"), "llama")
assert.equal(serverTypeOverride("auto"), undefined)
assert.equal(serverTypeOverride(undefined), undefined)
assert.equal(serverTypeOverride(42), undefined)
})
test("isStrataHealth detects the strata service field", () => {
assert.equal(isStrataHealth({ status: "ok", service: "strata" }), true)
assert.equal(isStrataHealth({ status: "ok" }), false)
assert.equal(isStrataHealth({ service: "llama" }), false)
assert.equal(isStrataHealth(null), false)
assert.equal(isStrataHealth("strata"), false)
})
test("mapStrataLive reading sets PP from prefill_tok_s_mean", () => {
const t = makeTracker()
mapStrataLive({ state: "reading", prefill_tok_s_mean: 1247.4, tok_s: null }, t)
assert.equal(t.isPrefilling, true)
assert.equal(t.isGenerating, false)
assert.equal(t.lastPrefillRate, 1247.4)
})
test("mapStrataLive generating sets TG from tok_s", () => {
const t = makeTracker()
mapStrataLive({ state: "generating", tok_s: 25.6, prefill_tok_s_mean: null }, t)
assert.equal(t.isGenerating, true)
assert.equal(t.isPrefilling, false)
assert.equal(t.lastGeneratedTps, 25.6)
})
test("mapStrataLive idle clears both phases", () => {
const t = makeTracker()
t.isGenerating = true
t.lastGeneratedTps = 42
mapStrataLive({ state: "idle", tok_s: null, prefill_tok_s_mean: null }, t)
assert.equal(t.isPrefilling, false)
assert.equal(t.isGenerating, false)
assert.equal(t.lastGeneratedTps, 0)
assert.equal(t.lastPrefillRate, 0)
})
test("mapStrataLive unloaded clears both phases", () => {
const t = makeTracker()
t.isPrefilling = true
t.lastPrefillRate = 99
mapStrataLive({ state: "unloaded" }, t)
assert.equal(t.isPrefilling, false)
assert.equal(t.isGenerating, false)
})
test("mapStrataLive tolerates missing live data", () => {
const t = makeTracker()
t.isGenerating = true
t.lastGeneratedTps = 10
mapStrataLive(null, t)
assert.equal(t.isGenerating, false)
assert.equal(t.lastGeneratedTps, 0)
})
test("mapStrataLive treats non-numeric rates as zero", () => {
const t = makeTracker()
mapStrataLive({ state: "generating", tok_s: null }, t)
assert.equal(t.isGenerating, true)
assert.equal(t.lastGeneratedTps, 0)
})
+117 -20
View File
@@ -2,7 +2,16 @@
import type { TextRenderable } from "@opentui/core"
import type { TuiPlugin, TuiPluginModule } from "@opencode-ai/plugin/tui"
import { onCleanup, createSignal } from "solid-js"
import { backoffDelayMs, processPoll, resolveServerUrls, selectSessionUrls } from "./src/stats.ts"
import {
backoffDelayMs,
processPoll,
resolveServerUrls,
selectSessionUrls,
serverTypeOverride,
isStrataHealth,
mapStrataLive,
type ServerType,
} from "./src/stats.ts"
type StreamSample = {
at: number
@@ -54,6 +63,30 @@ async function fetchSlots(baseUrl: string, model?: string): Promise<unknown | nu
}
}
async function fetchHealth(baseUrl: string): Promise<unknown | null> {
try {
const resp = await fetch(`${baseUrl}/health`)
if (!resp.ok) return null
return await resp.json()
} catch {
return null
}
}
// Strata reports the live rates under /metrics.live (prefill_tok_s_mean while
// reading, tok_s while generating) rather than in the minimal /slots stub.
async function fetchStrataLive(baseUrl: string): Promise<unknown | null> {
try {
const resp = await fetch(`${baseUrl}/metrics`)
if (!resp.ok) return null
const body = await resp.json()
if (body && typeof body === "object" && "live" in body) return (body as { live: unknown }).live
return null
} catch {
return null
}
}
function formatPps(value: number) {
if (!Number.isFinite(value) || value <= 0) return undefined
return `${Math.round(value)}`
@@ -179,6 +212,33 @@ const tui: TuiPlugin = async (api, options) => {
let llamaServerUrls: string[] = []
let llamaServerModel: string | undefined
const backoffByUrl = new Map<string, { failures: number; nextRetryAt: number }>()
const serverKindByUrl = new Map<string, ServerType>()
const forcedServerType = serverTypeOverride((options as Record<string, unknown> | undefined)?.serverType)
const resolveServerKind = async (baseUrl: string): Promise<ServerType> => {
if (forcedServerType) return forcedServerType
const cached = serverKindByUrl.get(baseUrl)
if (cached) return cached
const health = await fetchHealth(baseUrl)
if (health != null) {
const kind = isStrataHealth(health) ? "strata" : "llama"
serverKindByUrl.set(baseUrl, kind)
return kind
}
return "llama"
}
const noteServerFailure = (baseUrl: string) => {
const failures = (backoffByUrl.get(baseUrl)?.failures ?? 0) + 1
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
const jitter = Math.floor(Math.random() * (delay / 4))
backoffByUrl.set(baseUrl, {
failures,
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
})
tracker.failure = "n/a"
bump()
}
const pollMetrics = async () => {
if (!llamaServerUrls.length) return
@@ -211,17 +271,22 @@ const tui: TuiPlugin = async (api, options) => {
continue
}
if ((await resolveServerKind(baseUrl)) === "strata") {
const live = await fetchStrataLive(baseUrl)
if (live == null) {
noteServerFailure(baseUrl)
continue
}
backoffByUrl.delete(baseUrl)
tracker.failure = null
mapStrataLive(live, tracker)
bump()
continue
}
const slots = await fetchSlots(baseUrl, model)
if (slots == null) {
const failures = (backoff?.failures ?? 0) + 1
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
const jitter = Math.floor(Math.random() * (delay / 4))
backoffByUrl.set(baseUrl, {
failures,
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
})
tracker.failure = "n/a"
bump()
noteServerFailure(baseUrl)
continue
}
@@ -235,7 +300,7 @@ const tui: TuiPlugin = async (api, options) => {
tracker.failure = null
bump()
const result = processPoll(slotList, tracker)
processPoll(slotList, tracker)
bump()
}
}
@@ -436,6 +501,33 @@ const v2Setup = async (context: any) => {
let llamaServerUrls: string[] = []
let providerList: unknown[] = []
const backoffByUrl = new Map<string, { failures: number; nextRetryAt: number }>()
const serverKindByUrl = new Map<string, ServerType>()
const forcedServerType = serverTypeOverride((context.options as Record<string, unknown> | undefined)?.serverType)
const resolveServerKind = async (baseUrl: string): Promise<ServerType> => {
if (forcedServerType) return forcedServerType
const cached = serverKindByUrl.get(baseUrl)
if (cached) return cached
const health = await fetchHealth(baseUrl)
if (health != null) {
const kind = isStrataHealth(health) ? "strata" : "llama"
serverKindByUrl.set(baseUrl, kind)
return kind
}
return "llama"
}
const noteServerFailure = (baseUrl: string) => {
const failures = (backoffByUrl.get(baseUrl)?.failures ?? 0) + 1
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
const jitter = Math.floor(Math.random() * (delay / 4))
backoffByUrl.set(baseUrl, {
failures,
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
})
tracker.failure = "n/a"
bump()
}
const currentSessionID = (): string | undefined => {
const route = context.ui.router.current?.()
@@ -473,17 +565,22 @@ const v2Setup = async (context: any) => {
continue
}
if ((await resolveServerKind(baseUrl)) === "strata") {
const live = await fetchStrataLive(baseUrl)
if (live == null) {
noteServerFailure(baseUrl)
continue
}
backoffByUrl.delete(baseUrl)
tracker.failure = null
mapStrataLive(live, tracker)
bump()
continue
}
const slots = await fetchSlots(baseUrl, model)
if (slots == null) {
const failures = (backoff?.failures ?? 0) + 1
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
const jitter = Math.floor(Math.random() * (delay / 4))
backoffByUrl.set(baseUrl, {
failures,
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
})
tracker.failure = "n/a"
bump()
noteServerFailure(baseUrl)
continue
}