mirror of
https://git.sync.wtf/troed/oc-ls-stats.git
synced 2026-10-10 12:38:22 +03:00
feat: Strata support (detect via /health, read /metrics.live)
This commit is contained in:
1 parent
55f5e52a6c
commit
f127895ecd
4 files changed
+307
-23
No files matched your search
@@ -6,7 +6,7 @@ There are many plugins to show tokens per second, but the reason for this one is
|
||||
|
||||
Another motivation was to display the data of interest, but in a non-intrusive way with no UI elements jumping around. This plugin thus only shows a single numeric value, tokens per second, with an indicator as to whether the model is currently doing prompt processing or inference (token generation).
|
||||
|
||||
I'm using the llama-server /slots endpoint to get the needed data, which means if you connect opencode to another provider the plugin will just display "-" since it's not getting any data to display.
|
||||
I'm using the llama-server /slots endpoint to get the needed data. Strata (a llama.cpp-derived server with its own HTTP layer) is also supported: it exposes live prefill/decode rates under /metrics, and the plugin detects it automatically. If you connect opencode to an unrelated provider the plugin will just display "-" since it's not getting any data to display.
|
||||
|
||||
Note: As explained in further detail below the data that's displayed has to be deduced from llama-server's output. Sometimes the plugin might display PP for prompt processing while in reality the model is doing TG. If additional developments are made to the llama-server output data the plugin might be able to discern between them in a better way, but I think is as good as it gets for now.
|
||||
|
||||
@@ -63,9 +63,21 @@ If your llama-server provider is named differently, configure the endpoint expli
|
||||
|
||||
A `baseURL`-style value ending in `/v1` is also accepted.
|
||||
|
||||
The plugin detects each server's type automatically by probing `GET /health`: a Strata server reports `"service": "strata"`. You can bypass detection when needed with a `serverType` option:
|
||||
|
||||
```json
|
||||
{
|
||||
"plugin": [
|
||||
["@troed/oc-ls-stats@latest", { "serverType": "strata" }]
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
`serverType` accepts `"llama"` or `"strata"`; omit it (or set `"auto"`) for automatic detection.
|
||||
|
||||
### Slot Polling
|
||||
|
||||
Every 500ms, the plugin polls `GET /slots?model=<model>` on each discovered server. The model parameter is required by the `/slots` endpoint. If the model cannot be discovered from the current route's session, the plugin skips polling.
|
||||
Every 500ms, the plugin polls each discovered server. For a llama.cpp server it calls `GET /slots?model=<model>` (the model parameter is required by that endpoint; if the model cannot be discovered from the current route's session, the plugin skips polling). For a Strata server it calls `GET /metrics` and reads the `live` object instead.
|
||||
|
||||
When a server returns an error (e.g. HTTP 502) or is unreachable, the plugin backs off with an exponentially increasing delay per server (1s, 2s, 4s, ...) capped at 10s, with a small random jitter to stagger multiple servers. Backoff resets as soon as the server responds successfully again.
|
||||
|
||||
@@ -106,6 +118,18 @@ During generation, the plugin calculates the instantaneous token generation rate
|
||||
|
||||
Slot reuse is tracked via `generateSlotId` to detect when a new generation starts on a different slot.
|
||||
|
||||
### Strata Support
|
||||
|
||||
Strata is a llama.cpp-derived server with its own Python HTTP layer. Its `/slots` route is a minimal llama-server compatibility stub (`id`, `n_ctx`, `is_processing` only) with no `next_token`, `id_task`, or `n_prompt_tokens`, so the delta-based classification above cannot produce a rate against it.
|
||||
|
||||
Strata instead reports the live rates directly in `GET /metrics`, under `live`:
|
||||
|
||||
- `state` — `reading` (prompt processing), `generating`, `idle`, or `unloaded`
|
||||
- `prefill_tok_s_mean` — prompt processing rate while `reading` (PP)
|
||||
- `tok_s` — instantaneous decode rate while `generating` (TG)
|
||||
|
||||
The plugin detects Strata per server URL from `GET /health` (`service: "strata"`) and maps these fields straight onto the display, so PP/TG are shown without delta estimation. Detection is cached per URL, and servers of both kinds can be polled side by side. Like the `/slots` requests, the plugin sends no API key; if Strata is configured to require one, it cannot read the stats.
|
||||
|
||||
## Limitations
|
||||
|
||||
### Progress Percentage
|
||||
|
||||
@@ -308,3 +308,67 @@ export function selectSessionUrls(
|
||||
}
|
||||
return []
|
||||
}
|
||||
|
||||
// --- Strata adapter ---------------------------------------------------------
|
||||
// Strata is a llama.cpp-derived server whose HTTP layer is Python. Its /slots
|
||||
// route is a minimal compatibility stub, but /metrics.live carries the live
|
||||
// rates directly: prefill_tok_s_mean while reading, tok_s while generating.
|
||||
|
||||
export type ServerType = "llama" | "strata"
|
||||
|
||||
/** Returns the explicit server type from the plugin option, or undefined for auto. */
|
||||
export function serverTypeOverride(value: unknown): ServerType | undefined {
|
||||
if (value === "llama" || value === "strata") return value
|
||||
return undefined
|
||||
}
|
||||
|
||||
/** True when a parsed /health body identifies the server as Strata. */
|
||||
export function isStrataHealth(health: unknown): boolean {
|
||||
if (!health || typeof health !== "object") return false
|
||||
return (health as Record<string, unknown>).service === "strata"
|
||||
}
|
||||
|
||||
function finiteOrZero(value: unknown): number {
|
||||
return typeof value === "number" && Number.isFinite(value) ? value : 0
|
||||
}
|
||||
|
||||
/**
|
||||
* Map a Strata `/metrics.live` object onto the shared tracker. Strata reports
|
||||
* the rates itself, so no delta classification is needed: `state` says which
|
||||
* phase is active and `prefill_tok_s_mean` / `tok_s` carry the rate.
|
||||
*/
|
||||
export function mapStrataLive(live: unknown, tracker: TrackerState): void {
|
||||
const l = live && typeof live === "object" ? (live as Record<string, unknown>) : {}
|
||||
const state = typeof l.state === "string" ? l.state : ""
|
||||
|
||||
if (state === "reading") {
|
||||
tracker.isPrefilling = true
|
||||
tracker.isGenerating = false
|
||||
tracker.prefillSlotId = 0
|
||||
tracker.lastPrefillRate = finiteOrZero(l.prefill_tok_s_mean)
|
||||
tracker.generateSlotId = null
|
||||
tracker.generatePrevNd = 0
|
||||
tracker.generateStartAt = 0
|
||||
tracker.lastGeneratedTps = 0
|
||||
} else if (state === "generating") {
|
||||
tracker.isGenerating = true
|
||||
tracker.isPrefilling = false
|
||||
tracker.generateSlotId = 0
|
||||
tracker.lastGeneratedTps = finiteOrZero(l.tok_s)
|
||||
tracker.prefillSlotId = null
|
||||
tracker.prefillCapturedTokens = null
|
||||
tracker.prefillStartAt = 0
|
||||
tracker.lastPrefillRate = 0
|
||||
} else {
|
||||
tracker.isPrefilling = false
|
||||
tracker.isGenerating = false
|
||||
tracker.prefillSlotId = null
|
||||
tracker.prefillCapturedTokens = null
|
||||
tracker.prefillStartAt = 0
|
||||
tracker.lastPrefillRate = 0
|
||||
tracker.generateSlotId = null
|
||||
tracker.generatePrevNd = 0
|
||||
tracker.generateStartAt = 0
|
||||
tracker.lastGeneratedTps = 0
|
||||
}
|
||||
}
|
||||
+100
-1
@@ -1,6 +1,14 @@
|
||||
import { test } from "node:test"
|
||||
import assert from "node:assert/strict"
|
||||
import { backoffDelayMs, resolveServerUrls, selectSessionUrls } from "./src/stats.ts"
|
||||
import {
|
||||
backoffDelayMs,
|
||||
resolveServerUrls,
|
||||
selectSessionUrls,
|
||||
serverTypeOverride,
|
||||
isStrataHealth,
|
||||
mapStrataLive,
|
||||
type TrackerState,
|
||||
} from "./src/stats.ts"
|
||||
|
||||
test("first failure retries after the base delay", () => {
|
||||
assert.equal(backoffDelayMs(1), 1000)
|
||||
@@ -289,3 +297,94 @@ test("V2 provider settings still excludes non-candidate providers", () => {
|
||||
const providers = [{ id: "llama-local", name: "llama.cpp", settings: { baseURL: "http://other:1" } }]
|
||||
assert.deepEqual(selectSessionUrls(["http://localhost:8080"], providers, "llama-local"), [])
|
||||
})
|
||||
|
||||
// --- Strata adapter ---------------------------------------------------------
|
||||
// Strata (a llama.cpp-derived server with a Python HTTP layer) exposes a
|
||||
// minimal /slots stub and puts the live rates in /metrics.live instead. The
|
||||
// plugin detects Strata per server URL and maps those rates directly, rather
|
||||
// than deriving PP/TG from /slots deltas.
|
||||
|
||||
function makeTracker(): TrackerState {
|
||||
return {
|
||||
prevNdBySlot: {},
|
||||
prevPromptTokensBySlot: {},
|
||||
isPrefilling: false,
|
||||
isGenerating: false,
|
||||
prefillSlotId: null,
|
||||
prefillCapturedTokens: null,
|
||||
prefillStartAt: 0,
|
||||
lastPrefillRate: 0,
|
||||
generateSlotId: null,
|
||||
generatePrevNd: 0,
|
||||
generateStartAt: 0,
|
||||
lastGeneratedTps: 0,
|
||||
}
|
||||
}
|
||||
|
||||
test("serverTypeOverride accepts only an explicit llama or strata", () => {
|
||||
assert.equal(serverTypeOverride("strata"), "strata")
|
||||
assert.equal(serverTypeOverride("llama"), "llama")
|
||||
assert.equal(serverTypeOverride("auto"), undefined)
|
||||
assert.equal(serverTypeOverride(undefined), undefined)
|
||||
assert.equal(serverTypeOverride(42), undefined)
|
||||
})
|
||||
|
||||
test("isStrataHealth detects the strata service field", () => {
|
||||
assert.equal(isStrataHealth({ status: "ok", service: "strata" }), true)
|
||||
assert.equal(isStrataHealth({ status: "ok" }), false)
|
||||
assert.equal(isStrataHealth({ service: "llama" }), false)
|
||||
assert.equal(isStrataHealth(null), false)
|
||||
assert.equal(isStrataHealth("strata"), false)
|
||||
})
|
||||
|
||||
test("mapStrataLive reading sets PP from prefill_tok_s_mean", () => {
|
||||
const t = makeTracker()
|
||||
mapStrataLive({ state: "reading", prefill_tok_s_mean: 1247.4, tok_s: null }, t)
|
||||
assert.equal(t.isPrefilling, true)
|
||||
assert.equal(t.isGenerating, false)
|
||||
assert.equal(t.lastPrefillRate, 1247.4)
|
||||
})
|
||||
|
||||
test("mapStrataLive generating sets TG from tok_s", () => {
|
||||
const t = makeTracker()
|
||||
mapStrataLive({ state: "generating", tok_s: 25.6, prefill_tok_s_mean: null }, t)
|
||||
assert.equal(t.isGenerating, true)
|
||||
assert.equal(t.isPrefilling, false)
|
||||
assert.equal(t.lastGeneratedTps, 25.6)
|
||||
})
|
||||
|
||||
test("mapStrataLive idle clears both phases", () => {
|
||||
const t = makeTracker()
|
||||
t.isGenerating = true
|
||||
t.lastGeneratedTps = 42
|
||||
mapStrataLive({ state: "idle", tok_s: null, prefill_tok_s_mean: null }, t)
|
||||
assert.equal(t.isPrefilling, false)
|
||||
assert.equal(t.isGenerating, false)
|
||||
assert.equal(t.lastGeneratedTps, 0)
|
||||
assert.equal(t.lastPrefillRate, 0)
|
||||
})
|
||||
|
||||
test("mapStrataLive unloaded clears both phases", () => {
|
||||
const t = makeTracker()
|
||||
t.isPrefilling = true
|
||||
t.lastPrefillRate = 99
|
||||
mapStrataLive({ state: "unloaded" }, t)
|
||||
assert.equal(t.isPrefilling, false)
|
||||
assert.equal(t.isGenerating, false)
|
||||
})
|
||||
|
||||
test("mapStrataLive tolerates missing live data", () => {
|
||||
const t = makeTracker()
|
||||
t.isGenerating = true
|
||||
t.lastGeneratedTps = 10
|
||||
mapStrataLive(null, t)
|
||||
assert.equal(t.isGenerating, false)
|
||||
assert.equal(t.lastGeneratedTps, 0)
|
||||
})
|
||||
|
||||
test("mapStrataLive treats non-numeric rates as zero", () => {
|
||||
const t = makeTracker()
|
||||
mapStrataLive({ state: "generating", tok_s: null }, t)
|
||||
assert.equal(t.isGenerating, true)
|
||||
assert.equal(t.lastGeneratedTps, 0)
|
||||
})
|
||||
@@ -2,7 +2,16 @@
|
||||
import type { TextRenderable } from "@opentui/core"
|
||||
import type { TuiPlugin, TuiPluginModule } from "@opencode-ai/plugin/tui"
|
||||
import { onCleanup, createSignal } from "solid-js"
|
||||
import { backoffDelayMs, processPoll, resolveServerUrls, selectSessionUrls } from "./src/stats.ts"
|
||||
import {
|
||||
backoffDelayMs,
|
||||
processPoll,
|
||||
resolveServerUrls,
|
||||
selectSessionUrls,
|
||||
serverTypeOverride,
|
||||
isStrataHealth,
|
||||
mapStrataLive,
|
||||
type ServerType,
|
||||
} from "./src/stats.ts"
|
||||
|
||||
type StreamSample = {
|
||||
at: number
|
||||
@@ -54,6 +63,30 @@ async function fetchSlots(baseUrl: string, model?: string): Promise<unknown | nu
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchHealth(baseUrl: string): Promise<unknown | null> {
|
||||
try {
|
||||
const resp = await fetch(`${baseUrl}/health`)
|
||||
if (!resp.ok) return null
|
||||
return await resp.json()
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
// Strata reports the live rates under /metrics.live (prefill_tok_s_mean while
|
||||
// reading, tok_s while generating) rather than in the minimal /slots stub.
|
||||
async function fetchStrataLive(baseUrl: string): Promise<unknown | null> {
|
||||
try {
|
||||
const resp = await fetch(`${baseUrl}/metrics`)
|
||||
if (!resp.ok) return null
|
||||
const body = await resp.json()
|
||||
if (body && typeof body === "object" && "live" in body) return (body as { live: unknown }).live
|
||||
return null
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
function formatPps(value: number) {
|
||||
if (!Number.isFinite(value) || value <= 0) return undefined
|
||||
return `${Math.round(value)}`
|
||||
@@ -179,6 +212,33 @@ const tui: TuiPlugin = async (api, options) => {
|
||||
let llamaServerUrls: string[] = []
|
||||
let llamaServerModel: string | undefined
|
||||
const backoffByUrl = new Map<string, { failures: number; nextRetryAt: number }>()
|
||||
const serverKindByUrl = new Map<string, ServerType>()
|
||||
const forcedServerType = serverTypeOverride((options as Record<string, unknown> | undefined)?.serverType)
|
||||
|
||||
const resolveServerKind = async (baseUrl: string): Promise<ServerType> => {
|
||||
if (forcedServerType) return forcedServerType
|
||||
const cached = serverKindByUrl.get(baseUrl)
|
||||
if (cached) return cached
|
||||
const health = await fetchHealth(baseUrl)
|
||||
if (health != null) {
|
||||
const kind = isStrataHealth(health) ? "strata" : "llama"
|
||||
serverKindByUrl.set(baseUrl, kind)
|
||||
return kind
|
||||
}
|
||||
return "llama"
|
||||
}
|
||||
|
||||
const noteServerFailure = (baseUrl: string) => {
|
||||
const failures = (backoffByUrl.get(baseUrl)?.failures ?? 0) + 1
|
||||
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
|
||||
const jitter = Math.floor(Math.random() * (delay / 4))
|
||||
backoffByUrl.set(baseUrl, {
|
||||
failures,
|
||||
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
|
||||
})
|
||||
tracker.failure = "n/a"
|
||||
bump()
|
||||
}
|
||||
|
||||
const pollMetrics = async () => {
|
||||
if (!llamaServerUrls.length) return
|
||||
@@ -211,17 +271,22 @@ const tui: TuiPlugin = async (api, options) => {
|
||||
continue
|
||||
}
|
||||
|
||||
if ((await resolveServerKind(baseUrl)) === "strata") {
|
||||
const live = await fetchStrataLive(baseUrl)
|
||||
if (live == null) {
|
||||
noteServerFailure(baseUrl)
|
||||
continue
|
||||
}
|
||||
backoffByUrl.delete(baseUrl)
|
||||
tracker.failure = null
|
||||
mapStrataLive(live, tracker)
|
||||
bump()
|
||||
continue
|
||||
}
|
||||
|
||||
const slots = await fetchSlots(baseUrl, model)
|
||||
if (slots == null) {
|
||||
const failures = (backoff?.failures ?? 0) + 1
|
||||
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
|
||||
const jitter = Math.floor(Math.random() * (delay / 4))
|
||||
backoffByUrl.set(baseUrl, {
|
||||
failures,
|
||||
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
|
||||
})
|
||||
tracker.failure = "n/a"
|
||||
bump()
|
||||
noteServerFailure(baseUrl)
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -235,7 +300,7 @@ const tui: TuiPlugin = async (api, options) => {
|
||||
tracker.failure = null
|
||||
bump()
|
||||
|
||||
const result = processPoll(slotList, tracker)
|
||||
processPoll(slotList, tracker)
|
||||
bump()
|
||||
}
|
||||
}
|
||||
@@ -436,6 +501,33 @@ const v2Setup = async (context: any) => {
|
||||
let llamaServerUrls: string[] = []
|
||||
let providerList: unknown[] = []
|
||||
const backoffByUrl = new Map<string, { failures: number; nextRetryAt: number }>()
|
||||
const serverKindByUrl = new Map<string, ServerType>()
|
||||
const forcedServerType = serverTypeOverride((context.options as Record<string, unknown> | undefined)?.serverType)
|
||||
|
||||
const resolveServerKind = async (baseUrl: string): Promise<ServerType> => {
|
||||
if (forcedServerType) return forcedServerType
|
||||
const cached = serverKindByUrl.get(baseUrl)
|
||||
if (cached) return cached
|
||||
const health = await fetchHealth(baseUrl)
|
||||
if (health != null) {
|
||||
const kind = isStrataHealth(health) ? "strata" : "llama"
|
||||
serverKindByUrl.set(baseUrl, kind)
|
||||
return kind
|
||||
}
|
||||
return "llama"
|
||||
}
|
||||
|
||||
const noteServerFailure = (baseUrl: string) => {
|
||||
const failures = (backoffByUrl.get(baseUrl)?.failures ?? 0) + 1
|
||||
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
|
||||
const jitter = Math.floor(Math.random() * (delay / 4))
|
||||
backoffByUrl.set(baseUrl, {
|
||||
failures,
|
||||
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
|
||||
})
|
||||
tracker.failure = "n/a"
|
||||
bump()
|
||||
}
|
||||
|
||||
const currentSessionID = (): string | undefined => {
|
||||
const route = context.ui.router.current?.()
|
||||
@@ -473,17 +565,22 @@ const v2Setup = async (context: any) => {
|
||||
continue
|
||||
}
|
||||
|
||||
if ((await resolveServerKind(baseUrl)) === "strata") {
|
||||
const live = await fetchStrataLive(baseUrl)
|
||||
if (live == null) {
|
||||
noteServerFailure(baseUrl)
|
||||
continue
|
||||
}
|
||||
backoffByUrl.delete(baseUrl)
|
||||
tracker.failure = null
|
||||
mapStrataLive(live, tracker)
|
||||
bump()
|
||||
continue
|
||||
}
|
||||
|
||||
const slots = await fetchSlots(baseUrl, model)
|
||||
if (slots == null) {
|
||||
const failures = (backoff?.failures ?? 0) + 1
|
||||
const delay = backoffDelayMs(failures, BACKOFF_BASE_MS, BACKOFF_MAX_MS)
|
||||
const jitter = Math.floor(Math.random() * (delay / 4))
|
||||
backoffByUrl.set(baseUrl, {
|
||||
failures,
|
||||
nextRetryAt: Date.now() + Math.min(delay + jitter, BACKOFF_MAX_MS),
|
||||
})
|
||||
tracker.failure = "n/a"
|
||||
bump()
|
||||
noteServerFailure(baseUrl)
|
||||
continue
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user