/*
* SysDeck - AI Gateway (klanker) Panel (v0.3.0)
* Author: Jeremy Anderson (https://dcos.net)
*
* UPSTREAM ATTRIBUTION: this panel is a client of klanker-gate — the
* "Frosty Deno" LLM gateway by TykoDev
* (https://github.com/TykoDev/klanker-gate, Apache-2.0, vendored
* unmodified at ../klanker-gate in the master tarball). klanker-gate
* is NOT SysDeck code — see klanker-gate/ATTRIBUTION.md and
* THIRD_PARTY.md. Zero upstream code is contained here.
*
* v0.3.0 NEW MODULE — operator view of the vendored klanker-gate
* service (master-build/klanker-gate): the Frosty Deno LLM gateway,
* REST on 127.0.0.1:8080, through bridge/klanker.py:
*
* status / providers / models / vkeys / runtime read polls
* logs (limit) → GET /api/logs?limit=N (request ring)
* analytics → GET /api/analytics (24h rollups)
* service (action) → systemctl klanker-gate.service
* journal (n) → journalctl -u klanker-gate -n N
* localstack → probes ollama/llama.cpp/koboldcpp/lmstudio/
* sglang/vllm on this host + wiring recipes (the
* gateway is NOT SaaS-only — all-local stacks are
* first-class: 5 keyless provider types upstream)
*
* Layout: gateway status card + stat grid (providers, keys, requests,
* spend, cache hit rate, avg latency), providers table, virtual keys
* table, recent requests table, model catalog, service control card
* with a journal viewer. Auto-refreshes every 5s; the refresh loop
* re-renders only the data containers — the service/journal cards are
* rendered once, so their state survives every tick. Upstream money
* is integer micro-USD everywhere; this panel divides by 1e6 and shows
* $X.XXXX. If the gateway is down the panel shows the bridge's error
* message verbatim (it carries the remediation hint) and keeps
* retrying, flipping back online on its own.
*/
export async function mount(panel, { bridge, EventBus }) {
panel.innerHTML = renderSkeleton();
const ctx = {
panel,
bridge,
EventBus,
laidOut: false, // full layout rendered (vs offline card)
firstRenderDone: false,
logsLimit: 25, // recent-requests ring size
};
// The status probe decides online vs offline — its error message
// carries the remediation hint, shown verbatim when offline.
let status = null;
let statusErr = null;
try {
status = await bridge.klanker.status();
} catch (err) {
statusErr = err;
}
if (statusErr || !status || status.ok === false) {
renderOffline(ctx, errMessage(statusErr, status));
EventBus.emit('klanker.offline', {});
} else {
await mountLayout(ctx, status);
}
// ── auto-refresh (5s) ─────────────────────────────────────────
// Re-fetches status + runtime + analytics + providers + vkeys +
// logs and re-renders only the data containers. Also watches for
// the gateway going away (→ offline card) or coming back (→ full
// layout).
if (panel._klankerInterval) clearInterval(panel._klankerInterval);
panel._klankerInterval = setInterval(() => { poll(ctx); }, 5000);
// Clean up the interval when the panel leaves the DOM
// (house pattern from netsec.js / fester.js).
// one observer per panel: a re-mount releases the previous one
// instead of stacking body-wide observers on every refresh
if (panel.klankerObserver) panel.klankerObserver.disconnect();
if (panel.klankerObserver) panel.klankerObserver.disconnect();
panel.klankerObserver = new MutationObserver(() => {
if (!document.body.contains(panel)) {
clearInterval(panel._klankerInterval);
panel.klankerObserver.disconnect();
}
});
panel.klankerObserver.observe(document.body, { childList: true, subtree: true });
}
// ── refresh loop ────────────────────────────────────────────────────
async function poll(ctx) {
let status = null;
let statusErr = null;
try {
status = await ctx.bridge.klanker.status();
} catch (err) {
statusErr = err;
}
const online = !statusErr && status && status.ok !== false;
if (online && !ctx.laidOut) {
// Gateway came back after an offline render — build the layout.
await mountLayout(ctx, status);
return;
}
if (!online && ctx.laidOut) {
// Gateway dropped — swap to the offline card.
renderOffline(ctx, errMessage(statusErr, status));
ctx.EventBus.emit('klanker.offline', {});
return;
}
if (!online) return; // still offline — the card is already shown
await refreshData(ctx, status);
}
async function refreshNow(ctx) {
// Immediate re-fetch after a user action (service control/manual).
let status = null;
let statusErr = null;
try {
status = await ctx.bridge.klanker.status();
} catch (err) {
statusErr = err;
}
if (!statusErr && status && status.ok !== false && ctx.laidOut) {
await refreshData(ctx, status);
}
}
async function refreshData(ctx, status) {
if (!ctx.laidOut) return;
const { bridge, panel } = ctx;
// Per-call try/catch: one failing endpoint must not sink the panel.
const runtimeResp = await safe(bridge.klanker.runtime());
const analyticsResp = await safe(bridge.klanker.analytics());
const providersResp = await safe(bridge.klanker.providers());
const vkeysResp = await safe(bridge.klanker.vkeys());
const logsResp = await safe(bridge.klanker.logs(ctx.logsLimit));
const statusEl = panel.querySelector('#klanker-status');
const providersEl = panel.querySelector('#klanker-providers');
const vkeysEl = panel.querySelector('#klanker-vkeys');
const logsEl = panel.querySelector('#klanker-logs');
if (statusEl) statusEl.innerHTML = renderStatus(status, runtimeResp, analyticsResp, providersResp, vkeysResp, logsResp);
if (providersEl) providersEl.innerHTML = renderProviders(providersResp, logsResp);
if (vkeysEl) vkeysEl.innerHTML = renderVkeys(vkeysResp, analyticsResp);
if (logsEl) logsEl.innerHTML = renderLogs(logsResp);
wireStatus(ctx);
if (!ctx.firstRenderDone) {
ctx.firstRenderDone = true;
const totals = analyticsTotals(analyticsResp);
ctx.EventBus.emit('klanker.loaded', { requests: totals.requests });
}
}
// The model catalog is fetched once per layout (it changes only when
// the operator edits provider configs — not on the 5s poll).
let modelsRespCache = null;
async function fetchModels(ctx) {
modelsRespCache = await safe(ctx.bridge.klanker.models());
const modelsEl = ctx.panel.querySelector('#klanker-models');
if (modelsEl) modelsEl.innerHTML = renderModels(modelsRespCache);
}
// ── layout ──────────────────────────────────────────────────────────
async function mountLayout(ctx, status) {
const { panel } = ctx;
ctx.laidOut = true;
const baseUrl = (status && status.base_url) || 'http://127.0.0.1:8080';
panel.innerHTML = `
`;
}
// Per-provider average latency over the recent request ring.
const sums = new Map();
for (const r of listLogs(logsResp)) {
if (!r.provider) continue;
const dur = Number(r.durationMs);
if (!Number.isFinite(dur)) continue;
const e = sums.get(r.provider) || { total: 0, n: 0 };
e.total += dur; e.n += 1;
sums.set(r.provider, e);
}
const rows = providers.map((p) => {
const models = Array.isArray(p.models) ? p.models.length : 0;
const enabled = p.enabled !== false;
const hasCreds = !!(p.hasApiKey || p.hasCloudCredentials);
const dot = enabled ? (hasCreds ? 'ok' : 'warn') : 'off';
const dotTitle = enabled
? (hasCreds ? 'enabled with credentials' : 'enabled but no credentials configured')
: 'disabled';
const s = sums.get(p.id);
const latency = s && s.n > 0 ? fmtMs(s.total / s.n) : '—';
return `
${healthDot(dot, dotTitle)}
${escapeHtml(p.id)}
${escapeHtml(p.type || '—')}
${n(models)}
${enabled ? 'enabled' : 'disabled'}
${latency}
`;
}).join('');
return `
Providers
Name
Kind
Models
Status
Latency (recent)
${rows || '
No providers configured — add one via the gateway admin UI.
'}
Health dot: green = enabled with credentials · amber = enabled without credentials · gray = disabled. Latency is the mean request duration over the recent ring below.
`;
}
function healthDot(state, title) {
const color = state === 'ok' ? 'var(--sysdeck-accent-success)'
: state === 'warn' ? 'var(--sysdeck-accent-warn)'
: 'var(--sysdeck-muted)';
return ``;
}
// ── virtual keys table ──────────────────────────────────────────────
function renderVkeys(vkeysResp, analyticsResp) {
const vkeys = listVkeys(vkeysResp);
if (vkeys == null) {
const msg = (vkeysResp && vkeysResp.error)
? vkeysResp.error
: 'virtual key list unavailable';
return `
Virtual Keys
${escapeHtml(msg)}
`;
}
// 24h attribution join from the analytics rollup (by virtual key).
const byKey = new Map();
if (analyticsResp && analyticsResp.ok !== false && Array.isArray(analyticsResp.byVirtualKey)) {
for (const v of analyticsResp.byVirtualKey) {
if (v && v.virtualKeyId != null) byKey.set(String(v.virtualKeyId), v);
}
}
const rows = vkeys.map((k) => {
const stats = byKey.get(String(k.id));
const scope = k.teamId
? `team ${escapeHtml(k.teamId)}`
: (k.allowedProviders || k.allowedModels)
? `${n((k.allowedProviders || []).length)} providers · ${n((k.allowedModels || []).length)} models`
: 'unrestricted';
const rl = k.rateLimit
? `${n(k.rateLimit.maxRequests)} / ${fmtWindowMs(k.rateLimit.windowMs)}`
: '—';
return `
No virtual keys — create one via the gateway admin UI.
'}
Requests / tokens / cost columns come from the 24h analytics rollup; they show — when the gateway has no keyed traffic yet. Token values are never stored or shown — only the last-4 hint.
Last ${logs.length} requests from the gateway's in-memory ring (FROSTY_LOG_STORE=pg enables the durable audit trail). Costs are micro-USD upstream, shown here in USD.
`;
}
// ── model catalog ───────────────────────────────────────────────────
function renderModels(modelsResp) {
if (!modelsResp) return '';
if (modelsResp.ok === false || modelsResp.error) {
return `
klanker-gate is not SaaS-only: ollama, LM Studio and SGLang are native keyless provider types; llama.cpp (llama-server), KoboldCpp and vLLM plug in through the generic openai-compatible type — the whole inference stack can run local, keyless, zero marginal cost. Rows are live probes of each backend's /v1/models from this host.
Backend
Provider type
Base URL (probed)
Status
Capabilities
${rows}
env wiring (gateway .env)
${escapeHtml(envLines)}
admin API — llama.cpp + koboldcpp accounts
${escapeHtml(data.admin_register_example || '')}
Env registers one account per type; to run llama.cpp and koboldcpp side by side, register each via POST /api/providers, then refresh-models auto-discovers its catalog. Port note: llama-server defaults to :8080 — the gateway's own port — so run it elsewhere (8081 here) or move the gateway.