SysDeck/klanker-gate/tests/integration/governance_api_test.ts

531 lines
17 KiB
TypeScript
Executable File

// Governance integration: virtual-key admission on /v1/*, rate limits,
// budgets, admin CRUD, and the metrics endpoint.
import { assert, assertEquals } from "@std/assert";
import { createHandler } from "../../apps/gateway/main.ts";
import {
type AppContext,
NullToolExecutor,
VERSION,
} from "../../apps/gateway/context.ts";
import { ConfigService } from "../../packages/config/src/service.ts";
import { ProviderManager } from "../../packages/providers/src/mod.ts";
import { Metrics } from "../../packages/telemetry/src/metrics.ts";
import { LogBus } from "../../packages/telemetry/src/logbus.ts";
import { VirtualKeyManager } from "../../packages/governance/src/virtual_keys.ts";
import { MCPRegistry } from "../../packages/mcp/src/registry.ts";
import { PluginManager } from "../../packages/plugins/src/lifecycle.ts";
import {
jsonResponse,
MockProvider,
openAIChatBody,
} from "../../packages/testing/src/mod.ts";
const base = "http://gateway.test";
async function makeContext(
mockUrl: string,
models: string[] = ["m1"],
): Promise<AppContext> {
return {
providers: new ProviderManager([{
id: "openai",
type: "openai",
apiKey: "k",
baseUrl: mockUrl,
enabled: true,
models,
priority: 0,
retry: { maxRetries: 0 },
}], "openai"),
metrics: new Metrics(),
logBus: new LogBus(),
virtualKeys: new VirtualKeyManager(),
mcp: new MCPRegistry(),
plugins: new PluginManager(),
toolExecutor: new NullToolExecutor(),
version: VERSION,
config: await ConfigService.open(":memory:"),
};
}
function chatRequest(token?: string): Request {
return new Request(`${base}/v1/chat/completions`, {
method: "POST",
headers: {
"Content-Type": "application/json",
...(token ? { Authorization: `Bearer ${token}` } : {}),
},
body: JSON.stringify({
model: "m1",
messages: [{ role: "user", content: "hi" }],
}),
});
}
Deno.test("governance: concurrent requests cannot exceed a lifetime request budget", async () => {
// Regression for the over-admission race: N requests fired at once must never
// admit more than maxRequests. Before the reserve-before-admit fix, they could
// all pass the read in check() during the async admit() window and each be let
// through, overshooting the budget by up to N-1.
const mock = new MockProvider(() => jsonResponse(openAIChatBody("ok")));
const ctx = await makeContext(mock.url);
ctx.virtualKeys.upsert({
id: "k",
name: "burst",
token: "vk-burst-concurrency",
enabled: true,
usedRequests: 0,
usedCostMicroUsd: 0,
budget: { maxRequests: 5 },
});
const handler = createHandler(ctx);
try {
const N = 25;
const results = await Promise.all(
Array.from(
{ length: N },
() => handler(chatRequest("vk-burst-concurrency")),
),
);
let admitted = 0;
let denied = 0;
for (const res of results) {
if (res.status === 200) admitted++;
else if (res.status === 402) denied++;
await res.body?.cancel();
}
assertEquals(admitted, 5); // exactly the budget — never more
assertEquals(denied, N - 5);
assertEquals(ctx.virtualKeys.get("k")!.usedRequests, 5);
} finally {
ctx.config?.close();
await mock.close();
}
});
Deno.test("governance: open access until a virtual key exists, then enforced", async (t) => {
const mock = new MockProvider(() => jsonResponse(openAIChatBody("ok")));
const ctx = await makeContext(mock.url);
const handler = createHandler(ctx);
let token = "";
try {
await t.step("no keys configured: /v1 is open", async () => {
const res = await handler(chatRequest());
assertEquals(res.status, 200);
await res.body?.cancel();
});
await t.step("create a virtual key via the admin API", async () => {
const res = await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
name: "team-a",
rateLimit: { maxRequests: 3, windowMs: 60_000 },
budget: { maxRequests: 5 },
}),
}),
);
assertEquals(res.status, 201);
const body = await res.json();
token = body.token;
assert(token.startsWith("vk-"));
// persisted to KV
assertEquals((await ctx.config!.listVirtualKeys()).length, 1);
});
await t.step("list shows only a token hint", async () => {
const res = await handler(new Request(`${base}/api/virtual-keys`));
const body = await res.json();
assertEquals(body.virtualKeys.length, 1);
assert(!("token" in body.virtualKeys[0]));
assert(!("tokenHash" in body.virtualKeys[0]));
// Hint reveals only the last 4 chars (leading "…"), never the token body.
assert(body.virtualKeys[0].tokenHint.startsWith("…"));
assert(!body.virtualKeys[0].tokenHint.includes(token));
});
await t.step("missing key is now rejected with 401", async () => {
const res = await handler(chatRequest());
assertEquals(res.status, 401);
assertEquals((await res.json()).error.code, "missing_virtual_key");
});
await t.step("valid key is admitted", async () => {
const res = await handler(chatRequest(token));
assertEquals(res.status, 200);
await res.body?.cancel();
});
await t.step("rate limit returns 429 with Retry-After", async () => {
// 1 request used above; limit is 3/min
await (await handler(chatRequest(token))).body?.cancel();
await (await handler(chatRequest(token))).body?.cancel();
const res = await handler(chatRequest(token));
assertEquals(res.status, 429);
assertEquals(res.headers.get("Retry-After"), "60");
assertEquals((await res.json()).error.code, "rate_limited");
});
await t.step("admin surface is not governed by virtual keys", async () => {
const res = await handler(new Request(`${base}/api/providers`));
assertEquals(res.status, 200);
await res.body?.cancel();
});
await t.step("disable via PUT locks the key out", async () => {
const list = await (await handler(
new Request(`${base}/api/virtual-keys`),
)).json();
const id = list.virtualKeys[0].id;
const res = await handler(
new Request(`${base}/api/virtual-keys/${id}`, {
method: "PUT",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ enabled: false }),
}),
);
assertEquals(res.status, 200);
const denied = await handler(chatRequest(token));
assertEquals(denied.status, 401);
await denied.body?.cancel();
});
await t.step(
"metrics endpoint exposes counters and latencies",
async () => {
const res = await handler(new Request(`${base}/metrics`));
assertEquals(res.status, 200);
const text = await res.text();
assert(text.includes("frosty_requests_total"));
assert(text.includes('route="/v1/chat/completions"'));
assert(text.includes("frosty_request_duration_ms"));
assert(text.includes("governance.denied.missing_virtual_key"));
},
);
} finally {
ctx.config!.close();
await mock.close();
}
});
Deno.test("governance: budget exhaustion returns 402", async () => {
const mock = new MockProvider(() => jsonResponse(openAIChatBody("ok")));
const ctx = await makeContext(mock.url);
ctx.virtualKeys.upsert({
id: "vk1",
name: "tiny-budget",
token: "vk-tiny-budget-token",
enabled: true,
budget: { maxRequests: 1 },
usedRequests: 0,
usedCostMicroUsd: 0,
});
const handler = createHandler(ctx);
try {
const first = await handler(chatRequest("vk-tiny-budget-token"));
assertEquals(first.status, 200);
await first.body?.cancel();
const second = await handler(chatRequest("vk-tiny-budget-token"));
assertEquals(second.status, 402);
assertEquals((await second.json()).error.code, "budget_exhausted");
} finally {
ctx.config!.close();
await mock.close();
}
});
Deno.test("governance: per-key model/provider scope enforcement", async (t) => {
const mock = new MockProvider(() => jsonResponse(openAIChatBody("ok")));
const ctx = await makeContext(mock.url, ["m1", "m2"]);
const handler = createHandler(ctx);
let token = "";
const chat = (model: string, tok: string, contentType = "application/json") =>
new Request(`${base}/v1/chat/completions`, {
method: "POST",
headers: {
"Content-Type": contentType,
Authorization: `Bearer ${tok}`,
},
body: JSON.stringify({
model,
messages: [{ role: "user", content: "hi" }],
}),
});
try {
await t.step("empty allowlist array is rejected at create", async () => {
const res = await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ name: "bad", allowedModels: [] }),
}),
);
assertEquals(res.status, 400);
});
await t.step("create a model-scoped key (allowedModels: m1)", async () => {
const res = await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ name: "scoped", allowedModels: ["m1"] }),
}),
);
assertEquals(res.status, 201);
token = (await res.json()).token;
});
await t.step("in-scope model is admitted", async () => {
const res = await handler(chat("m1", token));
assertEquals(res.status, 200);
await res.body?.cancel();
});
await t.step("in-scope model with a known provider prefix", async () => {
const res = await handler(chat("openai/m1", token));
assertEquals(res.status, 200);
await res.body?.cancel();
});
await t.step("out-of-scope model is refused with 403", async () => {
const res = await handler(chat("m2", token));
assertEquals(res.status, 403);
assertEquals((await res.json()).error.code, "model_not_permitted");
});
await t.step(
"a provider prefix cannot smuggle an out-of-scope model",
async () => {
const res = await handler(chat("openai/m2", token));
assertEquals(res.status, 403);
assertEquals((await res.json()).error.code, "model_not_permitted");
},
);
await t.step(
"a scoped key fails closed when the model is undeterminable",
async () => {
// multipart: the middleware cannot read a JSON model -> deny.
const res = await handler(
new Request(`${base}/v1/chat/completions`, {
method: "POST",
headers: {
"Content-Type": "multipart/form-data; boundary=xx",
Authorization: `Bearer ${token}`,
},
body: "--xx--",
}),
);
assertEquals(res.status, 403);
assertEquals((await res.json()).error.code, "model_not_permitted");
},
);
await t.step(
"a provider-scoped key refuses an off-list provider",
async () => {
const res = await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
name: "prov-scoped",
allowedProviders: ["anthropic"],
}),
}),
);
const provToken = (await res.json()).token;
// Model m1 resolves to provider "openai", which is not on the allowlist.
const denied = await handler(chat("m1", provToken));
assertEquals(denied.status, 403);
assertEquals(
(await denied.json()).error.code,
"provider_not_permitted",
);
},
);
await t.step(
"genai colon-tagged model cannot smuggle past a model scope",
async () => {
const res = await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
name: "genai-scoped",
allowedModels: ["m1"],
}),
}),
);
const gToken = (await res.json()).token;
const genai = (modelAction: string) =>
new Request(`${base}/genai/v1beta/models/${modelAction}`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${gToken}`,
},
body: JSON.stringify({ contents: [{ parts: [{ text: "hi" }] }] }),
});
// Dispatch splits on the LAST colon -> model "m1:pro"; enforcement must
// see the same, not the first-colon-truncated "m1".
const denied = await handler(genai("m1:pro:generateContent"));
assertEquals(denied.status, 403);
// A genai path answers in the GenAI error envelope (decision-log 86),
// where `code` is the HTTP status; the gateway's own code rides
// `details[].reason` so operators do not lose it.
const deniedBody = await denied.json() as {
error: {
code: number;
status: string;
details?: Array<{ reason?: string }>;
};
};
assertEquals(deniedBody.error.code, 403);
assertEquals(deniedBody.error.status, "PERMISSION_DENIED");
assertEquals(
deniedBody.error.details?.[0]?.reason,
"model_not_permitted",
);
// Exact in-scope model still admits.
const ok = await handler(genai("m1:generateContent"));
assertEquals(ok.status, 200);
await ok.body?.cancel();
},
);
await t.step(
"clearing the model scope via null re-opens all models",
async () => {
const list =
await (await handler(new Request(`${base}/api/virtual-keys`)))
.json();
const scopedKey = list.virtualKeys.find(
(k: { name: string }) => k.name === "scoped",
);
assertEquals(scopedKey.allowedModels, ["m1"]);
const put = await handler(
new Request(`${base}/api/virtual-keys/${scopedKey.id}`, {
method: "PUT",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ allowedModels: null }),
}),
);
assertEquals(put.status, 200);
assertEquals((await put.json()).allowedModels, undefined);
// m2 was refused before the clear; it is admitted now.
const admitted = await handler(chat("m2", token));
assertEquals(admitted.status, 200);
await admitted.body?.cancel();
},
);
} finally {
ctx.config!.close();
await mock.close();
}
});
Deno.test("governance: scoped key is not served by an off-list failover target", async (t) => {
const primary = new MockProvider(() =>
new Response(JSON.stringify({ error: "boom" }), {
status: 500,
headers: { "Content-Type": "application/json" },
})
);
const backup = new MockProvider(() => jsonResponse(openAIChatBody("from-B")));
const ctx: AppContext = {
providers: new ProviderManager([
{
id: "A",
type: "openai",
apiKey: "k",
baseUrl: primary.url,
enabled: true,
models: ["m"],
priority: 0,
retry: { maxRetries: 0 },
},
{
id: "B",
type: "openai",
apiKey: "k",
baseUrl: backup.url,
enabled: true,
models: ["m"],
priority: 0,
retry: { maxRetries: 0 },
},
], "A"),
metrics: new Metrics(),
logBus: new LogBus(),
virtualKeys: new VirtualKeyManager(),
mcp: new MCPRegistry(),
plugins: new PluginManager(),
toolExecutor: new NullToolExecutor(),
version: VERSION,
config: await ConfigService.open(":memory:"),
};
const handler = createHandler(ctx);
const chat = (tok: string) =>
new Request(`${base}/v1/chat/completions`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${tok}`,
},
body: JSON.stringify({
model: "m",
messages: [{ role: "user", content: "hi" }],
}),
});
try {
let unscoped = "";
let scoped = "";
await t.step("create an unscoped key and a key scoped to A", async () => {
unscoped = (await (await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ name: "open" }),
}),
)).json()).token;
scoped = (await (await handler(
new Request(`${base}/api/virtual-keys`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ name: "pin-A", allowedProviders: ["A"] }),
}),
)).json()).token;
});
await t.step(
"failover works for an unscoped key (A 500 -> B 200)",
async () => {
const res = await handler(chat(unscoped));
assertEquals(res.status, 200);
assertEquals((await res.json()).choices[0].message.content, "from-B");
},
);
await t.step("A-scoped key is NOT served by B on failover", async () => {
const res = await handler(chat(scoped));
assert(res.status >= 500, `expected 5xx all-failed, got ${res.status}`);
await res.body?.cancel();
});
} finally {
ctx.config!.close();
await primary.close();
await backup.close();
}
});