// D1 Stage 3: the media accounting channel written from the three media routes, // the per-process in-flight byte budget, and the C1 write gate. // // These drive the REAL route handlers registered on a bare Router rather than // through createHandler, because the channel is a WeakMap keyed on Request // identity and makeRequestLogger hands the router a CLONE of the request the // caller made (middleware.ts: `new Request(req, {headers})`). A test asserting // on its own Request object would read an empty channel and pass vacuously. The // full chain is exercised separately at the end, for the client-visible half. import { assert, assertEquals, assertRejects } from "@std/assert"; import { createHandler } from "../../apps/gateway/main.ts"; import { type AppContext, NullToolExecutor, VERSION, } from "../../apps/gateway/context.ts"; import { Router } from "../../packages/core/src/mod.ts"; import { registerAdvancedRoutes } from "../../apps/gateway/routes/advanced.ts"; import { mediaInflightReserved } from "../../apps/gateway/routes/advanced.ts"; import { ProviderError, ProviderManager, } from "../../packages/providers/src/mod.ts"; import { Metrics } from "../../packages/telemetry/src/metrics.ts"; import { LogBus } from "../../packages/telemetry/src/logbus.ts"; import { VirtualKeyManager } from "../../packages/governance/src/virtual_keys.ts"; import { GovernanceHierarchy } from "../../packages/governance/src/hierarchy.ts"; import { MCPRegistry } from "../../packages/mcp/src/registry.ts"; import { PluginManager } from "../../packages/plugins/src/lifecycle.ts"; import { getRequestDispatch, mergeRequestStatus, setRequestDispatch, } from "../../packages/telemetry/src/usage.ts"; import { MAX_MEDIA_INFLIGHT_BYTES, MAX_MEDIA_JSON_BYTES, MAX_TRANSCRIPTION_JSON_BYTES, MAX_TTS_INPUT_CHARS, MEDIA_BYTES_PER_IMAGE, } from "../../packages/contracts/src/mod.ts"; import { jsonResponse, MockProvider } from "../../packages/testing/src/mod.ts"; const base = "http://gateway.test"; function makeContext( mockUrl: string, retry: { maxRetries: number; initialDelayMs?: number } = { maxRetries: 0 }, ): AppContext { return { providers: new ProviderManager([{ id: "openai", type: "openai", apiKey: "sk-media", baseUrl: mockUrl, enabled: true, models: ["gpt-image-1", "tts-1", "whisper-1"], priority: 0, retry, }], "openai"), metrics: new Metrics(), logBus: new LogBus(), virtualKeys: new VirtualKeyManager(), hierarchy: new GovernanceHierarchy(), mcp: new MCPRegistry(), plugins: new PluginManager(), toolExecutor: new NullToolExecutor(), version: VERSION, }; } function mediaRouter(ctx: AppContext): Router { const router = new Router(); registerAdvancedRoutes(router, ctx); return router; } function imageRequest( body: Record, init?: RequestInit, ): Request { return new Request(`${base}/v1/images/generations`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: "openai/gpt-image-1", prompt: "fjord", ...body, }), ...init, }); } function speechRequest(input: string, init?: RequestInit): Request { return new Request(`${base}/v1/audio/speech`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: "openai/tts-1", input, voice: "alloy" }), ...init, }); } function transcriptionRequest(init?: RequestInit): Request { const boundary = "----frosty"; const payload = `--${boundary}\r\n` + 'Content-Disposition: form-data; name="model"\r\n\r\nwhisper-1\r\n' + `--${boundary}--\r\n`; return new Request(`${base}/v1/audio/transcriptions`, { method: "POST", headers: { "Content-Type": `multipart/form-data; boundary=${boundary}` }, body: payload, ...init, }); } /** Polls until `pred` holds. Used to observe a reservation that only exists * while a provider request is parked. */ async function until(pred: () => boolean, what: string): Promise { for (let i = 0; i < 1000; i++) { if (pred()) return; await new Promise((r) => setTimeout(r, 2)); } throw new Error(`timed out waiting for ${what}`); } /** A promise the test resolves to let a parked provider respond. */ function gate(): { wait: Promise; open: () => void } { let open = () => {}; const wait = new Promise((resolve) => { open = resolve; }); return { wait, open }; } // --------------------------------------------------------------------------- // D1-T1 / D1-T2 / D1-T3 / D1-T4 - the quantity written per surface // --------------------------------------------------------------------------- Deno.test("D1-T1..T4: each media surface writes its own quantity", async (t) => { const mock = new MockProvider((call) => { switch (call.path) { case "/images/generations": // Two images returned against n: 7, plus a token block, so imageCount // and tokens both come from the SAME parse. return jsonResponse({ created: 1, data: [{ b64_json: "aa" }, { b64_json: "bb" }], usage: { input_tokens: 11, output_tokens: 22, total_tokens: 33 }, }); case "/audio/speech": return new Response(new Uint8Array([1, 2, 3]), { headers: { "Content-Type": "audio/mpeg" }, }); case "/embeddings": return jsonResponse({ object: "list", data: [{ embedding: [0.1] }, { embedding: [0.2] }], model: "text-embedding-3-small", }); case "/audio/transcriptions": return jsonResponse({ text: "words", duration: 12.5 }); default: return jsonResponse({ error: call.path }, 500); } }); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { // D1-T1: `n: 7` requested, 2 delivered. The billed count is what arrived. await t.step("D1-T1: imageCount is data.length, never n", async () => { const req = imageRequest({ n: 7 }); const res = await router.handle(req); assertEquals(res.status, 200); assertEquals((await res.json()).data.length, 2); const channel = getRequestDispatch(req)!; assertEquals(channel.units, { imageCount: 2 }); assertEquals(channel.providerStatus, 200); assertEquals(channel.providerId, "openai"); assertEquals(channel.model, "gpt-image-1"); // M1: the token half comes from the same parse, not a second body sniff. assertEquals(channel.tokens, { prompt: 11, completion: 22, cached: 0, cacheCreation: 0, }); }); // D1-T2, route half: /v1/embeddings is not a media surface at all, so it // opens no channel - the strongest form of "records no imageCount". await t.step("D1-T2: an embeddings response records nothing", async () => { const req = new Request(`${base}/v1/embeddings`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: "openai/text-embedding-3-small", input: "hei", }), }); const res = await router.handle(req); assertEquals(res.status, 200); assertEquals((await res.json()).data.length, 2); assertEquals(getRequestDispatch(req), undefined); }); // D1-T3: code points, not UTF-16 units. "a😀b🎉c" is 5 code points and 7 // UTF-16 units, so `.length` would over-bill by 2 on a 5-character input. await t.step("D1-T3: characterCount counts code points", async () => { const input = "a\u{1F600}b\u{1F389}c"; assertEquals(input.length, 7); assertEquals([...input].length, 5); const req = speechRequest(input); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); assertEquals(getRequestDispatch(req)!.units, { characterCount: 5 }); assertEquals(getRequestDispatch(req)!.providerStatus, 200); // Audio carries no token block and OpenAI reports none, so tokens stay // absent rather than becoming a billed zero. assertEquals(getRequestDispatch(req)!.tokens, undefined); }); // D1-T4, route half: duration-bearing transcription. await t.step( "D1-T4: audioSeconds from a duration-bearing body", async () => { const req = transcriptionRequest(); const res = await router.handle(req); assertEquals(res.status, 200); assertEquals((await res.json()).text, "words"); assertEquals(getRequestDispatch(req)!.units, { audioSeconds: 12.5 }); assertEquals(getRequestDispatch(req)!.tokens, undefined); }, ); } finally { await mock.close(); } }); Deno.test("D1-T4: usage.seconds wins over duration on the wire", async () => { const mock = new MockProvider(() => jsonResponse({ text: "words", duration: 99, usage: { seconds: 4.25 } }) ); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { const req = transcriptionRequest(); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); assertEquals(getRequestDispatch(req)!.units, { audioSeconds: 4.25 }); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // D1-T27 - the C1 regression. A client must not be billed for a provider that // was never reached, and must not buy media by making the gateway fail. // --------------------------------------------------------------------------- /** * Stands in for BOTH halves that are not built yet: Stage 4's channel read at * the accounting sites, and P3's per-Mchar rate field on `UsageTokens` (which * `PricingService.costMicroUsd` does not yet accept - it has no media unit * fields at all). Everything else is real: the route, the OpenAI adapter, * mapDispatchError, VirtualKeyManager and GovernanceHierarchy including their * fail-closed admission chain. * * $15 per million characters is the design's own tts-1 figure. The design * measured the pre-fix over-bill at 30 000 000 micro-USD from a 2 000 000 * character input - but `MAX_TTS_INPUT_CHARS` (100 000), which B.4 introduced in * the same document and Stage 1 shipped, now refuses that request with a gateway * 400 before any dispatch. The largest admissible input is 100 000 characters, * so the largest reachable over-bill is 1 500 000 micro-USD per POST, repeatable * and still walking to team and customer. */ const TTS_MICRO_USD_PER_MCHAR = 15_000_000; /** The largest TTS input the gateway admits, i.e. the biggest single C1 loss. */ const MAX_INPUT = "x".repeat(MAX_TTS_INPUT_CHARS); const MAX_INPUT_MICRO_USD = 1_500_000; function settleFromChannelStandIn( req: Request, ctx: AppContext, keyId: string, ): void { const channel = getRequestDispatch(req); if (!channel) return; const characters = channel.units?.characterCount; if (characters === undefined) return; const cost = Math.round( (characters * TTS_MICRO_USD_PER_MCHAR) / 1_000_000, ); ctx.virtualKeys.recordCost(keyId, cost, false); const key = ctx.virtualKeys.get(keyId)!; ctx.hierarchy!.recordCost(key.teamId, cost, false); } Deno.test("D1-T27: a provider 400 and a refused connection each bill zero", async (t) => { // Arm A: the provider answers 400. mapDispatchError returns a Response // carrying the PROVIDER's 400, so a read-site `ok` gate cannot tell this // apart from a gateway 400 - which is exactly why the WRITE is gated. const mock = new MockProvider(() => jsonResponse({ error: { message: "voice not found" } }, 400) ); const arm = (mockUrl: string) => { const ctx = makeContext(mockUrl); ctx.hierarchy!.upsertCustomer({ id: "cust-1", name: "acme", enabled: true, budget: { maxCostUsd: 1 }, usedRequests: 0, usedCostMicroUsd: 0, }); ctx.hierarchy!.upsertTeam({ id: "team-1", name: "core", enabled: true, customerId: "cust-1", budget: { maxCostUsd: 1 }, usedRequests: 0, usedCostMicroUsd: 0, }); ctx.virtualKeys.upsert({ id: "vk-1", name: "media", token: "vk-media-c1-token", enabled: true, teamId: "team-1", budget: { maxCostUsd: 1 }, // 1 000 000 micro-USD usedRequests: 0, usedCostMicroUsd: 0, }); return { ctx, router: mediaRouter(ctx) }; }; const billed = (ctx: AppContext) => ({ key: ctx.virtualKeys.get("vk-1")!.usedCostMicroUsd, team: ctx.hierarchy!.getTeam("team-1")!.usedCostMicroUsd, customer: ctx.hierarchy!.getCustomer("cust-1")!.usedCostMicroUsd, }); try { await t.step( "provider 400: zero billed, providerStatus recorded", async () => { const { ctx, router } = arm(mock.url); const req = speechRequest(MAX_INPUT); const res = await router.handle(req); settleFromChannelStandIn(req, ctx, "vk-1"); // The client sees the provider's own status, which is the whole reason a // read-site gate cannot work. assertEquals(res.status, 400); await res.body?.cancel(); const channel = getRequestDispatch(req)!; // The evidence a provider was reached IS recorded ... assertEquals(channel.providerStatus, 400); // ... and no quantity was ever written, on any of the three surfaces. assertEquals(channel.units, undefined); assertEquals(channel.tokens, undefined); assertEquals(billed(ctx), { key: 0, team: 0, customer: 0 }); // The next admission is admitted, not 402: the key's own budget, the // team's and the customer's are all intact. assertEquals(ctx.virtualKeys.check("vk-media-c1-token", 0).ok, true); assertEquals(ctx.hierarchy!.checkChain("team-1").ok, true); }, ); await t.step( "refused connection: zero billed, no providerStatus", async () => { // Port 1 on loopback: nothing listens, so fetch rejects with a TypeError // and NO Response is ever produced - there is no status for a read-site // gate to inspect at all. const { ctx, router } = arm("http://127.0.0.1:1/v1"); const req = speechRequest(MAX_INPUT); await assertRejects(() => router.handle(req)); settleFromChannelStandIn(req, ctx, "vk-1"); const channel = getRequestDispatch(req)!; assertEquals(channel.providerStatus, undefined); assertEquals(channel.units, undefined); assertEquals(channel.tokens, undefined); assertEquals(billed(ctx), { key: 0, team: 0, customer: 0 }); assertEquals(ctx.virtualKeys.check("vk-media-c1-token", 0).ok, true); assertEquals(ctx.hierarchy!.checkChain("team-1").ok, true); }, ); // The positive control: the same composition DOES bill when the provider // returns 2xx. Without this the two assertions above would also pass on a // channel that never writes anything at all. await t.step("control: a 2xx provider bills all three tiers", async () => { const ok = new MockProvider(() => new Response(new Uint8Array([1]), { headers: { "Content-Type": "audio/mpeg" }, }) ); try { const { ctx, router } = arm(ok.url); const req = speechRequest(MAX_INPUT); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); settleFromChannelStandIn(req, ctx, "vk-1"); assertEquals(getRequestDispatch(req)!.units, { characterCount: MAX_TTS_INPUT_CHARS, }); assertEquals(billed(ctx), { key: MAX_INPUT_MICRO_USD, team: MAX_INPUT_MICRO_USD, customer: MAX_INPUT_MICRO_USD, }); // And now the budget IS exhausted - the cross-tenant denial the pre-fix // shape produced from a provider 400. assertEquals(ctx.virtualKeys.check("vk-media-c1-token", 0).ok, false); assertEquals(ctx.hierarchy!.checkChain("team-1").ok, false); } finally { await ok.close(); } }); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // D1-T6 - the pre-headers abort, on all three surfaces // --------------------------------------------------------------------------- Deno.test("D1-T6: an abort before provider headers bills nothing on any surface", async () => { const mock = new MockProvider(() => jsonResponse({ unreachable: true })); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { const aborted = (): RequestInit => { const ac = new AbortController(); ac.abort(); return { signal: ac.signal }; }; const cases: Array<[string, Request]> = [ ["speech", speechRequest("hei", aborted())], ["images", imageRequest({ n: 1 }, aborted())], ["transcriptions", transcriptionRequest(aborted())], ]; let expected = 0; for (const [label, req] of cases) { const before = mock.calls.length; // mapDispatchError rethrows an AbortError, so the route throws out to the // errorHandler. The channel is what matters here, not the envelope. await assertRejects(() => router.handle(req), Error, "", label); assertEquals(mock.calls.length, before, `${label}: no provider call`); const channel = getRequestDispatch(req)!; assertEquals(channel.providerStatus, undefined, label); assertEquals(channel.units, undefined, label); assertEquals(channel.tokens, undefined, label); expected += 1; assertEquals( ctx.metrics.get("media.abort_pre_headers"), expected, `${label}: media.abort_pre_headers increments`, ); } // Every reservation taken pre-dispatch was released on the throw path. assertEquals(mediaInflightReserved(), 0); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // D1-T14 / D1-T15 - the write contract, at every route call site // --------------------------------------------------------------------------- Deno.test("D1-T14: a forced double write is refused and counted, per call site", async () => { const mock = new MockProvider((call) => call.path === "/audio/speech" ? new Response(new Uint8Array([1]), { headers: { "Content-Type": "audio/mpeg" }, }) : jsonResponse({ created: 1, data: [{ b64_json: "aa" }], text: "w" }) ); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { // Every media route must pass ctx.metrics to setRequestDispatch. A site // that omitted it would leave accounting.dispatch_rewritten silent and // D1-T15 vacuous, so each site is driven independently here. const sites: Array<[string, () => Request]> = [ ["speech", () => speechRequest("hei")], ["images", () => imageRequest({ n: 1 })], ["transcriptions", () => transcriptionRequest()], ]; let expected = 0; for (const [label, make] of sites) { const req = make(); // Claim the channel first, so the ROUTE's write is the second one. assertEquals( setRequestDispatch(req, { providerId: "squatter", model: "planted" }), true, ); const res = await router.handle(req); assertEquals(res.status, 200, label); await res.body?.cancel(); expected += 1; assertEquals( ctx.metrics.get("accounting.dispatch_rewritten"), expected, `${label}: the refused rewrite is counted with the real ctx.metrics`, ); const channel = getRequestDispatch(req)!; assertEquals(channel.providerId, "squatter", `${label}: target intact`); assertEquals(channel.model, "planted", `${label}: model intact`); // The quantity still lands: a refused rewrite is not a refused merge. assert(channel.units !== undefined, `${label}: units still written`); } assertEquals(expected, 3); } finally { await mock.close(); } }); Deno.test("D1-T15: accounting.dispatch_rewritten stays 0 on normal traffic", async () => { const mock = new MockProvider((call) => call.path === "/audio/speech" ? new Response(new Uint8Array([1]), { headers: { "Content-Type": "audio/mpeg" }, }) : jsonResponse({ created: 1, data: [{ b64_json: "aa" }], text: "w", duration: 1, object: "list", }) ); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { const requests = [ speechRequest("hei"), imageRequest({ n: 2 }), transcriptionRequest(), imageRequest({ sampleCount: 3 }), speechRequest("moro"), new Request(`${base}/v1/embeddings`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: "openai/gpt-image-1", input: "x" }), }), ]; for (const req of requests) { const res = await router.handle(req); await res.body?.cancel(); } assertEquals(ctx.metrics.get("accounting.dispatch_rewritten"), 0); assertEquals(ctx.metrics.get("accounting.status_dropped"), 0); assertEquals(mediaInflightReserved(), 0); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // D1-T17 (over-cap half) - the 413 -> 502 remap // --------------------------------------------------------------------------- Deno.test("D1-T17: an over-cap provider body is 502, unbilled and counted", async () => { // A transcription reserves MAX_TRANSCRIPTION_JSON_BYTES (4 MiB); this body is // deliberately past it. readCappedBytes throws its REQUEST-shaped 413 and the // route must not let that reach the client. const oversized = "z".repeat(MAX_TRANSCRIPTION_JSON_BYTES + 1024); const mock = new MockProvider(() => jsonResponse({ text: oversized, duration: 5 }) ); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { const before = mediaInflightReserved(); const req = transcriptionRequest(); const res = await router.handle(req); assertEquals(res.status, 502); const body = await res.text(); const parsed = JSON.parse(body) as { error: { type: string; code: string; message: string }; }; assertEquals(parsed.error.type, "provider_error"); assertEquals(parsed.error.code, "media_body_cap_exceeded"); assert(parsed.error.message.includes(String(MAX_TRANSCRIPTION_JSON_BYTES))); // The 413's own wording describes a REQUEST body and must never surface on // a provider read. assert( !body.includes("Request body exceeds the maximum allowed size."), "the request-shaped 413 message must not reach a client", ); assertEquals(ctx.metrics.get("media.body_cap_exceeded"), 1); // The observable unbilled row: the provider WAS reached (200 recorded), and // no quantity was written, so nothing can be billed for it. const channel = getRequestDispatch(req)!; assertEquals(channel.providerStatus, 200); assertEquals(channel.units, undefined); assertEquals(channel.tokens, undefined); // D1-T23's cap path: the reservation came back. assertEquals(mediaInflightReserved(), before); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // D1-T18 - fetchGuarded pins the attempt count to one // --------------------------------------------------------------------------- Deno.test("D1-T18: a provider 429 on an image request is ONE upstream attempt", async () => { const mock = new MockProvider(() => jsonResponse({ error: { message: "slow down" } }, 429) ); // maxRetries: 3 is the shipped default, and it is what makes this assertion // non-vacuous: through fetchWithRetry this single request would cost four // paid attempts. const ctx = makeContext(mock.url, { maxRetries: 3, initialDelayMs: 1 }); const router = mediaRouter(ctx); try { const before = mock.calls.length; const imageReq = imageRequest({ n: 1 }); const res = await router.handle(imageReq); assertEquals(res.status, 429); await res.body?.cancel(); assertEquals( mock.calls.length - before, 1, "exactly one upstream attempt on a provider 429", ); // rawProxy signals a provider error by THROWING, so the provider's own // status only reaches the channel from dispatchBufferedMedia's catch. Pin // it: without this, a 4xx/5xx from a reached provider is indistinguishable // from a refused connection, which is the discrimination the C1 write gate // and the settle path both read. assertEquals(getRequestDispatch(imageReq)!.providerStatus, 429); assertEquals(getRequestDispatch(imageReq)!.units, undefined); // The same pin on the other two paid media surfaces. const speech = await router.handle(speechRequest("hei")); assertEquals(speech.status, 429); await speech.body?.cancel(); const transcriptionReq = transcriptionRequest(); const transcription = await router.handle(transcriptionReq); assertEquals(transcription.status, 429); await transcription.body?.cancel(); assertEquals(mock.calls.length - before, 3); // The second dispatchBufferedMedia route, same catch. assertEquals(getRequestDispatch(transcriptionReq)!.providerStatus, 429); assertEquals(getRequestDispatch(transcriptionReq)!.units, undefined); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // D1-T23 / D1-T28 - the in-flight byte budget and the reservation it takes // --------------------------------------------------------------------------- Deno.test("D1-T23/T28: the in-flight budget refuses pre-dispatch and always releases", async (t) => { const parked = gate(); const mock = new MockProvider(async (call) => { if (call.path === "/images/generations") { await parked.wait; return jsonResponse({ created: 1, data: [{ b64_json: "aa" }] }); } return jsonResponse({ text: "w", duration: 1 }); }); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { assertEquals(mediaInflightReserved(), 0); // D1-T28's reservation clause: `n: 10` reserves exactly the derived // per-response ceiling, so the reservation IS the read cap for it. const held = router.handle(imageRequest({ n: 10 })); await until(() => mediaInflightReserved() > 0, "the n=10 reservation"); await t.step("D1-T28: n = 10 reserves exactly MAX_MEDIA_JSON_BYTES", () => { assertEquals(mediaInflightReserved(), MAX_MEDIA_JSON_BYTES); assertEquals(MAX_MEDIA_JSON_BYTES, 10 * MEDIA_BYTES_PER_IMAGE); }); await t.step( "D1-T23: over budget is 429 + Retry-After, pre-dispatch", async () => { // 90 MiB held + 90 MiB requested is over the 128 MiB budget. const before = mock.calls.length; const res = await router.handle(imageRequest({ n: 10 })); assertEquals(res.status, 429); assertEquals(res.headers.get("Retry-After"), "1"); const body = await res.json() as { error: { type: string; code: string }; }; // The house pattern: governance_error, never rate_limit_error. assertEquals(body.error.type, "governance_error"); assertEquals(body.error.code, "media_inflight"); assertEquals( mock.calls.length, before, "refused BEFORE the provider is reached", ); assertEquals(ctx.metrics.get("media.inflight_rejected"), 1); // Nothing was spent, so there is nothing to bill and no channel row. assertEquals(mediaInflightReserved(), MAX_MEDIA_JSON_BYTES); }, ); await t.step( "D1-T23: a fitting request is still admitted alongside", async () => { // 38 MiB of budget remain; a 4 MiB transcription fits. const req = transcriptionRequest(); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); assertEquals(getRequestDispatch(req)!.units, { audioSeconds: 1 }); assertEquals(mediaInflightReserved(), MAX_MEDIA_JSON_BYTES); }, ); await t.step( "D1-T23: the reservation returns on the success path", async () => { parked.open(); const res = await held; assertEquals(res.status, 200); await res.body?.cancel(); assertEquals(mediaInflightReserved(), 0); }, ); } finally { parked.open(); await mock.close(); } }); Deno.test("D1-T23: the reservation returns on the throw path too", async () => { // Nothing listens on port 1, so the dispatch throws before any headers. const ctx = makeContext("http://127.0.0.1:1/v1"); const router = mediaRouter(ctx); const before = mediaInflightReserved(); await assertRejects(() => router.handle(imageRequest({ n: 10 }))); assertEquals(mediaInflightReserved(), before); await assertRejects(() => router.handle(transcriptionRequest())); assertEquals(mediaInflightReserved(), before); }); // --------------------------------------------------------------------------- // The in-flight budget's other half: a provider that never answers must not be // able to hold the budget against OTHER callers indefinitely. The budget is a // per-process resource, so a media path with no establishment timeout turns one // caller's hung provider into everybody's 429 - which is what Stage 3 made // reachable by adding the budget in the first place. // --------------------------------------------------------------------------- Deno.test("a never-answering provider releases the in-flight budget on the establishment timeout", async () => { // Raw TCP: the connection is accepted and not one byte of response is ever // sent, so not even headers arrive. A MockProvider cannot express this. const listener = Deno.listen({ hostname: "127.0.0.1", port: 0 }); const accepted: Deno.Conn[] = []; const accepting = (async () => { for await (const conn of listener) accepted.push(conn); })().catch(() => {}); const upstream = `http://127.0.0.1:${(listener.addr as Deno.NetAddr).port}`; const ctx: AppContext = { ...makeContext(upstream), providers: new ProviderManager([{ id: "azure", type: "azure", apiKey: "az", endpoint: upstream, apiVersion: "2024-06-01", enabled: true, models: ["dall-e-3"], priority: 0, // A short establishment budget so the release is observable in a test; // production takes FROSTY_HTTP_TIMEOUT_MS, else 120 s. network: { maxRetries: 3, timeoutSec: 0.3 }, }], "azure"), metrics: new Metrics(), }; const router = mediaRouter(ctx); const azureImage = () => new Request(`${base}/v1/images/generations`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: "azure/dall-e-3", prompt: "fjord", n: 1 }), }); // 14 x 9 MiB = 126 MiB of the 128 MiB budget: a 15th n=1 cannot fit. const n = Math.floor(MAX_MEDIA_INFLIGHT_BYTES / MEDIA_BYTES_PER_IMAGE); const stalled = Array.from( { length: n }, () => router.handle(azureImage()).catch((e: unknown) => e), ); try { await until( () => mediaInflightReserved() === n * MEDIA_BYTES_PER_IMAGE, "every stalled reservation", ); // A DIFFERENT caller, refused for a provider it never touched. const before = ctx.metrics.get("requests.images"); const refused = await router.handle(azureImage()); assertEquals(refused.status, 429); assertEquals( ((await refused.json()) as { error: { code: string } }).error.code, "media_inflight", ); assertEquals(ctx.metrics.get("media.inflight_rejected"), 1); assertEquals(ctx.metrics.get("requests.images"), before + 1); // And it is reclaimed: the establishment timeout ends every stalled // dispatch, and the `finally` gives every reservation back. Polled with a // bound rather than awaited: without a timeout on this path the dispatches // never settle at all, and an `await` on them would hang the suite instead // of failing it. await until( () => mediaInflightReserved() === 0, "the in-flight budget to be reclaimed", ); await Promise.allSettled(stalled); // The proof that the refusal was transient rather than terminal: the same // shape is admitted now, and no second refusal is counted. const admitted = await router.handle(azureImage()).catch((e: unknown) => e); assert( admitted instanceof DOMException && admitted.name === "TimeoutError", "admitted through to the provider, then timed out on its own budget", ); assertEquals(ctx.metrics.get("media.inflight_rejected"), 1); assertEquals(mediaInflightReserved(), 0); } finally { listener.close(); for (const conn of accepted) { try { conn.close(); } catch { /* already closed by the timeout */ } } await Promise.allSettled(stalled); await accepting; } }); // --------------------------------------------------------------------------- // onUsage - the surface whose reply cannot carry its own usage // --------------------------------------------------------------------------- Deno.test("onUsage: TTS token usage is reported and bounded route-side", async (t) => { // A fake rawProxy standing in for GeminiAdapter.speech: it reports the usage // block through the callback, exactly as the real one now does. let reportNext: { prompt?: number; completion?: number; total?: number } = {}; const ctx = makeContext("http://127.0.0.1:1/v1"); const target = ctx.providers.resolve("openai/tts-1"); target.adapter.rawProxy = (_path, _req, context) => { context?.onUsage?.(reportNext); return Promise.resolve( new Response(new Uint8Array([1]), { headers: { "Content-Type": "audio/wav" }, }), ); }; const router = mediaRouter(ctx); await t.step("a reported block reaches the channel", async () => { reportNext = { prompt: 31, completion: 7, total: 38 }; const req = speechRequest("hei"); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); const channel = getRequestDispatch(req)!; assertEquals(channel.tokens, { prompt: 31, completion: 7, cached: 0, cacheCreation: 0, }); // The character quantity is unaffected: a token-priced TTS vendor still // gets its characters counted. assertEquals(channel.units, { characterCount: 3 }); }); await t.step( "negative, fractional, Infinity and 1e308 are clamped", async () => { for ( const [reported, want] of [ [{ prompt: -5, completion: 3.9 }, { prompt: 0, completion: 3 }], [{ prompt: Infinity, completion: 1 }, { prompt: 0, completion: 1 }], [{ prompt: 1e308, completion: NaN }, { prompt: Number.MAX_SAFE_INTEGER, completion: 0, }], ] as const ) { reportNext = reported; const req = speechRequest("hei"); const res = await router.handle(req); await res.body?.cancel(); assertEquals(getRequestDispatch(req)!.tokens, { ...want, cached: 0, cacheCreation: 0, }); } }, ); await t.step("no reported block leaves tokens absent", async () => { let called = false; target.adapter.rawProxy = (_path, _req, _context) => { called = true; return Promise.resolve( new Response(new Uint8Array([1]), { headers: { "Content-Type": "audio/wav" }, }), ); }; const req = speechRequest("hei"); const res = await router.handle(req); await res.body?.cancel(); assert(called); assertEquals(getRequestDispatch(req)!.tokens, undefined); }); }); // --------------------------------------------------------------------------- // The client-visible half, through the real middleware chain // --------------------------------------------------------------------------- Deno.test("the 429 and the 502 survive the real middleware chain", async () => { const parked = gate(); const mock = new MockProvider(async (call) => { if (call.path === "/images/generations") { await parked.wait; return jsonResponse({ created: 1, data: [{ b64_json: "aa" }] }); } return jsonResponse({ text: "z".repeat(MAX_TRANSCRIPTION_JSON_BYTES + 8) }); }); const ctx = makeContext(mock.url); const handler = createHandler(ctx); try { const held = handler(imageRequest({ n: 10 })); await until(() => mediaInflightReserved() > 0, "the held reservation"); const refused = await handler(imageRequest({ n: 10 })); assertEquals(refused.status, 429); assertEquals(refused.headers.get("Retry-After"), "1"); assertEquals( ((await refused.json()) as { error: { code: string } }).error.code, "media_inflight", ); const overCap = await handler(transcriptionRequest()); assertEquals(overCap.status, 502); const text = await overCap.text(); assert(!text.includes("Request body exceeds the maximum allowed size.")); assertEquals( (JSON.parse(text) as { error: { code: string } }).error.code, "media_body_cap_exceeded", ); parked.open(); const ok = await held; assertEquals(ok.status, 200); await ok.body?.cancel(); assertEquals(mediaInflightReserved(), 0); } finally { parked.open(); await mock.close(); } }); // --------------------------------------------------------------------------- // B.4's second clause: `upstream.ok`, not merely "after the headers" // --------------------------------------------------------------------------- Deno.test("B.4: a !ok provider Response records the status and no quantity", async (t) => { // Every rawProxy and generateImage in this repo THROWS a ProviderError on a // non-2xx (condition 15 requires it), so `upstream.ok` is defence in depth // against an adapter that returns one instead. Without an arm like this, // deleting the guard is unkillable by any test - which is exactly how a write // gate quietly stops gating. const ctx = makeContext("http://127.0.0.1:1/v1"); const target = ctx.providers.resolve("openai/tts-1"); target.adapter.rawProxy = (path) => Promise.resolve( path === "/audio/speech" ? new Response(new Uint8Array([9]), { status: 500, headers: { "Content-Type": "audio/mpeg" }, }) // A body that WOULD count if the gate were absent: two images and a // duration, under a 503. : jsonResponse( { created: 1, data: [{ b64_json: "a" }, { b64_json: "b" }], duration: 3, }, 503, ), ); const router = mediaRouter(ctx); await t.step("speech", async () => { const req = speechRequest("hei"); const res = await router.handle(req); assertEquals(res.status, 500); await res.body?.cancel(); const channel = getRequestDispatch(req)!; assertEquals(channel.providerStatus, 500); assertEquals(channel.units, undefined); assertEquals(channel.tokens, undefined); }); await t.step("images", async () => { const req = imageRequest({ n: 1 }); const res = await router.handle(req); assertEquals(res.status, 503); await res.body?.cancel(); const channel = getRequestDispatch(req)!; assertEquals(channel.providerStatus, 503); assertEquals(channel.units, undefined); assertEquals(channel.tokens, undefined); assertEquals(mediaInflightReserved(), 0); }); await t.step("transcriptions", async () => { const req = transcriptionRequest(); const res = await router.handle(req); assertEquals(res.status, 503); await res.body?.cancel(); const channel = getRequestDispatch(req)!; assertEquals(channel.providerStatus, 503); assertEquals(channel.units, undefined); assertEquals(mediaInflightReserved(), 0); }); }); // --------------------------------------------------------------------------- // The native generateImage branch - the fourth channel call site // --------------------------------------------------------------------------- Deno.test("native generateImage writes the channel from the adapter's return", async (t) => { const ctx = makeContext("http://127.0.0.1:1/v1"); const target = ctx.providers.resolve("openai/gpt-image-1"); let thrown: unknown; target.adapter.generateImage = () => { if (thrown) return Promise.reject(thrown); return Promise.resolve({ created: 1, data: [{ b64_json: "a" }, { b64_json: "b" }, { b64_json: "c" }], usage: { input_tokens: 4, output_tokens: 5, total_tokens: 9 }, }); }; const router = mediaRouter(ctx); await t.step("units and tokens come from the resolved result", async () => { const req = imageRequest({ n: 7 }); const res = await router.handle(req); assertEquals(res.status, 200); assertEquals((await res.json()).data.length, 3); const channel = getRequestDispatch(req)!; // Three delivered against n: 7, exactly as on the rawProxy branch. assertEquals(channel.units, { imageCount: 3 }); assertEquals(channel.tokens, { prompt: 4, completion: 5, cached: 0, cacheCreation: 0, }); // Every generateImage adapter throws on a non-2xx, so a resolved call is // the evidence a provider was reached. The typed return cannot carry the // exact code, so 200 stands in for it. assertEquals(channel.providerStatus, 200); // This branch buffers inside the adapter, so it takes no reservation. assertEquals(mediaInflightReserved(), 0); }); // The native branch's own setRequestDispatch site - one of the three in // advanced.ts, reached from four route paths - driven with the real // ctx.metrics. await t.step("a forced double write here is counted too", async () => { const before = ctx.metrics.get("accounting.dispatch_rewritten"); const req = imageRequest({ n: 1 }); setRequestDispatch(req, { providerId: "squatter", model: "planted" }); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); assertEquals( ctx.metrics.get("accounting.dispatch_rewritten"), before + 1, ); assertEquals(getRequestDispatch(req)!.providerId, "squatter"); }); await t.step("a throwing adapter writes no quantity", async () => { thrown = new ProviderError(429, "Too Many Requests", '{"error":"slow"}'); const req = imageRequest({ n: 1 }); const res = await router.handle(req); assertEquals(res.status, 429); await res.body?.cancel(); const channel = getRequestDispatch(req)!; assertEquals(channel.units, undefined); assertEquals(channel.tokens, undefined); // The status still lands: the provider WAS reached and refused, which is a // different row from a provider that was never reached at all. assertEquals(channel.providerStatus, 429); thrown = undefined; }); }); // --------------------------------------------------------------------------- // B.4 row 4: the native row's "survives an after-headers abort". The three // steps above never drive an abort, and the branch used to be the only media // dispatch with the client's abort still attached across the provider call. // --------------------------------------------------------------------------- Deno.test("native generateImage survives a client abort after the provider committed", async (t) => { // A REAL native adapter shape: it owns its own fetch, reads the whole body and // returns a parsed object. A stub that ignored `signal` would pass every arm // below vacuously, so the abort has to reach a real in-flight HTTP read. const mock = new MockProvider(() => new Response( new ReadableStream({ async start(controller) { // Status + headers are already committed; only the body is stalled, // which is the after-headers window B.4 row 4 is about. await new Promise((r) => setTimeout(r, 200)); controller.enqueue( new TextEncoder().encode( JSON.stringify({ created: 1, data: [{ b64_json: "a" }, { b64_json: "b" }], }), ), ); controller.close(); }, }), { status: 200, headers: { "Content-Type": "application/json" } }, ) ); const ctx = makeContext(mock.url); const target = ctx.providers.resolve("openai/gpt-image-1"); let reachedProvider = 0; target.adapter.generateImage = async (_req, context) => { reachedProvider++; const upstream = await fetch(`${mock.url}/images:predict`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ prompt: "fjord" }), signal: context?.signal, }); return await upstream.json(); }; const router = mediaRouter(ctx); try { await t.step("no abort: two images counted", async () => { const req = imageRequest({ n: 2 }); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); const channel = getRequestDispatch(req)!; assertEquals(channel.units, { imageCount: 2 }); assertEquals(channel.providerStatus, 200); }); await t.step( "abort mid-body: the provider's work is still counted", async () => { const before = reachedProvider; const ac = new AbortController(); const req = imageRequest({ n: 2 }, { signal: ac.signal }); const inflight = router.handle(req); // Well inside the 200 ms body stall, and after the provider committed a // 200. The provider renders and charges either way; the only thing an // abort here can change is whether the gateway counts it. setTimeout(() => ac.abort(), 60); const res = await inflight; assertEquals(res.status, 200); await res.body?.cancel(); assertEquals(reachedProvider - before, 1); const channel = getRequestDispatch(req)!; assertEquals(channel.units, { imageCount: 2 }); assertEquals(channel.providerStatus, 200); // A settled dispatch is not a pre-headers abort. assertEquals(ctx.metrics.get("media.abort_pre_headers"), 0); }, ); await t.step( "already aborted at entry: the provider is never reached", async () => { const before = reachedProvider; const ac = new AbortController(); ac.abort(); const req = imageRequest({ n: 2 }, { signal: ac.signal }); // Detaching AFTER the dispatch must not become "never detach": an // abandoned request still has to cost the provider nothing. await assertRejects(() => router.handle(req)); assertEquals( reachedProvider - before, 1, "the adapter is entered, but its fetch is refused by the pre-aborted signal", ); assertEquals(getRequestDispatch(req)!.units, undefined); assertEquals(getRequestDispatch(req)!.providerStatus, undefined); assertEquals(ctx.metrics.get("media.abort_pre_headers"), 1); }, ); } finally { await mock.close(); } }); // --------------------------------------------------------------------------- // Condition 13's TTS clause: zero reservation, asserted while in flight // --------------------------------------------------------------------------- Deno.test("TTS holds no in-flight reservation at all", async (t) => { const parked = gate(); let seenDuringFlight = -1; const mock = new MockProvider((call) => { // Sampled inside the provider handler, i.e. strictly between the route's // reservation point and its release. A reservation taken for TTS would be // visible here and nowhere else. seenDuringFlight = mediaInflightReserved(); parked.open(); if (call.path === "/audio/speech") { return new Response(new Uint8Array([1]), { headers: { "Content-Type": "audio/mpeg" }, }); } return jsonResponse({ created: 1, data: [{ b64_json: "aa" }] }); }); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { assertEquals(mediaInflightReserved(), 0); const req = speechRequest("hei"); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); await parked.wait; assertEquals( seenDuringFlight, 0, "TTS streams through untouched, so it reserves nothing", ); assertEquals(mediaInflightReserved(), 0); assertEquals(getRequestDispatch(req)!.units, { characterCount: 3 }); // The control, sampled at the SAME point by the SAME handler: without it a // zero here would also be what a broken sampler reads. await t.step( "a buffered surface sampled the same way is non-zero", async () => { seenDuringFlight = -1; const image = await router.handle(imageRequest({ n: 2 })); assertEquals(image.status, 200); await image.body?.cancel(); assertEquals(seenDuringFlight, 2 * MEDIA_BYTES_PER_IMAGE); assertEquals(mediaInflightReserved(), 0); }, ); } finally { await mock.close(); } }); Deno.test("a dropped provider status is counted, never silent", async () => { const mock = new MockProvider(() => jsonResponse({ text: "w", duration: 2 })); const ctx = makeContext(mock.url); const router = mediaRouter(ctx); try { // mergeRequestStatus is first-write-wins, so a pre-recorded status makes the // route's own write a no-op. That is a silently unbilled row unless counted. const req = transcriptionRequest(); setRequestDispatch(req, { providerId: "openai", model: "whisper-1" }); mergeRequestStatus(req, 418); const res = await router.handle(req); assertEquals(res.status, 200); await res.body?.cancel(); assertEquals(ctx.metrics.get("accounting.status_dropped"), 1); assertEquals(getRequestDispatch(req)!.providerStatus, 418); } finally { await mock.close(); } });