# ============================================================================= # Frosty Deno - full configuration surface # ============================================================================= # Every operator knob the gateway reads, grouped the same way as # docs/reference/environment-variables.md so the two can be diffed. For the # smallest config that boots, use `.env.example.dev` instead. # # cp .env.example .env # docker compose up -d postgres # REQUIRED (section 4) # deno task setup && deno task dev # # Conventions in this file: # * A blank value means "leave the feature off" - it is never a placeholder to # be filled in blindly. Secrets ship blank on purpose. # * A value that IS filled in is either a real default or an inert format # example (a URL shape, a model list), safe to copy as-is. # * Anything absent from this file has a safe default. Adding a new knob means # a bounded parse, a row in the reference doc above, and a line here. # # Sections # 1 Provider credentials and provider catalogs # 2 Azure OpenAI, Bedrock, and Vertex AI # 3 Generic compatible endpoints # 4 Core gateway and PostgreSQL <- required # 5 Worker topology and shared governance # 6 Admin protection and origin control # 7 Cache and vector store # 8 MCP and Code Mode # 9 Logging, analytics display, and observability # 10 Pricing sync, encryption, plugins, HTTP client # 11 Not operator knobs (harness, migration, supervisor-set) # ============================================================================= # 1. Provider credentials and provider catalogs # ============================================================================= # Any subset. A provider registers itself at boot when its credential is set, # and stays absent otherwise, so a blank key is a supported state rather than a # broken one. Persisted config from /api/providers overlays these and WINS on an # id collision. # OpenAI-wire vendors. OPENAI_API_KEY= ANTHROPIC_API_KEY= GEMINI_API_KEY= OPENROUTER_API_KEY= GROQ_API_KEY= MISTRAL_API_KEY= XAI_API_KEY= PERPLEXITY_API_KEY= CEREBRAS_API_KEY= NEBIUS_API_KEY= PARASAIL_API_KEY= # Native adapters (their own wire formats, translated to canonical internally). HF_TOKEN= COHERE_API_KEY= # Audio only: /v1/audio/speech and /v1/audio/transcriptions. ELEVENLABS_API_KEY= # Ollama needs no key; it registers on BASE_URL alone. MODELS is a # comma-separated catalog, because Ollama's tag list is host-specific. OLLAMA_BASE_URL= OLLAMA_MODELS= # ============================================================================= # 2. Azure OpenAI, Bedrock, and Vertex AI # ============================================================================= # These three do not take a bare API key alone: each needs its own coordinates # before a model name can be resolved to an endpoint. # Azure routes per DEPLOYMENT, not per model, so the deployment list is what # makes models addressable. Comma-separated. AZURE_OPENAI_API_KEY= AZURE_OPENAI_ENDPOINT=https://my-resource.openai.azure.com AZURE_OPENAI_API_VERSION=2024-02-15-preview AZURE_OPENAI_DEPLOYMENTS=gpt-4o-deployment,gpt-35-deployment # AWS Bedrock (SigV4). SESSION_TOKEN only for temporary credentials. AWS_REGION=us-east-1 AWS_ACCESS_KEY_ID= AWS_SECRET_ACCESS_KEY= AWS_SESSION_TOKEN= BEDROCK_MODELS= # Vertex AI. The service-account JSON goes in as ONE line, quotes intact. VERTEX_PROJECT_ID= VERTEX_LOCATION=us-central1 VERTEX_SERVICE_ACCOUNT_JSON= VERTEX_MODELS=gemini-2.5-pro # ============================================================================= # 3. Generic compatible endpoints # ============================================================================= # Point Frosty at any OpenAI-wire or Anthropic-wire server. Each stays off until # its BASE_URL is set; the API key is optional because many local servers take # none. # Any OpenAI-wire server (vLLM, llama.cpp, TGI, ...). Id: `openai-compatible`. OPENAI_COMPAT_BASE_URL= OPENAI_COMPAT_API_KEY= OPENAI_COMPAT_DEFAULT_MODEL= # Any Anthropic Messages-wire server. Id: `anthropic-compatible`. ANTHROPIC_COMPAT_BASE_URL= ANTHROPIC_COMPAT_API_KEY= ANTHROPIC_COMPAT_DEFAULT_MODEL= # LM Studio (local OpenAI-compatible server; its default base URL shown). LMSTUDIO_BASE_URL=http://localhost:1234/v1 LMSTUDIO_API_KEY= LMSTUDIO_DEFAULT_MODEL= # ============================================================================= # 4. Core gateway and PostgreSQL # ============================================================================= PORT=8080 # Provider account used for model names without a `provider/` prefix. Must match # a registered id from section 1-3. FROSTY_DEFAULT_PROVIDER=openai # --- PostgreSQL: the gateway's ONE stateful dependency, and it is REQUIRED ---- # Holds the control-plane config, governance counters, the request-log trail, # the L2 response cache, and the pgvector embedding index. There is no fallback: # an unreachable or unset URL ABORTS BOOT rather than serving with empty # governance state (decision-log 61). Deno KV, which used to hold this, is # retired - migrate an existing data/frosty.kv with # `deno task migrate:kv-pg -- --commit`. # # docker compose up -d postgres # # Credentials are required; the Compose service sets user/password/db to # `frosty`. Use `localhost` from the host, `postgres` from inside Compose. FROSTY_PG_URL=postgres://frosty:frosty@localhost:5432/frosty # Session-stable connection used ONLY for LISTEN (cross-process cache # invalidation). Defaults to FROSTY_PG_URL, which is correct until a pooler sits # in between: LISTEN through PgBouncer's transaction mode stops delivering # SILENTLY. With `--profile pgbouncer` up, point FROSTY_PG_URL at :6432 and # leave this one on :5432. FROSTY_PG_DIRECT_URL= # Connections held PER PROCESS (default 8, max 100). The number PostgreSQL sees # is this times FROSTY_WORKERS, plus one LISTEN connection per process. FROSTY_PG_POOL_SIZE= # Table holding the semantic cache's embedding vectors. FROSTY_PG_TABLE=frosty_vectors # ============================================================================= # 5. Worker topology and shared governance # ============================================================================= # Worker processes sharing one port through SO_REUSEPORT. Unset or 1 = single # process. LINUX/macOS ONLY - Windows has no SO_REUSEPORT and the second bind # fails with os error 10048, so the gateway logs why and serves single-process # instead. See docs/guides/multi-process.md. FROSTY_WORKERS= # Fleet-wide rate limiting. Fixed rate/token windows live in an in-process Map # by default, which is exact for ONE process and admits N times the limit across # N. `auto` (default) moves them to a shared PostgreSQL authority only when # FROSTY_WORKERS>1, because that reservation costs ~1.8 ms per governed request # versus ~1 us in memory. Set `on` when running separate REPLICAS (auto cannot # see those); `off` accepts N-times-the-limit. auto|on|off FROSTY_SHARED_RATE_LIMIT= # How often each process re-reads durable config, in ms (0-3600000, 0 disables). # Config changes normally arrive over LISTEN/NOTIFY within milliseconds; this # poll is the backstop that bounds staleness when a notification is lost, so a # revoked virtual key stops working even then. Default 30000. FROSTY_CONFIG_RECONCILE_MS= # ============================================================================= # 6. Admin protection and origin control # ============================================================================= # Unset = explicit local admin mode (no auth on /api/*). Set to require # "Authorization: Bearer " on all /api/* config routes. FROSTY_ADMIN_TOKEN= # Admin origin guard allow-list (DNS-rebind defense). localhost, 127.0.0.1, and # ::1 are always allowed; add your public host(s) here (comma-separated) when # the gateway is reachable beyond localhost. Applies to every /api/* request. FROSTY_ALLOWED_HOSTS= # ============================================================================= # 7. Cache and vector store # ============================================================================= # Unset = off, "exact" = exact-match, "semantic" adds embedding similarity via # the embed model below (the provider must support embeddings). FROSTY_CACHE= FROSTY_CACHE_TTL_MS= # Sent VERBATIM and case-sensitive, so it must match an id the provider serves # (check GET /v1/models). Lookups are fail-open, so a wrong id reduces the cache # to exact-match only; every failed lookup warns `semantic cache lookup degraded # to miss`, and frosty_cache_events_total stays at 100% result="miss". FROSTY_CACHE_EMBED_MODEL=text-embedding-3-small # Similarity vectors live in-process unless FROSTY_VECTOR_STORE=pgvector puts # them in the same PostgreSQL as everything else, which is also what makes them # survive a restart. The Redis (RediSearch) option was removed: Compose # provisioned it and no shipped configuration ever selected it, so it was a # dependency that served zero requests (decision-log 60). FROSTY_VECTOR_STORE= # ============================================================================= # 8. MCP and Code Mode # ============================================================================= # MCP client transports are configured per server via POST /api/mcp/clients # ("transport": "http-sse" (default) | "streamable-http" | "auto" | "stdio"). # Servers that only speak streamable-http need an explicit transport (or "auto" # for spec-order negotiation). # # stdio (subprocess) MCP servers are DISABLED by default. Enabling them requires # BOTH this flag ("1" or "true") AND running with --allow-run, which is outside # the standard permission set on purpose (see permissions.md). FROSTY_MCP_ALLOW_STDIO= # Background MCP health sweeps (0/unset = on-demand via /api/mcp/health only). FROSTY_MCP_HEALTH_INTERVAL_MS= # Code Mode is default-off behind TWO independent gates. The VFS metadata # surface is live and inert; the sandboxed executor is HARD-OFF and # experimental. off|on (default off). The executor additionally requires a # per-request `x-frosty-code-mode: run` header and still refuses (501) because # the run primitive is intentionally stubbed. FROSTY_CODE_MODE=off FROSTY_CODE_MODE_VFS=on # ============================================================================= # 9. Logging, analytics display, and observability # ============================================================================= # Durable request-log store, in the same PostgreSQL as the rest of the state. ON # by default. Disable with FROSTY_LOG_STORE=off. The legacy value `kv` is still # accepted and means "on" - it named the retired Deno KV backend, and rejecting # it would break existing .env files over a backend that is gone. FROSTY_LOG_STORE=pg FROSTY_LOG_STORE_MAX=5000 # Paths kept OUT of the Logs dashboard trail (live stream + durable store). The # container healthcheck and the Prometheus scrape hit the gateway on a fixed # interval, so without this they accumulate until they are ~99% of the capped # trail and real requests get pruned away. Console access logging is NOT # affected: `docker logs` still shows every request. Blank uses the default # below; set to `off` to log everything; `/prefix/*` matches a subtree. FROSTY_LOG_EXCLUDE_PATHS=/healthz,/metrics,/favicon.ico # Request/response CONTENT capture in the durable log store (OPT-IN; default OFF # for privacy; secrets are never captured). on to enable. FROSTY_LOG_CONTENT= # Display currency: cost is accounted in USD internally; the Control UI presents # euros by multiplying by this EUR-per-USD rate (default 0.92). Operator-only. FROSTY_EUR_RATE=0.92 # OpenTelemetry OTLP/HTTP trace export. Unset = off. For Docker Compose use # http://otel-collector:4318 and start `--profile observability`; spans go to # Tempo for drill-down and to Prometheus as RED metrics. OTEL_EXPORTER_OTLP_ENDPOINT= OTEL_FLUSH_INTERVAL_MS=5000 # Distinct models admitted as the `frosty.metrics.model` span-metric label # before the rest fold to "other". Bounds Prometheus series growth; traces keep # the real model either way. Non-positive/unparseable falls back to the default. FROSTY_OTEL_MODEL_CARDINALITY_CAP=11 # ============================================================================= # 10. Pricing sync, encryption, plugins, HTTP client # ============================================================================= # LiteLLM pricing sync. Opt-in and DEFAULT OFF (offline/no-outbound default): # set FROSTY_PRICING_SYNC=on to fetch model prices + metadata at boot and # refresh on an interval. Operator /api/pricing overrides always win over synced # prices. POST /api/pricing/force-sync triggers a sync on demand even when this # is off. FROSTY_PRICING_SYNC= # Refresh cadence (ms); default 24h, floored at 60s. FROSTY_PRICING_SYNC_INTERVAL_MS=86400000 # Source URL for the price list (server-side only; never taken from a request). FROSTY_PRICING_URL=https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json # --- Config secret encryption-at-rest (OPT-IN; default OFF = plaintext) ------- # Set a base64-encoded 32-byte key (preferred) OR a strong passphrase (PBKDF2). # When set, provider API keys / AWS secret+session / Vertex SA-JSON / proxy # password / CA cert / virtual-key tokens / MCP header values are AES-256-GCM # encrypted in PostgreSQL. Fail-closed: once a store has encrypted data, an # unset or wrong key REFUSES boot. The key is effectively set-once (rotate via # _OLD). # Generate: `deno eval "console.log(btoa(String.fromCharCode(...crypto.getRandomValues(new Uint8Array(32)))))"` FROSTY_ENCRYPTION_KEY= # Rotation only: set to the previous key alongside a new FROSTY_ENCRYPTION_KEY # to rewrap the data-encryption key offline (data is not re-encrypted). FROSTY_ENCRYPTION_KEY_OLD= # JSON-repair plugin (OPT-IN; default OFF). Repairs invalid-JSON model output # post-response and, for streams, post-completion via the reconstructed message # (the live client stream is never mutated). on|1|true|yes to enable. FROSTY_JSON_REPAIR= # Request mocker: short-circuits upstream calls with synthetic responses, for # offline dev/demo/load-testing. OPT-IN and OFF unless set to on|1|true|yes - # leave it empty for a realistic deployment. Rules come from # FROSTY_MOCKER_CONFIG (inline JSON starting with `{`, or a file path). FROSTY_MOCKER= FROSTY_MOCKER_CONFIG= # Provider HTTP client: default per-request timeout in ms (default 120000; 0 # disables). Never total-caps an in-progress SSE/eventstream. FROSTY_NO_PROXY # takes comma-separated bypass patterns (*, .example.com, *.example.com, or an # exact host). FROSTY_HTTP_TIMEOUT_MS=120000 FROSTY_NO_PROXY= # ============================================================================= # 11. Not operator knobs # ============================================================================= # Listed for parity with the code, so a reader who greps for one of these finds # out why it is not above. Do NOT set these in a deployment .env. # # FROSTY_BASE_URL target for scripts/full_suite.ts and the browser # harness; defaults to http://localhost:8080 # FROSTY_BENCH_TARGET scripts/load-bench.ts only # FROSTY_BENCH_KEY scripts/load-bench.ts only # FROSTY_BENCH_MODEL scripts/load-bench.ts only # FROSTY_KV_PATH read ONLY by scripts/migrate_kv_to_pg.ts, to find an # existing data/frosty.kv to migrate. It configures # nothing at runtime; Deno KV is retired. # FROSTY_WORKER_ROLE set BY the supervisor on each child it spawns # FROSTY_WORKER_INDEX set BY the supervisor on each child it spawns