329 lines
15 KiB
Plaintext
Executable File
329 lines
15 KiB
Plaintext
Executable File
# =============================================================================
|
|
# Frosty Deno - full configuration surface
|
|
# =============================================================================
|
|
# Every operator knob the gateway reads, grouped the same way as
|
|
# docs/reference/environment-variables.md so the two can be diffed. For the
|
|
# smallest config that boots, use `.env.example.dev` instead.
|
|
#
|
|
# cp .env.example .env
|
|
# docker compose up -d postgres # REQUIRED (section 4)
|
|
# deno task setup && deno task dev
|
|
#
|
|
# Conventions in this file:
|
|
# * A blank value means "leave the feature off" - it is never a placeholder to
|
|
# be filled in blindly. Secrets ship blank on purpose.
|
|
# * A value that IS filled in is either a real default or an inert format
|
|
# example (a URL shape, a model list), safe to copy as-is.
|
|
# * Anything absent from this file has a safe default. Adding a new knob means
|
|
# a bounded parse, a row in the reference doc above, and a line here.
|
|
#
|
|
# Sections
|
|
# 1 Provider credentials and provider catalogs
|
|
# 2 Azure OpenAI, Bedrock, and Vertex AI
|
|
# 3 Generic compatible endpoints
|
|
# 4 Core gateway and PostgreSQL <- required
|
|
# 5 Worker topology and shared governance
|
|
# 6 Admin protection and origin control
|
|
# 7 Cache and vector store
|
|
# 8 MCP and Code Mode
|
|
# 9 Logging, analytics display, and observability
|
|
# 10 Pricing sync, encryption, plugins, HTTP client
|
|
# 11 Not operator knobs (harness, migration, supervisor-set)
|
|
|
|
# =============================================================================
|
|
# 1. Provider credentials and provider catalogs
|
|
# =============================================================================
|
|
# Any subset. A provider registers itself at boot when its credential is set,
|
|
# and stays absent otherwise, so a blank key is a supported state rather than a
|
|
# broken one. Persisted config from /api/providers overlays these and WINS on an
|
|
# id collision.
|
|
|
|
# OpenAI-wire vendors.
|
|
OPENAI_API_KEY=
|
|
ANTHROPIC_API_KEY=
|
|
GEMINI_API_KEY=
|
|
OPENROUTER_API_KEY=
|
|
GROQ_API_KEY=
|
|
MISTRAL_API_KEY=
|
|
XAI_API_KEY=
|
|
PERPLEXITY_API_KEY=
|
|
CEREBRAS_API_KEY=
|
|
NEBIUS_API_KEY=
|
|
PARASAIL_API_KEY=
|
|
|
|
# Native adapters (their own wire formats, translated to canonical internally).
|
|
HF_TOKEN=
|
|
COHERE_API_KEY=
|
|
# Audio only: /v1/audio/speech and /v1/audio/transcriptions.
|
|
ELEVENLABS_API_KEY=
|
|
|
|
# Ollama needs no key; it registers on BASE_URL alone. MODELS is a
|
|
# comma-separated catalog, because Ollama's tag list is host-specific.
|
|
OLLAMA_BASE_URL=
|
|
OLLAMA_MODELS=
|
|
|
|
# =============================================================================
|
|
# 2. Azure OpenAI, Bedrock, and Vertex AI
|
|
# =============================================================================
|
|
# These three do not take a bare API key alone: each needs its own coordinates
|
|
# before a model name can be resolved to an endpoint.
|
|
|
|
# Azure routes per DEPLOYMENT, not per model, so the deployment list is what
|
|
# makes models addressable. Comma-separated.
|
|
AZURE_OPENAI_API_KEY=
|
|
AZURE_OPENAI_ENDPOINT=https://my-resource.openai.azure.com
|
|
AZURE_OPENAI_API_VERSION=2024-02-15-preview
|
|
AZURE_OPENAI_DEPLOYMENTS=gpt-4o-deployment,gpt-35-deployment
|
|
|
|
# AWS Bedrock (SigV4). SESSION_TOKEN only for temporary credentials.
|
|
AWS_REGION=us-east-1
|
|
AWS_ACCESS_KEY_ID=
|
|
AWS_SECRET_ACCESS_KEY=
|
|
AWS_SESSION_TOKEN=
|
|
BEDROCK_MODELS=
|
|
|
|
# Vertex AI. The service-account JSON goes in as ONE line, quotes intact.
|
|
VERTEX_PROJECT_ID=
|
|
VERTEX_LOCATION=us-central1
|
|
VERTEX_SERVICE_ACCOUNT_JSON=
|
|
VERTEX_MODELS=gemini-2.5-pro
|
|
|
|
# =============================================================================
|
|
# 3. Generic compatible endpoints
|
|
# =============================================================================
|
|
# Point Frosty at any OpenAI-wire or Anthropic-wire server. Each stays off until
|
|
# its BASE_URL is set; the API key is optional because many local servers take
|
|
# none.
|
|
|
|
# Any OpenAI-wire server (vLLM, llama.cpp, TGI, ...). Id: `openai-compatible`.
|
|
OPENAI_COMPAT_BASE_URL=
|
|
OPENAI_COMPAT_API_KEY=
|
|
OPENAI_COMPAT_DEFAULT_MODEL=
|
|
|
|
# Any Anthropic Messages-wire server. Id: `anthropic-compatible`.
|
|
ANTHROPIC_COMPAT_BASE_URL=
|
|
ANTHROPIC_COMPAT_API_KEY=
|
|
ANTHROPIC_COMPAT_DEFAULT_MODEL=
|
|
|
|
# LM Studio (local OpenAI-compatible server; its default base URL shown).
|
|
LMSTUDIO_BASE_URL=http://localhost:1234/v1
|
|
LMSTUDIO_API_KEY=
|
|
LMSTUDIO_DEFAULT_MODEL=
|
|
|
|
# =============================================================================
|
|
# 4. Core gateway and PostgreSQL
|
|
# =============================================================================
|
|
PORT=8080
|
|
|
|
# Provider account used for model names without a `provider/` prefix. Must match
|
|
# a registered id from section 1-3.
|
|
FROSTY_DEFAULT_PROVIDER=openai
|
|
|
|
# --- PostgreSQL: the gateway's ONE stateful dependency, and it is REQUIRED ----
|
|
# Holds the control-plane config, governance counters, the request-log trail,
|
|
# the L2 response cache, and the pgvector embedding index. There is no fallback:
|
|
# an unreachable or unset URL ABORTS BOOT rather than serving with empty
|
|
# governance state (decision-log 61). Deno KV, which used to hold this, is
|
|
# retired - migrate an existing data/frosty.kv with
|
|
# `deno task migrate:kv-pg -- --commit`.
|
|
#
|
|
# docker compose up -d postgres
|
|
#
|
|
# Credentials are required; the Compose service sets user/password/db to
|
|
# `frosty`. Use `localhost` from the host, `postgres` from inside Compose.
|
|
FROSTY_PG_URL=postgres://frosty:frosty@localhost:5432/frosty
|
|
|
|
# Session-stable connection used ONLY for LISTEN (cross-process cache
|
|
# invalidation). Defaults to FROSTY_PG_URL, which is correct until a pooler sits
|
|
# in between: LISTEN through PgBouncer's transaction mode stops delivering
|
|
# SILENTLY. With `--profile pgbouncer` up, point FROSTY_PG_URL at :6432 and
|
|
# leave this one on :5432.
|
|
FROSTY_PG_DIRECT_URL=
|
|
|
|
# Connections held PER PROCESS (default 8, max 100). The number PostgreSQL sees
|
|
# is this times FROSTY_WORKERS, plus one LISTEN connection per process.
|
|
FROSTY_PG_POOL_SIZE=
|
|
|
|
# Table holding the semantic cache's embedding vectors.
|
|
FROSTY_PG_TABLE=frosty_vectors
|
|
|
|
# =============================================================================
|
|
# 5. Worker topology and shared governance
|
|
# =============================================================================
|
|
# Worker processes sharing one port through SO_REUSEPORT. Unset or 1 = single
|
|
# process. LINUX/macOS ONLY - Windows has no SO_REUSEPORT and the second bind
|
|
# fails with os error 10048, so the gateway logs why and serves single-process
|
|
# instead. See docs/guides/multi-process.md.
|
|
FROSTY_WORKERS=
|
|
|
|
# Fleet-wide rate limiting. Fixed rate/token windows live in an in-process Map
|
|
# by default, which is exact for ONE process and admits N times the limit across
|
|
# N. `auto` (default) moves them to a shared PostgreSQL authority only when
|
|
# FROSTY_WORKERS>1, because that reservation costs ~1.8 ms per governed request
|
|
# versus ~1 us in memory. Set `on` when running separate REPLICAS (auto cannot
|
|
# see those); `off` accepts N-times-the-limit. auto|on|off
|
|
FROSTY_SHARED_RATE_LIMIT=
|
|
|
|
# How often each process re-reads durable config, in ms (0-3600000, 0 disables).
|
|
# Config changes normally arrive over LISTEN/NOTIFY within milliseconds; this
|
|
# poll is the backstop that bounds staleness when a notification is lost, so a
|
|
# revoked virtual key stops working even then. Default 30000.
|
|
FROSTY_CONFIG_RECONCILE_MS=
|
|
|
|
# =============================================================================
|
|
# 6. Admin protection and origin control
|
|
# =============================================================================
|
|
# Unset = explicit local admin mode (no auth on /api/*). Set to require
|
|
# "Authorization: Bearer <token>" on all /api/* config routes.
|
|
FROSTY_ADMIN_TOKEN=
|
|
|
|
# Admin origin guard allow-list (DNS-rebind defense). localhost, 127.0.0.1, and
|
|
# ::1 are always allowed; add your public host(s) here (comma-separated) when
|
|
# the gateway is reachable beyond localhost. Applies to every /api/* request.
|
|
FROSTY_ALLOWED_HOSTS=
|
|
|
|
# =============================================================================
|
|
# 7. Cache and vector store
|
|
# =============================================================================
|
|
# Unset = off, "exact" = exact-match, "semantic" adds embedding similarity via
|
|
# the embed model below (the provider must support embeddings).
|
|
FROSTY_CACHE=
|
|
FROSTY_CACHE_TTL_MS=
|
|
|
|
# Sent VERBATIM and case-sensitive, so it must match an id the provider serves
|
|
# (check GET /v1/models). Lookups are fail-open, so a wrong id reduces the cache
|
|
# to exact-match only; every failed lookup warns `semantic cache lookup degraded
|
|
# to miss`, and frosty_cache_events_total stays at 100% result="miss".
|
|
FROSTY_CACHE_EMBED_MODEL=text-embedding-3-small
|
|
|
|
# Similarity vectors live in-process unless FROSTY_VECTOR_STORE=pgvector puts
|
|
# them in the same PostgreSQL as everything else, which is also what makes them
|
|
# survive a restart. The Redis (RediSearch) option was removed: Compose
|
|
# provisioned it and no shipped configuration ever selected it, so it was a
|
|
# dependency that served zero requests (decision-log 60).
|
|
FROSTY_VECTOR_STORE=
|
|
|
|
# =============================================================================
|
|
# 8. MCP and Code Mode
|
|
# =============================================================================
|
|
# MCP client transports are configured per server via POST /api/mcp/clients
|
|
# ("transport": "http-sse" (default) | "streamable-http" | "auto" | "stdio").
|
|
# Servers that only speak streamable-http need an explicit transport (or "auto"
|
|
# for spec-order negotiation).
|
|
#
|
|
# stdio (subprocess) MCP servers are DISABLED by default. Enabling them requires
|
|
# BOTH this flag ("1" or "true") AND running with --allow-run, which is outside
|
|
# the standard permission set on purpose (see permissions.md).
|
|
FROSTY_MCP_ALLOW_STDIO=
|
|
|
|
# Background MCP health sweeps (0/unset = on-demand via /api/mcp/health only).
|
|
FROSTY_MCP_HEALTH_INTERVAL_MS=
|
|
|
|
# Code Mode is default-off behind TWO independent gates. The VFS metadata
|
|
# surface is live and inert; the sandboxed executor is HARD-OFF and
|
|
# experimental. off|on (default off). The executor additionally requires a
|
|
# per-request `x-frosty-code-mode: run` header and still refuses (501) because
|
|
# the run primitive is intentionally stubbed.
|
|
FROSTY_CODE_MODE=off
|
|
FROSTY_CODE_MODE_VFS=on
|
|
|
|
# =============================================================================
|
|
# 9. Logging, analytics display, and observability
|
|
# =============================================================================
|
|
# Durable request-log store, in the same PostgreSQL as the rest of the state. ON
|
|
# by default. Disable with FROSTY_LOG_STORE=off. The legacy value `kv` is still
|
|
# accepted and means "on" - it named the retired Deno KV backend, and rejecting
|
|
# it would break existing .env files over a backend that is gone.
|
|
FROSTY_LOG_STORE=pg
|
|
FROSTY_LOG_STORE_MAX=5000
|
|
|
|
# Paths kept OUT of the Logs dashboard trail (live stream + durable store). The
|
|
# container healthcheck and the Prometheus scrape hit the gateway on a fixed
|
|
# interval, so without this they accumulate until they are ~99% of the capped
|
|
# trail and real requests get pruned away. Console access logging is NOT
|
|
# affected: `docker logs` still shows every request. Blank uses the default
|
|
# below; set to `off` to log everything; `/prefix/*` matches a subtree.
|
|
FROSTY_LOG_EXCLUDE_PATHS=/healthz,/metrics,/favicon.ico
|
|
|
|
# Request/response CONTENT capture in the durable log store (OPT-IN; default OFF
|
|
# for privacy; secrets are never captured). on to enable.
|
|
FROSTY_LOG_CONTENT=
|
|
|
|
# Display currency: cost is accounted in USD internally; the Control UI presents
|
|
# euros by multiplying by this EUR-per-USD rate (default 0.92). Operator-only.
|
|
FROSTY_EUR_RATE=0.92
|
|
|
|
# OpenTelemetry OTLP/HTTP trace export. Unset = off. For Docker Compose use
|
|
# http://otel-collector:4318 and start `--profile observability`; spans go to
|
|
# Tempo for drill-down and to Prometheus as RED metrics.
|
|
OTEL_EXPORTER_OTLP_ENDPOINT=
|
|
OTEL_FLUSH_INTERVAL_MS=5000
|
|
|
|
# Distinct models admitted as the `frosty.metrics.model` span-metric label
|
|
# before the rest fold to "other". Bounds Prometheus series growth; traces keep
|
|
# the real model either way. Non-positive/unparseable falls back to the default.
|
|
FROSTY_OTEL_MODEL_CARDINALITY_CAP=11
|
|
|
|
# =============================================================================
|
|
# 10. Pricing sync, encryption, plugins, HTTP client
|
|
# =============================================================================
|
|
# LiteLLM pricing sync. Opt-in and DEFAULT OFF (offline/no-outbound default):
|
|
# set FROSTY_PRICING_SYNC=on to fetch model prices + metadata at boot and
|
|
# refresh on an interval. Operator /api/pricing overrides always win over synced
|
|
# prices. POST /api/pricing/force-sync triggers a sync on demand even when this
|
|
# is off.
|
|
FROSTY_PRICING_SYNC=
|
|
# Refresh cadence (ms); default 24h, floored at 60s.
|
|
FROSTY_PRICING_SYNC_INTERVAL_MS=86400000
|
|
# Source URL for the price list (server-side only; never taken from a request).
|
|
FROSTY_PRICING_URL=https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json
|
|
|
|
# --- Config secret encryption-at-rest (OPT-IN; default OFF = plaintext) -------
|
|
# Set a base64-encoded 32-byte key (preferred) OR a strong passphrase (PBKDF2).
|
|
# When set, provider API keys / AWS secret+session / Vertex SA-JSON / proxy
|
|
# password / CA cert / virtual-key tokens / MCP header values are AES-256-GCM
|
|
# encrypted in PostgreSQL. Fail-closed: once a store has encrypted data, an
|
|
# unset or wrong key REFUSES boot. The key is effectively set-once (rotate via
|
|
# _OLD).
|
|
# Generate: `deno eval "console.log(btoa(String.fromCharCode(...crypto.getRandomValues(new Uint8Array(32)))))"`
|
|
FROSTY_ENCRYPTION_KEY=
|
|
# Rotation only: set to the previous key alongside a new FROSTY_ENCRYPTION_KEY
|
|
# to rewrap the data-encryption key offline (data is not re-encrypted).
|
|
FROSTY_ENCRYPTION_KEY_OLD=
|
|
|
|
# JSON-repair plugin (OPT-IN; default OFF). Repairs invalid-JSON model output
|
|
# post-response and, for streams, post-completion via the reconstructed message
|
|
# (the live client stream is never mutated). on|1|true|yes to enable.
|
|
FROSTY_JSON_REPAIR=
|
|
|
|
# Request mocker: short-circuits upstream calls with synthetic responses, for
|
|
# offline dev/demo/load-testing. OPT-IN and OFF unless set to on|1|true|yes -
|
|
# leave it empty for a realistic deployment. Rules come from
|
|
# FROSTY_MOCKER_CONFIG (inline JSON starting with `{`, or a file path).
|
|
FROSTY_MOCKER=
|
|
FROSTY_MOCKER_CONFIG=
|
|
|
|
# Provider HTTP client: default per-request timeout in ms (default 120000; 0
|
|
# disables). Never total-caps an in-progress SSE/eventstream. FROSTY_NO_PROXY
|
|
# takes comma-separated bypass patterns (*, .example.com, *.example.com, or an
|
|
# exact host).
|
|
FROSTY_HTTP_TIMEOUT_MS=120000
|
|
FROSTY_NO_PROXY=
|
|
|
|
# =============================================================================
|
|
# 11. Not operator knobs
|
|
# =============================================================================
|
|
# Listed for parity with the code, so a reader who greps for one of these finds
|
|
# out why it is not above. Do NOT set these in a deployment .env.
|
|
#
|
|
# FROSTY_BASE_URL target for scripts/full_suite.ts and the browser
|
|
# harness; defaults to http://localhost:8080
|
|
# FROSTY_BENCH_TARGET scripts/load-bench.ts only
|
|
# FROSTY_BENCH_KEY scripts/load-bench.ts only
|
|
# FROSTY_BENCH_MODEL scripts/load-bench.ts only
|
|
# FROSTY_KV_PATH read ONLY by scripts/migrate_kv_to_pg.ts, to find an
|
|
# existing data/frosty.kv to migrate. It configures
|
|
# nothing at runtime; Deno KV is retired.
|
|
# FROSTY_WORKER_ROLE set BY the supervisor on each child it spawns
|
|
# FROSTY_WORKER_INDEX set BY the supervisor on each child it spawns
|