# Frosty gateway + PostgreSQL. Requires Compose v2.20+ for `include`. # # PostgreSQL is NOT optional and is not behind a profile: the gateway keeps all # durable state in it (config, governance counters, request logs, the L2 # response cache, and the pgvector embedding index) and refuses to start without # it (decision-log 61). Deno KV, which used to hold that state, is gone. # # docker compose up -d # gateway + postgres # docker compose --profile pgbouncer up -d # + transaction pooler # docker compose --profile observability up -d # + metrics + traces # deno task test:live # drives postgres itself include: - path: ./deploy/vector-stores/docker-compose.yml services: gateway: build: . ports: - "8080:8080" depends_on: postgres: condition: service_healthy # The full env surface of .env.example, forwarded as VAR=${VAR:-} so # anything set in the shell or .env reaches the container and unset stays # empty. PORT is pinned to what the healthcheck expects. Adding a knob to # .env.example means adding it here too. environment: # --- Gateway core --- - PORT=8080 # Worker processes sharing :8080 through SO_REUSEPORT. Linux only - see # docs/guides/multi-process.md. 0 or unset = single process. - FROSTY_WORKERS=${FROSTY_WORKERS:-} # Fleet-wide rate/token windows. auto = on when FROSTY_WORKERS>1. - FROSTY_SHARED_RATE_LIMIT=${FROSTY_SHARED_RATE_LIMIT:-} # Backstop poll for config propagation when a NOTIFY is lost (ms). - FROSTY_CONFIG_RECONCILE_MS=${FROSTY_CONFIG_RECONCILE_MS:-} - LOG_LEVEL=${LOG_LEVEL:-} - FROSTY_DEFAULT_PROVIDER=${FROSTY_DEFAULT_PROVIDER:-} - FROSTY_ADMIN_TOKEN=${FROSTY_ADMIN_TOKEN:-} - FROSTY_ALLOWED_HOSTS=${FROSTY_ALLOWED_HOSTS:-} # --- First-party provider keys --- - OPENAI_API_KEY=${OPENAI_API_KEY:-} - ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY:-} - GEMINI_API_KEY=${GEMINI_API_KEY:-} - OPENROUTER_API_KEY=${OPENROUTER_API_KEY:-} # Azure OpenAI - AZURE_OPENAI_API_KEY=${AZURE_OPENAI_API_KEY:-} - AZURE_OPENAI_ENDPOINT=${AZURE_OPENAI_ENDPOINT:-} - AZURE_OPENAI_API_VERSION=${AZURE_OPENAI_API_VERSION:-} - AZURE_OPENAI_DEPLOYMENTS=${AZURE_OPENAI_DEPLOYMENTS:-} # --- Wave-3 provider breadth (OpenAI-wire vendors + native adapters) --- - GROQ_API_KEY=${GROQ_API_KEY:-} - MISTRAL_API_KEY=${MISTRAL_API_KEY:-} - XAI_API_KEY=${XAI_API_KEY:-} - PERPLEXITY_API_KEY=${PERPLEXITY_API_KEY:-} - CEREBRAS_API_KEY=${CEREBRAS_API_KEY:-} - NEBIUS_API_KEY=${NEBIUS_API_KEY:-} - PARASAIL_API_KEY=${PARASAIL_API_KEY:-} - HF_TOKEN=${HF_TOKEN:-} - COHERE_API_KEY=${COHERE_API_KEY:-} - ELEVENLABS_API_KEY=${ELEVENLABS_API_KEY:-} - OLLAMA_BASE_URL=${OLLAMA_BASE_URL:-} - OLLAMA_MODELS=${OLLAMA_MODELS:-} # --- Generic compat endpoints (set the BASE_URL to enable each) --- - OPENAI_COMPAT_BASE_URL=${OPENAI_COMPAT_BASE_URL:-} - OPENAI_COMPAT_API_KEY=${OPENAI_COMPAT_API_KEY:-} - OPENAI_COMPAT_DEFAULT_MODEL=${OPENAI_COMPAT_DEFAULT_MODEL:-} - ANTHROPIC_COMPAT_BASE_URL=${ANTHROPIC_COMPAT_BASE_URL:-} - ANTHROPIC_COMPAT_API_KEY=${ANTHROPIC_COMPAT_API_KEY:-} - ANTHROPIC_COMPAT_DEFAULT_MODEL=${ANTHROPIC_COMPAT_DEFAULT_MODEL:-} - LMSTUDIO_BASE_URL=${LMSTUDIO_BASE_URL:-} - LMSTUDIO_API_KEY=${LMSTUDIO_API_KEY:-} - LMSTUDIO_DEFAULT_MODEL=${LMSTUDIO_DEFAULT_MODEL:-} # --- AWS Bedrock (SigV4) --- - AWS_REGION=${AWS_REGION:-} - AWS_ACCESS_KEY_ID=${AWS_ACCESS_KEY_ID:-} - AWS_SECRET_ACCESS_KEY=${AWS_SECRET_ACCESS_KEY:-} - AWS_SESSION_TOKEN=${AWS_SESSION_TOKEN:-} - BEDROCK_MODELS=${BEDROCK_MODELS:-} # --- Google Vertex AI --- - VERTEX_PROJECT_ID=${VERTEX_PROJECT_ID:-} - VERTEX_LOCATION=${VERTEX_LOCATION:-} - VERTEX_SERVICE_ACCOUNT_JSON=${VERTEX_SERVICE_ACCOUNT_JSON:-} - VERTEX_MODELS=${VERTEX_MODELS:-} # --- Response cache (semantic cache uses the embed model + vector store) --- - FROSTY_CACHE=${FROSTY_CACHE:-} - FROSTY_CACHE_TTL_MS=${FROSTY_CACHE_TTL_MS:-} - FROSTY_CACHE_EMBED_MODEL=${FROSTY_CACHE_EMBED_MODEL:-} - FROSTY_VECTOR_STORE=${FROSTY_VECTOR_STORE:-} # --- PostgreSQL: the one stateful dependency --- # An unset URL resolves to the `postgres` Compose service by name. A # localhost FROSTY_PG_URL in .env (for host-native `deno task dev`) will # override this and make the container dial its own empty localhost:5432, # so inside Compose keep FROSTY_PG_URL pointed at the service name. - FROSTY_PG_URL=${FROSTY_PG_URL:-postgres://frosty:frosty@postgres:5432/frosty} # Session-stable connection for LISTEN. Only needs to differ from the # pooled URL when PgBouncer is in front. - FROSTY_PG_DIRECT_URL=${FROSTY_PG_DIRECT_URL:-} - FROSTY_PG_POOL_SIZE=${FROSTY_PG_POOL_SIZE:-} - FROSTY_PG_TABLE=${FROSTY_PG_TABLE:-frosty_vectors} # --- Durable request-log store (ON by default; set FROSTY_LOG_STORE=off to disable) --- - FROSTY_LOG_STORE=${FROSTY_LOG_STORE:-pg} - FROSTY_LOG_STORE_MAX=${FROSTY_LOG_STORE_MAX:-} # Probe paths excluded from the dashboard trail. This compose file runs a # 15s healthcheck and the `observability` profile scrapes /metrics, which # together would otherwise own the whole capped trail. Blank = default # (/healthz,/metrics,/favicon.ico); `off` logs everything. - FROSTY_LOG_EXCLUDE_PATHS=${FROSTY_LOG_EXCLUDE_PATHS:-} # --- Display currency (EUR-per-USD rate; micro-USD stays canonical) --- - FROSTY_EUR_RATE=${FROSTY_EUR_RATE:-0.92} # --- MCP clients --- - FROSTY_MCP_ALLOW_STDIO=${FROSTY_MCP_ALLOW_STDIO:-} - FROSTY_MCP_HEALTH_INTERVAL_MS=${FROSTY_MCP_HEALTH_INTERVAL_MS:-} # --- OpenTelemetry traces (unset = no export). With the `observability` # profile up, set OTEL_EXPORTER_OTLP_ENDPOINT=http://otel-collector:4318 # in .env (or the shell) to turn export on. It is NOT defaulted to the # collector, because without that profile the host does not exist. --- - OTEL_EXPORTER_OTLP_ENDPOINT=${OTEL_EXPORTER_OTLP_ENDPOINT:-} - OTEL_FLUSH_INTERVAL_MS=${OTEL_FLUSH_INTERVAL_MS:-} # Bounds distinct model values promoted to span-derived metric labels. - FROSTY_OTEL_MODEL_CARDINALITY_CAP=${FROSTY_OTEL_MODEL_CARDINALITY_CAP:-} # --- Provider HTTP client --- - FROSTY_HTTP_TIMEOUT_MS=${FROSTY_HTTP_TIMEOUT_MS:-} - FROSTY_NO_PROXY=${FROSTY_NO_PROXY:-} # --- Config secret encryption-at-rest (opt-in; fail-closed once set) --- - FROSTY_ENCRYPTION_KEY=${FROSTY_ENCRYPTION_KEY:-} - FROSTY_ENCRYPTION_KEY_OLD=${FROSTY_ENCRYPTION_KEY_OLD:-} # --- Optional plugins (both default OFF) --- - FROSTY_JSON_REPAIR=${FROSTY_JSON_REPAIR:-} - FROSTY_MOCKER=${FROSTY_MOCKER:-} - FROSTY_MOCKER_CONFIG=${FROSTY_MOCKER_CONFIG:-} # --- Request/response CONTENT capture in the log store (opt-in) --- - FROSTY_LOG_CONTENT=${FROSTY_LOG_CONTENT:-} # --- MCP Code Mode (default off; executor additionally gated) --- - FROSTY_CODE_MODE=${FROSTY_CODE_MODE:-} - FROSTY_CODE_MODE_VFS=${FROSTY_CODE_MODE_VFS:-} # --- LiteLLM pricing sync (opt-in, default off) --- - FROSTY_PRICING_SYNC=${FROSTY_PRICING_SYNC:-} - FROSTY_PRICING_URL=${FROSTY_PRICING_URL:-} - FROSTY_PRICING_SYNC_INTERVAL_MS=${FROSTY_PRICING_SYNC_INTERVAL_MS:-} volumes: # Durable state lives in PostgreSQL now; this volume only holds # process-local scratch that the --allow-write=data permission covers. - frosty-data:/app/data healthcheck: # 127.0.0.1, NOT localhost: inside the container `localhost` resolves to # ::1 first and Deno.serve binds IPv4 only, so the IPv6 attempt is refused # and the container is reported unhealthy while it is serving fine. test: ["CMD-SHELL", "wget -qO- http://127.0.0.1:8080/healthz || exit 1"] interval: 15s timeout: 5s retries: 3 # --- Observability (profile: observability) -------------------------------- # docker compose --profile observability up -d # # gateway --OTLP--> otel-collector --+--> tempo-distributor --> MinIO (S3) # +--> spanmetrics :8889 --> prometheus # grafana --> prometheus (metrics) + tempo-query-frontend (traces) # # Traces need OTEL_EXPORTER_OTLP_ENDPOINT=http://otel-collector:4318 on the # gateway; metrics need nothing. Bind-mount paths are relative to this file. # # Grafana http://localhost:3000 (admin/admin; anonymous read-only) # Prometheus http://localhost:9090 # Tempo API http://localhost:3200 # MinIO http://localhost:9001 (console) prometheus: image: prom/prometheus:v2.53.0 profiles: ["observability"] restart: unless-stopped command: - --config.file=/etc/prometheus/prometheus.yml - --storage.tsdb.path=/prometheus - --storage.tsdb.retention.time=15d - --web.enable-lifecycle volumes: - ./deploy/observability/prometheus.yml:/etc/prometheus/prometheus.yml:ro - prometheus-data:/prometheus ports: - "9090:9090" grafana: image: grafana/grafana:11.1.0 profiles: ["observability"] restart: unless-stopped environment: - GF_SECURITY_ADMIN_USER=${GF_ADMIN_USER:-admin} - GF_SECURITY_ADMIN_PASSWORD=${GF_ADMIN_PASSWORD:-admin} - GF_USERS_ALLOW_SIGN_UP=false # Anonymous access is READ-ONLY: dashboards without a login, no Explore. # Trace drill-down goes through the provisioned `frosty-observer` account # created by grafana-provision below. - GF_AUTH_ANONYMOUS_ENABLED=true - GF_AUTH_ANONYMOUS_ORG_ROLE=Viewer - GF_DASHBOARDS_DEFAULT_HOME_DASHBOARD_PATH=/etc/grafana/dashboards/frosty-gateway.json volumes: - ./deploy/observability/provisioning/datasources:/etc/grafana/provisioning/datasources:ro - ./deploy/observability/provisioning/dashboards:/etc/grafana/provisioning/dashboards:ro # The whole directory, so adding a dashboard needs no compose change. - ./deploy/observability/dashboards:/etc/grafana/dashboards:ro - grafana-data:/var/lib/grafana ports: - "3000:3000" healthcheck: test: [ "CMD-SHELL", "wget -qO- http://127.0.0.1:3000/api/health | grep -q ok || exit 1", ] interval: 5s timeout: 5s retries: 30 depends_on: - prometheus # Creates the `frosty-observer` account (Editor) so trace drill-down via # Explore has a named, scoped principal instead of granting every anonymous # visitor edit rights. Grafana OSS cannot define custom RBAC roles - that API # is Enterprise-licensed - so a pre-provisioned account is the OSS equivalent. # Runs once, is idempotent, and exits. grafana-provision: image: curlimages/curl:8.11.1 profiles: ["observability"] restart: "no" depends_on: grafana: condition: service_healthy environment: - GF_ADMIN_USER=${GF_ADMIN_USER:-admin} - GF_ADMIN_PASSWORD=${GF_ADMIN_PASSWORD:-admin} - GF_OBSERVER_USER=${GF_OBSERVER_USER:-frosty-observer} - GF_OBSERVER_PASSWORD=${GF_OBSERVER_PASSWORD:-frosty-observer} entrypoint: ["/bin/sh", "/provision/grafana-observer.sh"] volumes: - ./deploy/observability/grafana-observer.sh:/provision/grafana-observer.sh:ro # Receives OTLP spans and fans them out: Tempo stores them, the spanmetrics # connector aggregates them onto :8889 for Prometheus, debug prints them. otel-collector: image: otel/opentelemetry-collector-contrib:0.109.0 profiles: ["observability"] restart: unless-stopped command: ["--config=/etc/otelcol-contrib/config.yaml"] volumes: - ./deploy/observability/otel-collector-config.yaml:/etc/otelcol-contrib/config.yaml:ro ports: - "4318:4318" # OTLP HTTP (gateway exports here) - "4317:4317" # OTLP gRPC - "8888:8888" # collector telemetry (job: otel-collector) - "8889:8889" # spanmetrics output (job: otel-spanmetrics) depends_on: - tempo-distributor # --- Trace storage: MinIO (S3) + Tempo in distributed mode ----------------- # Blocks live in the object store, so no container owns the trace data. # FROSTY_TEMPO_STORAGE_PATH sets where those objects land on the host. minio: image: minio/minio:RELEASE.2025-09-07T16-13-09Z profiles: ["observability"] restart: unless-stopped command: ["server", "/data", "--console-address", ":9001"] environment: - MINIO_ROOT_USER=${TEMPO_S3_ACCESS_KEY:-tempo} - MINIO_ROOT_PASSWORD=${TEMPO_S3_SECRET_KEY:-tempo-secret} volumes: - ${FROSTY_TEMPO_STORAGE_PATH:-./data/tempo}:/data ports: - "9000:9000" # S3 API - "9001:9001" # console healthcheck: test: ["CMD-SHELL", "mc ready local || exit 1"] interval: 5s timeout: 5s retries: 30 # Creates the traces bucket, then exits. minio-init: image: minio/mc:RELEASE.2025-08-13T08-35-41Z profiles: ["observability"] restart: "no" depends_on: minio: condition: service_healthy entrypoint: - /bin/sh - -c - > mc alias set tempo http://minio:9000 "${TEMPO_S3_ACCESS_KEY:-tempo}" "${TEMPO_S3_SECRET_KEY:-tempo-secret}" && mc mb --ignore-existing "tempo/${TEMPO_S3_BUCKET:-tempo-traces}" # The five Tempo components. Same image and config file; only `-target` and # the published ports differ. `-config.expand-env` resolves the ${...} in # tempo.yaml, which is how the bucket and credentials get in. tempo-distributor: image: &tempo-image grafana/tempo:2.9.0 profiles: ["observability"] restart: unless-stopped command: &tempo-cmd - -config.file=/etc/tempo/tempo.yaml - -config.expand-env=true - -target=distributor environment: &tempo-env - TEMPO_S3_BUCKET=${TEMPO_S3_BUCKET:-tempo-traces} - TEMPO_S3_ACCESS_KEY=${TEMPO_S3_ACCESS_KEY:-tempo} - TEMPO_S3_SECRET_KEY=${TEMPO_S3_SECRET_KEY:-tempo-secret} - TEMPO_BLOCK_RETENTION=${TEMPO_BLOCK_RETENTION:-72h} volumes: &tempo-volumes - ./deploy/observability/tempo.yaml:/etc/tempo/tempo.yaml:ro depends_on: minio-init: condition: service_completed_successfully tempo-ingester: image: *tempo-image profiles: ["observability"] restart: unless-stopped command: - -config.file=/etc/tempo/tempo.yaml - -config.expand-env=true - -target=ingester environment: *tempo-env volumes: - ./deploy/observability/tempo.yaml:/etc/tempo/tempo.yaml:ro # WAL only - blocks go to S3. Node-local and drained on flush. - tempo-wal:/var/tempo user: root depends_on: minio-init: condition: service_completed_successfully tempo-querier: image: *tempo-image profiles: ["observability"] restart: unless-stopped command: - -config.file=/etc/tempo/tempo.yaml - -config.expand-env=true - -target=querier environment: *tempo-env volumes: *tempo-volumes depends_on: - tempo-query-frontend # Grafana's Tempo datasource points here. tempo-query-frontend: image: *tempo-image profiles: ["observability"] restart: unless-stopped command: - -config.file=/etc/tempo/tempo.yaml - -config.expand-env=true - -target=query-frontend environment: *tempo-env volumes: *tempo-volumes ports: - "3200:3200" # query API + /ready depends_on: minio-init: condition: service_completed_successfully tempo-compactor: image: *tempo-image profiles: ["observability"] restart: unless-stopped command: - -config.file=/etc/tempo/tempo.yaml - -config.expand-env=true - -target=compactor environment: *tempo-env volumes: *tempo-volumes depends_on: minio-init: condition: service_completed_successfully # Tempo images are distroless (no /bin/sh), so a CMD-SHELL healthcheck would # report a healthy service as unhealthy. Probe from the host instead: # `curl localhost:3200/ready`. volumes: frosty-data: prometheus-data: grafana-data: # Ingester write-ahead log only; trace blocks live in MinIO, whose data # directory is the bind mount at FROSTY_TEMPO_STORAGE_PATH. tempo-wal: