-
Notifications
You must be signed in to change notification settings - Fork 0
Configuration
OBEY API Gateway uses a single YAML configuration file. On first run, a default config is created automatically if none exists.
The config file is resolved in this order:
-
--configCLI flag (highest priority) -
CONFIG_PATHenvironment variable -
./config.yamlin working directory
# CLI flag
ai-gateway --config C:\path\to\config.yaml
# Environment variable
$env:CONFIG_PATH = "C:\path\to\config.yaml"
ai-gatewayThe entire configuration can be managed visually through the Admin Panel:

server:
host: "0.0.0.0"
port: 8080
request_timeout_seconds: 30
providers:
- name: "openai"
type: "openai"
base_url: "https://api.openai.com/v1"
api_key_env: "OPENAI_API_KEY"
timeout_seconds: 30
model_groups:
- name: "gpt-4-group"
models:
- provider: "openai"
model: "gpt-4"
priority: 1server:
host: "0.0.0.0" # Bind address
port: 8080 # Listen port
request_timeout_seconds: 30 # Global request timeout
max_request_size_mb: 10 # Max request body sizetls:
enabled: true
cert_path: "./cert.pem"
key_path: "./key.pem"See Providers for full details on each provider type.
providers:
- name: "openai"
type: "openai" # openai | ollama | bedrock | groq | together | vllm | lmstudio | nvidia_nim
base_url: "https://api.openai.com/v1"
api_key_env: "OPENAI_API_KEY" # Env var name, resolved at runtime
timeout_seconds: 30 # Legacy total timeout
ttfb_timeout_seconds: 30 # Time-to-first-byte timeout
total_timeout_seconds: 300 # Total request timeout
max_connections: 100 # Connection pool limit
rate_limit_per_minute: 60 # 0 = unlimited
custom_headers:
X-Custom: "${MY_ENV_VAR}" # Supports env var substitutionmodel_groups:
- name: "gpt-4-group"
version_fallback_enabled: false # Try older versions on failure
models:
- provider: "openai"
model: "gpt-4"
cost_per_million_input_tokens: 10.0
cost_per_million_output_tokens: 30.0
priority: 100 # Lower = higher prioritycircuit_breaker:
failure_threshold: 3 # Failures before opening
backoff_sequence_seconds: [5, 10, 20, 40, 300]
success_threshold: 1 # Successes to close
retry:
max_retries_per_provider: 1
backoff_sequence_seconds: [1, 2, 4]logging:
level: "info" # trace | debug | info | warn | error
database_path: "./logs.db" # SQLite log file
request_body_logging: false
response_body_logging: false
max_body_size_bytes: 10000
excluded_fields: ["api_key", "authorization"]
retention_days: 30
cleanup_schedule_hours: 24admin:
enabled: true
path: "/admin"
auth:
enabled: false
username_env: "ADMIN_USERNAME" # Env var for username
password_env: "ADMIN_PASSWORD" # Env var for passworddashboard:
enabled: true
path: "/dashboard"
metrics_update_interval_seconds: 1cors:
enabled: false
allowed_origins: ["*"]
allowed_methods: ["GET", "POST", "OPTIONS"]
allowed_headers: ["Content-Type", "Authorization"]exact_cache:
enabled: true
max_entries: 5000
ttl_seconds: 3600
temperature_threshold: 0.15semantic_cache:
enabled: true
qdrant_url: "http://localhost:6334" # gRPC port
collection_name: "ai_gateway_cache"
similarity_threshold: 0.95
embedding_provider: "openai"
embedding_model: "text-embedding-3-small"
ttl_seconds: 3600
max_cache_size: 10000streaming:
emit_early_event: true # Synthetic role event before upstream responds
keepalive_interval_seconds: 5 # SSE keep-alive (0 = disabled)
passthrough_enabled: true # True SSE relay for capable providers
chunk_timeout_seconds: 60 # Max gap between SSE chunks
retry_on_truncation: true # Failover on finish_reason=lengthprometheus:
enabled: true
path: "/metrics"virtual_keys:
enforcement: disabled # disabled | optional | required
database_path: "./keys.db"loop_detection:
enabled: false # Opt-in agent loop detection
session_timeout_minutes: 30
max_sessions: 10000
history_depth: 5
thresholds:
warn_confidence: 0.30
throttle_confidence: 0.50
inject_confidence: 0.70
hardstop_confidence: 0.90
weights:
content_similarity: 0.25
tool_call_repetition: 0.20
response_stagnation: 0.15
token_velocity: 0.10
error_cycling: 0.15
context_growth: 0.10
cost_velocity: 0.05
throttle_delay_seconds: 2
injection_strategy: system_prompt_appendSee Agent Loop Detection for full details.
structured_output:
enabled: true
max_retries: 1
retry_temperature: 0
passthrough_providers: [openai]See Structured Output for full details.
guardrails:
max_reinjection_entries: 256
providers: []
pipelines: []
global_default_pipeline: null
bindings: {}See Guardrail Pipelines for full details.
tool_compression:
enabled: false
level: medium # low | medium | high | max
progressive_disclosure: false
pruning:
enabled: false
min_requests: 5
feedback_loop:
enabled: true
error_threshold: 0.10See Tool Definition Compression for full details.
smart_routing:
enabled: false
classifier: heuristic # heuristic | ml | llm | composite | jev | laya
cost_quality_threshold: 0.5
tier_boundaries:
fast_max: 0.3
balanced_max: 0.7
cascade:
enabled: false
max_escalations: 2See Smart Model Routing for full details.
cache_aware_routing:
enabled: false # Master switch (default: false)
stickiness_ttl_seconds: 300 # TTL for prefix → provider affinity (0 disables stickiness)
default_cache_min_tokens: 1024 # Fallback min prefix tokens eligible for caching
cost_sort_hit_rate: 0.0 # Assumed hit rate (0.0–1.0) for cache-aware cost sortSee Cache-Aware Routing for full details.
reasoning_compat:
enabled: true # Master switch (default: true)
strip_on_model_change: true # Strip reasoning state on any model change
attribute_reasoning_cost: true # Attribute reasoning-token spend in cost/metrics
conversation_model_affinity: true # Track prefix → resolved model for attribution
effort_budget_map: # reasoning_effort → thinking.budget_tokens (all ≥ 1024)
minimal: 1024
low: 2048
medium: 8192
high: 16384
xhigh: 32768See Reasoning Compatibility for full details.
context:
enabled: true # Automatic context truncation on overflow
truncation_strategy: remove_oldest # remove_oldest | sliding_window
sliding_window_size: 10 # Messages kept for sliding_window
capabilities_cache_ttl_seconds: 3600
max_truncation_retries: 3
default_context_window: 32768 # Used when model capabilities are unknowncompression:
enabled: false # Disabled unless explicitly enabledSee Token Compression for full details and per-level pipelines.
memory:
enabled: falseSee Persistent Memory for full details.
codex_search:
enabled: true # Defaults to enabled when a Codex provider exists
output_to_chat: true # Append search results to visible chat history
base_url: "https://chatgpt.com/backend-api/codex/alpha/search"
timeout_seconds: 15 # 1–120
max_iterations: 5 # 1–20See OAuth & Codex for full details.
# Additional model identifiers that accept Codex `xhigh` reasoning effort
xhigh_models_allowlist:
- "my-custom-o-series-model"
# Additional model identifiers that accept Codex reasoning parameters
reasoning_models_allowlist:
- "my-custom-reasoning-model"first_launch_completed: false
tray:
show_notifications: true
auto_open_browser: true
splash_duration_ms: 3000| Variable | Purpose |
|---|---|
CONFIG_PATH |
Override config file location |
AI_GATEWAY_DATA_DIR |
Override secrets/master-key directory (recommended for Docker) |
OPENAI_API_KEY |
Provider API key (name matches api_key_env in config) |
ADMIN_USERNAME |
Admin panel username |
ADMIN_PASSWORD |
Admin panel password |
RUST_LOG |
Tracing filter (info, debug, ai_gateway=trace) |
OAUTH_CALLBACK_BIND_HOST |
Bind address for OAuth callback (Docker: 0.0.0.0) |
The api_key_env field is resolved at runtime:
- First tried as an environment variable name (e.g.,
OPENAI_API_KEY→ looks up$env:OPENAI_API_KEY) - If the env var doesn't exist, the literal string value is used as the key
Custom headers support ${ENV_VAR} syntax:
providers:
- name: "custom"
type: "openai"
custom_headers:
X-API-Token: "${MY_SECRET_TOKEN}"Provider URLs are automatically normalized:
- Trailing
/is stripped -
/v1is appended if not already present (for OpenAI-compatible providers)
Configuration can be reloaded without restarting:
curl -X POST http://localhost:8080/admin/config/reloadOr use the Admin Panel UI. Circuit breakers are reset on config reload.
- Providers — detailed provider configuration
- Routing & Failover — how requests are routed
- Security — TLS, encryption, authentication