159 lines
5.5 KiB
Text
159 lines
5.5 KiB
Text
# LLM Inference Configuration
|
|
# Copy this file to .env and fill in ALL required values
|
|
# The application will fail to start if required variables are missing
|
|
|
|
# =============================================================================
|
|
# REQUIRED: Backend Selection (no defaults - must be set explicitly)
|
|
# =============================================================================
|
|
|
|
# Default backend: litellm, vllm, or llamacpp
|
|
LLM_DEFAULT_BACKEND=litellm
|
|
|
|
# Enable optional backends (true/false)
|
|
LLM_ENABLE_VLLM=false
|
|
LLM_ENABLE_LLAMACPP=false
|
|
|
|
# =============================================================================
|
|
# REQUIRED: API Keys (at least one for litellm backend)
|
|
# =============================================================================
|
|
|
|
# OpenRouter API key (access 100+ models via single API)
|
|
# Get one at: https://openrouter.ai/
|
|
OPENROUTER_API_KEY=
|
|
|
|
# OpenAI API key
|
|
# Get one at: https://platform.openai.com/
|
|
OPENAI_API_KEY=
|
|
|
|
# Anthropic API key
|
|
# Get one at: https://console.anthropic.com/
|
|
ANTHROPIC_API_KEY=
|
|
|
|
# =============================================================================
|
|
# REQUIRED for vLLM/llama.cpp profiles: Model Configuration
|
|
# =============================================================================
|
|
|
|
# Directory containing model files (absolute path recommended)
|
|
MODELS_DIR=/path/to/models
|
|
|
|
# vLLM model to load (Hugging Face model ID). Platform primary model (offer).
|
|
# Start vLLM with `--served-model-name qwen3.5` so callers can use the alias.
|
|
VLLM_MODEL=Qwen/Qwen3.5-35B-A3B
|
|
|
|
# llama.cpp model file (GGUF format, filename only - must be in MODELS_DIR)
|
|
LLAMACPP_MODEL=qwen3.5-35b-a3b.Q4_K_M.gguf
|
|
|
|
# Default model alias used when a request omits `model`.
|
|
LLM_DEFAULT_MODEL=qwen3.5
|
|
|
|
# Optional: map friendly alias -> real served model id (JSON). Leave unset if
|
|
# the backend already serves the model under the alias name (e.g. via
|
|
# vLLM --served-model-name qwen3.5).
|
|
# LLM_MODEL_ALIASES={"qwen3.5":"Qwen/Qwen3.5-35B-A3B"}
|
|
|
|
# llama.cpp performance settings
|
|
LLAMACPP_THREADS=4
|
|
LLAMACPP_PARALLEL=1
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: API Authentication
|
|
# =============================================================================
|
|
|
|
# API tokens for Bearer authentication (comma-separated)
|
|
# If not set, authentication is disabled and all endpoints are public
|
|
# If set, requests to protected endpoints require: Authorization: Bearer <token>
|
|
# Health endpoints (/health, /ready) are always public
|
|
# LLM_API_TOKENS=token1,token2,token3
|
|
|
|
# =============================================================================
|
|
# REQUIRED: External URL (for OpenAPI spec and catalog integration)
|
|
# =============================================================================
|
|
|
|
# External URL where this API is reachable (used in OpenAPI spec and /v1/info)
|
|
LLM_EXTERNAL_URL=http://localhost:14011
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: API Server Settings
|
|
# =============================================================================
|
|
|
|
# Server binding
|
|
# LLM_HOST=0.0.0.0
|
|
# LLM_PORT=14011
|
|
|
|
# Default model when not specified in request
|
|
# LLM_DEFAULT_MODEL=gpt-3.5-turbo
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: Timeout and Retry Settings
|
|
# =============================================================================
|
|
|
|
# Request timeout for LLM calls (seconds)
|
|
# LLM_REQUEST_TIMEOUT=120.0
|
|
|
|
# Connection timeout (seconds)
|
|
# LLM_CONNECT_TIMEOUT=10.0
|
|
|
|
# Maximum retry attempts for transient failures
|
|
# LLM_MAX_RETRIES=3
|
|
|
|
# Minimum wait between retries (seconds)
|
|
# LLM_RETRY_MIN_WAIT=1.0
|
|
|
|
# Maximum wait between retries (seconds)
|
|
# LLM_RETRY_MAX_WAIT=60.0
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: Rate Limiting and Concurrency
|
|
# =============================================================================
|
|
|
|
# Requests per second limit
|
|
# LLM_RATE_LIMIT_RPS=10.0
|
|
|
|
# Maximum burst size for rate limiting
|
|
# LLM_RATE_LIMIT_BURST=20
|
|
|
|
# Maximum concurrent completion requests
|
|
# LLM_MAX_CONCURRENT_COMPLETIONS=10
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: Logging
|
|
# =============================================================================
|
|
|
|
# Log level: DEBUG, INFO, WARNING, ERROR, CRITICAL
|
|
# LLM_LOG_LEVEL=INFO
|
|
|
|
# Enable JSON logging format (true/false)
|
|
# LLM_LOG_JSON=false
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: Nginx Proxy Timeouts (for Docker Compose deployment)
|
|
# =============================================================================
|
|
|
|
# Nginx connection timeout
|
|
# NGINX_CONNECT_TIMEOUT=60s
|
|
|
|
# Nginx send timeout
|
|
# NGINX_SEND_TIMEOUT=120s
|
|
|
|
# Nginx read timeout (should be >= LLM_REQUEST_TIMEOUT)
|
|
# NGINX_READ_TIMEOUT=600s
|
|
|
|
# =============================================================================
|
|
# OPTIONAL: Backend-Specific URLs
|
|
# =============================================================================
|
|
|
|
# vLLM server URL (internal Docker network)
|
|
# LLM_VLLM_BASE_URL=http://didiAI-vllm-qwen3.5:14001
|
|
|
|
# llama.cpp server URL (single server, legacy)
|
|
# LLM_LLAMACPP_BASE_URL=http://didiAI-llm-llamacpp:8080
|
|
|
|
# llama.cpp server URLs (comma-separated, load balanced round-robin with failover)
|
|
# If set, overrides LLM_LLAMACPP_BASE_URL
|
|
# LLM_LLAMACPP_BASE_URLS=http://10.11.10.43:14001,http://10.11.10.18:14001
|
|
|
|
# Health check interval for llama.cpp servers (seconds)
|
|
# LLM_LLAMACPP_HEALTH_CHECK_INTERVAL=30
|
|
|
|
# vLLM API key (if authentication is enabled on vLLM server)
|
|
# LLM_VLLM_API_KEY=
|