Livrare LOT 1 - Didi
This commit is contained in:
commit
5380c3fc63
990 changed files with 133308 additions and 0 deletions
131
ai_platform/modules/embeddings/deploy/docker-compose.yml
Normal file
131
ai_platform/modules/embeddings/deploy/docker-compose.yml
Normal file
|
|
@ -0,0 +1,131 @@
|
|||
# Embeddings Module - Docker Compose Configuration
|
||||
#
|
||||
# Port Allocation (x41xx = Embeddings):
|
||||
# 14100 - Embeddings API (Prod)
|
||||
# 54100 - Embeddings API (Dev)
|
||||
# 14101/54101 - vLLM Embedding Server
|
||||
# 14110/54110 - llama.cpp Embedding Server
|
||||
#
|
||||
# Profiles:
|
||||
# api - API server only (uses external embedding servers)
|
||||
# vllm - API + vLLM server (GPU required)
|
||||
# llamacpp - API + llama.cpp server
|
||||
#
|
||||
# Naming Convention: didiAI-{module}-{service}
|
||||
#
|
||||
# Network:
|
||||
# Uses deploy_default network (shared with other modules)
|
||||
|
||||
networks:
|
||||
deploy_default:
|
||||
external: true
|
||||
|
||||
services:
|
||||
# ==========================================================================
|
||||
# Embeddings API Server
|
||||
# ==========================================================================
|
||||
embeddings-api:
|
||||
container_name: didiAI-embeddings-api
|
||||
image: didiai-embeddings-api
|
||||
build:
|
||||
context: ..
|
||||
dockerfile: deploy/Dockerfile
|
||||
ports:
|
||||
- "${EMB_PORT:-14100}:${EMB_PORT:-14100}"
|
||||
networks:
|
||||
- deploy_default
|
||||
environment:
|
||||
- EMB_PORT=${EMB_PORT:-14100}
|
||||
- EMB_EXTERNAL_URL=${EMB_EXTERNAL_URL}
|
||||
- EMB_DEFAULT_BACKEND=${EMB_DEFAULT_BACKEND}
|
||||
- EMB_ENABLE_VLLM=${EMB_ENABLE_VLLM}
|
||||
- EMB_ENABLE_LLAMACPP=${EMB_ENABLE_LLAMACPP}
|
||||
- EMB_VLLM_BASE_URL=http://didiAI-embeddings-vllm:14101
|
||||
- EMB_LLAMACPP_BASE_URL=http://didiAI-embeddings-llamacpp:8080
|
||||
- EMB_API_TOKENS=${EMB_API_TOKENS:-}
|
||||
- EMB_DASHBOARD_URL=${EMB_DASHBOARD_URL:-http://didiAI-dashboard:51300}
|
||||
healthcheck:
|
||||
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:${EMB_PORT:-14100}/health')"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 30s
|
||||
restart: unless-stopped
|
||||
profiles:
|
||||
- api
|
||||
- vllm
|
||||
- llamacpp
|
||||
|
||||
# ==========================================================================
|
||||
# vLLM Embedding Server
|
||||
# ==========================================================================
|
||||
# Embedding model using vLLM with --task embed (v0.8.x syntax)
|
||||
vllm-embed:
|
||||
container_name: didiAI-embeddings-vllm
|
||||
image: vllm/vllm-openai:v0.8.5
|
||||
ports:
|
||||
- "${EMB_VLLM_PORT:-14101}:14101"
|
||||
networks:
|
||||
- deploy_default
|
||||
volumes:
|
||||
- ${HF_CACHE_DIR:-/cai2_ds_storage/hf_cache}:/root/.cache/huggingface
|
||||
environment:
|
||||
- HF_HOME=/root/.cache/huggingface
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- CUDA_VISIBLE_DEVICES=${EMB_VLLM_GPU:-0}
|
||||
command: >
|
||||
--model ${EMB_VLLM_MODEL:-BAAI/bge-m3}
|
||||
--host 0.0.0.0
|
||||
--port 14101
|
||||
--task embed
|
||||
--trust-remote-code
|
||||
--max-model-len ${EMB_VLLM_MAX_LEN:-8192}
|
||||
--gpu-memory-utilization ${EMB_VLLM_GPU_UTIL:-0.50}
|
||||
--disable-log-requests
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids: ['${EMB_VLLM_GPU:-0}']
|
||||
capabilities: [gpu]
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:14101/health"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 10
|
||||
start_period: 300s
|
||||
restart: unless-stopped
|
||||
profiles:
|
||||
- vllm
|
||||
|
||||
# ==========================================================================
|
||||
# llama.cpp Embedding Server
|
||||
# ==========================================================================
|
||||
# Embedding model using llama.cpp with --embedding
|
||||
llamacpp-embed:
|
||||
container_name: didiAI-embeddings-llamacpp
|
||||
image: ghcr.io/ggml-org/llama.cpp:server-b4769
|
||||
ports:
|
||||
- "${EMB_LLAMACPP_PORT:-14110}:8080"
|
||||
networks:
|
||||
- deploy_default
|
||||
volumes:
|
||||
- ${MODELS_DIR:-/cai2_ds_storage/models}:/models:ro
|
||||
command: >
|
||||
--model /models/${EMB_LLAMACPP_MODEL:-bge-m3-q4_k_m.gguf}
|
||||
--host 0.0.0.0
|
||||
--port 8080
|
||||
--embedding
|
||||
--ctx-size ${EMB_LLAMACPP_CTX:-8192}
|
||||
--threads ${EMB_LLAMACPP_THREADS:-4}
|
||||
--parallel ${EMB_LLAMACPP_PARALLEL:-4}
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:8080/health"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 60s
|
||||
restart: unless-stopped
|
||||
profiles:
|
||||
- llamacpp
|
||||
Loading…
Add table
Add a link
Reference in a new issue