# LLM Inference API Server # Multi-stage build for smaller final image FROM python:3.11.12-slim AS builder # Install uv COPY --from=ghcr.io/astral-sh/uv:0.10 /uv /usr/local/bin/uv WORKDIR /app # Copy project files (uv.lock not committed; resolved at build time) COPY pyproject.toml README.md ./ COPY src/ ./src/ # Install dependencies RUN uv sync --no-dev # Production image FROM python:3.11.12-slim WORKDIR /app # Copy virtual environment from builder COPY --from=builder /app/.venv /app/.venv # Copy source code COPY src/ ./src/ # Set environment variables ENV PATH="/app/.venv/bin:$PATH" ENV PYTHONUNBUFFERED=1 ENV LLM_HOST=0.0.0.0 ENV LLM_PORT=14011 # Health check (using Python since curl not available in slim image) HEALTHCHECK --interval=30s --timeout=10s --retries=3 \ CMD python -c "import urllib.request; urllib.request.urlopen('http://localhost:14011/health')" || exit 1 # Expose port EXPOSE 14011 # Run the server CMD ["python", "-m", "llm_inference.cli"]