diff --git a/Dockerfile b/Dockerfile index a8b6358..4228c2c 100644 --- a/Dockerfile +++ b/Dockerfile @@ -43,9 +43,22 @@ COPY backend ./backend COPY locales ./locales COPY package.json ./ -# Install backend deps against the final pyproject.toml/uv.lock +# Install backend deps against the final pyproject.toml/uv.lock, then strip the +# NVIDIA CUDA runtime libraries that torch drags in (via camel-oasis -> +# sentence-transformers). The server has no GPU and all LLM calls go through an +# API, so CUDA libs are pure bloat (many GB). torch itself stays (CPU runtime) +# but without the nvidia-* packages the image is dramatically smaller. We then +# verify the stripped env still imports torch/sentence-transformers and the app +# so the build fails loudly instead of a silent runtime break. RUN cd backend && uv sync --frozen --no-dev \ - && uv run --frozen python -c "import psycopg" || (echo "FATAL: psycopg not installed — package sync missing the Postgres driver" >&2 && exit 1) + && uv pip uninstall \ + nvidia-cublas-cu12 nvidia-cuda-cupti-cu12 nvidia-cuda-nvrtc-cu12 \ + nvidia-cuda-runtime-cu12 nvidia-cudnn-cu12 nvidia-cufft-cu12 \ + nvidia-cufile-cu12 nvidia-curand-cu12 nvidia-cusolver-cu12 \ + nvidia-cusparse-cu12 nvidia-cusparselt-cu12 nvidia-nccl-cu12 \ + nvidia-nvjitlink-cu12 nvidia-nvshmem-cu12 nvidia-nvtx-cu12 triton \ + && uv run --frozen python -c "import psycopg, torch; import sentence_transformers; from app import create_app; print('deps-after-strip-ok')" \ + || (echo "FATAL: dependency import failed after CUDA strip (torch/sentence-transformers/app). Revisit the nvidia-* uninstall list." >&2 && exit 1) # Make the migration-runner entrypoint executable RUN chmod +x /app/backend/docker_entrypoint.sh