# llama-server built from source with SM 120 (Blackwell / RTX 5070) CUDA kernels. # Multi-stage: ~8GB devel toolkit stays in builder, runtime image is ~2GB. # ── Builder ────────────────────────────────────────────────────────────── FROM nvidia/cuda:12.8.1-devel-ubuntu24.04 AS builder RUN apt-get update && apt-get install -y --no-install-recommends \ cmake git build-essential curl ca-certificates \ && rm -rf /var/lib/apt/lists/* WORKDIR /build # Pin to a release tag for reproducibility ARG LLAMA_CPP_VERSION=b5460 RUN git clone --depth 1 --branch ${LLAMA_CPP_VERSION} \ https://github.com/ggerganov/llama.cpp.git WORKDIR /build/llama.cpp # CUDA driver symbols (cuMemCreate, etc.) are resolved at runtime by nvidia-container-runtime. # --allow-shlib-undefined lets the linker accept unresolved refs in libggml-cuda.so. RUN cmake -B build \ -DCMAKE_BUILD_TYPE=Release \ -DCMAKE_CUDA_ARCHITECTURES=120 \ -DGGML_CUDA=ON \ -DGGML_CUDA_FORCE_CUBLAS=ON \ -DLLAMA_BUILD_SERVER=ON \ -DLLAMA_CURL=OFF \ -DCMAKE_EXE_LINKER_FLAGS="-Wl,--allow-shlib-undefined" \ && cmake --build build --target llama-server -j$(nproc) # ── Runtime ────────────────────────────────────────────────────────────── FROM nvidia/cuda:12.8.1-runtime-ubuntu24.04 RUN apt-get update && apt-get install -y --no-install-recommends \ curl ca-certificates libgomp1 \ && rm -rf /var/lib/apt/lists/* # Copy server binary and its shared libraries (libggml-*.so) COPY --from=builder /build/llama.cpp/build/bin/llama-server /usr/local/bin/llama-server COPY --from=builder /build/llama.cpp/build/bin/lib*.so /usr/local/lib/ RUN ldconfig RUN useradd -u 1000 -m llama 2>/dev/null || true USER 1000 ENTRYPOINT ["llama-server"]