Files
AI-Profile-Router/platform/docker/llama-cpp/Dockerfile
T

42 lines
2.0 KiB
Docker

ARG CUDA_VERSION=12.8.1
ARG UBUNTU_VERSION=24.04
FROM nvidia/cuda:${CUDA_VERSION}-devel-ubuntu${UBUNTU_VERSION} AS build
ARG DEBIAN_FRONTEND=noninteractive
ARG LLAMA_CPP_COMMIT
ARG CUDA_ARCHITECTURES="86;120"
RUN apt-get update && apt-get install -y --no-install-recommends \
ca-certificates cmake git libcurl4-openssl-dev ninja-build pkg-config && \
rm -rf /var/lib/apt/lists/*
# CUDA's devel image keeps libcuda.so.1 in its compatibility directory, which
# is not part of the default linker search path. llama.cpp's shared CUDA
# backend links successfully without this, but the final server link then
# cannot resolve CUDA driver symbols such as cuMemCreate. Register it for the
# build stage; at runtime the NVIDIA container runtime supplies the host driver.
RUN printf '%s\n' /usr/local/cuda/compat >/etc/ld.so.conf.d/cuda-compat.conf && \
ldconfig
RUN test -n "$LLAMA_CPP_COMMIT"
RUN git clone --filter=blob:none https://github.com/ggml-org/llama.cpp /src/llama.cpp && \
git -C /src/llama.cpp checkout "$LLAMA_CPP_COMMIT"
RUN cmake -S /src/llama.cpp -B /src/llama.cpp/build -G Ninja \
-DCMAKE_BUILD_TYPE=Release \
-DGGML_CUDA=ON \
-DGGML_NATIVE=OFF \
-DCMAKE_CUDA_ARCHITECTURES="$CUDA_ARCHITECTURES" && \
cmake --build /src/llama.cpp/build --target llama-server -j "$(nproc)"
FROM nvidia/cuda:${CUDA_VERSION}-runtime-ubuntu${UBUNTU_VERSION}
ARG DEBIAN_FRONTEND=noninteractive
ARG LLAMA_CPP_COMMIT
RUN apt-get update && apt-get install -y --no-install-recommends \
ca-certificates curl libcurl4 libgomp1 python3 && \
rm -rf /var/lib/apt/lists/* && \
useradd --system --uid 10003 --home /nonexistent --shell /usr/sbin/nologin llama
COPY --from=build /src/llama.cpp/build/bin/ /opt/llama/bin/
COPY platform/web-search/web_search_mcp.py /opt/mike-ai/mcp/web_search_mcp.py
LABEL org.opencontainers.image.source="https://github.com/ggml-org/llama.cpp" \
com.mike-ai.llama-cpp-commit="$LLAMA_CPP_COMMIT"
ENV LD_LIBRARY_PATH=/opt/llama/bin
USER 10003:10003
ENTRYPOINT ["/opt/llama/bin/llama-server"]