Dockerfile: fix build on ubuntu 24.04
- ppa:kisak/kisak does not exist (repo is kisak-mesa) and add-apt-repository failed outright; noble-updates already ships mesa 25.2.x with gfx1151 radv, so drop the PPA and use the distro mesa - pin a modern shaderc/glslc (v2026.3) from conda-forge: noble's glslc (shaderc 2023.8) predates GL_NV_cooperative_matrix2 and the fork's coopmat2 vulkan kernels; the pinned stack is self-contained (rpath) - add missing build deps this fork needs: pkg-config, spirv-headers (find_package(SPIRV-Headers) in ggml-vulkan), libssl-dev (httplib TLS), unzip/zstd for the glslc bundle - copy all shared libs from bin/ (fork produces libmtmd, *-impl libs and backend .so's the old glob missed) + register /app via ld.so.conf - drop LLAMA_CURL (deprecated no-op in this fork) - CMD is now bare /app/llama-server so the image can be driven by an external runtime (llamaswap); host/port stay overridable via LLAMA_ARG_HOST/LLAMA_ARG_PORT env. compose passes the recommended flags Verified by a clean-room build of the strix-halo-vulkan branch with noble's toolchain + headers (gcc 13, vulkan-headers 1.3.275, openssl 3.0.13): all three targets link and resolve against stock noble runtime libs, llama-server/cli/bench boot successfully.
This commit is contained in:
+51
-22
@@ -1,21 +1,48 @@
|
|||||||
# haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv)
|
# haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv)
|
||||||
# builds the nathanw1014 strix-halo-vulkan engine and serves with the
|
# builds the nathanw1014 strix-halo-vulkan engine and serves with the
|
||||||
# recommended flags. models are mounted, not baked in.
|
# recommended flags. models are mounted, not baked in.
|
||||||
|
#
|
||||||
|
# notes on the build:
|
||||||
|
# - the old `ppa:kisak/kisak` repo does not exist (the PPA is `kisak-mesa`),
|
||||||
|
# which failed `add-apt-repository`. noble-updates already ships
|
||||||
|
# mesa 25.2.x (radv supports gfx1151), so no PPA is needed.
|
||||||
|
# - noble's glslc (shaderc 2023.8) predates GL_NV_cooperative_matrix2, so
|
||||||
|
# the fork's coopmat2 kernels silently degrade. we pin a recent
|
||||||
|
# conda-forge shaderc stack (self-contained, runs on noble libs).
|
||||||
|
# - LLAMA_CURL is a deprecated no-op option in this fork; dropped.
|
||||||
|
# - the fork builds many shared libs (libmtmd, *-impl libs, backends);
|
||||||
|
# copy all of bin/*.so*, not just libggml*/libllama*.
|
||||||
|
|
||||||
# ---- stage 1: build ----
|
# ---- stage 1: build ----
|
||||||
FROM ubuntu:24.04 AS build
|
FROM ubuntu:24.04 AS build
|
||||||
ENV DEBIAN_FRONTEND=noninteractive
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
RUN apt-get update && apt-get install -y \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
build-essential cmake ninja-build git ccache \
|
build-essential cmake ninja-build git ccache pkg-config \
|
||||||
libvulkan-dev glslc vulkan-tools \
|
ca-certificates curl unzip zstd \
|
||||||
libcurl4-openssl-dev \
|
libvulkan-dev spirv-headers libssl-dev \
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
# modern glslc (shaderc v2026.3) + its libs, pinned conda-forge builds.
|
||||||
|
# glslc uses rpath $ORIGIN/../lib, so /opt/glslc/{bin,lib} is self-contained.
|
||||||
|
RUN set -eux; \
|
||||||
|
mkdir -p /opt/glslc /tmp/gc; cd /tmp/gc; \
|
||||||
|
for u in \
|
||||||
|
https://conda.anaconda.org/conda-forge/linux-64/shaderc-2026.3-hcebf71c_1.conda \
|
||||||
|
https://conda.anaconda.org/conda-forge/linux-64/glslang-16.5.0-h980caa0_2.conda \
|
||||||
|
https://conda.anaconda.org/conda-forge/linux-64/spirv-tools-2026.3-h7148c6a_1.conda; do \
|
||||||
|
curl -fL --retry 3 -o pkg.conda "$u"; \
|
||||||
|
unzip -o -q pkg.conda "pkg-*.tar.zst"; \
|
||||||
|
tar --zstd -xf pkg-*.tar.zst -C /opt/glslc; \
|
||||||
|
rm -f pkg.conda pkg-*.tar.zst; \
|
||||||
|
done; \
|
||||||
|
rm -rf /tmp/gc; \
|
||||||
|
/opt/glslc/bin/glslc --version | head -1
|
||||||
|
|
||||||
RUN git clone --depth 1 -b strix-halo-vulkan \
|
RUN git clone --depth 1 -b strix-halo-vulkan \
|
||||||
https://github.com/Nathanw1014/llama.cpp /src/engine
|
https://github.com/Nathanw1014/llama.cpp /src/engine
|
||||||
RUN cmake -B /src/engine/build -S /src/engine \
|
RUN cmake -B /src/engine/build -S /src/engine -G Ninja \
|
||||||
-DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \
|
-DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \
|
||||||
-DLLAMA_CURL=ON \
|
-DVulkan_GLSLC_EXECUTABLE=/opt/glslc/bin/glslc \
|
||||||
&& cmake --build /src/engine/build --parallel $(nproc) \
|
&& cmake --build /src/engine/build --parallel $(nproc) \
|
||||||
--target llama-server llama-cli llama-bench
|
--target llama-server llama-cli llama-bench
|
||||||
|
|
||||||
@@ -23,22 +50,24 @@ RUN cmake -B /src/engine/build -S /src/engine \
|
|||||||
FROM ubuntu:24.04
|
FROM ubuntu:24.04
|
||||||
ENV DEBIAN_FRONTEND=noninteractive
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
|
|
||||||
# add kisak ppa for recent mesa/radv (gfx1151 needs >= 24.x)
|
# stock noble mesa (25.2.x in noble-updates) — radv gfx1151 supported,
|
||||||
RUN apt-get update && apt-get install -y software-properties-common gpg-agent \
|
# no PPA needed
|
||||||
&& add-apt-repository -y ppa:kisak/kisak \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
&& apt-get update && apt-get install -y \
|
|
||||||
mesa-vulkan-drivers vulkan-tools libvulkan1 \
|
mesa-vulkan-drivers vulkan-tools libvulkan1 \
|
||||||
libcurl4 \
|
libssl3t64 libgomp1 libstdc++6 ca-certificates \
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
COPY --from=build /src/engine/build/bin/llama-server /app/llama-server
|
COPY --from=build /src/engine/build/bin/llama-server /app/llama-server
|
||||||
COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli
|
COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli
|
||||||
COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench
|
COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench
|
||||||
COPY --from=build /src/engine/build/bin/libggml*.so* /app/
|
COPY --from=build /src/engine/build/bin/ /tmp/engine-bin/
|
||||||
COPY --from=build /src/engine/build/bin/libllama*.so* /app/
|
RUN set -eux; \
|
||||||
|
for f in /tmp/engine-bin/*.so*; do \
|
||||||
RUN ldconfig /app 2>/dev/null; true
|
[ -e "$f" ] || continue; \
|
||||||
ENV LD_LIBRARY_PATH=/app
|
if [ -L "$f" ]; then cp -a "$f" /app/; else cp -aL "$f" /app/; fi; \
|
||||||
|
done; \
|
||||||
|
rm -rf /tmp/engine-bin; \
|
||||||
|
echo /app > /etc/ld.so.conf.d/app.conf; ldconfig
|
||||||
|
|
||||||
# models volume
|
# models volume
|
||||||
VOLUME /models
|
VOLUME /models
|
||||||
@@ -46,9 +75,9 @@ VOLUME /models
|
|||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
EXPOSE 8080
|
EXPOSE 8080
|
||||||
|
|
||||||
CMD ["/app/llama-server", \
|
# bare binary — pass all args (model, ctx, spec decode, ...) from the
|
||||||
"-m", "/models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf", \
|
# container runtime / llamaswap. bind defaults kept env-configurable:
|
||||||
"-ngl", "999", "-fa", "on", \
|
# override with -e LLAMA_ARG_HOST=... / -e LLAMA_ARG_PORT=...
|
||||||
"-ctk", "q8_0", "-ctv", "q8_0", \
|
ENV LLAMA_ARG_HOST=0.0.0.0 \
|
||||||
"-c", "32768", "-ub", "2048", "-t", "4", \
|
LLAMA_ARG_PORT=8080
|
||||||
"--jinja", "--host", "0.0.0.0", "--port", "8080"]
|
CMD ["/app/llama-server"]
|
||||||
|
|||||||
+10
-4
@@ -16,8 +16,14 @@ services:
|
|||||||
- "8080:8080"
|
- "8080:8080"
|
||||||
environment:
|
environment:
|
||||||
- LD_LIBRARY_PATH=/app
|
- LD_LIBRARY_PATH=/app
|
||||||
# override CMD to change context, quant, or add mtp:
|
# the image CMD is bare (/app/llama-server) — pass the model + flags:
|
||||||
# docker compose run qwen38-flash-next \
|
command: >-
|
||||||
# /app/llama-server -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf \
|
-m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf
|
||||||
# -c 131072 -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \
|
-ngl 999 -fa on -ctk q8_0 -ctv q8_0
|
||||||
|
-c 32768 -ub 2048 -t 4 --jinja
|
||||||
|
# for mtp speculative decoding:
|
||||||
|
# command: >-
|
||||||
|
# -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf
|
||||||
|
# -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf
|
||||||
# --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
|
# --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
|
||||||
|
# -ngl 999 -fa on -ctk q8_0 -ctv q8_0 -c 32768 -ub 2048 -t 4 --jinja
|
||||||
|
|||||||
Reference in New Issue
Block a user