diff --git a/Dockerfile b/Dockerfile index 914e2dd..e251585 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,21 +1,48 @@ # haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv) # builds the nathanw1014 strix-halo-vulkan engine and serves with the # recommended flags. models are mounted, not baked in. +# +# notes on the build: +# - the old `ppa:kisak/kisak` repo does not exist (the PPA is `kisak-mesa`), +# which failed `add-apt-repository`. noble-updates already ships +# mesa 25.2.x (radv supports gfx1151), so no PPA is needed. +# - noble's glslc (shaderc 2023.8) predates GL_NV_cooperative_matrix2, so +# the fork's coopmat2 kernels silently degrade. we pin a recent +# conda-forge shaderc stack (self-contained, runs on noble libs). +# - LLAMA_CURL is a deprecated no-op option in this fork; dropped. +# - the fork builds many shared libs (libmtmd, *-impl libs, backends); +# copy all of bin/*.so*, not just libggml*/libllama*. # ---- stage 1: build ---- FROM ubuntu:24.04 AS build ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update && apt-get install -y \ - build-essential cmake ninja-build git ccache \ - libvulkan-dev glslc vulkan-tools \ - libcurl4-openssl-dev \ +RUN apt-get update && apt-get install -y --no-install-recommends \ + build-essential cmake ninja-build git ccache pkg-config \ + ca-certificates curl unzip zstd \ + libvulkan-dev spirv-headers libssl-dev \ && rm -rf /var/lib/apt/lists/* +# modern glslc (shaderc v2026.3) + its libs, pinned conda-forge builds. +# glslc uses rpath $ORIGIN/../lib, so /opt/glslc/{bin,lib} is self-contained. +RUN set -eux; \ + mkdir -p /opt/glslc /tmp/gc; cd /tmp/gc; \ + for u in \ + https://conda.anaconda.org/conda-forge/linux-64/shaderc-2026.3-hcebf71c_1.conda \ + https://conda.anaconda.org/conda-forge/linux-64/glslang-16.5.0-h980caa0_2.conda \ + https://conda.anaconda.org/conda-forge/linux-64/spirv-tools-2026.3-h7148c6a_1.conda; do \ + curl -fL --retry 3 -o pkg.conda "$u"; \ + unzip -o -q pkg.conda "pkg-*.tar.zst"; \ + tar --zstd -xf pkg-*.tar.zst -C /opt/glslc; \ + rm -f pkg.conda pkg-*.tar.zst; \ + done; \ + rm -rf /tmp/gc; \ + /opt/glslc/bin/glslc --version | head -1 + RUN git clone --depth 1 -b strix-halo-vulkan \ https://github.com/Nathanw1014/llama.cpp /src/engine -RUN cmake -B /src/engine/build -S /src/engine \ +RUN cmake -B /src/engine/build -S /src/engine -G Ninja \ -DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \ - -DLLAMA_CURL=ON \ + -DVulkan_GLSLC_EXECUTABLE=/opt/glslc/bin/glslc \ && cmake --build /src/engine/build --parallel $(nproc) \ --target llama-server llama-cli llama-bench @@ -23,22 +50,24 @@ RUN cmake -B /src/engine/build -S /src/engine \ FROM ubuntu:24.04 ENV DEBIAN_FRONTEND=noninteractive -# add kisak ppa for recent mesa/radv (gfx1151 needs >= 24.x) -RUN apt-get update && apt-get install -y software-properties-common gpg-agent \ - && add-apt-repository -y ppa:kisak/kisak \ - && apt-get update && apt-get install -y \ +# stock noble mesa (25.2.x in noble-updates) — radv gfx1151 supported, +# no PPA needed +RUN apt-get update && apt-get install -y --no-install-recommends \ mesa-vulkan-drivers vulkan-tools libvulkan1 \ - libcurl4 \ + libssl3t64 libgomp1 libstdc++6 ca-certificates \ && rm -rf /var/lib/apt/lists/* COPY --from=build /src/engine/build/bin/llama-server /app/llama-server COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench -COPY --from=build /src/engine/build/bin/libggml*.so* /app/ -COPY --from=build /src/engine/build/bin/libllama*.so* /app/ - -RUN ldconfig /app 2>/dev/null; true -ENV LD_LIBRARY_PATH=/app +COPY --from=build /src/engine/build/bin/ /tmp/engine-bin/ +RUN set -eux; \ + for f in /tmp/engine-bin/*.so*; do \ + [ -e "$f" ] || continue; \ + if [ -L "$f" ]; then cp -a "$f" /app/; else cp -aL "$f" /app/; fi; \ + done; \ + rm -rf /tmp/engine-bin; \ + echo /app > /etc/ld.so.conf.d/app.conf; ldconfig # models volume VOLUME /models @@ -46,9 +75,9 @@ VOLUME /models WORKDIR /app EXPOSE 8080 -CMD ["/app/llama-server", \ - "-m", "/models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf", \ - "-ngl", "999", "-fa", "on", \ - "-ctk", "q8_0", "-ctv", "q8_0", \ - "-c", "32768", "-ub", "2048", "-t", "4", \ - "--jinja", "--host", "0.0.0.0", "--port", "8080"] +# bare binary — pass all args (model, ctx, spec decode, ...) from the +# container runtime / llamaswap. bind defaults kept env-configurable: +# override with -e LLAMA_ARG_HOST=... / -e LLAMA_ARG_PORT=... +ENV LLAMA_ARG_HOST=0.0.0.0 \ + LLAMA_ARG_PORT=8080 +CMD ["/app/llama-server"] diff --git a/docker-compose.yml b/docker-compose.yml index e426906..0ae2a67 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -16,8 +16,14 @@ services: - "8080:8080" environment: - LD_LIBRARY_PATH=/app - # override CMD to change context, quant, or add mtp: - # docker compose run qwen38-flash-next \ - # /app/llama-server -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf \ - # -c 131072 -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \ + # the image CMD is bare (/app/llama-server) — pass the model + flags: + command: >- + -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf + -ngl 999 -fa on -ctk q8_0 -ctv q8_0 + -c 32768 -ub 2048 -t 4 --jinja + # for mtp speculative decoding: + # command: >- + # -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf + # -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf # --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75 + # -ngl 999 -fa on -ctk q8_0 -ctv q8_0 -c 32768 -ub 2048 -t 4 --jinja