# haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv)
# builds the nathanw1014 strix-halo-vulkan engine and serves with the
# recommended flags. models are mounted, not baked in.
#
# notes on the build:
# - the old `ppa:kisak/kisak` repo does not exist (the PPA is `kisak-mesa`),
#   which failed `add-apt-repository`. noble-updates already ships
#   mesa 25.2.x (radv supports gfx1151), so no PPA is needed.
# - noble's glslc (shaderc 2023.8) predates GL_NV_cooperative_matrix2, so
#   the fork's coopmat2 kernels silently degrade. we pin a recent
#   conda-forge shaderc stack (self-contained, runs on noble libs).
# - LLAMA_CURL is a deprecated no-op option in this fork; dropped.
# - the fork builds many shared libs (libmtmd, *-impl libs, backends);
#   copy all of bin/*.so*, not just libggml*/libllama*.

# ---- stage 1: build ----
FROM ubuntu:24.04 AS build
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update && apt-get install -y --no-install-recommends \
    build-essential cmake ninja-build git ccache pkg-config \
    ca-certificates curl unzip zstd \
    libvulkan-dev spirv-headers libssl-dev \
    && rm -rf /var/lib/apt/lists/*

# modern glslc (shaderc v2026.3) + its libs, pinned conda-forge builds.
# glslc uses rpath $ORIGIN/../lib, so /opt/glslc/{bin,lib} is self-contained.
RUN set -eux; \
    mkdir -p /opt/glslc /tmp/gc; cd /tmp/gc; \
    for u in \
      https://conda.anaconda.org/conda-forge/linux-64/shaderc-2026.3-hcebf71c_1.conda \
      https://conda.anaconda.org/conda-forge/linux-64/glslang-16.5.0-h980caa0_2.conda \
      https://conda.anaconda.org/conda-forge/linux-64/spirv-tools-2026.3-h7148c6a_1.conda; do \
        curl -fL --retry 3 -o pkg.conda "$u"; \
        unzip -o -q pkg.conda "pkg-*.tar.zst"; \
        tar --zstd -xf pkg-*.tar.zst -C /opt/glslc; \
        rm -f pkg.conda pkg-*.tar.zst; \
    done; \
    rm -rf /tmp/gc; \
    /opt/glslc/bin/glslc --version | head -1

RUN git clone --depth 1 -b strix-halo-vulkan \
    https://github.com/Nathanw1014/llama.cpp /src/engine
RUN cmake -B /src/engine/build -S /src/engine -G Ninja \
    -DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \
    -DVulkan_GLSLC_EXECUTABLE=/opt/glslc/bin/glslc \
    && cmake --build /src/engine/build --parallel $(nproc) \
       --target llama-server llama-cli llama-bench

# ---- stage 2: runtime ----
FROM ubuntu:24.04
ENV DEBIAN_FRONTEND=noninteractive

# stock noble mesa (25.2.x in noble-updates) — radv gfx1151 supported,
# no PPA needed
RUN apt-get update && apt-get install -y --no-install-recommends \
    mesa-vulkan-drivers vulkan-tools libvulkan1 \
    libssl3t64 libgomp1 libstdc++6 ca-certificates \
    && rm -rf /var/lib/apt/lists/*

COPY --from=build /src/engine/build/bin/llama-server /app/llama-server
COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli
COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench
COPY --from=build /src/engine/build/bin/ /tmp/engine-bin/
RUN set -eux; \
    for f in /tmp/engine-bin/*.so*; do \
      [ -e "$f" ] || continue; \
      if [ -L "$f" ]; then cp -a "$f" /app/; else cp -aL "$f" /app/; fi; \
    done; \
    rm -rf /tmp/engine-bin; \
    echo /app > /etc/ld.so.conf.d/app.conf; ldconfig

# models volume
VOLUME /models

WORKDIR /app
EXPOSE 8080

# bare binary — pass all args (model, ctx, spec decode, ...) from the
# container runtime / llamaswap. bind defaults kept env-configurable:
# override with -e LLAMA_ARG_HOST=... / -e LLAMA_ARG_PORT=...
ENV LLAMA_ARG_HOST=0.0.0.0 \
    LLAMA_ARG_PORT=8080
CMD ["/app/llama-server"]
