services: qwen38-flash-next: build: . image: haloq38flash:latest container_name: qwen38-flash-next devices: - /dev/dri group_add: - video - render security_opt: - seccomp=unconfined volumes: - /mnt/ssd2/models/qwen38-flash-next:/models:ro ports: - "8080:8080" environment: - LD_LIBRARY_PATH=/app # the image CMD is bare (/app/llama-server) — pass the model + flags: command: >- -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf -ngl 999 -fa on -ctk q8_0 -ctv q8_0 -c 32768 -ub 2048 -t 4 --jinja # for mtp speculative decoding: # command: >- # -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf # -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf # --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75 # -ngl 999 -fa on -ctk q8_0 -ctv q8_0 -c 32768 -ub 2048 -t 4 --jinja