From 0527bbf1494815677c4189360038999ba352934c Mon Sep 17 00:00:00 2001 From: Roger Oriol Date: Tue, 18 Aug 2026 19:35:21 +0200 Subject: [PATCH] fix qwen3.8 deployment --- llamacpp/README.md | 6 ++-- llamacpp/deployment-qwen38-27b.yaml | 43 +++++++++++++++++++++++------ llamacpp/pv.yaml | 2 +- 3 files changed, 39 insertions(+), 12 deletions(-) diff --git a/llamacpp/README.md b/llamacpp/README.md index 1bcf6fd..b8680a9 100644 --- a/llamacpp/README.md +++ b/llamacpp/README.md @@ -13,13 +13,13 @@ Ollama endpoint. |---|---|---|---| | `qwen3.8-27b` | Qwen3.8-27B with MTP | `Q4_K_M` primary, `Q4_0` draft, 196k context, q8_0 K/V cache | `llamacpp-qwen38-27b.llamacpp:80` | -The active Deployment uses llama.cpp's Hugging Face downloader for both model -repositories: +An initContainer downloads both model files atomically before llama-server +starts: - Primary: `ggml-org/Qwen3.8-27B-GGUF:Q4_K_M` - Draft: `ggml-org/Qwen3.8-27B-GGUF:Q4_0` -The model cache is stored on the shared hostPath PVC at +The model files are stored on the shared hostPath PVC at `/data/llamacpp/models` on the NUCBox. The server is configured with `--spec-default --spec-type draft-mtp`, `--reasoning-preserve`, `--fit off`, and `--agent`. diff --git a/llamacpp/deployment-qwen38-27b.yaml b/llamacpp/deployment-qwen38-27b.yaml index eae6479..52b866a 100644 --- a/llamacpp/deployment-qwen38-27b.yaml +++ b/llamacpp/deployment-qwen38-27b.yaml @@ -1,6 +1,6 @@ # Qwen3.8-27B with MTP speculative decoding served by llama.cpp. -# The primary and draft GGUFs are downloaded by llama-server into the shared -# persistent llama.cpp cache on the NUCBox Radeon 8060S. +# The primary and draft GGUFs are downloaded by an initContainer into the +# shared model PVC before llama-server starts. --- apiVersion: apps/v1 kind: Deployment @@ -27,15 +27,41 @@ spec: nodeSelector: kubernetes.io/arch: amd64 hardware: high-memory + initContainers: + - name: fetch-models + image: alpine:3.20 + command: ["/bin/sh", "-c"] + args: + - | + set -eu + apk add --no-cache curl + for entry in \ + "https://huggingface.co/ggml-org/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-Q4_K_M.gguf|Qwen3.8-27B-Q4_K_M.gguf" \ + "https://huggingface.co/ggml-org/Qwen3.8-27B-GGUF/resolve/main/mtp-Qwen3.8-27B-Q4_0.gguf|mtp-Qwen3.8-27B-Q4_0.gguf"; do + url=${entry%%|*} + file=${entry##*|} + if [ -s "/models/$file" ]; then + echo "$file already present - skipping download." + continue + fi + echo "Downloading $file ..." + curl -fL --retry 5 --retry-delay 5 -C - \ + -o "/models/$file.partial" "$url" + mv "/models/$file.partial" "/models/$file" + echo "Download complete: $(ls -lh "/models/$file")" + done + volumeMounts: + - name: models + mountPath: /models containers: - name: llama-server image: ghcr.io/ggml-org/llama.cpp:server-vulkan imagePullPolicy: IfNotPresent args: - - -hf - - ggml-org/Qwen3.8-27B-GGUF:Q4_K_M - - -hfd - - ggml-org/Qwen3.8-27B-GGUF:Q4_0 + - -m + - /models/Qwen3.8-27B-Q4_K_M.gguf + - --model-draft + - /models/mtp-Qwen3.8-27B-Q4_0.gguf - --spec-default - --spec-type - draft-mtp @@ -79,14 +105,15 @@ spec: httpGet: path: /health port: 8080 - initialDelaySeconds: 180 + initialDelaySeconds: 600 periodSeconds: 30 failureThreshold: 5 securityContext: privileged: true volumeMounts: - name: models - mountPath: /root/.cache/llama.cpp + mountPath: /models + readOnly: true - name: dri mountPath: /dev/dri volumes: diff --git a/llamacpp/pv.yaml b/llamacpp/pv.yaml index b9e2961..45b27d2 100644 --- a/llamacpp/pv.yaml +++ b/llamacpp/pv.yaml @@ -10,7 +10,7 @@ # # IMPORTANT: capacity is only metadata for a hostPath volume — k8s does NOT # enforce it and bumping it does NOT add physical disk space. The active Qwen -# primary and draft GGUFs are downloaded into this cache, so the hostPath +# primary and draft GGUFs are downloaded into this directory, so the hostPath # filesystem must have enough free space for both models. apiVersion: v1 kind: PersistentVolume