# Qwen3.8-27B with MTP speculative decoding served by llama.cpp. # The primary and draft GGUFs are downloaded by an initContainer into the # shared model PVC before llama-server starts. --- apiVersion: apps/v1 kind: Deployment metadata: name: llamacpp-qwen38-27b namespace: llamacpp labels: app: llamacpp model: qwen3.8-27b spec: replicas: 1 strategy: type: Recreate selector: matchLabels: app: llamacpp model: qwen3.8-27b template: metadata: labels: app: llamacpp model: qwen3.8-27b spec: nodeSelector: kubernetes.io/arch: amd64 hardware: high-memory initContainers: - name: fetch-models image: alpine:3.20 command: ["/bin/sh", "-c"] args: - | set -eu apk add --no-cache curl for entry in \ "https://huggingface.co/ggml-org/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-Q4_K_M.gguf|Qwen3.8-27B-Q4_K_M.gguf" \ "https://huggingface.co/ggml-org/Qwen3.8-27B-GGUF/resolve/main/mtp-Qwen3.8-27B-Q4_0.gguf|mtp-Qwen3.8-27B-Q4_0.gguf"; do url=${entry%%|*} file=${entry##*|} if [ -s "/models/$file" ]; then echo "$file already present - skipping download." continue fi echo "Downloading $file ..." curl -fL --retry 5 --retry-delay 5 -C - \ -o "/models/$file.partial" "$url" mv "/models/$file.partial" "/models/$file" echo "Download complete: $(ls -lh "/models/$file")" done volumeMounts: - name: models mountPath: /models containers: - name: llama-server image: ghcr.io/ggml-org/llama.cpp:server-vulkan imagePullPolicy: IfNotPresent args: - -m - /models/Qwen3.8-27B-Q4_K_M.gguf - --model-draft - /models/mtp-Qwen3.8-27B-Q4_0.gguf - --spec-default - --spec-type - draft-mtp - --ctx-size - "196608" - --cache-type-k - q8_0 - --cache-type-v - q8_0 - --reasoning-preserve - --fit - "off" - --agent - --chat-template-kwargs - '{"reasoning_effort":"medium"}' - --alias - qwen3.8-27b - --host - 0.0.0.0 - --port - "8080" - --jinja ports: - name: http containerPort: 8080 resources: requests: cpu: "1000m" memory: 2Gi limits: cpu: "4000m" memory: 24Gi readinessProbe: httpGet: path: /health port: 8080 initialDelaySeconds: 30 periodSeconds: 10 failureThreshold: 6 livenessProbe: httpGet: path: /health port: 8080 initialDelaySeconds: 600 periodSeconds: 30 failureThreshold: 5 securityContext: privileged: true volumeMounts: - name: models mountPath: /models readOnly: true - name: dri mountPath: /dev/dri volumes: - name: models persistentVolumeClaim: claimName: llamacpp-models - name: dri hostPath: path: /dev/dri type: Directory --- apiVersion: v1 kind: Service metadata: name: llamacpp-qwen38-27b namespace: llamacpp labels: app: llamacpp model: qwen3.8-27b spec: type: ClusterIP selector: app: llamacpp model: qwen3.8-27b ports: - name: http port: 80 targetPort: 8080