# Qwen3.5-4B (MTP, UD-Q4_K_XL) served by llama.cpp for low-latency assistants. # The model is fully offloaded to the NUCBox Radeon 8060S. Model weights are # stored on the shared llamacpp-models hostPath PVC. --- apiVersion: apps/v1 kind: Deployment metadata: name: llamacpp-qwen35-4b namespace: llamacpp labels: app: llamacpp model: qwen3.5-4b spec: replicas: 1 strategy: type: Recreate selector: matchLabels: app: llamacpp model: qwen3.5-4b template: metadata: labels: app: llamacpp model: qwen3.5-4b spec: nodeSelector: kubernetes.io/arch: amd64 hardware: high-memory initContainers: - name: fetch-model image: alpine:3.20 command: ["/bin/sh", "-c"] args: - | set -e if [ -s "/models/$MODEL_FILE" ]; then echo "Model $MODEL_FILE already present — skipping download." exit 0 fi apk add --no-cache curl echo "Downloading $MODEL_FILE from $MODEL_URL ..." curl -fL --retry 5 --retry-delay 5 -o "/models/$MODEL_FILE.partial" "$MODEL_URL" mv "/models/$MODEL_FILE.partial" "/models/$MODEL_FILE" echo "Download complete: $(ls -lh /models/$MODEL_FILE)" env: - name: MODEL_URL value: "https://huggingface.co/unsloth/Qwen3.5-4B-MTP-GGUF/resolve/main/Qwen3.5-4B-UD-Q4_K_XL.gguf" - name: MODEL_FILE value: "Qwen3.5-4B-UD-Q4_K_XL.gguf" volumeMounts: - name: models mountPath: /models containers: - name: llama-server image: ghcr.io/ggml-org/llama.cpp:server-vulkan imagePullPolicy: IfNotPresent args: - -m - /models/Qwen3.5-4B-UD-Q4_K_XL.gguf - --alias - qwen3.5-4b - --host - 0.0.0.0 - --port - "8080" - --jinja - -ngl - "999" - -c - "65536" - -np - "1" - --cont-batching - --cache-type-k - q8_0 - --cache-type-v - q8_0 - --threads - "8" ports: - name: http containerPort: 8080 resources: requests: cpu: "1000m" memory: 2Gi limits: cpu: "4000m" memory: 8Gi readinessProbe: httpGet: path: /health port: 8080 initialDelaySeconds: 30 periodSeconds: 10 failureThreshold: 6 livenessProbe: httpGet: path: /health port: 8080 initialDelaySeconds: 120 periodSeconds: 30 failureThreshold: 5 securityContext: privileged: true volumeMounts: - name: models mountPath: /models readOnly: true - name: dri mountPath: /dev/dri volumes: - name: models persistentVolumeClaim: claimName: llamacpp-models - name: dri hostPath: path: /dev/dri type: Directory --- apiVersion: v1 kind: Service metadata: name: llamacpp-qwen35-4b namespace: llamacpp labels: app: llamacpp model: qwen3.5-4b spec: type: ClusterIP selector: app: llamacpp model: qwen3.5-4b ports: - name: http port: 80 targetPort: 8080