deploy qwen3.5 4b

This commit is contained in:
Roger Oriol
2026-08-02 11:54:58 +02:00
parent 171178f7d0
commit d928297b64
4 changed files with 151 additions and 1 deletions

View File

@@ -52,6 +52,12 @@ data:
model: openai/qwen3.6-27b
api_base: http://llamacpp-qwen36-27b.llamacpp/v1
api_key: "sk-no-auth"
# Small, fast model intended for Home Assistant voice Assist.
- model_name: qwen3.5-4b
litellm_params:
model: openai/qwen3.5-4b
api_base: http://llamacpp-qwen35-4b.llamacpp/v1
api_key: "sk-no-auth"
router_settings:
fallbacks:
- deepseek-v4-flash-0731: [qwen3.6-27b]

View File

@@ -15,6 +15,7 @@ to the NUCBox (`nodeSelector: {kubernetes.io/arch: amd64, hardware: high-memory}
| Alias | Model | GGUF | Service | Args ref |
|--------------------------|--------------------------------|-------------------------------------------------------|------------------------------------------------------|----------|
| `deepseek-v4-flash-0731` | DeepSeek-V4-Flash-0731 (MoE) | unsloth/DeepSeek-V4-Flash-0731-GGUF (UD-IQ1_M, ~87 GiB) | `llamacpp-deepseek-v4-flash-0731.llamacpp:80` | [args-deepseek-v4-flash-0731.md](args-deepseek-v4-flash-0731.md) |
| `qwen3.5-4b` | Qwen3.5-4B (MTP) | unsloth/Qwen3.5-4B-MTP-GGUF (`UD-Q4_K_XL`) | `llamacpp-qwen35-4b.llamacpp:80` | — |
DeepSeek-V4-Flash-0731 is a Mixture-of-Experts model (256 experts, 6 active per
token) with MLA attention, so only a small fraction of the weights is computed
@@ -150,3 +151,8 @@ to VRAM).
directory recursively.)
4. Check the VRAM budget table above — at ~87 GiB this model nearly fills the
90 GiB pool on its own, so co-locating another large model is not possible.
The `qwen3.5-4b` alias is exposed through LiteLLM as a low-latency option for
Home Assistant voice Assist. Select this model in the Home Assistant
conversation/voice Assist provider configuration; the LiteLLM endpoint is
`http://litellm-service.litellm:80/v1` from inside the cluster.

View File

@@ -0,0 +1,138 @@
# Qwen3.5-4B (MTP, UD-Q4_K_XL) served by llama.cpp for low-latency assistants.
# The model is fully offloaded to the NUCBox Radeon 8060S. Model weights are
# stored on the shared llamacpp-models hostPath PVC.
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: llamacpp-qwen35-4b
namespace: llamacpp
labels:
app: llamacpp
model: qwen3.5-4b
spec:
replicas: 1
strategy:
type: Recreate
selector:
matchLabels:
app: llamacpp
model: qwen3.5-4b
template:
metadata:
labels:
app: llamacpp
model: qwen3.5-4b
spec:
nodeSelector:
kubernetes.io/arch: amd64
hardware: high-memory
initContainers:
- name: fetch-model
image: alpine:3.20
command: ["/bin/sh", "-c"]
args:
- |
set -e
if [ -s "/models/$MODEL_FILE" ]; then
echo "Model $MODEL_FILE already present — skipping download."
exit 0
fi
apk add --no-cache curl
echo "Downloading $MODEL_FILE from $MODEL_URL ..."
curl -fL --retry 5 --retry-delay 5 -o "/models/$MODEL_FILE.partial" "$MODEL_URL"
mv "/models/$MODEL_FILE.partial" "/models/$MODEL_FILE"
echo "Download complete: $(ls -lh /models/$MODEL_FILE)"
env:
- name: MODEL_URL
value: "https://huggingface.co/unsloth/Qwen3.5-4B-MTP-GGUF/resolve/main/Qwen3.5-4B-UD-Q4_K_XL.gguf"
- name: MODEL_FILE
value: "Qwen3.5-4B-UD-Q4_K_XL.gguf"
volumeMounts:
- name: models
mountPath: /models
containers:
- name: llama-server
image: ghcr.io/ggml-org/llama.cpp:server-vulkan
imagePullPolicy: IfNotPresent
args:
- -m
- /models/Qwen3.5-4B-UD-Q4_K_XL.gguf
- --alias
- qwen3.5-4b
- --host
- 0.0.0.0
- --port
- "8080"
- --jinja
- -ngl
- "999"
- -c
- "8192"
- -np
- "1"
- --cont-batching
- --cache-type-k
- q8_0
- --cache-type-v
- q8_0
- --threads
- "8"
ports:
- name: http
containerPort: 8080
resources:
requests:
cpu: "1000m"
memory: 2Gi
limits:
cpu: "4000m"
memory: 8Gi
readinessProbe:
httpGet:
path: /health
port: 8080
initialDelaySeconds: 30
periodSeconds: 10
failureThreshold: 6
livenessProbe:
httpGet:
path: /health
port: 8080
initialDelaySeconds: 120
periodSeconds: 30
failureThreshold: 5
securityContext:
privileged: true
volumeMounts:
- name: models
mountPath: /models
readOnly: true
- name: dri
mountPath: /dev/dri
volumes:
- name: models
persistentVolumeClaim:
claimName: llamacpp-models
- name: dri
hostPath:
path: /dev/dri
type: Directory
---
apiVersion: v1
kind: Service
metadata:
name: llamacpp-qwen35-4b
namespace: llamacpp
labels:
app: llamacpp
model: qwen3.5-4b
spec:
type: ClusterIP
selector:
app: llamacpp
model: qwen3.5-4b
ports:
- name: http
port: 80
targetPort: 8080

View File

@@ -71,7 +71,7 @@ spec:
- -c
- "131072"
- -np
- "2"
- "1"
- --cont-batching
- --cache-type-k
- q8_0