remove qwen 3.5 4b and disable thinking in agent crons

This commit is contained in:
Roger Oriol
2026-08-03 10:58:29 +02:00
parent 89516d9cc3
commit 9da3fb3195
5 changed files with 16 additions and 150 deletions

View File

@@ -15,7 +15,6 @@ to the NUCBox (`nodeSelector: {kubernetes.io/arch: amd64, hardware: high-memory}
| Alias | Model | GGUF | Service | Args ref |
|--------------------------|--------------------------------|-------------------------------------------------------|------------------------------------------------------|----------|
| `deepseek-v4-flash-0731` | DeepSeek-V4-Flash-0731 (MoE) | unsloth/DeepSeek-V4-Flash-0731-GGUF (UD-IQ1_M, ~87 GiB) | `llamacpp-deepseek-v4-flash-0731.llamacpp:80` | [args-deepseek-v4-flash-0731.md](args-deepseek-v4-flash-0731.md) |
| `qwen3.5-4b` | Qwen3.5-4B (MTP) | unsloth/Qwen3.5-4B-MTP-GGUF (`UD-Q4_K_XL`) | `llamacpp-qwen35-4b.llamacpp:80` | — |
DeepSeek-V4-Flash-0731 is a Mixture-of-Experts model (256 experts, 6 active per
token) with MLA attention, so only a small fraction of the weights is computed
@@ -151,8 +150,3 @@ to VRAM).
directory recursively.)
4. Check the VRAM budget table above — at ~87 GiB this model nearly fills the
90 GiB pool on its own, so co-locating another large model is not possible.
The `qwen3.5-4b` alias is exposed through LiteLLM as a low-latency option for
Home Assistant voice Assist. Select this model in the Home Assistant
conversation/voice Assist provider configuration; the LiteLLM endpoint is
`http://litellm-service.litellm:80/v1` from inside the cluster.

View File

@@ -1,138 +0,0 @@
# Qwen3.5-4B (MTP, UD-Q4_K_XL) served by llama.cpp for low-latency assistants.
# The model is fully offloaded to the NUCBox Radeon 8060S. Model weights are
# stored on the shared llamacpp-models hostPath PVC.
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: llamacpp-qwen35-4b
namespace: llamacpp
labels:
app: llamacpp
model: qwen3.5-4b
spec:
replicas: 1
strategy:
type: Recreate
selector:
matchLabels:
app: llamacpp
model: qwen3.5-4b
template:
metadata:
labels:
app: llamacpp
model: qwen3.5-4b
spec:
nodeSelector:
kubernetes.io/arch: amd64
hardware: high-memory
initContainers:
- name: fetch-model
image: alpine:3.20
command: ["/bin/sh", "-c"]
args:
- |
set -e
if [ -s "/models/$MODEL_FILE" ]; then
echo "Model $MODEL_FILE already present — skipping download."
exit 0
fi
apk add --no-cache curl
echo "Downloading $MODEL_FILE from $MODEL_URL ..."
curl -fL --retry 5 --retry-delay 5 -o "/models/$MODEL_FILE.partial" "$MODEL_URL"
mv "/models/$MODEL_FILE.partial" "/models/$MODEL_FILE"
echo "Download complete: $(ls -lh /models/$MODEL_FILE)"
env:
- name: MODEL_URL
value: "https://huggingface.co/unsloth/Qwen3.5-4B-MTP-GGUF/resolve/main/Qwen3.5-4B-UD-Q4_K_XL.gguf"
- name: MODEL_FILE
value: "Qwen3.5-4B-UD-Q4_K_XL.gguf"
volumeMounts:
- name: models
mountPath: /models
containers:
- name: llama-server
image: ghcr.io/ggml-org/llama.cpp:server-vulkan
imagePullPolicy: IfNotPresent
args:
- -m
- /models/Qwen3.5-4B-UD-Q4_K_XL.gguf
- --alias
- qwen3.5-4b
- --host
- 0.0.0.0
- --port
- "8080"
- --jinja
- -ngl
- "999"
- -c
- "65536"
- -np
- "1"
- --cont-batching
- --cache-type-k
- q8_0
- --cache-type-v
- q8_0
- --threads
- "8"
ports:
- name: http
containerPort: 8080
resources:
requests:
cpu: "1000m"
memory: 2Gi
limits:
cpu: "4000m"
memory: 8Gi
readinessProbe:
httpGet:
path: /health
port: 8080
initialDelaySeconds: 30
periodSeconds: 10
failureThreshold: 6
livenessProbe:
httpGet:
path: /health
port: 8080
initialDelaySeconds: 120
periodSeconds: 30
failureThreshold: 5
securityContext:
privileged: true
volumeMounts:
- name: models
mountPath: /models
readOnly: true
- name: dri
mountPath: /dev/dri
volumes:
- name: models
persistentVolumeClaim:
claimName: llamacpp-models
- name: dri
hostPath:
path: /dev/dri
type: Directory
---
apiVersion: v1
kind: Service
metadata:
name: llamacpp-qwen35-4b
namespace: llamacpp
labels:
app: llamacpp
model: qwen3.5-4b
spec:
type: ClusterIP
selector:
app: llamacpp
model: qwen3.5-4b
ports:
- name: http
port: 80
targetPort: 8080