diff --git a/litellm/litellm.yaml b/litellm/litellm.yaml index 9bf00ae..81b78d6 100644 --- a/litellm/litellm.yaml +++ b/litellm/litellm.yaml @@ -52,6 +52,12 @@ data: model: openai/qwen3.6-27b api_base: http://llamacpp-qwen36-27b.llamacpp/v1 api_key: "sk-no-auth" + # Small, fast model intended for Home Assistant voice Assist. + - model_name: qwen3.5-4b + litellm_params: + model: openai/qwen3.5-4b + api_base: http://llamacpp-qwen35-4b.llamacpp/v1 + api_key: "sk-no-auth" router_settings: fallbacks: - deepseek-v4-flash-0731: [qwen3.6-27b] diff --git a/llamacpp/README.md b/llamacpp/README.md index a37316a..1de7ece 100644 --- a/llamacpp/README.md +++ b/llamacpp/README.md @@ -15,6 +15,7 @@ to the NUCBox (`nodeSelector: {kubernetes.io/arch: amd64, hardware: high-memory} | Alias | Model | GGUF | Service | Args ref | |--------------------------|--------------------------------|-------------------------------------------------------|------------------------------------------------------|----------| | `deepseek-v4-flash-0731` | DeepSeek-V4-Flash-0731 (MoE) | unsloth/DeepSeek-V4-Flash-0731-GGUF (UD-IQ1_M, ~87 GiB) | `llamacpp-deepseek-v4-flash-0731.llamacpp:80` | [args-deepseek-v4-flash-0731.md](args-deepseek-v4-flash-0731.md) | +| `qwen3.5-4b` | Qwen3.5-4B (MTP) | unsloth/Qwen3.5-4B-MTP-GGUF (`UD-Q4_K_XL`) | `llamacpp-qwen35-4b.llamacpp:80` | — | DeepSeek-V4-Flash-0731 is a Mixture-of-Experts model (256 experts, 6 active per token) with MLA attention, so only a small fraction of the weights is computed @@ -150,3 +151,8 @@ to VRAM). directory recursively.) 4. Check the VRAM budget table above — at ~87 GiB this model nearly fills the 90 GiB pool on its own, so co-locating another large model is not possible. + +The `qwen3.5-4b` alias is exposed through LiteLLM as a low-latency option for +Home Assistant voice Assist. Select this model in the Home Assistant +conversation/voice Assist provider configuration; the LiteLLM endpoint is +`http://litellm-service.litellm:80/v1` from inside the cluster. diff --git a/llamacpp/deployment-qwen35-4b.yaml b/llamacpp/deployment-qwen35-4b.yaml new file mode 100644 index 0000000..90240fe --- /dev/null +++ b/llamacpp/deployment-qwen35-4b.yaml @@ -0,0 +1,138 @@ +# Qwen3.5-4B (MTP, UD-Q4_K_XL) served by llama.cpp for low-latency assistants. +# The model is fully offloaded to the NUCBox Radeon 8060S. Model weights are +# stored on the shared llamacpp-models hostPath PVC. +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: llamacpp-qwen35-4b + namespace: llamacpp + labels: + app: llamacpp + model: qwen3.5-4b +spec: + replicas: 1 + strategy: + type: Recreate + selector: + matchLabels: + app: llamacpp + model: qwen3.5-4b + template: + metadata: + labels: + app: llamacpp + model: qwen3.5-4b + spec: + nodeSelector: + kubernetes.io/arch: amd64 + hardware: high-memory + initContainers: + - name: fetch-model + image: alpine:3.20 + command: ["/bin/sh", "-c"] + args: + - | + set -e + if [ -s "/models/$MODEL_FILE" ]; then + echo "Model $MODEL_FILE already present — skipping download." + exit 0 + fi + apk add --no-cache curl + echo "Downloading $MODEL_FILE from $MODEL_URL ..." + curl -fL --retry 5 --retry-delay 5 -o "/models/$MODEL_FILE.partial" "$MODEL_URL" + mv "/models/$MODEL_FILE.partial" "/models/$MODEL_FILE" + echo "Download complete: $(ls -lh /models/$MODEL_FILE)" + env: + - name: MODEL_URL + value: "https://huggingface.co/unsloth/Qwen3.5-4B-MTP-GGUF/resolve/main/Qwen3.5-4B-UD-Q4_K_XL.gguf" + - name: MODEL_FILE + value: "Qwen3.5-4B-UD-Q4_K_XL.gguf" + volumeMounts: + - name: models + mountPath: /models + containers: + - name: llama-server + image: ghcr.io/ggml-org/llama.cpp:server-vulkan + imagePullPolicy: IfNotPresent + args: + - -m + - /models/Qwen3.5-4B-UD-Q4_K_XL.gguf + - --alias + - qwen3.5-4b + - --host + - 0.0.0.0 + - --port + - "8080" + - --jinja + - -ngl + - "999" + - -c + - "8192" + - -np + - "1" + - --cont-batching + - --cache-type-k + - q8_0 + - --cache-type-v + - q8_0 + - --threads + - "8" + ports: + - name: http + containerPort: 8080 + resources: + requests: + cpu: "1000m" + memory: 2Gi + limits: + cpu: "4000m" + memory: 8Gi + readinessProbe: + httpGet: + path: /health + port: 8080 + initialDelaySeconds: 30 + periodSeconds: 10 + failureThreshold: 6 + livenessProbe: + httpGet: + path: /health + port: 8080 + initialDelaySeconds: 120 + periodSeconds: 30 + failureThreshold: 5 + securityContext: + privileged: true + volumeMounts: + - name: models + mountPath: /models + readOnly: true + - name: dri + mountPath: /dev/dri + volumes: + - name: models + persistentVolumeClaim: + claimName: llamacpp-models + - name: dri + hostPath: + path: /dev/dri + type: Directory +--- +apiVersion: v1 +kind: Service +metadata: + name: llamacpp-qwen35-4b + namespace: llamacpp + labels: + app: llamacpp + model: qwen3.5-4b +spec: + type: ClusterIP + selector: + app: llamacpp + model: qwen3.5-4b + ports: + - name: http + port: 80 + targetPort: 8080 diff --git a/llamacpp/deployment-qwen36-27b.yaml b/llamacpp/deployment-qwen36-27b.yaml index f1faeb0..ec9f8b3 100644 --- a/llamacpp/deployment-qwen36-27b.yaml +++ b/llamacpp/deployment-qwen36-27b.yaml @@ -71,7 +71,7 @@ spec: - -c - "131072" - -np - - "2" + - "1" - --cont-batching - --cache-type-k - q8_0