From 7a141642b5ab16ad80e73ff7967d572050b139eb Mon Sep 17 00:00:00 2001 From: Roger Oriol Date: Wed, 19 Aug 2026 00:36:48 +0200 Subject: [PATCH] deploy lfm2.5 model --- home-manager/configmap.yaml | 6 +- litellm/litellm.yaml | 5 + llamacpp/README.md | 1 + llamacpp/deployment-lfm25-26b.yaml | 142 +++++++++++++++++++++++++++++ 4 files changed, 151 insertions(+), 3 deletions(-) create mode 100644 llamacpp/deployment-lfm25-26b.yaml diff --git a/home-manager/configmap.yaml b/home-manager/configmap.yaml index 5fa8236..57ef990 100644 --- a/home-manager/configmap.yaml +++ b/home-manager/configmap.yaml @@ -9,18 +9,18 @@ data: config.yaml: | model: provider: openai-api - default: qwen3.8-27b + default: lfm2.5-2.6b base_url: "http://litellm-service.litellm:80/v1" api_mode: chat_completions auxiliary: compression: provider: openai-api - model: qwen3.8-27b + model: lfm2.5-2.6b base_url: "http://litellm-service.litellm:80/v1" title_generation: provider: openai-api - model: qwen3.8-27b + model: lfm2.5-2.6b base_url: "http://litellm-service.litellm:80/v1" terminal: diff --git a/litellm/litellm.yaml b/litellm/litellm.yaml index ab59818..bbb8046 100644 --- a/litellm/litellm.yaml +++ b/litellm/litellm.yaml @@ -47,6 +47,11 @@ data: model: openai/qwen3.8-27b api_base: http://llamacpp-qwen38-27b.llamacpp/v1 api_key: "sk-no-auth" + - model_name: lfm2.5-2.6b + litellm_params: + model: openai/lfm2.5-2.6b + api_base: http://llamacpp-lfm25-26b.llamacpp/v1 + api_key: "sk-no-auth" litellm_settings: #set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production callbacks: ["arize_phoenix"] diff --git a/llamacpp/README.md b/llamacpp/README.md index b8680a9..f7abcdc 100644 --- a/llamacpp/README.md +++ b/llamacpp/README.md @@ -12,6 +12,7 @@ Ollama endpoint. | Alias | Model | Configuration | Service | |---|---|---|---| | `qwen3.8-27b` | Qwen3.8-27B with MTP | `Q4_K_M` primary, `Q4_0` draft, 196k context, q8_0 K/V cache | `llamacpp-qwen38-27b.llamacpp:80` | +| `lfm2.5-2.6b` | LiquidAI LFM2.5-2.6B with speculative decoding | `Q4_K_M` primary, `Q4_0` draft, 121k context, temperature 0.1 | `llamacpp-lfm25-26b.llamacpp:80` | An initContainer downloads both model files atomically before llama-server starts: diff --git a/llamacpp/deployment-lfm25-26b.yaml b/llamacpp/deployment-lfm25-26b.yaml new file mode 100644 index 0000000..41fe4df --- /dev/null +++ b/llamacpp/deployment-lfm25-26b.yaml @@ -0,0 +1,142 @@ +# LiquidAI LFM2.5-2.6B with speculative decoding served by llama.cpp. +# The primary and draft GGUFs are downloaded into the shared model PVC before +# llama-server starts. +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: llamacpp-lfm25-26b + namespace: llamacpp + labels: + app: llamacpp + model: lfm2.5-2.6b +spec: + replicas: 1 + strategy: + type: Recreate + selector: + matchLabels: + app: llamacpp + model: lfm2.5-2.6b + template: + metadata: + labels: + app: llamacpp + model: lfm2.5-2.6b + spec: + nodeSelector: + kubernetes.io/arch: amd64 + hardware: high-memory + initContainers: + - name: fetch-models + image: alpine:3.20 + command: ["/bin/sh", "-c"] + args: + - | + set -eu + apk add --no-cache curl + for entry in \ + "https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_K_M.gguf|LFM2.5-2.6B-Q4_K_M.gguf" \ + "https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_0.gguf|LFM2.5-2.6B-Q4_0.gguf"; do + url=${entry%%|*} + file=${entry##*|} + if [ -s "/models/$file" ]; then + echo "$file already present - skipping download." + continue + fi + echo "Downloading $file ..." + curl -fL --retry 5 --retry-delay 5 -C - \ + -o "/models/$file.partial" "$url" + mv "/models/$file.partial" "/models/$file" + echo "Download complete: $(ls -lh "/models/$file")" + done + volumeMounts: + - name: models + mountPath: /models + containers: + - name: llama-server + image: ghcr.io/ggml-org/llama.cpp:server-vulkan + imagePullPolicy: IfNotPresent + args: + - -m + - /models/LFM2.5-2.6B-Q4_K_M.gguf + - --model-draft + - /models/LFM2.5-2.6B-Q4_0.gguf + - --spec-default + - --spec-type + - draft + - --ctx-size + - "121000" + - --temp + - "0.1" + - --cache-type-k + - q8_0 + - --cache-type-v + - q8_0 + - --fit + - "off" + - --alias + - lfm2.5-2.6b + - --host + - 0.0.0.0 + - --port + - "8080" + - --jinja + ports: + - name: http + containerPort: 8080 + resources: + requests: + cpu: "500m" + memory: 2Gi + limits: + cpu: "4000m" + memory: 12Gi + readinessProbe: + httpGet: + path: /health + port: 8080 + initialDelaySeconds: 30 + periodSeconds: 10 + failureThreshold: 6 + livenessProbe: + httpGet: + path: /health + port: 8080 + initialDelaySeconds: 180 + periodSeconds: 30 + failureThreshold: 5 + securityContext: + privileged: true + volumeMounts: + - name: models + mountPath: /models + readOnly: true + - name: dri + mountPath: /dev/dri + volumes: + - name: models + persistentVolumeClaim: + claimName: llamacpp-models + - name: dri + hostPath: + path: /dev/dri + type: Directory +--- +apiVersion: v1 +kind: Service +metadata: + name: llamacpp-lfm25-26b + namespace: llamacpp + labels: + app: llamacpp + model: lfm2.5-2.6b +spec: + type: ClusterIP + selector: + app: llamacpp + model: lfm2.5-2.6b + ports: + - name: http + port: 80 + targetPort: 8080