From 70bc47328036b58a8de9621060867291cd708e15 Mon Sep 17 00:00:00 2001 From: Roger Oriol Date: Sat, 22 Aug 2026 11:41:42 +0200 Subject: [PATCH] fix lfm 2.5 --- llamacpp/deployment-lfm25-26b.yaml | 19 +++++++++---------- 1 file changed, 9 insertions(+), 10 deletions(-) diff --git a/llamacpp/deployment-lfm25-26b.yaml b/llamacpp/deployment-lfm25-26b.yaml index b6a2126..fef65ae 100644 --- a/llamacpp/deployment-lfm25-26b.yaml +++ b/llamacpp/deployment-lfm25-26b.yaml @@ -1,5 +1,5 @@ -# LiquidAI LFM2.5-2.6B with speculative decoding served by llama.cpp. -# The primary and draft GGUFs are downloaded into the shared model PVC before +# LiquidAI LFM2.5-2.6B served by llama.cpp (Vulkan, no speculative decoding). +# The primary GGUF is downloaded into the shared model PVC before # llama-server starts. --- apiVersion: apps/v1 @@ -36,8 +36,7 @@ spec: set -eu apk add --no-cache curl for entry in \ - "https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_K_M.gguf|LFM2.5-2.6B-Q4_K_M.gguf" \ - "https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_0.gguf|LFM2.5-2.6B-Q4_0.gguf"; do + "https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_K_M.gguf|LFM2.5-2.6B-Q4_K_M.gguf"; do url=${entry%%|*} file=${entry##*|} if [ -s "/models/$file" ]; then @@ -60,13 +59,13 @@ spec: args: - -m - /models/LFM2.5-2.6B-Q4_K_M.gguf - - --model-draft - - /models/LFM2.5-2.6B-Q4_0.gguf - - --spec-default - - --spec-type - - draft-simple + # Draft model + speculative decoding removed: the spec path crashes the + # server (GGML_ASSERT slot.spec_i_batch...) when the KV cache is under + # pressure, taking all in-flight requests (incl. home-manager crons) down. - --ctx-size - - "121000" + - "80000" + - --parallel + - "2" - --temp - "0.1" - --cache-type-k