forked from roger/k3s-cluster
deploy lfm2.5 model
This commit is contained in:
@@ -9,18 +9,18 @@ data:
|
|||||||
config.yaml: |
|
config.yaml: |
|
||||||
model:
|
model:
|
||||||
provider: openai-api
|
provider: openai-api
|
||||||
default: qwen3.8-27b
|
default: lfm2.5-2.6b
|
||||||
base_url: "http://litellm-service.litellm:80/v1"
|
base_url: "http://litellm-service.litellm:80/v1"
|
||||||
api_mode: chat_completions
|
api_mode: chat_completions
|
||||||
|
|
||||||
auxiliary:
|
auxiliary:
|
||||||
compression:
|
compression:
|
||||||
provider: openai-api
|
provider: openai-api
|
||||||
model: qwen3.8-27b
|
model: lfm2.5-2.6b
|
||||||
base_url: "http://litellm-service.litellm:80/v1"
|
base_url: "http://litellm-service.litellm:80/v1"
|
||||||
title_generation:
|
title_generation:
|
||||||
provider: openai-api
|
provider: openai-api
|
||||||
model: qwen3.8-27b
|
model: lfm2.5-2.6b
|
||||||
base_url: "http://litellm-service.litellm:80/v1"
|
base_url: "http://litellm-service.litellm:80/v1"
|
||||||
|
|
||||||
terminal:
|
terminal:
|
||||||
|
|||||||
@@ -47,6 +47,11 @@ data:
|
|||||||
model: openai/qwen3.8-27b
|
model: openai/qwen3.8-27b
|
||||||
api_base: http://llamacpp-qwen38-27b.llamacpp/v1
|
api_base: http://llamacpp-qwen38-27b.llamacpp/v1
|
||||||
api_key: "sk-no-auth"
|
api_key: "sk-no-auth"
|
||||||
|
- model_name: lfm2.5-2.6b
|
||||||
|
litellm_params:
|
||||||
|
model: openai/lfm2.5-2.6b
|
||||||
|
api_base: http://llamacpp-lfm25-26b.llamacpp/v1
|
||||||
|
api_key: "sk-no-auth"
|
||||||
litellm_settings:
|
litellm_settings:
|
||||||
#set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production
|
#set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production
|
||||||
callbacks: ["arize_phoenix"]
|
callbacks: ["arize_phoenix"]
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ Ollama endpoint.
|
|||||||
| Alias | Model | Configuration | Service |
|
| Alias | Model | Configuration | Service |
|
||||||
|---|---|---|---|
|
|---|---|---|---|
|
||||||
| `qwen3.8-27b` | Qwen3.8-27B with MTP | `Q4_K_M` primary, `Q4_0` draft, 196k context, q8_0 K/V cache | `llamacpp-qwen38-27b.llamacpp:80` |
|
| `qwen3.8-27b` | Qwen3.8-27B with MTP | `Q4_K_M` primary, `Q4_0` draft, 196k context, q8_0 K/V cache | `llamacpp-qwen38-27b.llamacpp:80` |
|
||||||
|
| `lfm2.5-2.6b` | LiquidAI LFM2.5-2.6B with speculative decoding | `Q4_K_M` primary, `Q4_0` draft, 121k context, temperature 0.1 | `llamacpp-lfm25-26b.llamacpp:80` |
|
||||||
|
|
||||||
An initContainer downloads both model files atomically before llama-server
|
An initContainer downloads both model files atomically before llama-server
|
||||||
starts:
|
starts:
|
||||||
|
|||||||
142
llamacpp/deployment-lfm25-26b.yaml
Normal file
142
llamacpp/deployment-lfm25-26b.yaml
Normal file
@@ -0,0 +1,142 @@
|
|||||||
|
# LiquidAI LFM2.5-2.6B with speculative decoding served by llama.cpp.
|
||||||
|
# The primary and draft GGUFs are downloaded into the shared model PVC before
|
||||||
|
# llama-server starts.
|
||||||
|
---
|
||||||
|
apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
metadata:
|
||||||
|
name: llamacpp-lfm25-26b
|
||||||
|
namespace: llamacpp
|
||||||
|
labels:
|
||||||
|
app: llamacpp
|
||||||
|
model: lfm2.5-2.6b
|
||||||
|
spec:
|
||||||
|
replicas: 1
|
||||||
|
strategy:
|
||||||
|
type: Recreate
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: llamacpp
|
||||||
|
model: lfm2.5-2.6b
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app: llamacpp
|
||||||
|
model: lfm2.5-2.6b
|
||||||
|
spec:
|
||||||
|
nodeSelector:
|
||||||
|
kubernetes.io/arch: amd64
|
||||||
|
hardware: high-memory
|
||||||
|
initContainers:
|
||||||
|
- name: fetch-models
|
||||||
|
image: alpine:3.20
|
||||||
|
command: ["/bin/sh", "-c"]
|
||||||
|
args:
|
||||||
|
- |
|
||||||
|
set -eu
|
||||||
|
apk add --no-cache curl
|
||||||
|
for entry in \
|
||||||
|
"https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_K_M.gguf|LFM2.5-2.6B-Q4_K_M.gguf" \
|
||||||
|
"https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_0.gguf|LFM2.5-2.6B-Q4_0.gguf"; do
|
||||||
|
url=${entry%%|*}
|
||||||
|
file=${entry##*|}
|
||||||
|
if [ -s "/models/$file" ]; then
|
||||||
|
echo "$file already present - skipping download."
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
echo "Downloading $file ..."
|
||||||
|
curl -fL --retry 5 --retry-delay 5 -C - \
|
||||||
|
-o "/models/$file.partial" "$url"
|
||||||
|
mv "/models/$file.partial" "/models/$file"
|
||||||
|
echo "Download complete: $(ls -lh "/models/$file")"
|
||||||
|
done
|
||||||
|
volumeMounts:
|
||||||
|
- name: models
|
||||||
|
mountPath: /models
|
||||||
|
containers:
|
||||||
|
- name: llama-server
|
||||||
|
image: ghcr.io/ggml-org/llama.cpp:server-vulkan
|
||||||
|
imagePullPolicy: IfNotPresent
|
||||||
|
args:
|
||||||
|
- -m
|
||||||
|
- /models/LFM2.5-2.6B-Q4_K_M.gguf
|
||||||
|
- --model-draft
|
||||||
|
- /models/LFM2.5-2.6B-Q4_0.gguf
|
||||||
|
- --spec-default
|
||||||
|
- --spec-type
|
||||||
|
- draft
|
||||||
|
- --ctx-size
|
||||||
|
- "121000"
|
||||||
|
- --temp
|
||||||
|
- "0.1"
|
||||||
|
- --cache-type-k
|
||||||
|
- q8_0
|
||||||
|
- --cache-type-v
|
||||||
|
- q8_0
|
||||||
|
- --fit
|
||||||
|
- "off"
|
||||||
|
- --alias
|
||||||
|
- lfm2.5-2.6b
|
||||||
|
- --host
|
||||||
|
- 0.0.0.0
|
||||||
|
- --port
|
||||||
|
- "8080"
|
||||||
|
- --jinja
|
||||||
|
ports:
|
||||||
|
- name: http
|
||||||
|
containerPort: 8080
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
cpu: "500m"
|
||||||
|
memory: 2Gi
|
||||||
|
limits:
|
||||||
|
cpu: "4000m"
|
||||||
|
memory: 12Gi
|
||||||
|
readinessProbe:
|
||||||
|
httpGet:
|
||||||
|
path: /health
|
||||||
|
port: 8080
|
||||||
|
initialDelaySeconds: 30
|
||||||
|
periodSeconds: 10
|
||||||
|
failureThreshold: 6
|
||||||
|
livenessProbe:
|
||||||
|
httpGet:
|
||||||
|
path: /health
|
||||||
|
port: 8080
|
||||||
|
initialDelaySeconds: 180
|
||||||
|
periodSeconds: 30
|
||||||
|
failureThreshold: 5
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
volumeMounts:
|
||||||
|
- name: models
|
||||||
|
mountPath: /models
|
||||||
|
readOnly: true
|
||||||
|
- name: dri
|
||||||
|
mountPath: /dev/dri
|
||||||
|
volumes:
|
||||||
|
- name: models
|
||||||
|
persistentVolumeClaim:
|
||||||
|
claimName: llamacpp-models
|
||||||
|
- name: dri
|
||||||
|
hostPath:
|
||||||
|
path: /dev/dri
|
||||||
|
type: Directory
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Service
|
||||||
|
metadata:
|
||||||
|
name: llamacpp-lfm25-26b
|
||||||
|
namespace: llamacpp
|
||||||
|
labels:
|
||||||
|
app: llamacpp
|
||||||
|
model: lfm2.5-2.6b
|
||||||
|
spec:
|
||||||
|
type: ClusterIP
|
||||||
|
selector:
|
||||||
|
app: llamacpp
|
||||||
|
model: lfm2.5-2.6b
|
||||||
|
ports:
|
||||||
|
- name: http
|
||||||
|
port: 80
|
||||||
|
targetPort: 8080
|
||||||
Reference in New Issue
Block a user