142 lines
3.5 KiB
YAML
142 lines
3.5 KiB
YAML
# LiquidAI LFM2.5-2.6B served by llama.cpp (Vulkan, no speculative decoding).
|
|
# The primary GGUF is downloaded into the shared model PVC before
|
|
# llama-server starts.
|
|
---
|
|
apiVersion: apps/v1
|
|
kind: Deployment
|
|
metadata:
|
|
name: llamacpp-lfm25-26b
|
|
namespace: llamacpp
|
|
labels:
|
|
app: llamacpp
|
|
model: lfm2.5-2.6b
|
|
spec:
|
|
replicas: 1
|
|
strategy:
|
|
type: Recreate
|
|
selector:
|
|
matchLabels:
|
|
app: llamacpp
|
|
model: lfm2.5-2.6b
|
|
template:
|
|
metadata:
|
|
labels:
|
|
app: llamacpp
|
|
model: lfm2.5-2.6b
|
|
spec:
|
|
nodeSelector:
|
|
kubernetes.io/arch: amd64
|
|
hardware: high-memory
|
|
initContainers:
|
|
- name: fetch-models
|
|
image: alpine:3.20
|
|
command: ["/bin/sh", "-c"]
|
|
args:
|
|
- |
|
|
set -eu
|
|
apk add --no-cache curl
|
|
for entry in \
|
|
"https://huggingface.co/LiquidAI/LFM2.5-2.6B-GGUF/resolve/main/LFM2.5-2.6B-Q4_K_M.gguf|LFM2.5-2.6B-Q4_K_M.gguf"; do
|
|
url=${entry%%|*}
|
|
file=${entry##*|}
|
|
if [ -s "/models/$file" ]; then
|
|
echo "$file already present - skipping download."
|
|
continue
|
|
fi
|
|
echo "Downloading $file ..."
|
|
curl -fL --retry 5 --retry-delay 5 -C - \
|
|
-o "/models/$file.partial" "$url"
|
|
mv "/models/$file.partial" "/models/$file"
|
|
echo "Download complete: $(ls -lh "/models/$file")"
|
|
done
|
|
volumeMounts:
|
|
- name: models
|
|
mountPath: /models
|
|
containers:
|
|
- name: llama-server
|
|
image: ghcr.io/ggml-org/llama.cpp:server-vulkan
|
|
imagePullPolicy: IfNotPresent
|
|
args:
|
|
- -m
|
|
- /models/LFM2.5-2.6B-Q4_K_M.gguf
|
|
# Draft model + speculative decoding removed: the spec path crashes the
|
|
# server (GGML_ASSERT slot.spec_i_batch...) when the KV cache is under
|
|
# pressure, taking all in-flight requests (incl. home-manager crons) down.
|
|
- --ctx-size
|
|
- "80000"
|
|
- --parallel
|
|
- "2"
|
|
- --temp
|
|
- "0.1"
|
|
- --cache-type-k
|
|
- q8_0
|
|
- --cache-type-v
|
|
- q8_0
|
|
- --fit
|
|
- "off"
|
|
- --alias
|
|
- lfm2.5-2.6b
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --jinja
|
|
ports:
|
|
- name: http
|
|
containerPort: 8080
|
|
resources:
|
|
requests:
|
|
cpu: "500m"
|
|
memory: 2Gi
|
|
limits:
|
|
cpu: "4000m"
|
|
memory: 12Gi
|
|
readinessProbe:
|
|
httpGet:
|
|
path: /health
|
|
port: 8080
|
|
initialDelaySeconds: 30
|
|
periodSeconds: 10
|
|
failureThreshold: 6
|
|
livenessProbe:
|
|
httpGet:
|
|
path: /health
|
|
port: 8080
|
|
initialDelaySeconds: 180
|
|
periodSeconds: 30
|
|
failureThreshold: 5
|
|
securityContext:
|
|
privileged: true
|
|
volumeMounts:
|
|
- name: models
|
|
mountPath: /models
|
|
readOnly: true
|
|
- name: dri
|
|
mountPath: /dev/dri
|
|
volumes:
|
|
- name: models
|
|
persistentVolumeClaim:
|
|
claimName: llamacpp-models
|
|
- name: dri
|
|
hostPath:
|
|
path: /dev/dri
|
|
type: Directory
|
|
---
|
|
apiVersion: v1
|
|
kind: Service
|
|
metadata:
|
|
name: llamacpp-lfm25-26b
|
|
namespace: llamacpp
|
|
labels:
|
|
app: llamacpp
|
|
model: lfm2.5-2.6b
|
|
spec:
|
|
type: ClusterIP
|
|
selector:
|
|
app: llamacpp
|
|
model: lfm2.5-2.6b
|
|
ports:
|
|
- name: http
|
|
port: 80
|
|
targetPort: 8080
|