new llamacpp service

This commit is contained in:
Roger Oriol
2026-07-22 23:54:52 +02:00
parent 5f7f1bd52a
commit 0143ebefd9
7 changed files with 319 additions and 11 deletions

47
llamacpp/pv.yaml Normal file
View File

@@ -0,0 +1,47 @@
# Shared model-weight storage for all llama.cpp pods.
#
# All llamacpp pods are pinned to the NUCBox (roger-nucbox-evo-x2) via
# nodeSelector, so a single hostPath PV on that node is correct and matches the
# existing postgres hostPath pattern. GGUF files are large (10s of GB); baking
# them into images would be wasteful, and an initContainer downloads them
# idempotently on first boot instead.
#
# nodeAffinity keeps the PV bound to the NUCBox even if labels change later.
apiVersion: v1
kind: PersistentVolume
metadata:
name: llamacpp-models
labels:
type: local
app: llamacpp
spec:
storageClassName: manual
capacity:
storage: 100Gi
accessModes:
- ReadWriteMany
hostPath:
path: /data/llamacpp/models
nodeAffinity:
required:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: In
values:
- roger-nucbox-evo-x2
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: llamacpp-models
namespace: llamacpp
labels:
app: llamacpp
spec:
storageClassName: manual
accessModes:
- ReadWriteMany
resources:
requests:
storage: 100Gi