55 lines
1.6 KiB
YAML
55 lines
1.6 KiB
YAML
# Shared model-weight storage for all llama.cpp pods.
|
|
#
|
|
# All llamacpp pods are pinned to the NUCBox (roger-nucbox-evo-x2) via
|
|
# nodeSelector, so a single hostPath PV on that node is correct and matches the
|
|
# existing postgres hostPath pattern. GGUF files are large (10s of GB); baking
|
|
# them into images would be wasteful, and an initContainer downloads them
|
|
# idempotently on first boot instead.
|
|
#
|
|
# nodeAffinity keeps the PV bound to the NUCBox even if labels change later.
|
|
#
|
|
# IMPORTANT: capacity is only metadata for a hostPath volume — k8s does NOT
|
|
# enforce it and bumping it does NOT add physical disk space. The
|
|
# DeepSeek-V4-Flash-0731 UD-IQ1_M GGUF is ~87 GiB across 3 shards, so the
|
|
# hostPath filesystem (/data on the NUCBox) must physically have ~95 GiB free.
|
|
# The fetch-model initContainer checks free space and fails loudly if the disk
|
|
# is too small; expanding the disk is a host operation, not a manifest change.
|
|
apiVersion: v1
|
|
kind: PersistentVolume
|
|
metadata:
|
|
name: llamacpp-models
|
|
labels:
|
|
type: local
|
|
app: llamacpp
|
|
spec:
|
|
storageClassName: manual
|
|
capacity:
|
|
storage: 200Gi
|
|
accessModes:
|
|
- ReadWriteMany
|
|
hostPath:
|
|
path: /data/llamacpp/models
|
|
nodeAffinity:
|
|
required:
|
|
nodeSelectorTerms:
|
|
- matchExpressions:
|
|
- key: kubernetes.io/hostname
|
|
operator: In
|
|
values:
|
|
- roger-nucbox-evo-x2
|
|
---
|
|
apiVersion: v1
|
|
kind: PersistentVolumeClaim
|
|
metadata:
|
|
name: llamacpp-models
|
|
namespace: llamacpp
|
|
labels:
|
|
app: llamacpp
|
|
spec:
|
|
storageClassName: manual
|
|
accessModes:
|
|
- ReadWriteMany
|
|
resources:
|
|
requests:
|
|
storage: 200Gi
|