# Shared model-weight storage for all llama.cpp pods. # # All llamacpp pods are pinned to the NUCBox (roger-nucbox-evo-x2) via # nodeSelector, so a single hostPath PV on that node is correct and matches the # existing postgres hostPath pattern. GGUF files are large (10s of GB); baking # them into images would be wasteful, and an initContainer downloads them # idempotently on first boot instead. # # nodeAffinity keeps the PV bound to the NUCBox even if labels change later. # # IMPORTANT: capacity is only metadata for a hostPath volume — k8s does NOT # enforce it and bumping it does NOT add physical disk space. The # DeepSeek-V4-Flash-0731 UD-IQ1_M GGUF is ~87 GiB across 3 shards, so the # hostPath filesystem (/data on the NUCBox) must physically have ~95 GiB free. # The fetch-model initContainer checks free space and fails loudly if the disk # is too small; expanding the disk is a host operation, not a manifest change. apiVersion: v1 kind: PersistentVolume metadata: name: llamacpp-models labels: type: local app: llamacpp spec: storageClassName: manual capacity: storage: 100Gi accessModes: - ReadWriteMany hostPath: path: /data/llamacpp/models nodeAffinity: required: nodeSelectorTerms: - matchExpressions: - key: kubernetes.io/hostname operator: In values: - roger-nucbox-evo-x2 --- apiVersion: v1 kind: PersistentVolumeClaim metadata: name: llamacpp-models namespace: llamacpp labels: app: llamacpp spec: storageClassName: manual accessModes: - ReadWriteMany resources: requests: storage: 100Gi