From 173a4935ab04ffbaef14ba8a6198314921a2b69e Mon Sep 17 00:00:00 2001 From: riotpiaole <19826264+Riotpiaole@users.noreply.github.com> Date: Mon, 22 Jun 2026 12:23:12 -0700 Subject: [PATCH] fix: pin Kafka broker storage to a 2-replica class sized for current cluster The default "longhorn" StorageClass requests 3 replicas, but the homelab cluster currently has only 2 schedulable nodes, so the 3rd replica could never be scheduled and volumes stayed permanently degraded. Adds longhorn-kafka (numberOfReplicas: 2) and reduces broker PVC size so 3 broker volumes' replicas fit within each node's remaining Longhorn scheduling headroom. Revert to "longhorn" once a 3rd node joins. --- .../kafka-cluster/templates/storageclass.yaml | 22 +++++++++++++++++++ k8s/charts/kafka-cluster/values.yaml | 8 +++++-- k8s/environments/homelab.yaml | 15 +++++++++---- 3 files changed, 39 insertions(+), 6 deletions(-) create mode 100644 k8s/charts/kafka-cluster/templates/storageclass.yaml diff --git a/k8s/charts/kafka-cluster/templates/storageclass.yaml b/k8s/charts/kafka-cluster/templates/storageclass.yaml new file mode 100644 index 0000000..026982d --- /dev/null +++ b/k8s/charts/kafka-cluster/templates/storageclass.yaml @@ -0,0 +1,22 @@ +{{- if eq .Values.nodePool.storage.class "longhorn-kafka" }} +# The default "longhorn" StorageClass requests 3 replicas, but the homelab +# cluster currently has only 2 schedulable nodes -- Longhorn places at most +# one replica per node, so the 3rd replica can never be scheduled and the +# volume runs permanently "degraded". This class pins replica count to the +# actual node count. Bump back to "longhorn" (or numberOfReplicas: "3" here) +# once the 3rd node joins. +apiVersion: storage.k8s.io/v1 +kind: StorageClass +metadata: + name: longhorn-kafka +provisioner: driver.longhorn.io +allowVolumeExpansion: true +reclaimPolicy: Delete +volumeBindingMode: Immediate +parameters: + numberOfReplicas: "2" + staleReplicaTimeout: "30" + fromBackup: "" + fsType: "ext4" + dataLocality: "disabled" +{{- end }} diff --git a/k8s/charts/kafka-cluster/values.yaml b/k8s/charts/kafka-cluster/values.yaml index 269a0ed..09f7ccf 100644 --- a/k8s/charts/kafka-cluster/values.yaml +++ b/k8s/charts/kafka-cluster/values.yaml @@ -4,8 +4,12 @@ namespace: sqs nodePool: replicas: 3 storage: - class: longhorn - sizeGi: 50 + class: longhorn-kafka + # Longhorn's per-node scheduling budget on the current 2-node cluster has + # only ~36Gi of headroom left (other PVCs already reserve the rest), and + # each node hosts one replica of all 3 broker volumes -- so 3 * sizeGi + # must fit in that headroom. Revisit once the 3rd node joins. + sizeGi: 10 resources: memory: 5Gi cpu: "2" diff --git a/k8s/environments/homelab.yaml b/k8s/environments/homelab.yaml index f805eee..5ac66a3 100644 --- a/k8s/environments/homelab.yaml +++ b/k8s/environments/homelab.yaml @@ -7,8 +7,15 @@ kafkaCluster: nodePool: replicas: 3 storage: - class: longhorn - sizeGi: 50 + # longhorn-kafka pins numberOfReplicas to 2 to match the current + # 2-node cluster (see kafka-cluster chart's storageclass.yaml). + # Switch back to "longhorn" (3 replicas) once the 3rd node joins. + class: longhorn-kafka + # Longhorn's per-node scheduling budget on the current 2-node cluster + # only has ~36Gi of headroom (other PVCs reserve the rest), and each + # node hosts one replica of all 3 broker volumes, so 3 * sizeGi must + # fit in that headroom. Revisit once the 3rd node joins. + sizeGi: 10 resources: memory: 5Gi cpu: "2" @@ -21,5 +28,5 @@ managementService: ingress: host: kmsvc.homelab.internal clusterIssuer: homelab-ca - authentikIssuerURL: "" # cluster/KMSVC_AUTHENTIK_ISSUER_URL — fill in once task 0 is done - authentikAudience: "" # cluster/KMSVC_AUTHENTIK_AUDIENCE + authentikIssuerURL: "https://authentik.riotpiao.homelab.com/application/o/kafaka/" + authentikAudience: "QI0gPtR99ar8VvhK8Tqox4SDkTKzbNU7lbgwBNSc"