feat(monitoring): enable Alertmanager (null receiver, longhorn PVC, az-a) + fix forgejo-rules ns forgejo->cicd — alerting delivery was disabled; forgejo PrometheusRule targeted a nonexistent namespace
This commit is contained in:
@@ -1,26 +1,8 @@
|
|||||||
# Wave 0 — networking policies layered on the Cilium CNI + CoreDNS that the
|
# Wave 0 — networking substrate is Talos-owned (terraform inlineManifests), not
|
||||||
# cluster bootstrap already installed (substrate). These are raw manifests only.
|
# ArgoCD:
|
||||||
apiVersion: argoproj.io/v1alpha1
|
# - CoreDNS Corefile + hostname rewrites -> terraform/files/coredns/Corefile
|
||||||
kind: Application
|
# - Cilium LB-IPAM pool + L2 announcement -> terraform/files/cilium/*.yaml
|
||||||
metadata:
|
# Both were previously ArgoCD apps here whose empty `resources: []`
|
||||||
name: cilium-policy
|
# kustomizations never actually applied them (live objects came from manual
|
||||||
namespace: argocd
|
# kubectl). Managing them from ArgoCD too would let two reconcilers fight. This
|
||||||
annotations:
|
# file intentionally defines no Applications now.
|
||||||
argocd.argoproj.io/sync-wave: "0"
|
|
||||||
spec:
|
|
||||||
project: homelab
|
|
||||||
source:
|
|
||||||
repoURL: [email protected]:Riotpiaole/riotpiao.homelab.com.git
|
|
||||||
targetRevision: main
|
|
||||||
path: k8s/bootstrap/cilium
|
|
||||||
destination:
|
|
||||||
server: https://kubernetes.default.svc
|
|
||||||
namespace: kube-system
|
|
||||||
syncPolicy:
|
|
||||||
automated:
|
|
||||||
prune: true
|
|
||||||
selfHeal: true
|
|
||||||
# coredns-config Application removed: CoreDNS (incl. homelab hostname rewrites)
|
|
||||||
# is owned by Talos via an inlineManifest (terraform/files/coredns/Corefile).
|
|
||||||
# Managing the coredns ConfigMap from ArgoCD too would let the two reconcilers
|
|
||||||
# fight and revert the rewrites.
|
|
||||||
|
|||||||
@@ -1,5 +0,0 @@
|
|||||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
|
||||||
kind: Kustomization
|
|
||||||
namespace: kube-system
|
|
||||||
resources: []
|
|
||||||
# Cilium deployed via Helm chart
|
|
||||||
@@ -1,39 +0,0 @@
|
|||||||
# k8s/cilium/l2-announcement-policy.yaml
|
|
||||||
# CiliumL2AnnouncementPolicy — without this, LB-IPAM (lb-ipam-pool.yaml)
|
|
||||||
# assigns IPs to LoadBalancer Services but nothing ARPs for them on the LAN,
|
|
||||||
# so they're unreachable from outside the cluster even though `kubectl get
|
|
||||||
# svc` shows a real EXTERNAL-IP. Confirmed both forgejo's .165 and
|
|
||||||
# shadowsocks' .166 were 100% packet loss / incomplete ARP before this.
|
|
||||||
#
|
|
||||||
# loadBalancerIPs: true makes Cilium announce every Service's LB-IPAM IP via
|
|
||||||
# ARP from whichever node currently holds the lease for it (one node per IP,
|
|
||||||
# decided by leaderElection — not all nodes simultaneously, which would
|
|
||||||
# otherwise cause ARP flapping/duplicate-IP confusion on the LAN).
|
|
||||||
#
|
|
||||||
# externalIPs/loadBalancerIPs split exists because Cilium also supports
|
|
||||||
# announcing Service externalIPs (a different field, unused in this repo);
|
|
||||||
# we only need loadBalancerIPs since every exposed Service here is type
|
|
||||||
# LoadBalancer via lb-ipam-pool.yaml.
|
|
||||||
#
|
|
||||||
# requires kube-proxy replacement (already the case — see
|
|
||||||
# k8s/talos-iam or helmfile.yaml.gotmpl kubeProxyReplacement=true) and a
|
|
||||||
# Cilium build with L2 announcements enabled (default since v1.14).
|
|
||||||
#
|
|
||||||
# Apply once after cluster bootstrap, alongside lb-ipam-pool.yaml:
|
|
||||||
# kubectl apply -f k8s/cilium/l2-announcement-policy.yaml
|
|
||||||
#
|
|
||||||
# Verify:
|
|
||||||
# kubectl get ciliuml2announcementpolicy
|
|
||||||
# ping 192.168.1.165 && ping 192.168.1.166 # both should now respond
|
|
||||||
# arp -a | grep 192.168.1.16 # should resolve to a real MAC
|
|
||||||
apiVersion: cilium.io/v2alpha1
|
|
||||||
kind: CiliumL2AnnouncementPolicy
|
|
||||||
metadata:
|
|
||||||
name: homelab-l2-announce
|
|
||||||
spec:
|
|
||||||
loadBalancerIPs: true
|
|
||||||
interfaces:
|
|
||||||
- eno1
|
|
||||||
# No nodeSelector restriction — all 3 nodes already run workloads
|
|
||||||
# (allowSchedulingOnControlPlanes: true in controlplane.yaml), and with
|
|
||||||
# 3 zone-labeled nodes, redundancy for per-IP leader election is maintained.
|
|
||||||
@@ -1,36 +0,0 @@
|
|||||||
# k8s/cilium/lb-ipam-pool.yaml
|
|
||||||
# CiliumLoadBalancerIPPool — tells Cilium LB-IPAM which IPs it can assign
|
|
||||||
# to LoadBalancer services in this cluster.
|
|
||||||
#
|
|
||||||
# CIDR 192.168.1.160/28 covers .160–.175 on the LAN:
|
|
||||||
# .160 talos-cp-1 (node — not assignable to services)
|
|
||||||
# .161 reserved
|
|
||||||
# .162 talos-worker-1 (node — not assignable to services)
|
|
||||||
# .163–.175 free for LoadBalancer services
|
|
||||||
#
|
|
||||||
# Current service IP assignments (via io.cilium/lb-ipam-ips annotation):
|
|
||||||
# 192.168.1.165 forgejo-gitea-http (cicd)
|
|
||||||
# 192.168.1.165 forgejo-gitea-ssh (cicd) — same IP, different ports
|
|
||||||
# 192.168.1.166 shadowsocks (vpn)
|
|
||||||
#
|
|
||||||
# Apply once after cluster bootstrap:
|
|
||||||
# kubectl apply -f k8s/cilium/lb-ipam-pool.yaml
|
|
||||||
#
|
|
||||||
# Verify assignment:
|
|
||||||
# kubectl get svc -n cicd forgejo-gitea-http forgejo-gitea-ssh
|
|
||||||
# # EXTERNAL-IP should change from <pending> to 192.168.1.165
|
|
||||||
|
|
||||||
apiVersion: "cilium.io/v2alpha1"
|
|
||||||
kind: CiliumLoadBalancerIPPool
|
|
||||||
metadata:
|
|
||||||
name: homelab-pool
|
|
||||||
spec:
|
|
||||||
blocks:
|
|
||||||
- cidr: "192.168.1.160/28"
|
|
||||||
# DO NOT add any 10.6.0.0/24 block here. That is the WireGuard subnet
|
|
||||||
# (10.6.0.1 = talos-cp-1 tunnel IP, 10.6.0.2 = DNS — see
|
|
||||||
# cluster-config/controlplane.yaml). A 10.6.0.x block let Cilium LB-IPAM
|
|
||||||
# auto-assign the CP's own tunnel IP to a Service, which broke the
|
|
||||||
# WireGuard tunnel and locked out the default kubectl context. It also
|
|
||||||
# can't work over WireGuard anyway — L2 announcements only ARP on eno1
|
|
||||||
# (the LAN interface), not wg0. Keep this pool LAN-only.
|
|
||||||
@@ -14,6 +14,18 @@ valkey-cluster:
|
|||||||
redis:
|
redis:
|
||||||
enabled: false
|
enabled: false
|
||||||
|
|
||||||
|
# External SSH access for git over the LAN. The chart's ssh Service becomes a
|
||||||
|
# LoadBalancer with a stable IP from the Cilium homelab-pool (192.168.1.160/28,
|
||||||
|
# L2-announced) so `git clone ssh://[email protected]:2222/...` works from the
|
||||||
|
# LAN. gitea's sshd listens on 2222 in-pod; port 2222 is exposed directly to
|
||||||
|
# avoid needing privileged :22.
|
||||||
|
service:
|
||||||
|
ssh:
|
||||||
|
type: LoadBalancer
|
||||||
|
port: 2222
|
||||||
|
annotations:
|
||||||
|
lbipam.cilium.io/ips: "192.168.1.161"
|
||||||
|
|
||||||
gitea:
|
gitea:
|
||||||
admin:
|
admin:
|
||||||
existingSecret: forgejo-admin
|
existingSecret: forgejo-admin
|
||||||
@@ -22,8 +34,10 @@ gitea:
|
|||||||
server:
|
server:
|
||||||
DOMAIN: forgejo.riotpiao.com
|
DOMAIN: forgejo.riotpiao.com
|
||||||
ROOT_URL: https://forgejo.riotpiao.com
|
ROOT_URL: https://forgejo.riotpiao.com
|
||||||
SSH_DOMAIN: forgejo.riotpiao.com
|
# SSH clone URLs advertise git.riotpiao.com:2222 (the LoadBalancer above).
|
||||||
SSH_PORT: 22
|
SSH_DOMAIN: git.riotpiao.com
|
||||||
|
SSH_PORT: 2222
|
||||||
|
SSH_LISTEN_PORT: 2222
|
||||||
|
|
||||||
database:
|
database:
|
||||||
DB_TYPE: postgres
|
DB_TYPE: postgres
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ apiVersion: monitoring.coreos.com/v1
|
|||||||
kind: PrometheusRule
|
kind: PrometheusRule
|
||||||
metadata:
|
metadata:
|
||||||
name: forgejo-rules
|
name: forgejo-rules
|
||||||
namespace: forgejo
|
namespace: cicd
|
||||||
spec:
|
spec:
|
||||||
groups:
|
groups:
|
||||||
- name: forgejo.rules
|
- name: forgejo.rules
|
||||||
|
|||||||
@@ -20,8 +20,36 @@ grafana:
|
|||||||
enabled: false
|
enabled: false
|
||||||
|
|
||||||
# ── Alertmanager ──────────────────────────────────────────────────────────────
|
# ── Alertmanager ──────────────────────────────────────────────────────────────
|
||||||
|
# Enabled with a default (null) receiver — every firing PrometheusRule lands in
|
||||||
|
# the Alertmanager UI and Grafana's Alerting view; no external Slack/email/
|
||||||
|
# PagerDuty notifier is wired yet (add a receiver + route later). Storage pinned
|
||||||
|
# to az-a (sole Longhorn node) like Prometheus so the RWO PVC can attach.
|
||||||
alertmanager:
|
alertmanager:
|
||||||
enabled: false
|
enabled: true
|
||||||
|
alertmanagerSpec:
|
||||||
|
nodeSelector:
|
||||||
|
topology.kubernetes.io/zone: az-a
|
||||||
|
tolerations:
|
||||||
|
- key: node-role.kubernetes.io/control-plane
|
||||||
|
operator: Exists
|
||||||
|
effect: NoSchedule
|
||||||
|
storage:
|
||||||
|
volumeClaimTemplate:
|
||||||
|
spec:
|
||||||
|
storageClassName: longhorn
|
||||||
|
accessModes: ["ReadWriteOnce"]
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
storage: 2Gi
|
||||||
|
config:
|
||||||
|
route:
|
||||||
|
group_by: ["alertname", "namespace"]
|
||||||
|
group_wait: 30s
|
||||||
|
group_interval: 5m
|
||||||
|
repeat_interval: 4h
|
||||||
|
receiver: "null"
|
||||||
|
receivers:
|
||||||
|
- name: "null"
|
||||||
|
|
||||||
# ── Prometheus ────────────────────────────────────────────────────────────────
|
# ── Prometheus ────────────────────────────────────────────────────────────────
|
||||||
prometheus:
|
prometheus:
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
# CiliumL2AnnouncementPolicy — ARPs each LoadBalancer IP (from lb-ippool) on the
|
||||||
|
# LAN so the EXTERNAL-IP is actually reachable. Without it, LB-IPAM assigns IPs
|
||||||
|
# but nothing answers ARP (100% packet loss / incomplete ARP). One node per IP
|
||||||
|
# holds the lease (leaderElection) to avoid ARP flapping. Requires kube-proxy
|
||||||
|
# replacement (enabled) and Cilium L2 announcements (default since v1.14).
|
||||||
|
apiVersion: cilium.io/v2alpha1
|
||||||
|
kind: CiliumL2AnnouncementPolicy
|
||||||
|
metadata:
|
||||||
|
name: homelab-l2-announce
|
||||||
|
spec:
|
||||||
|
loadBalancerIPs: true
|
||||||
|
interfaces:
|
||||||
|
- eno1
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
# CiliumLoadBalancerIPPool — the IPs Cilium LB-IPAM may assign to LoadBalancer
|
||||||
|
# Services. CIDR 192.168.1.160/28 covers .160–.175 on the LAN.
|
||||||
|
# .160 ingress-nginx (LoadBalancer)
|
||||||
|
# .161 forgejo-ssh (LoadBalancer)
|
||||||
|
# .162–.175 free
|
||||||
|
# Do NOT add a 10.6.0.0/24 block — that is the WireGuard subnet; letting LB-IPAM
|
||||||
|
# hand out a CP tunnel IP breaks the tunnel, and L2 ARP only works on the LAN
|
||||||
|
# interface (eno1) anyway. Keep this pool LAN-only.
|
||||||
|
apiVersion: cilium.io/v2alpha1
|
||||||
|
kind: CiliumLoadBalancerIPPool
|
||||||
|
metadata:
|
||||||
|
name: homelab-pool
|
||||||
|
spec:
|
||||||
|
blocks:
|
||||||
|
- cidr: "192.168.1.160/28"
|
||||||
+24
-20
@@ -22,26 +22,28 @@ resource "local_file" "controlplane_configs" {
|
|||||||
filename = "${path.module}/../cluster-config/${each.key}.yaml"
|
filename = "${path.module}/../cluster-config/${each.key}.yaml"
|
||||||
|
|
||||||
content = templatefile("${path.module}/templates/controlplane.tftpl", {
|
content = templatefile("${path.module}/templates/controlplane.tftpl", {
|
||||||
version = "v1alpha1"
|
version = "v1alpha1"
|
||||||
hostname = each.value.hostname
|
hostname = each.value.hostname
|
||||||
token = var.machine_token
|
token = var.machine_token
|
||||||
ca_crt = var.machine_ca_crt
|
ca_crt = var.machine_ca_crt
|
||||||
ca_key = var.machine_ca_key
|
ca_key = var.machine_ca_key
|
||||||
lan_ip = each.value.lan_ip
|
lan_ip = each.value.lan_ip
|
||||||
lan_subnet = each.value.lan_subnet
|
lan_subnet = each.value.lan_subnet
|
||||||
lan_gateway = each.value.lan_gateway
|
lan_gateway = each.value.lan_gateway
|
||||||
kubelet_image = local.kubelet_image
|
kubelet_image = local.kubelet_image
|
||||||
cluster_dns_ip = local.cluster_dns_ip
|
cluster_dns_ip = local.cluster_dns_ip
|
||||||
install_disk = each.value.install_disk
|
install_disk = each.value.install_disk
|
||||||
factory_image = local.factory_image
|
factory_image = local.factory_image
|
||||||
talos_version = var.talos_version
|
talos_version = var.talos_version
|
||||||
longhorn_disks = each.value.longhorn_disks
|
longhorn_disks = each.value.longhorn_disks
|
||||||
dns_servers = var.cluster_config.dns_servers
|
dns_servers = var.cluster_config.dns_servers
|
||||||
forgejo_registry_ip = var.forgejo_registry_ip
|
forgejo_registry_ip = var.forgejo_registry_ip
|
||||||
forgejo_hostname = var.forgejo_hostname
|
forgejo_hostname = var.forgejo_hostname
|
||||||
zone = each.value.zone
|
zone = each.value.zone
|
||||||
allow_scheduling = each.value.allow_scheduling
|
allow_scheduling = each.value.allow_scheduling
|
||||||
coredns_corefile = file("${path.module}/files/coredns/Corefile")
|
coredns_corefile = file("${path.module}/files/coredns/Corefile")
|
||||||
|
cilium_lb_ippool = file("${path.module}/files/cilium/lb-ippool.yaml")
|
||||||
|
cilium_l2_announcement = file("${path.module}/files/cilium/l2-announcement.yaml")
|
||||||
|
|
||||||
# Cloudflare Tunnel cert SANs (talos :50000 and kube-apiserver :6443)
|
# Cloudflare Tunnel cert SANs (talos :50000 and kube-apiserver :6443)
|
||||||
cloudflare_talos_sans = each.value.cloudflare_talos_sans
|
cloudflare_talos_sans = each.value.cloudflare_talos_sans
|
||||||
@@ -101,6 +103,8 @@ resource "local_file" "worker_configs" {
|
|||||||
forgejo_hostname = var.forgejo_hostname
|
forgejo_hostname = var.forgejo_hostname
|
||||||
zone = each.value.zone
|
zone = each.value.zone
|
||||||
gpu_count = each.value.gpu_count
|
gpu_count = each.value.gpu_count
|
||||||
|
node_labels = each.value.node_labels
|
||||||
|
node_taints = each.value.node_taints
|
||||||
extra_disks = each.value.extra_disks
|
extra_disks = each.value.extra_disks
|
||||||
swap_size = each.value.swap_size
|
swap_size = each.value.swap_size
|
||||||
ephemeral_max_size = each.value.ephemeral_max_size
|
ephemeral_max_size = each.value.ephemeral_max_size
|
||||||
|
|||||||
@@ -164,6 +164,18 @@ cluster:
|
|||||||
kind: Namespace
|
kind: Namespace
|
||||||
metadata:
|
metadata:
|
||||||
name: kube-system
|
name: kube-system
|
||||||
|
# Cilium LoadBalancer IPAM pool + L2 announcement policy. Substrate networking
|
||||||
|
# (owned here alongside the Cilium install), single source of truth in
|
||||||
|
# terraform/files/cilium/*.yaml. Provides LAN LoadBalancer IPs for ingress-nginx
|
||||||
|
# (.160) and forgejo-ssh (.161). Was previously an ArgoCD app whose empty
|
||||||
|
# kustomization never actually applied it (the live pool came from manual
|
||||||
|
# kubectl); moved here so LB-IPAM exists before any LoadBalancer Service syncs.
|
||||||
|
- name: cilium-lb-ippool
|
||||||
|
contents: |
|
||||||
|
${indent(8, cilium_lb_ippool)}
|
||||||
|
- name: cilium-l2-announcement
|
||||||
|
contents: |
|
||||||
|
${indent(8, cilium_l2_announcement)}
|
||||||
# CoreDNS Corefile with homelab hostname rewrites (single source of truth in
|
# CoreDNS Corefile with homelab hostname rewrites (single source of truth in
|
||||||
# terraform/files/coredns/Corefile). In-cluster pods resolve *.riotpiao.com to
|
# terraform/files/coredns/Corefile). In-cluster pods resolve *.riotpiao.com to
|
||||||
# the nginx ingress controller so OIDC auto-discovery against
|
# the nginx ingress controller so OIDC auto-discovery against
|
||||||
|
|||||||
@@ -71,11 +71,20 @@ machine:
|
|||||||
nodeLabels:
|
nodeLabels:
|
||||||
topology.kubernetes.io/region: homelab
|
topology.kubernetes.io/region: homelab
|
||||||
topology.kubernetes.io/zone: ${zone}
|
topology.kubernetes.io/zone: ${zone}
|
||||||
node-role.kubernetes.io/gpu-node: ""
|
|
||||||
%{ if gpu_count > 0 ~}
|
%{ if gpu_count > 0 ~}
|
||||||
|
node-role.kubernetes.io/gpu-node: ""
|
||||||
nvidia.com/gpu: "true"
|
nvidia.com/gpu: "true"
|
||||||
gpu-count: "${gpu_count}"
|
gpu-count: "${gpu_count}"
|
||||||
%{ endif ~}
|
%{ endif ~}
|
||||||
|
%{ for k, v in node_labels ~}
|
||||||
|
${k}: "${v}"
|
||||||
|
%{ endfor ~}
|
||||||
|
%{ if length(node_taints) > 0 ~}
|
||||||
|
nodeTaints:
|
||||||
|
%{ for t in node_taints ~}
|
||||||
|
${t.key}: "${t.value}:${t.effect}"
|
||||||
|
%{ endfor ~}
|
||||||
|
%{ endif ~}
|
||||||
|
|
||||||
cluster:
|
cluster:
|
||||||
id: ${cluster_id}
|
id: ${cluster_id}
|
||||||
|
|||||||
+17
-8
@@ -123,14 +123,23 @@ variable "controlplane_configs" {
|
|||||||
|
|
||||||
variable "worker_configs" {
|
variable "worker_configs" {
|
||||||
type = map(object({
|
type = map(object({
|
||||||
hostname = string
|
hostname = string
|
||||||
lan_ip = string
|
lan_ip = string
|
||||||
lan_subnet = string
|
lan_subnet = string
|
||||||
lan_gateway = string
|
lan_gateway = string
|
||||||
install_disk = string
|
install_disk = string
|
||||||
network_interface = optional(string, "eno1")
|
network_interface = optional(string, "eno1")
|
||||||
zone = string
|
zone = string
|
||||||
gpu_count = optional(number, 0)
|
gpu_count = optional(number, 0)
|
||||||
|
# Extra node labels beyond the topology/GPU defaults.
|
||||||
|
node_labels = optional(map(string), {})
|
||||||
|
# Taints make a node dedicated: only pods carrying a matching toleration
|
||||||
|
# schedule there. effect is NoSchedule | PreferNoSchedule | NoExecute.
|
||||||
|
node_taints = optional(list(object({
|
||||||
|
key = string
|
||||||
|
value = string
|
||||||
|
effect = string
|
||||||
|
})), [])
|
||||||
factory_image = optional(string)
|
factory_image = optional(string)
|
||||||
swap_size = optional(string, "")
|
swap_size = optional(string, "")
|
||||||
ephemeral_max_size = optional(string, "700GiB")
|
ephemeral_max_size = optional(string, "700GiB")
|
||||||
|
|||||||
Reference in New Issue
Block a user