feat(helmfile): remove phase2 releases (cloudnative-pg, loki, grafana, prometheus, forgejo, forgejo-runner) — ArgoCD-managed
This commit is contained in:
@@ -175,132 +175,6 @@ releases:
|
||||
- -f
|
||||
- k8s/ingress/ingress.yaml
|
||||
|
||||
# ── CloudNativePG (centralized database) ──────────────────────────────────────
|
||||
# Single HA cluster (1 primary + 2 replicas) serving Authentik + story-crater-backend.
|
||||
# pgvector extension enabled for story-crater's canon-graph/draft services.
|
||||
- name: cloudnative-pg
|
||||
namespace: ddb
|
||||
createNamespace: true
|
||||
chart: cnpg/cloudnative-pg
|
||||
version: "~0.20"
|
||||
values:
|
||||
- k8s/ddb/cnpg-values.yaml
|
||||
needs:
|
||||
- ingress-nginx/ingress-nginx # wait for ingress-nginx to be ready
|
||||
hooks:
|
||||
- events: ["postsync"]
|
||||
command: bash
|
||||
args:
|
||||
- -c
|
||||
- |
|
||||
# Wait for operator to be ready
|
||||
kubectl rollout status deploy/cloudnative-pg -n ddb --timeout=120s
|
||||
|
||||
# Apply the CNPG Cluster CR with pgvector support
|
||||
# Note: Using unquoted EOF so bash can expand variables
|
||||
kubectl apply -f - <<EOF
|
||||
apiVersion: postgresql.cnpg.io/v1
|
||||
kind: Cluster
|
||||
metadata:
|
||||
name: ddb-cluster
|
||||
namespace: ddb
|
||||
spec:
|
||||
instances: 3
|
||||
|
||||
# Image with pgvector extension
|
||||
imageName: ghcr.io/cloudnative-pg/postgresql:16.2
|
||||
|
||||
# Bootstrap: minimal setup, just vector extension
|
||||
# App users (authentik, story_crater) created by post-sync Job with Vault credentials
|
||||
bootstrap:
|
||||
initdb:
|
||||
database: postgres
|
||||
owner: postgres
|
||||
postInitApplicationSQL:
|
||||
- "CREATE EXTENSION IF NOT EXISTS vector;"
|
||||
|
||||
# Storage: 10Gi per instance (3 replicas = 30Gi total, backed by Longhorn)
|
||||
storage:
|
||||
size: 10Gi
|
||||
storageClass: longhorn
|
||||
|
||||
# Primary keeps 7 days of WAL for point-in-time recovery
|
||||
postgresql:
|
||||
parameters:
|
||||
max_parallel_workers_per_gather: "4"
|
||||
max_parallel_workers: "4"
|
||||
shared_buffers: "256MB"
|
||||
wal_sender_timeout: "900"
|
||||
wal_receiver_timeout: "900"
|
||||
|
||||
# ── Pod health probes (tuned for 5+ second network latency) ──────────────────────
|
||||
# Kubernetes probes to detect PostgreSQL readiness and liveness
|
||||
# Default timeout: 5s — too aggressive for high-latency clusters
|
||||
# Updated to 60s to tolerate WiFi bridge latency spikes
|
||||
podTemplateSpec:
|
||||
spec:
|
||||
affinity:
|
||||
# ── AZ Mapping (pod ordinal → node) ──────────────────────────────────────────
|
||||
# ddb-cluster-1 (ordinal 0) → talos-cp-1 (az-a)
|
||||
# ddb-cluster-2 (ordinal 1) → talos-worker-1 (az-b)
|
||||
# ddb-cluster-3 (ordinal 2) → talos-worker-2 (az-c)
|
||||
podAntiAffinity:
|
||||
preferredDuringSchedulingIgnoredDuringExecution:
|
||||
- weight: 100
|
||||
podAffinityTerm:
|
||||
labelSelector:
|
||||
matchExpressions:
|
||||
- key: cnpg.io/cluster
|
||||
operator: In
|
||||
values:
|
||||
- ddb-cluster
|
||||
topologyKey: kubernetes.io/hostname
|
||||
containers:
|
||||
- name: postgres
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
port: 8000
|
||||
path: /healthz
|
||||
initialDelaySeconds: 0
|
||||
timeoutSeconds: 60
|
||||
periodSeconds: 10
|
||||
failureThreshold: 3
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
port: 8000
|
||||
path: /readyz
|
||||
initialDelaySeconds: 0
|
||||
timeoutSeconds: 60
|
||||
periodSeconds: 10
|
||||
failureThreshold: 3
|
||||
startupProbe:
|
||||
httpGet:
|
||||
port: 8000
|
||||
path: /healthz
|
||||
initialDelaySeconds: 0
|
||||
timeoutSeconds: 60
|
||||
periodSeconds: 10
|
||||
failureThreshold: 360
|
||||
EOF
|
||||
|
||||
# Wait for CNPG cluster to be Ready (both condition and pod replicas)
|
||||
kubectl wait cluster/ddb-cluster -n ddb --for=condition=Ready --timeout=600s 2>/dev/null || true
|
||||
kubectl wait pod -n ddb -l cnpg.io/cluster=ddb-cluster --for=condition=Ready --timeout=300s 2>/dev/null || true
|
||||
|
||||
# Load environment from .env file (safe variable expansion)
|
||||
set -a
|
||||
[ -f .env ] && source .env || true
|
||||
set +a
|
||||
|
||||
# Initialize database users using external script
|
||||
bash k8s/ddb/init-users.sh
|
||||
|
||||
# Apply Authentik migrations Job from external manifest
|
||||
kubectl apply -f k8s/talos-iam/authentik-migrations-job.yaml
|
||||
|
||||
# Wait for migrations Job to complete before proceeding
|
||||
kubectl wait --for=condition=complete job/authentik-migrations -n iam --timeout=300s 2>/dev/null || true
|
||||
|
||||
# ── IAM (Authentik + Vault) ───────────────────────────────────────────────────
|
||||
- name: authentik
|
||||
namespace: iam
|
||||
@@ -419,189 +293,6 @@ releases:
|
||||
# Loki uses minio.storage.svc.cluster.local:9000 with Longhorn-replicated storage.
|
||||
# Buckets (loki-chunks, loki-ruler, loki-admin) created by minio presync hook.
|
||||
|
||||
- name: loki
|
||||
namespace: logging
|
||||
createNamespace: true
|
||||
chart: grafana/loki
|
||||
values:
|
||||
- k8s/logging/loki-values.yaml
|
||||
set:
|
||||
- name: loki.storage.s3.accessKeyId
|
||||
value: {{ env "MINIO_ROOT_USER" }}
|
||||
- name: loki.storage.s3.secretAccessKey
|
||||
value: {{ env "MINIO_ROOT_PASSWORD" }}
|
||||
- name: loki.storage.bucketNames.chunks
|
||||
value: loki-chunks
|
||||
- name: loki.storage.bucketNames.ruler
|
||||
value: loki-ruler
|
||||
- name: loki.storage.bucketNames.admin
|
||||
value: loki-admin
|
||||
hooks:
|
||||
- events: ["presync"]
|
||||
command: bash
|
||||
args:
|
||||
- -c
|
||||
- |
|
||||
bash k8s/base/namespace-setup.sh logging
|
||||
# Pre-create buckets BEFORE Loki pod starts — prevents NoSuchBucket errors
|
||||
bash k8s/storage/minio-bucket-init.sh storage loki-chunks loki-ruler loki-admin
|
||||
needs:
|
||||
- storage/minio
|
||||
|
||||
- name: grafana
|
||||
namespace: logging
|
||||
chart: grafana/grafana
|
||||
values:
|
||||
- k8s/logging/grafana-values.yaml
|
||||
set:
|
||||
- name: adminPassword
|
||||
value: {{ env "GRAFANA_ADMIN_PASSWORD" }}
|
||||
- name: env.GF_AUTH_GENERIC_OAUTH_CLIENT_SECRET
|
||||
value: {{ env "GRAFANA_OIDC_CLIENT_SECRET" }}
|
||||
needs:
|
||||
- logging/loki
|
||||
hooks:
|
||||
- events: ["postsync"]
|
||||
command: kubectl
|
||||
args:
|
||||
- apply
|
||||
- -f
|
||||
- k8s/monitoring/dashboards/
|
||||
|
||||
# ── Monitoring (kube-prometheus-stack) ───────────────────────────────────────
|
||||
|
||||
- name: prometheus
|
||||
namespace: monitoring
|
||||
createNamespace: true
|
||||
chart: prometheus-community/kube-prometheus-stack
|
||||
values:
|
||||
- k8s/monitoring/prometheus-values.yaml
|
||||
hooks:
|
||||
- events: ["presync"]
|
||||
command: bash
|
||||
args:
|
||||
- -c
|
||||
- |
|
||||
bash k8s/base/namespace-setup.sh monitoring
|
||||
- events: ["postsync"]
|
||||
command: bash
|
||||
args:
|
||||
- -c
|
||||
- |
|
||||
# Raw manifests that use the PrometheusRule/ServiceMonitor CRDs this
|
||||
# release just installed — must apply after the operator is up.
|
||||
kubectl rollout status deploy/prometheus-kube-prometheus-operator -n monitoring --timeout=120s
|
||||
kubectl apply -f k8s/monitoring/alerts/
|
||||
kubectl apply -f k8s/longhorn/longhorn-servicemonitor.yaml
|
||||
|
||||
# ── CI/CD (Forgejo + runner + Argo CD) ───────────────────────────────────────
|
||||
# Required env vars:
|
||||
# FORGEJO_ADMIN_PASSWORD — bootstrap admin password (set once, stored in .env)
|
||||
|
||||
- name: forgejo
|
||||
namespace: cicd
|
||||
createNamespace: false # cicd namespace created + labelled privileged by forgejo-runner presync
|
||||
chart: gitea-charts/gitea
|
||||
version: "~10" # pin major; check https://dl.gitea.com/charts/ for current
|
||||
values:
|
||||
- k8s/talos-ci-cd/forgejo-values.yaml
|
||||
set:
|
||||
- name: gitea.admin.password
|
||||
value: {{ env "FORGEJO_ADMIN_PASSWORD" }}
|
||||
- name: gitea.config.oauth2.ENABLED
|
||||
value: "true"
|
||||
- name: gitea.config.oauth2.PROVIDER
|
||||
value: "openidconnect"
|
||||
- name: gitea.config.oauth2.OPENID_CONNECT_DISCOVERY_URL
|
||||
value: "https://authentik.{{ env "CLUSTER_DOMAIN" }}/application/o/forgejo/.well-known/openid-configuration"
|
||||
- name: gitea.config.oauth2.CLIENT_ID
|
||||
value: "forgejo"
|
||||
- name: gitea.config.oauth2.CLIENT_SECRET
|
||||
value: {{ env "FORGEJO_OIDC_CLIENT_SECRET" }}
|
||||
needs:
|
||||
- iam/authentik
|
||||
hooks:
|
||||
- events: ["presync"]
|
||||
command: bash
|
||||
args:
|
||||
- -c
|
||||
- |
|
||||
bash k8s/base/namespace-setup.sh cicd
|
||||
# CA trust: read from cert-manager source-of-truth (never hardcoded)
|
||||
CA_PEM=$(kubectl get secret homelab-ca-secret -n cert-manager \
|
||||
-o jsonpath='{.data.tls\.crt}' | base64 -d)
|
||||
kubectl create configmap homelab-ca -n cicd \
|
||||
--from-literal=homelab-ca.crt="$CA_PEM" \
|
||||
--dry-run=client -o yaml | kubectl apply -f -
|
||||
kubectl apply -f k8s/ingress/ingress.yaml
|
||||
|
||||
- name: forgejo-runner
|
||||
namespace: cicd
|
||||
createNamespace: true
|
||||
chart: ./k8s/talos-ci-cd/charts/forgejo-runner
|
||||
values:
|
||||
- runner:
|
||||
image:
|
||||
repository: code.forgejo.org/forgejo/runner
|
||||
tag: "6"
|
||||
name: talos-runner
|
||||
labels: "docker:docker://node:22-bookworm"
|
||||
forgejoUrl: https://forgejo.{{ env "CLUSTER_DOMAIN" }}
|
||||
tokenSecret: runner-token
|
||||
resources:
|
||||
requests:
|
||||
cpu: 100m
|
||||
memory: 256Mi
|
||||
limits:
|
||||
cpu: "2"
|
||||
memory: 4Gi
|
||||
dind:
|
||||
image:
|
||||
repository: docker
|
||||
tag: "27-dind"
|
||||
resources:
|
||||
requests:
|
||||
cpu: 100m
|
||||
memory: 256Mi
|
||||
limits:
|
||||
cpu: "2"
|
||||
memory: 4Gi
|
||||
persistence:
|
||||
reg:
|
||||
storageClass: longhorn
|
||||
size: 1Gi
|
||||
dind:
|
||||
storageClass: longhorn
|
||||
size: 30Gi
|
||||
tolerations:
|
||||
- key: node-role.kubernetes.io/control-plane
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
needs:
|
||||
- cicd/forgejo
|
||||
hooks:
|
||||
- events: ["presync"]
|
||||
command: bash
|
||||
args:
|
||||
- -c
|
||||
- |
|
||||
bash k8s/base/namespace-setup.sh cicd
|
||||
# CA secret so the runner can verify Forgejo's TLS cert
|
||||
kubectl -n cicd create secret generic homelab-ca \
|
||||
--from-file=ca.crt=k8s/forge/pki/ca.crt \
|
||||
--dry-run=client -o yaml | kubectl apply -f -
|
||||
# Wait for Forgejo to be ready, then fetch + store the runner token
|
||||
kubectl -n cicd rollout status deploy/forgejo --timeout=120s
|
||||
TOKEN=$(kubectl -n cicd exec deploy/forgejo -- \
|
||||
forgejo actions generate-runner-token 2>/dev/null | tr -d '\r\n')
|
||||
kubectl -n cicd create secret generic runner-token \
|
||||
--from-literal=token="$TOKEN" \
|
||||
--dry-run=client -o yaml | kubectl apply -f -
|
||||
echo "Runner token stored in cicd/runner-token"
|
||||
- events: ["postsync"]
|
||||
command: kubectl
|
||||
args: ["apply", "-f", "k8s/forge/runner-gc-cronjob.yaml"]
|
||||
|
||||
# ── CI/CD (Argo CD) ──────────────────────────────────────────────────────────
|
||||
# Forgejo + runner are bootstrapped via kubectl apply (forge/forgejo.yaml, forge/runner.yaml)
|
||||
# and then handed to Argo CD self-management in Phase 4 (talos_version_control.html).
|
||||
|
||||
Reference in New Issue
Block a user