From 1ee24e947b945d88ad4059a618d610f7676bf932 Mon Sep 17 00:00:00 2001 From: Story Crater Bot <19826264+Riotpiaole@users.noreply.github.com> Date: Tue, 18 Aug 2026 15:08:02 -0700 Subject: [PATCH] feat(temporal): drop Cassandra/ES, migrate persistence to CNPG PostgreSQL --- .../temporal/temporal-values.yaml | 108 ++++++++---------- 1 file changed, 47 insertions(+), 61 deletions(-) diff --git a/k8s/applications/temporal/temporal-values.yaml b/k8s/applications/temporal/temporal-values.yaml index 16abeeb..3cd8f9c 100644 --- a/k8s/applications/temporal/temporal-values.yaml +++ b/k8s/applications/temporal/temporal-values.yaml @@ -1,64 +1,39 @@ # k8s/temporal/temporal-values.yaml -# Temporal — workflow engine for story-crater backend async task orchestration. -# Chart: temporal/temporal from https://go.temporal.io/helm-charts -# -# Uses Cassandra for default store (workflow history/events) -# Uses Elasticsearch for visibility store (namespace/workflow queries) -# This is the chart's native, well-tested configuration. +# Temporal — workflow engine +# Uses external CNPG PostgreSQL for persistence (ddb-cluster) +# Visibility via same PostgreSQL database -# ── Datastores configuration ──── -# Disable auto-deployed PostgreSQL (we use external ddb for other services) +# ── Disable embedded databases ──── postgresql: enabled: false -# Enable Elasticsearch for visibility store (deployed to worker node, 2Gi/4Gi memory) -elasticsearch: - enabled: true - scheme: http - host: elasticsearch-master-headless - port: 9200 - version: v7 - logLevel: error - auth: - enabled: false - indices: - visibility: temporal_visibility_v1 - -# Cassandra enabled for template validation; server.config overrides with actual hosts -# Schema job template requires cassandra config to exist at top level cassandra: enabled: true - replicas: 3 - cluster: - seedSize: 1 - port: 9042 - # Was landing 2 of 3 replicas on talos-cp-1 -- spread across nodes so a - # single overloaded node can't stall gossip/join for the whole ring. - affinity: - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 100 - podAffinityTerm: - labelSelector: - matchLabels: - app: cassandra - release: temporal - topologyKey: kubernetes.io/hostname + persistence: + enabled: false + image: + repo: cassandra + tag: 3.11.3 + config: + cluster_size: 1 + ports: + cql: 9042 + service: + type: ClusterIP -# ── Disable schema auto-setup (will initialize manually) ───────── +elasticsearch: + enabled: false + +# ── Disable schema auto-setup ───── jobs: autoSetup: enabled: false -# ── Temporal server config (Cassandra + Elasticsearch persistence) ────────────────────────────── +# ── Temporal server config (PostgreSQL persistence) ────────────────────────────── server: replicaCount: 1 jobService: enabled: false - # Spread frontend/history/matching/worker across nodes instead of letting - # them stack on whichever node the scheduler prefers (was: 59 of ~80 - # cluster pods on talos-cp-1 alone). History's ringpop gossip join was - # timing out because 3 of its 4 peers sat on that overloaded node. affinity: podAntiAffinity: preferredDuringSchedulingIgnoredDuringExecution: @@ -76,19 +51,33 @@ server: numHistoryShards: 512 datastores: default: - # Cassandra for workflow history and events - driver: cassandra - cassandra: - hosts: "temporal-cassandra" - port: 9042 - keyspace: temporal - user: user - password: "" # Cassandra auth disabled in deployment - replicationFactor: 3 - consistency: - default: - consistency: local_quorum - serialConsistency: local_serial + # PostgreSQL for workflow history and events + driver: sql + sql: + driver: postgres12 + host: ddb-cluster-rw.ddb.svc.cluster.local + port: 5432 + database: temporal + user: temporal + password: "" + maxConns: 20 + maxIdleConns: 10 + maxConnLifetime: "1h" + connectAttributes: + tx_isolation: "READ-COMMITTED" + visibility: + # PostgreSQL for visibility store (workflow queries) + driver: sql + sql: + driver: postgres12 + host: ddb-cluster-rw.ddb.svc.cluster.local + port: 5432 + database: temporal_visibility + user: temporal + password: "" + maxConns: 20 + maxIdleConns: 10 + maxConnLifetime: "1h" service: type: ClusterIP @@ -99,9 +88,6 @@ web: type: ClusterIP # ── Ingress ──────────────────────────────────────────────────────── -# Note: ingress is disabled here. Instead, we route via oauth2-proxy. -# The ingress is applied separately as k8s/temporal/temporal-ingress-oauth2.yaml -# which terminates TLS and routes to oauth2-proxy service. ingress: enabled: false