fix(temporal): switch to PostgreSQL (CNPG ddb-cluster) instead of broken Cassandra/ES setup

This commit is contained in:
Story Crater Bot
2026-07-21 13:41:12 -07:00
parent 4ea25620dd
commit e82c4b36a4
+37 -66
View File
@@ -1,65 +1,28 @@
# k8s/temporal/temporal-values.yaml
# Temporal — workflow engine for story-crater backend async task orchestration.
# Chart: temporal/temporal from https://go.temporal.io/helm-charts
#
# Uses Cassandra for default store (workflow history/events)
# Uses Elasticsearch for visibility store (namespace/workflow queries)
# This is the chart's native, well-tested configuration.
# Temporal — workflow engine
# Uses external CNPG PostgreSQL for persistence (ddb-cluster)
# Visibility via same PostgreSQL database
# ── Datastores configuration ────
# Disable auto-deployed PostgreSQL (we use external ddb for other services)
# ── Disable embedded databases ────
postgresql:
enabled: false
# Enable Elasticsearch for visibility store (deployed to worker node, 2Gi/4Gi memory)
elasticsearch:
enabled: true
scheme: http
host: elasticsearch-master-headless
port: 9200
version: v7
logLevel: error
auth:
enabled: false
indices:
visibility: temporal_visibility_v1
replicas: 1
# Cassandra enabled for template validation; server.config overrides with actual hosts
# Schema job template requires cassandra config to exist at top level
cassandra:
enabled: true
replicas: 3
cluster:
seedSize: 1
port: 9042
# Was landing 2 of 3 replicas on talos-cp-1 -- spread across nodes so a
# single overloaded node can't stall gossip/join for the whole ring.
affinity:
podAntiAffinity:
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
podAffinityTerm:
labelSelector:
matchLabels:
app: cassandra
release: temporal
topologyKey: kubernetes.io/hostname
enabled: false
# ── Disable schema auto-setup (will initialize manually) ─────────
elasticsearch:
enabled: false
# ── Disable schema auto-setup ─────
jobs:
autoSetup:
enabled: false
# ── Temporal server config (Cassandra + Elasticsearch persistence) ──────────────────────────────
# ── Temporal server config (PostgreSQL persistence) ──────────────────────────────
server:
replicaCount: 1
jobService:
enabled: false
# Spread frontend/history/matching/worker across nodes instead of letting
# them stack on whichever node the scheduler prefers (was: 59 of ~80
# cluster pods on talos-cp-1 alone). History's ringpop gossip join was
# timing out because 3 of its 4 peers sat on that overloaded node.
affinity:
podAntiAffinity:
preferredDuringSchedulingIgnoredDuringExecution:
@@ -77,22 +40,33 @@ server:
numHistoryShards: 512
datastores:
default:
# Cassandra for workflow history and events
driver: cassandra
cassandra:
hosts:
- "temporal-cassandra-0.temporal-cassandra.temporal"
- "temporal-cassandra-1.temporal-cassandra.temporal"
- "temporal-cassandra-2.temporal-cassandra.temporal"
port: 9042
keyspace: temporal
user: user
password: "" # Cassandra auth disabled in deployment
replicationFactor: 3
consistency:
default:
consistency: local_quorum
serialConsistency: local_serial
# PostgreSQL for workflow history and events
driver: sql
sql:
driver: postgres12
host: ddb-cluster-rw.ddb.svc.cluster.local
port: 5432
database: temporal
user: temporal
password: ""
maxConns: 20
maxIdleConns: 10
maxConnLifetime: "1h"
connectAttributes:
tx_isolation: "READ-COMMITTED"
visibility:
# PostgreSQL for visibility store (workflow queries)
driver: sql
sql:
driver: postgres12
host: ddb-cluster-rw.ddb.svc.cluster.local
port: 5432
database: temporal_visibility
user: temporal
password: ""
maxConns: 20
maxIdleConns: 10
maxConnLifetime: "1h"
service:
type: ClusterIP
@@ -103,9 +77,6 @@ web:
type: ClusterIP
# ── Ingress ────────────────────────────────────────────────────────
# Note: ingress is disabled here. Instead, we route via oauth2-proxy.
# The ingress is applied separately as k8s/temporal/temporal-ingress-oauth2.yaml
# which terminates TLS and routes to oauth2-proxy service.
ingress:
enabled: false