# logging/loki-values.yaml # Grafana Loki — log aggregation backend for the homelab. # Deployed in SingleBinary mode: one pod handles ingest, query, and compaction. # Chunks are stored in MinIO (S3-compatible) — no local PVC needed for log data. # # MinIO credentials are injected at deploy time via helmfile --set: # loki.storage.s3.accessKeyId ← MINIO_ROOT_USER # loki.storage.s3.secretAccessKey ← MINIO_ROOT_PASSWORD # The placeholder values below are overridden and never used. # ── Deployment mode ─────────────────────────────────────────────────────────── # SingleBinary collapses all Loki components (ingester, querier, compactor, ruler) # into one Deployment. Simpler ops for a homelab — no inter-component networking # or separate scaling to worry about. The tradeoff is no horizontal scaling. deploymentMode: SingleBinary loki: # auth_enabled: false skips tenant header (X-Scope-OrgID) enforcement. # All Promtail → Loki traffic is internal; multi-tenancy adds no value here. auth_enabled: false # ── Replication ───────────────────────────────────────────────────────────── # replication_factor: 1 — single replica, no write quorum needed. # Higher values require multiple ingesters (only valid outside SingleBinary). commonConfig: replication_factor: 1 # ── Storage backend ────────────────────────────────────────────────────────── # s3 type with s3ForcePathStyle: MinIO exposes buckets as paths # (http://host:9000/bucket) not subdomains (http://bucket.host:9000). # insecure: true — MinIO in this cluster has no TLS; traffic stays in-cluster. # Three buckets: chunks (log data), ruler (recording/alerting rules), admin (index). storage: type: s3 s3: endpoint: minio.storage.svc.cluster.local:80 # operator svc maps :80 -> pod :9000 region: us-east-1 # MinIO ignores region but Loki's S3 client requires it s3ForcePathStyle: true insecure: true # Interpolated at runtime from the loki-s3-creds Secret (SOPS-CMP emitted), # injected as env vars via extraEnvFrom below. Requires -config.expand-env. access_key_id: "${access_key_id}" secret_access_key: "${secret_access_key}" bucketNames: chunks: loki-chunks ruler: loki-ruler admin: loki-admin # ── Schema ─────────────────────────────────────────────────────────────────── # v13 + TSDB is the current recommended schema (Loki 3.x). # from: sets the date after which this schema applies — logs before this date # would use a previous schema config (none exists here, so all logs use v13). # period: 24h means one index table per day in the object store. schemaConfig: configs: - from: "2024-01-01" store: boltdb-shipper object_store: s3 schema: v13 index: prefix: index_ period: 24h # ── Ingester ───────────────────────────────────────────────────────────────── # Controls how log chunks are buffered before being flushed to MinIO. # chunk_idle_period: flush a chunk if no new logs arrive for 3m (reduces # open chunk count). chunk_retain_period: keep flushed chunks in memory # briefly so late-arriving out-of-order logs can still be appended. # WAL persists in-memory chunks to disk — required for boltdb-shipper. ingester: chunk_idle_period: 3m chunk_block_size: 262144 chunk_retain_period: 1m wal: dir: /var/loki/wal # ── Compactor ──────────────────────────────────────────────────────────────── # Merges small index files written by ingesters into larger ones, and # enforces retention by deleting chunks older than retention_period. # retention_delete_delay: waits 2h after marking chunks for deletion before # actually removing them — safety window if a query is still reading them. compactor: working_directory: /var/loki/compactor compaction_interval: 10m retention_enabled: true retention_delete_delay: 2h retention_delete_worker_count: 150 delete_request_store: s3 # ── Limits ─────────────────────────────────────────────────────────────────── # retention_period: 10 days. Homelab — no long-term log storage needed. # ingestion_rate_mb / burst: rate limits per tenant (single tenant here). # 4 MB/s steady, 6 MB/s burst — plenty for a 3-node cluster. # max_query_series: caps how many unique label combinations a single query # can return — prevents runaway cardinality queries from OOMing the pod. # max_query_lookback: hard cap matching retention_period (no point querying # further back than what's stored). # allow_structured_metadata: false — required for boltdb-shipper index store. limits_config: retention_period: 240h ingestion_rate_mb: 4 ingestion_burst_size_mb: 6 max_query_series: 5000 max_query_lookback: 240h max_label_names_per_series: 30 allow_structured_metadata: false # Query timeout: increased to 120s to tolerate 5+ second network latency spikes # Default: 30s — too aggressive when pod-to-pod latency hits 5-10s query_timeout: 120s # ── Single binary pod ───────────────────────────────────────────────────────── singleBinary: replicas: 1 # Interpolate ${access_key_id}/${secret_access_key} in the S3 config from the # loki-s3-creds Secret (SOPS-CMP emitted). extraArgs: - "-config.expand-env=true" extraEnvFrom: - secretRef: name: loki-s3-creds resources: requests: cpu: 100m memory: 256Mi limits: cpu: 1000m memory: 512Mi # Persistent storage for indices + WAL. Durable log chunks live in MinIO. persistence: enabled: true storageClassName: longhorn size: 5Gi # Pin to az-a (talos-cp-1) — sole Longhorn node; otherwise the PVC fails to # attach on cp-2/cp-3. nodeSelector: topology.kubernetes.io/zone: az-a # ── Disable micro-service replicas ─────────────────────────────────────────── # In SingleBinary mode the chart still templates read/write/backend Deployments # unless explicitly set to 0. Setting replicas: 0 keeps them out of the cluster. read: replicas: 0 write: replicas: 0 backend: replicas: 0 # Nginx gateway is only needed for multi-replica deployments that split read # and write paths. Not used in SingleBinary. gateway: enabled: false # Disable the bundled MinIO subchart — we run our own minio-az-a/az-b releases # in the storage namespace with site replication. minio: enabled: false # ── Monitoring ──────────────────────────────────────────────────────────────── # Self-monitoring ships a Grafana Agent operator to scrape Loki's own metrics. # We use kube-prometheus-stack for that instead — avoid running two agents. # lokiCanary sends synthetic log lines to verify the write→read pipeline; # useful in production, too noisy for a homelab. monitoring: selfMonitoring: enabled: false grafanaAgent: installOperator: false lokiCanary: enabled: false serviceMonitor: enabled: false test: enabled: false