feat(api): add Kong RED metrics for LLM routes

Cluster-wide prometheus KongClusterPlugin (kong-metrics.yaml) +
chart-native ServiceMonitor (kong-values.yaml) expose
kong_http_requests_total/kong_latency_bucket/kong_bandwidth_bytes for
every route, LLM and otherwise. Dashboard filters to route=~"llm-.*"
for request rate, error rate, p95 upstream latency, and bandwidth.
Token-count metrics still need ai-proxy-advanced (Enterprise-only);
not attempted.
This commit is contained in:
Story Crater Bot
2026-08-16 07:01:15 -07:00
parent 3724cd3cdb
commit 0c52dec155
4 changed files with 40 additions and 7 deletions
+19
View File
@@ -0,0 +1,19 @@
# Cluster-wide Kong Prometheus plugin -- `global: "true"` label makes the
# ingress controller apply it to every route on this Kong instance, so all
# five LLM routes (ornith/reasoning/qwen/embeddings/rerank) get RED metrics
# without touching llm-routes.yaml. Scraped via kong-values.yaml's
# serviceMonitor (status listener, already on by chart default at :8100).
apiVersion: configuration.konghq.com/v1
kind: KongClusterPlugin
metadata:
name: prometheus
annotations:
kubernetes.io/ingress.class: kong
labels:
global: "true"
plugin: prometheus
config:
status_code_metrics: true
latency_metrics: true
bandwidth_metrics: true
upstream_health_metrics: true
+10
View File
@@ -118,6 +118,16 @@ podDisruptionBudget:
enabled: true
minAvailable: 1
# Status listener (metrics/health) is on by default at :8100 (chart default,
# verified via `helm show values`). This just wires the ServiceMonitor the
# chart already knows how to generate for it, so kong_http_requests_total /
# kong_latency_* / kong_bandwidth_bytes land in Prometheus. Paired with the
# cluster-wide `prometheus` KongClusterPlugin in kong-metrics.yaml.
serviceMonitor:
enabled: true
labels:
release: kube-prometheus-stack
# Spread the two replicas across nodes; `ScheduleAnyway` so a single-node
# situation degrades to co-location instead of leaving a pod Pending.
topologySpreadConstraints:
+1
View File
@@ -6,6 +6,7 @@ kind: Kustomization
# or it is silently dropped with no error and no drift shown.
resources:
- ingress.yaml
- kong-metrics.yaml
- llm-routes.yaml
- model-auth.yaml
# No top-level `namespace:` transformer on purpose: ingress.yaml sets its own