Author SHA1 Message Date
rock 179e12b9a9 feat: vLLM Prometheus metrics scraping (#37)
Add ServiceMonitor for llm-serving namespace. Wire prometheus.io annotations to all vLLM pods (reasoning, ornith, embeddings, reranker). Scrape /metrics@8080 every 30s with proper relabeling.
2026-09-11 10:46:15 +09:00
rock 7bd9f83fa7 fix(cnpg): scale paperless-db, immich-db, gotify-db to 3 instances (#30)
Co-authored-by: rock <[email protected]>
2026-09-10 22:56:41 +00:00
8 changed files with 65 additions and 3 deletions
+1 -1
View File
@@ -8,7 +8,7 @@ metadata:
annotations:
argocd.argoproj.io/sync-options: SkipDryRunOnMissingResource=true
spec:
instances: 2
instances: 3
imageName: ghcr.io/cloudnative-pg/postgresql:16.2
bootstrap:
initdb:
+1 -1
View File
@@ -18,7 +18,7 @@ metadata:
annotations:
argocd.argoproj.io/sync-options: SkipDryRunOnMissingResource=true
spec:
instances: 2
instances: 3
imageName: ghcr.io/cloudnative-pg/postgresql:18-minimal-trixie
postgresql:
extensions:
+8
View File
@@ -45,6 +45,14 @@ spec:
volumeMounts:
- mountPath: /mnt/models
name: models
podMetadata:
annotations:
prometheus.io/scrape: "true"
prometheus.io/port: "8080"
prometheus.io/path: "/metrics"
labels:
app.kubernetes.io/name: llm-embeddings
app.kubernetes.io/part-of: llm-serving
maxReplicas: 1
minReplicas: 1
nodeSelector:
+8
View File
@@ -83,6 +83,14 @@ spec:
volumeMounts:
- mountPath: /mnt/models
name: models
podMetadata:
annotations:
prometheus.io/scrape: "true"
prometheus.io/port: "8080"
prometheus.io/path: "/metrics"
labels:
app.kubernetes.io/name: llm-ornith
app.kubernetes.io/part-of: llm-serving
deploymentStrategy:
type: Recreate
# 1 replica -- ornith:35b only. qwen2.5:3b moved to CPU on cp-2.
+8
View File
@@ -104,6 +104,14 @@ spec:
name: models
- mountPath: /dev/shm
name: shm
podMetadata:
annotations:
prometheus.io/scrape: "true"
prometheus.io/port: "8080"
prometheus.io/path: "/metrics"
labels:
app.kubernetes.io/name: llm-reasoning
app.kubernetes.io/part-of: llm-serving
deploymentStrategy:
type: Recreate
maxReplicas: 1
+8
View File
@@ -45,6 +45,14 @@ spec:
volumeMounts:
- mountPath: /mnt/models
name: models
podMetadata:
annotations:
prometheus.io/scrape: "true"
prometheus.io/port: "8080"
prometheus.io/path: "/metrics"
labels:
app.kubernetes.io/name: llm-reranker
app.kubernetes.io/part-of: llm-serving
maxReplicas: 1
minReplicas: 1
nodeSelector:
+1 -1
View File
@@ -11,7 +11,7 @@ metadata:
annotations:
argocd.argoproj.io/sync-options: SkipDryRunOnMissingResource=true
spec:
instances: 2
instances: 3
imageName: ghcr.io/cloudnative-pg/postgresql:16.2
bootstrap:
initdb:
@@ -0,0 +1,30 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: vllm
namespace: monitoring
labels:
app.kubernetes.io/name: vllm
app.kubernetes.io/part-of: llm-serving
spec:
namespaceSelector:
matchNames:
- llm-serving
selector:
matchLabels:
app.kubernetes.io/part-of: llm-serving
endpoints:
- port: http
interval: 30s
scrapeTimeout: 10s
path: /metrics
scheme: http
relabelings:
- sourceLabels: [__meta_kubernetes_namespace]
targetLabel: namespace
- sourceLabels: [__meta_kubernetes_pod_name]
targetLabel: pod
- sourceLabels: [__meta_kubernetes_service_name]
targetLabel: service
- sourceLabels: [__meta_kubernetes_pod_label_app_kubernetes_io_name]
targetLabel: app