From 5d7b9acd439949dcbdde8e48178dcb114f084b31 Mon Sep 17 00:00:00 2001 From: Story Crater Bot <19826264+Riotpiaole@users.noreply.github.com> Date: Tue, 18 Aug 2026 18:25:12 -0700 Subject: [PATCH] feat(llm-serving): scale ornith to 2 replicas instead of a dedicated grm GPU reasoning keeps its 2 GPUs untouched. verifier's freed GPU goes to a second ornith replica instead of a standalone qwen-only pod -- both replicas load ornith:35b + qwen2.5:3b-instruct, k8s Service load-balances across them, so 2 concurrent implementer-style calls get independent instances. --- k8s/apps/api/llm-routes.yaml | 17 +++-- k8s/apps/llm-serving/grm.yaml | 96 ------------------------- k8s/apps/llm-serving/kustomization.yaml | 1 - k8s/apps/llm-serving/ornith.yaml | 24 ++++--- 4 files changed, 24 insertions(+), 114 deletions(-) delete mode 100644 k8s/apps/llm-serving/grm.yaml diff --git a/k8s/apps/api/llm-routes.yaml b/k8s/apps/api/llm-routes.yaml index 29c6f0f..e33d3fe 100644 --- a/k8s/apps/api/llm-routes.yaml +++ b/k8s/apps/api/llm-routes.yaml @@ -7,14 +7,13 @@ # has a `path: k8s/apps/api` source) so all gateway config stays in one place. # # ── Model -> upstream map (verified live) ─────────────────────────────────── -# reasoning -> reasoning-predictor vLLM, DeepSeek-R1-Distill-32B -# ornith:35b -> ornith-predictor Ollama -# qwen2.5:3b-instruct -> grm-predictor Ollama (dedicated GPU -- -# retired verifier-predictor's -# vLLM PRM slot; verification/ -# judge traffic no longer -# contends with ornith:35b's -# planner/implementer traffic) +# reasoning -> reasoning-predictor vLLM, DeepSeek-R1-Distill-32B, 2 replicas +# ornith:35b -> ornith-predictor Ollama, 2 replicas (retired verifier- +# qwen2.5:3b-instruct -> ornith-predictor Ollama predictor's vLLM PRM slot to get +# the 2nd GPU) -- k8s Service load-balances +# across both, each replica loads both +# models, so 2 concurrent implementer-style +# calls each land on an independent instance # nomic-embed-text-v2 -> embeddings-predictor TEI # bge-reranker-base -> reranker-predictor TEI # @@ -214,7 +213,7 @@ spec: pathType: Prefix backend: service: - name: grm-predictor + name: ornith-predictor port: number: 80 --- diff --git a/k8s/apps/llm-serving/grm.yaml b/k8s/apps/llm-serving/grm.yaml deleted file mode 100644 index 1e8a904..0000000 --- a/k8s/apps/llm-serving/grm.yaml +++ /dev/null @@ -1,96 +0,0 @@ -apiVersion: serving.kserve.io/v1beta1 -kind: InferenceService -metadata: - annotations: - serving.kserve.io/deploymentMode: RawDeployment - # Same Kong-timeout rationale as ornith.yaml: these configure the Service - # KServe generates, not the Ingress, and matter once OLLAMA_KEEP_ALIVE=-1 - # stops covering a cold load after a pod restart. - konghq.com/connect-timeout: "10000" - konghq.com/read-timeout: "3600000" - konghq.com/write-timeout: "3600000" - labels: - app.kubernetes.io/name: llm-grm - app.kubernetes.io/part-of: llm-serving - name: grm - namespace: llm-serving -spec: - predictor: - containers: - - command: - - /bin/sh - - -c - - 'set -e - - ollama serve & - - SERVE_PID=$! - - until ollama list >/dev/null 2>&1; do sleep 2; done - - ollama pull qwen2.5:3b-instruct - - ollama run qwen2.5:3b-instruct "ok" >/dev/null 2>&1 || true - - wait $SERVE_PID - - ' - env: - - name: OLLAMA_HOST - value: 0.0.0.0:8080 - - name: OLLAMA_MODELS - value: /mnt/models/ollama - - name: OLLAMA_CONTEXT_LENGTH - value: '32768' - - name: OLLAMA_KEEP_ALIVE - value: '-1' - - name: OLLAMA_NUM_PARALLEL - value: '1' - # Room for a second verification/reward model alongside qwen2.5:3b - # without a redeploy -- matches ornith.yaml's pattern, one dedicated - # GPU now free for it instead of contending with ornith:35b's. - - name: OLLAMA_MAX_LOADED_MODELS - value: '2' - image: ollama/ollama:0.32.9@sha256:1685741456770df6e3cceb2a945a5f75e020f658d1701509668d6f4688f1dd3f - name: kserve-container - ports: - - containerPort: 8080 - protocol: TCP - readinessProbe: - exec: - command: - - /bin/sh - - -c - - ollama ps 2>/dev/null | grep -q qwen2.5 - periodSeconds: 10 - resources: - limits: - cpu: '16' - memory: 16Gi - nvidia.com/gpu: '1' - requests: - cpu: '8' - memory: 8Gi - nvidia.com/gpu: '1' - startupProbe: - exec: - command: - - /bin/sh - - -c - - ollama ps 2>/dev/null | grep -q qwen2.5 - failureThreshold: 120 - periodSeconds: 15 - volumeMounts: - - mountPath: /mnt/models - name: models - deploymentStrategy: - type: Recreate - maxReplicas: 1 - minReplicas: 1 - nodeSelector: - kubernetes.io/hostname: worker-1 - runtimeClassName: nvidia - volumes: - - name: models - persistentVolumeClaim: - claimName: llm-models diff --git a/k8s/apps/llm-serving/kustomization.yaml b/k8s/apps/llm-serving/kustomization.yaml index 338cfd7..d646e65 100644 --- a/k8s/apps/llm-serving/kustomization.yaml +++ b/k8s/apps/llm-serving/kustomization.yaml @@ -11,7 +11,6 @@ kind: Kustomization # in tens of seconds, not a rolling update. resources: - embeddings.yaml - - grm.yaml - ornith.yaml - reasoning.yaml - reranker.yaml diff --git a/k8s/apps/llm-serving/ornith.yaml b/k8s/apps/llm-serving/ornith.yaml index 0293f96..a4ab0d2 100644 --- a/k8s/apps/llm-serving/ornith.yaml +++ b/k8s/apps/llm-serving/ornith.yaml @@ -38,8 +38,12 @@ spec: ollama pull ornith:35b + ollama pull qwen2.5:3b-instruct + ollama run ornith:35b "ok" >/dev/null 2>&1 || true + ollama run qwen2.5:3b-instruct "ok" >/dev/null 2>&1 || true + wait $SERVE_PID ' @@ -54,11 +58,8 @@ spec: value: '-1' - name: OLLAMA_NUM_PARALLEL value: '1' - # qwen2.5:3b-instruct moved to its own dedicated GPU (llm-serving/grm.yaml) - # so verification/judge traffic no longer contends with ornith:35b's - # planner/implementer traffic on this one -- single model here now. - name: OLLAMA_MAX_LOADED_MODELS - value: '1' + value: '2' image: ollama/ollama:0.32.9@sha256:1685741456770df6e3cceb2a945a5f75e020f658d1701509668d6f4688f1dd3f name: kserve-container ports: @@ -69,7 +70,8 @@ spec: command: - /bin/sh - -c - - ollama ps 2>/dev/null | grep -q ornith + - ollama ps 2>/dev/null | grep -q ornith && ollama ps 2>/dev/null | + grep -q qwen2.5 periodSeconds: 10 resources: limits: @@ -85,7 +87,8 @@ spec: command: - /bin/sh - -c - - ollama ps 2>/dev/null | grep -q ornith + - ollama ps 2>/dev/null | grep -q ornith && ollama ps 2>/dev/null | + grep -q qwen2.5 failureThreshold: 120 periodSeconds: 15 volumeMounts: @@ -93,8 +96,13 @@ spec: name: models deploymentStrategy: type: Recreate - maxReplicas: 1 - minReplicas: 1 + # 2 replicas -- each its own GPU, each loading both ornith:35b and + # qwen2.5:3b-instruct -- so 2 concurrent implementer-style calls each + # get an independent instance instead of contending on one, at the + # cost of judge/qwen traffic still sharing whichever replica an + # implementer call also lands on. + maxReplicas: 2 + minReplicas: 2 nodeSelector: kubernetes.io/hostname: worker-1 runtimeClassName: nvidia