apiVersion: serving.kserve.io/v1beta1 kind: InferenceService metadata: annotations: serving.kserve.io/deploymentMode: RawDeployment # Same Kong-timeout rationale as ornith.yaml: these configure the Service # KServe generates, not the Ingress, and matter once OLLAMA_KEEP_ALIVE=-1 # stops covering a cold load after a pod restart. konghq.com/connect-timeout: "10000" konghq.com/read-timeout: "3600000" konghq.com/write-timeout: "3600000" labels: app.kubernetes.io/name: llm-grm app.kubernetes.io/part-of: llm-serving name: grm namespace: llm-serving spec: predictor: containers: - command: - /bin/sh - -c - 'set -e ollama serve & SERVE_PID=$! until ollama list >/dev/null 2>&1; do sleep 2; done ollama pull qwen2.5:3b-instruct ollama run qwen2.5:3b-instruct "ok" >/dev/null 2>&1 || true wait $SERVE_PID ' env: - name: OLLAMA_HOST value: 0.0.0.0:8080 - name: OLLAMA_MODELS value: /mnt/models/ollama - name: OLLAMA_CONTEXT_LENGTH value: '32768' - name: OLLAMA_KEEP_ALIVE value: '-1' - name: OLLAMA_NUM_PARALLEL value: '1' # Room for a second verification/reward model alongside qwen2.5:3b # without a redeploy -- matches ornith.yaml's pattern, one dedicated # GPU now free for it instead of contending with ornith:35b's. - name: OLLAMA_MAX_LOADED_MODELS value: '2' image: ollama/ollama:0.32.9@sha256:1685741456770df6e3cceb2a945a5f75e020f658d1701509668d6f4688f1dd3f name: kserve-container ports: - containerPort: 8080 protocol: TCP readinessProbe: exec: command: - /bin/sh - -c - ollama ps 2>/dev/null | grep -q qwen2.5 periodSeconds: 10 resources: limits: cpu: '16' memory: 16Gi nvidia.com/gpu: '1' requests: cpu: '8' memory: 8Gi nvidia.com/gpu: '1' startupProbe: exec: command: - /bin/sh - -c - ollama ps 2>/dev/null | grep -q qwen2.5 failureThreshold: 120 periodSeconds: 15 volumeMounts: - mountPath: /mnt/models name: models deploymentStrategy: type: Recreate maxReplicas: 1 minReplicas: 1 nodeSelector: kubernetes.io/hostname: worker-1 runtimeClassName: nvidia volumes: - name: models persistentVolumeClaim: claimName: llm-models