# Action engine — Ornith-1.0-35B on Ollama. # # Why not vLLM like the other two: Ornith is # Qwen3_5MoeForConditionalGeneration (Qwen3.5 MoE, 256 experts / 8 active, # hybrid attention — 30 linear_attention + 10 full_attention layers). vLLM's # Qwen3.5 support landed 2026-07-29, AFTER vLLM dropped Volta (sm_70) at # v0.11.1. No vLLM build has both, so Ornith cannot run on vLLM on a V100. # # Ollama ships `ornith:35b` in its library and runs a llama-server runner # underneath, which keeps Volta support. q4 is ~21GB — fits one 32GB V100 with # room for KV. # # OLLAMA_KEEP_ALIVE=-1 is load-bearing: the harness calls this every loop # iteration, and Ollama's default is to evict an idle model after 5m, which # would add a ~21GB reload to a random future request. apiVersion: serving.kserve.io/v1beta1 kind: InferenceService metadata: name: ornith labels: app.kubernetes.io/name: llm-ornith app.kubernetes.io/part-of: llm-serving spec: predictor: minReplicas: 1 # Recreate, not the default RollingUpdate: GPUs are allocated exactly 4/4, # so a surge pod has no card to claim and sits Pending while the old pod is # never torn down — a deadlock. Recreate tears down first, accepting a brief # gap during updates. deploymentStrategy: type: Recreate maxReplicas: 1 nodeSelector: kubernetes.io/hostname: worker-1 runtimeClassName: nvidia containers: - name: kserve-container image: ollama/ollama:0.32.9@sha256:1685741456770df6e3cceb2a945a5f75e020f658d1701509668d6f4688f1dd3f # `ollama serve` does not pull models, and `ollama pull` needs a running # server — so background the server, wait for it, pull, then hand the # foreground back to serve. command: - /bin/sh - -c - | set -e ollama serve & SERVE_PID=$! until ollama list >/dev/null 2>&1; do sleep 2; done ollama pull ornith:35b ollama pull qwen2.5:3b-instruct wait $SERVE_PID env: # Match the port the other two engines use. - name: OLLAMA_HOST value: "0.0.0.0:8080" - name: OLLAMA_MODELS value: /mnt/models/ollama # Ollama defaults to a 4096 context, far too small for an agentic # coding model. Ornith's hybrid attention means only 10 of its 40 # layers hold a conventional KV cache, so 32K is affordable in the # ~11GiB left after its 21GB of weights. - name: OLLAMA_CONTEXT_LENGTH value: "32768" # Never evict — this model is on the harness's hot path. - name: OLLAMA_KEEP_ALIVE value: "-1" # Serial agent loop; no benefit from parallel slots. - name: OLLAMA_NUM_PARALLEL value: "1" # 2, so ornith and the small utility model stay co-resident on GPU2 # instead of evicting one another on every alternating request. - name: OLLAMA_MAX_LOADED_MODELS value: "2" ports: - containerPort: 8080 protocol: TCP resources: requests: cpu: "8" memory: 8Gi nvidia.com/gpu: "1" limits: cpu: "16" memory: 16Gi nvidia.com/gpu: "1" volumeMounts: - name: models mountPath: /mnt/models # Probes must confirm the MODEL is present, not just that the server # answers. Ollama's `GET /` returns 200 ("Ollama is running") the moment # `ollama serve` binds — which is before the ~21GB pull finishes. An # httpGet probe would therefore mark this pod Ready with no model # loaded, and KServe would route traffic to it. startupProbe: exec: command: ["/bin/sh", "-c", "ollama list 2>/dev/null | grep -q ornith && ollama list 2>/dev/null | grep -q qwen2.5"] periodSeconds: 15 failureThreshold: 120 readinessProbe: exec: command: ["/bin/sh", "-c", "ollama list 2>/dev/null | grep -q ornith && ollama list 2>/dev/null | grep -q qwen2.5"] periodSeconds: 10 volumes: - name: models persistentVolumeClaim: claimName: llm-models