# qwen2.5:3b-instruct on CPU (talos-cp-2, 144GB RAM, 24 cores). # Moved off GPU to free a V100 for ComfyUI. Latency ~10x slower # than GPU but sufficient for lightweight tasks (summarization, # classification, quick answers). apiVersion: apps/v1 kind: Deployment metadata: name: qwen-cpu namespace: llm-serving labels: app: qwen-cpu app.kubernetes.io/name: qwen-cpu app.kubernetes.io/part-of: llm-serving spec: replicas: 1 strategy: type: Recreate selector: matchLabels: app: qwen-cpu template: metadata: labels: app: qwen-cpu app.kubernetes.io/name: qwen-cpu app.kubernetes.io/part-of: llm-serving spec: nodeSelector: kubernetes.io/hostname: talos-cp-2 tolerations: - key: node-role.kubernetes.io/control-plane operator: Exists effect: NoSchedule containers: - name: ollama image: ollama/ollama:0.32.9@sha256:1685741456770df6e3cceb2a945a5f75e020f658d1701509668d6f4688f1dd3f command: ["/bin/sh", "-c"] args: - | ollama serve & SERVE_PID=$! until ollama list >/dev/null 2>&1; do sleep 2; done ollama pull qwen2.5:3b-instruct ollama run qwen2.5:3b-instruct "ok" >/dev/null 2>&1 || true wait $SERVE_PID env: - name: OLLAMA_HOST value: "0.0.0.0:8080" - name: OLLAMA_MODELS value: /root/.ollama/models - name: OLLAMA_CONTEXT_LENGTH value: "32768" - name: OLLAMA_KEEP_ALIVE value: "-1" - name: OLLAMA_MAX_LOADED_MODELS value: "1" - name: OLLAMA_NUM_PARALLEL value: "2" ports: - containerPort: 8080 protocol: TCP readinessProbe: exec: command: ["/bin/sh", "-c", "ollama ps 2>/dev/null | grep -q qwen2.5"] periodSeconds: 10 startupProbe: exec: command: ["/bin/sh", "-c", "ollama ps 2>/dev/null | grep -q qwen2.5"] failureThreshold: 60 periodSeconds: 10 resources: requests: cpu: "4" memory: 4Gi limits: cpu: "8" memory: 8Gi volumeMounts: - mountPath: /root/.ollama name: ollama-data volumes: - name: ollama-data persistentVolumeClaim: claimName: qwen-cpu-data --- apiVersion: v1 kind: Service metadata: name: qwen-cpu namespace: llm-serving labels: app: qwen-cpu app.kubernetes.io/part-of: llm-serving spec: selector: app: qwen-cpu ports: - port: 80 targetPort: 8080 protocol: TCP --- # Small PVC for qwen2.5:3b model weights (~1.9GB). # Separate from llm-models PVC which is pinned to worker-1. apiVersion: v1 kind: PersistentVolumeClaim metadata: name: qwen-cpu-data namespace: llm-serving spec: accessModes: - ReadWriteOnce storageClassName: longhorn resources: requests: storage: 5Gi