diff --git a/k8s/apps/agent-pod/configmap.yaml b/k8s/apps/agent-pod/configmap.yaml index 174a1ae..4437f48 100644 --- a/k8s/apps/agent-pod/configmap.yaml +++ b/k8s/apps/agent-pod/configmap.yaml @@ -8,8 +8,8 @@ data: "theme": "light", "compaction": { "enabled": true, - "reserveTokens": 8192, - "keepRecentTokens": 12000 + "reserveTokens": 16000, + "keepRecentTokens": 6000 } } kind: ConfigMap diff --git a/k8s/apps/agent-pod/coordinator-configmap.yaml b/k8s/apps/agent-pod/coordinator-configmap.yaml index 9e550a6..3f873ce 100644 --- a/k8s/apps/agent-pod/coordinator-configmap.yaml +++ b/k8s/apps/agent-pod/coordinator-configmap.yaml @@ -39,7 +39,6 @@ data: // with fs.watch instead of polling. const fs = require("node:fs"); const path = require("node:path"); - const crypto = require("node:crypto"); const { spawn } = require("node:child_process"); const WORK_DIR = process.env.HUB_WORK_DIR || path.join(require("node:os").tmpdir(), "agent-harness-work"); @@ -523,6 +522,11 @@ data: // running concurrently (see runCoordinator). async function runRepoPipeline({ repoId, repo, baseBranch, tasks, branchName }, pipelineSession) { const cwd = path.join(WORK_DIR, repoId); + // repoId is a slug derived from the repo URL now (see slugFor), not a + // fresh UUID -- reusable across separate `runCoordinator` invocations + // against the same repo, so a stale clone from a prior run has to be + // wiped before this one starts, not merged into. + fs.rmSync(cwd, { recursive: true, force: true }); fs.mkdirSync(cwd, { recursive: true }); const pool = {}; @@ -678,20 +682,40 @@ data: // How many repos can be mid-flight at once. Each repo gets its own clone // and its own 4-agent pool (planner/investigator/implementer/judge), so // this is now the real concurrency knob -- tasks within one repo are - // already serialized against that repo's pool (see runPhase). Backends - // that queue rather than reject concurrent requests to one loaded model - // (e.g. Ollama) still process this many repos' worth of stages one at a - // time internally even if REPO_CONCURRENCY says otherwise; bump it once - // there's more than one model instance to actually parallelize against. + // already serialized against that repo's pool (see runPhase). The backend + // (homelab-ornith) actually runs 2 GPU replicas behind one Kubernetes + // Service, each with its own copy of the model loaded (see homelab's + // k8s/apps/llm-serving/ornith.yaml) -- so up to 2 concurrent LLM calls get + // real independent instances; a 3rd+ concurrent call queues inside + // whichever replica the Service's own load-balancing lands it on (each + // replica runs OLLAMA_NUM_PARALLEL=1). REPO_CONCURRENCY above 2 is still + // useful (more repos in flight overlaps git/file work, not just LLM calls) + // but past 2 simultaneous LLM calls, extra concurrency mostly means queueing + // rather than added throughput -- bump the backend's replica count to + // change that, not this constant. const REPO_CONCURRENCY = Number(process.env.REPO_CONCURRENCY) || 3; + // repoId is the repo's own name, not a random id -- it's what every role + // session's --name is built from (see runOnPool: `${repoId}-${role}`), so + // agent-manager's own session list groups naturally by repo ("portfolio- + // planner", "portfolio-judge", "poiman-planner", ...) instead of by opaque + // UUID. Takes the last path segment of the URL, strips a trailing `.git`, + // and sanitizes anything that isn't safe in a tmux session name / directory + // name / git branch name. Two different repos that happen to share a + // basename (e.g. two orgs' "portfolio") would collide -- not handled, since + // nothing about this harness's usage has needed more than one org per run. + function slugFor(repoUrl) { + const last = repoUrl.replace(/\/+$/, "").split("/").pop() || repoUrl; + return last.replace(/\.git$/, "").replace(/[^a-zA-Z0-9._-]/g, "-"); + } + // Top-level entry point: runs every repo in `repos` to completion, up to // REPO_CONCURRENCY at a time. Returns a map of repoId -> final // pipelineSession, one per repo, independent of how the others fared. async function runCoordinator({ repos, base, tasks, branchName }) { const sessions = {}; await runConcurrent(repos, REPO_CONCURRENCY, async (repoUrl) => { - const repoId = crypto.randomUUID(); + const repoId = slugFor(repoUrl); const pipelineSession = { id: repoId, repo: repoUrl, diff --git a/k8s/apps/agent-pod/deployment.yaml b/k8s/apps/agent-pod/deployment.yaml index 13a65e2..8a944e4 100644 --- a/k8s/apps/agent-pod/deployment.yaml +++ b/k8s/apps/agent-pod/deployment.yaml @@ -77,6 +77,16 @@ spec: value: /usr/local/bin/agent-manager - name: HUB_WORK_DIR value: /root/agent-harness-work + # planner/investigator/implementer stay on the default + # (homelab-ornith/ornith:35b, pi's settings.json default). Judge + # moves to the separate homelab-reasoning backend (DeepSeek-R1, + # its own 2 GPU replicas) so judge calls stop contending with the + # other 3 roles for the 2 ornith pods -- an entire role's worth + # of traffic moves onto otherwise-idle capacity instead. + - name: JUDGE_PROVIDER + value: homelab-reasoning + - name: JUDGE_MODEL + value: reasoning ports: - containerPort: 9090 resources: