fix(agent-pod): sync coordinator.js (slug repoId), tighten compaction, route judge to reasoning model
This commit is contained in:
@@ -8,8 +8,8 @@ data:
|
|||||||
"theme": "light",
|
"theme": "light",
|
||||||
"compaction": {
|
"compaction": {
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"reserveTokens": 8192,
|
"reserveTokens": 16000,
|
||||||
"keepRecentTokens": 12000
|
"keepRecentTokens": 6000
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
kind: ConfigMap
|
kind: ConfigMap
|
||||||
|
|||||||
@@ -39,7 +39,6 @@ data:
|
|||||||
// with fs.watch instead of polling.
|
// with fs.watch instead of polling.
|
||||||
const fs = require("node:fs");
|
const fs = require("node:fs");
|
||||||
const path = require("node:path");
|
const path = require("node:path");
|
||||||
const crypto = require("node:crypto");
|
|
||||||
const { spawn } = require("node:child_process");
|
const { spawn } = require("node:child_process");
|
||||||
|
|
||||||
const WORK_DIR = process.env.HUB_WORK_DIR || path.join(require("node:os").tmpdir(), "agent-harness-work");
|
const WORK_DIR = process.env.HUB_WORK_DIR || path.join(require("node:os").tmpdir(), "agent-harness-work");
|
||||||
@@ -523,6 +522,11 @@ data:
|
|||||||
// running concurrently (see runCoordinator).
|
// running concurrently (see runCoordinator).
|
||||||
async function runRepoPipeline({ repoId, repo, baseBranch, tasks, branchName }, pipelineSession) {
|
async function runRepoPipeline({ repoId, repo, baseBranch, tasks, branchName }, pipelineSession) {
|
||||||
const cwd = path.join(WORK_DIR, repoId);
|
const cwd = path.join(WORK_DIR, repoId);
|
||||||
|
// repoId is a slug derived from the repo URL now (see slugFor), not a
|
||||||
|
// fresh UUID -- reusable across separate `runCoordinator` invocations
|
||||||
|
// against the same repo, so a stale clone from a prior run has to be
|
||||||
|
// wiped before this one starts, not merged into.
|
||||||
|
fs.rmSync(cwd, { recursive: true, force: true });
|
||||||
fs.mkdirSync(cwd, { recursive: true });
|
fs.mkdirSync(cwd, { recursive: true });
|
||||||
const pool = {};
|
const pool = {};
|
||||||
|
|
||||||
@@ -678,20 +682,40 @@ data:
|
|||||||
// How many repos can be mid-flight at once. Each repo gets its own clone
|
// How many repos can be mid-flight at once. Each repo gets its own clone
|
||||||
// and its own 4-agent pool (planner/investigator/implementer/judge), so
|
// and its own 4-agent pool (planner/investigator/implementer/judge), so
|
||||||
// this is now the real concurrency knob -- tasks within one repo are
|
// this is now the real concurrency knob -- tasks within one repo are
|
||||||
// already serialized against that repo's pool (see runPhase). Backends
|
// already serialized against that repo's pool (see runPhase). The backend
|
||||||
// that queue rather than reject concurrent requests to one loaded model
|
// (homelab-ornith) actually runs 2 GPU replicas behind one Kubernetes
|
||||||
// (e.g. Ollama) still process this many repos' worth of stages one at a
|
// Service, each with its own copy of the model loaded (see homelab's
|
||||||
// time internally even if REPO_CONCURRENCY says otherwise; bump it once
|
// k8s/apps/llm-serving/ornith.yaml) -- so up to 2 concurrent LLM calls get
|
||||||
// there's more than one model instance to actually parallelize against.
|
// real independent instances; a 3rd+ concurrent call queues inside
|
||||||
|
// whichever replica the Service's own load-balancing lands it on (each
|
||||||
|
// replica runs OLLAMA_NUM_PARALLEL=1). REPO_CONCURRENCY above 2 is still
|
||||||
|
// useful (more repos in flight overlaps git/file work, not just LLM calls)
|
||||||
|
// but past 2 simultaneous LLM calls, extra concurrency mostly means queueing
|
||||||
|
// rather than added throughput -- bump the backend's replica count to
|
||||||
|
// change that, not this constant.
|
||||||
const REPO_CONCURRENCY = Number(process.env.REPO_CONCURRENCY) || 3;
|
const REPO_CONCURRENCY = Number(process.env.REPO_CONCURRENCY) || 3;
|
||||||
|
|
||||||
|
// repoId is the repo's own name, not a random id -- it's what every role
|
||||||
|
// session's --name is built from (see runOnPool: `${repoId}-${role}`), so
|
||||||
|
// agent-manager's own session list groups naturally by repo ("portfolio-
|
||||||
|
// planner", "portfolio-judge", "poiman-planner", ...) instead of by opaque
|
||||||
|
// UUID. Takes the last path segment of the URL, strips a trailing `.git`,
|
||||||
|
// and sanitizes anything that isn't safe in a tmux session name / directory
|
||||||
|
// name / git branch name. Two different repos that happen to share a
|
||||||
|
// basename (e.g. two orgs' "portfolio") would collide -- not handled, since
|
||||||
|
// nothing about this harness's usage has needed more than one org per run.
|
||||||
|
function slugFor(repoUrl) {
|
||||||
|
const last = repoUrl.replace(/\/+$/, "").split("/").pop() || repoUrl;
|
||||||
|
return last.replace(/\.git$/, "").replace(/[^a-zA-Z0-9._-]/g, "-");
|
||||||
|
}
|
||||||
|
|
||||||
// Top-level entry point: runs every repo in `repos` to completion, up to
|
// Top-level entry point: runs every repo in `repos` to completion, up to
|
||||||
// REPO_CONCURRENCY at a time. Returns a map of repoId -> final
|
// REPO_CONCURRENCY at a time. Returns a map of repoId -> final
|
||||||
// pipelineSession, one per repo, independent of how the others fared.
|
// pipelineSession, one per repo, independent of how the others fared.
|
||||||
async function runCoordinator({ repos, base, tasks, branchName }) {
|
async function runCoordinator({ repos, base, tasks, branchName }) {
|
||||||
const sessions = {};
|
const sessions = {};
|
||||||
await runConcurrent(repos, REPO_CONCURRENCY, async (repoUrl) => {
|
await runConcurrent(repos, REPO_CONCURRENCY, async (repoUrl) => {
|
||||||
const repoId = crypto.randomUUID();
|
const repoId = slugFor(repoUrl);
|
||||||
const pipelineSession = {
|
const pipelineSession = {
|
||||||
id: repoId,
|
id: repoId,
|
||||||
repo: repoUrl,
|
repo: repoUrl,
|
||||||
|
|||||||
@@ -77,6 +77,16 @@ spec:
|
|||||||
value: /usr/local/bin/agent-manager
|
value: /usr/local/bin/agent-manager
|
||||||
- name: HUB_WORK_DIR
|
- name: HUB_WORK_DIR
|
||||||
value: /root/agent-harness-work
|
value: /root/agent-harness-work
|
||||||
|
# planner/investigator/implementer stay on the default
|
||||||
|
# (homelab-ornith/ornith:35b, pi's settings.json default). Judge
|
||||||
|
# moves to the separate homelab-reasoning backend (DeepSeek-R1,
|
||||||
|
# its own 2 GPU replicas) so judge calls stop contending with the
|
||||||
|
# other 3 roles for the 2 ornith pods -- an entire role's worth
|
||||||
|
# of traffic moves onto otherwise-idle capacity instead.
|
||||||
|
- name: JUDGE_PROVIDER
|
||||||
|
value: homelab-reasoning
|
||||||
|
- name: JUDGE_MODEL
|
||||||
|
value: reasoning
|
||||||
ports:
|
ports:
|
||||||
- containerPort: 9090
|
- containerPort: 9090
|
||||||
resources:
|
resources:
|
||||||
|
|||||||
Reference in New Issue
Block a user