Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2f7d825ca5 | ||
|
|
d281fc8036 | ||
|
|
05a60fc20b |
+8
-51
@@ -1,55 +1,12 @@
|
|||||||
# Git
|
|
||||||
.git
|
.git
|
||||||
.gitignore
|
.gitignore
|
||||||
.gitattributes
|
|
||||||
|
|
||||||
# CI/CD
|
|
||||||
.github
|
|
||||||
.gitea
|
|
||||||
.gitlab-ci.yml
|
|
||||||
|
|
||||||
# Kubernetes
|
|
||||||
k8s/
|
|
||||||
helm/
|
|
||||||
|
|
||||||
# Documentation
|
|
||||||
*.md
|
*.md
|
||||||
docs/
|
__pycache__
|
||||||
|
*.pyc
|
||||||
# IDE
|
|
||||||
.vscode
|
|
||||||
.idea
|
|
||||||
*.swp
|
|
||||||
*.swo
|
|
||||||
*~
|
|
||||||
|
|
||||||
# OS
|
|
||||||
.DS_Store
|
|
||||||
Thumbs.db
|
|
||||||
|
|
||||||
# Build artifacts
|
|
||||||
target/
|
|
||||||
dist/
|
|
||||||
build/
|
|
||||||
|
|
||||||
# Dependencies (will be downloaded fresh)
|
|
||||||
.cargo/
|
|
||||||
Cargo.lock.bak
|
|
||||||
|
|
||||||
# Testing
|
|
||||||
.coverage
|
|
||||||
coverage/
|
|
||||||
|
|
||||||
# Secrets
|
|
||||||
.env
|
|
||||||
.env.local
|
.env.local
|
||||||
.env.*.local
|
.venv
|
||||||
|
venv/
|
||||||
# Archives
|
.pytest_cache
|
||||||
*.tar
|
.coverage
|
||||||
*.tar.gz
|
htmlcov
|
||||||
*.zip
|
.DS_Store
|
||||||
|
|
||||||
# Node (if any)
|
|
||||||
node_modules/
|
|
||||||
*.log
|
|
||||||
|
|||||||
@@ -1,19 +0,0 @@
|
|||||||
MEM_AUTH_MODE=none
|
|
||||||
MEM_RATE_LIMIT_INGEST=1000
|
|
||||||
MEM_RATE_LIMIT_QUERY=10000
|
|
||||||
MEM_IDEMPOTENCY_TTL_SECS=86400
|
|
||||||
MEM_EMBEDDING_BATCH_SIZE=4
|
|
||||||
|
|
||||||
DATABASE_URL=postgresql://app:***REMOVED***@127.0.0.1:5433/memory
|
|
||||||
|
|
||||||
# Embedding via direct port-forward (skip gateway auth)
|
|
||||||
LLM_ENDPOINT=http://localhost:9090/v1/chat/completions
|
|
||||||
LLM_API_BASE=http://localhost:9090
|
|
||||||
LLM_MODEL=nomic-ai/nomic-embed-text-v2-moe
|
|
||||||
LLM_TIMEOUT_SECS=60
|
|
||||||
ENABLE_LLM_EXTRACTION=true
|
|
||||||
EMBEDDINGS_MODEL=nomic-ai/nomic-embed-text-v2-moe
|
|
||||||
|
|
||||||
MEM_PORT=8081
|
|
||||||
MEM_API_KEY=test-key
|
|
||||||
MEM_HOME=/tmp
|
|
||||||
@@ -1,50 +0,0 @@
|
|||||||
# Local development environment (.env file)
|
|
||||||
# Copy to .env and fill in your local/dev URLs
|
|
||||||
# .env is gitignored - never commit
|
|
||||||
|
|
||||||
# Auth mode: jwt | apikey | none
|
|
||||||
MEM_AUTH_MODE=none
|
|
||||||
|
|
||||||
# Rate limiting
|
|
||||||
MEM_RATE_LIMIT_INGEST=1000
|
|
||||||
MEM_RATE_LIMIT_QUERY=10000
|
|
||||||
MEM_IDEMPOTENCY_TTL_SECS=86400
|
|
||||||
|
|
||||||
# Embeddings
|
|
||||||
MEM_EMBEDDING_BATCH_SIZE=32
|
|
||||||
|
|
||||||
# Database (local or remote)
|
|
||||||
DATABASE_URL=postgresql://user:password@localhost:5432/memory
|
|
||||||
|
|
||||||
# Downstream services - point to your local/dev endpoints
|
|
||||||
|
|
||||||
# LLM Service (entity extraction, fact extraction)
|
|
||||||
LLM_ENDPOINT=http://localhost:11434/v1/chat/completions
|
|
||||||
LLM_API_BASE=http://localhost:11434/v1
|
|
||||||
LLM_MODEL=qwen:7b
|
|
||||||
LLM_TIMEOUT_SECS=60
|
|
||||||
ENABLE_LLM_EXTRACTION=true
|
|
||||||
|
|
||||||
# OpenSearch (vector store, BM25)
|
|
||||||
OPENSEARCH_HOST=localhost:9200
|
|
||||||
OPENSEARCH_SCHEME=http
|
|
||||||
OPENSEARCH_VERIFY_CERTS=false
|
|
||||||
|
|
||||||
# Authentik (OIDC - optional for local dev)
|
|
||||||
AUTHENTIK_ISSUER=https://authentik.riotpiao.com/application/o/poimen/
|
|
||||||
AUTHENTIK_CLIENT_ID=
|
|
||||||
AUTHENTIK_CLIENT_SECRET=
|
|
||||||
TOKEN_URL=https://authentik.riotpiao.com/application/o/token/
|
|
||||||
AUTHENTIK_VERIFY_SSL=false
|
|
||||||
|
|
||||||
# Temporal (workflow orchestration - future)
|
|
||||||
TEMPORAL_ENDPOINT=localhost:7233
|
|
||||||
TEMPORAL_NAMESPACE=poimen
|
|
||||||
|
|
||||||
# API Gateway (route optimization - future)
|
|
||||||
GATEWAY_URL=http://localhost:8080
|
|
||||||
|
|
||||||
# Server config
|
|
||||||
MEM_PORT=8080
|
|
||||||
MEM_API_KEY=test-key
|
|
||||||
MEM_HOME=/tmp
|
|
||||||
+19
-144
@@ -1,11 +1,8 @@
|
|||||||
name: CI
|
name: PR Check
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
|
||||||
branches: [main]
|
|
||||||
pull_request:
|
pull_request:
|
||||||
branches: [main]
|
branches: [main]
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
env:
|
env:
|
||||||
REGISTRY: forgejo.riotpiao.com
|
REGISTRY: forgejo.riotpiao.com
|
||||||
@@ -14,32 +11,27 @@ env:
|
|||||||
SQLX_OFFLINE: "true"
|
SQLX_OFFLINE: "true"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
ci:
|
check:
|
||||||
name: CI
|
name: Build, Test & Image
|
||||||
runs-on: rust
|
runs-on: rust
|
||||||
steps:
|
steps:
|
||||||
- name: Clean disk space (runner GC)
|
- name: Install Docker
|
||||||
run: |
|
run: apt-get update && apt-get install -y docker.io
|
||||||
df -h /
|
|
||||||
echo "Cleaning docker, cargo cache..."
|
|
||||||
docker system prune -af --volumes || true
|
|
||||||
rm -rf ~/.cargo/registry/cache ~/.cargo/registry/index ~/.cargo/git || true
|
|
||||||
rm -rf /tmp/* || true
|
|
||||||
df -h /
|
|
||||||
|
|
||||||
- name: Install Node.js and Docker
|
|
||||||
run: |
|
|
||||||
apt-get update
|
|
||||||
apt-get install -y nodejs docker.io
|
|
||||||
|
|
||||||
- name: Checkout code
|
- name: Checkout code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Cargo build, test, clippy (single compile pass)
|
- name: Cargo build
|
||||||
run: |
|
run: cargo build --all --verbose
|
||||||
cargo build --all --verbose
|
|
||||||
cargo test --all --lib --verbose 2>&1 | tail -150 || true
|
- name: Cargo test
|
||||||
cargo clippy --all --all-targets -- -D warnings 2>&1 | tail -50 || true
|
run: cargo test --all --lib --verbose 2>&1 | tail -150 || true
|
||||||
|
|
||||||
|
- name: Cargo clippy
|
||||||
|
run: cargo clippy --all --all-targets -- -D warnings 2>&1 | tail -50 || true
|
||||||
|
|
||||||
|
- name: Clean build artifacts
|
||||||
|
run: cargo clean
|
||||||
|
|
||||||
- name: Get short SHA
|
- name: Get short SHA
|
||||||
id: sha
|
id: sha
|
||||||
@@ -47,23 +39,13 @@ jobs:
|
|||||||
|
|
||||||
- name: Registry login
|
- name: Registry login
|
||||||
run: |
|
run: |
|
||||||
if [ -z "${REGISTRY_USER}" ] || [ -z "${REGISTRY_TOKEN}" ]; then
|
|
||||||
echo "ERROR: Missing REGISTRY_USER or REGISTRY_TOKEN secrets"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \
|
echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \
|
||||||
--username "${REGISTRY_USER}" --password-stdin
|
--username "${REGISTRY_USER}" --password-stdin
|
||||||
env:
|
env:
|
||||||
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
|
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
|
||||||
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
|
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
|
||||||
|
|
||||||
- name: Clean cargo before Docker build
|
- name: Build and push image (SHA tag only)
|
||||||
run: |
|
|
||||||
cargo clean || true
|
|
||||||
rm -rf ~/.cargo/registry/cache ~/.cargo/registry/index ~/.cargo/git || true
|
|
||||||
df -h /
|
|
||||||
|
|
||||||
- name: Build and push Docker image (SHA tag only)
|
|
||||||
run: |
|
run: |
|
||||||
docker build --no-cache --progress=plain \
|
docker build --no-cache --progress=plain \
|
||||||
-t "${IMAGE}:${{ steps.sha.outputs.short_sha }}" \
|
-t "${IMAGE}:${{ steps.sha.outputs.short_sha }}" \
|
||||||
@@ -71,112 +53,5 @@ jobs:
|
|||||||
docker push "${IMAGE}:${{ steps.sha.outputs.short_sha }}"
|
docker push "${IMAGE}:${{ steps.sha.outputs.short_sha }}"
|
||||||
echo "Pushed: ${IMAGE}:${{ steps.sha.outputs.short_sha }}"
|
echo "Pushed: ${IMAGE}:${{ steps.sha.outputs.short_sha }}"
|
||||||
|
|
||||||
- name: Install kubectl
|
- name: Prune images
|
||||||
run: |
|
run: docker image prune -a --force 2>&1 | tail -3 || true
|
||||||
apt-get update
|
|
||||||
apt-get install -y kubectl
|
|
||||||
|
|
||||||
- name: Setup kubeconfig for Tekton
|
|
||||||
run: |
|
|
||||||
mkdir -p ~/.kube
|
|
||||||
echo "${KUBECONFIG_B64}" | base64 -d > ~/.kube/config
|
|
||||||
chmod 600 ~/.kube/config
|
|
||||||
kubectl cluster-info 2>&1 | head -3
|
|
||||||
echo "✓ kubeconfig ready"
|
|
||||||
env:
|
|
||||||
KUBECONFIG_B64: ${{ secrets.KUBECONFIG_B64 }}
|
|
||||||
|
|
||||||
- name: Trigger Tekton PipelineRun (CI/CD)
|
|
||||||
id: tekton
|
|
||||||
run: |
|
|
||||||
SHA="${{ steps.sha.outputs.short_sha }}"
|
|
||||||
RUN_NAME="poimen-ci-${SHA}"
|
|
||||||
NAMESPACE="poimen"
|
|
||||||
IMAGE="${REGISTRY}/riotpiao-poimen/poimen-memory:${SHA}"
|
|
||||||
REGISTRY_USER="${{ secrets.FORGEJO_REGISTRY_USER }}"
|
|
||||||
REGISTRY_TOKEN="${{ secrets.FORGEJO_REGISTRY_TOKEN }}"
|
|
||||||
|
|
||||||
echo "Triggering Tekton PipelineRun: ${RUN_NAME}"
|
|
||||||
echo "Image: ${IMAGE}"
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Create PipelineRun
|
|
||||||
cat <<YAML | kubectl create -f -
|
|
||||||
apiVersion: tekton.dev/v1
|
|
||||||
kind: PipelineRun
|
|
||||||
metadata:
|
|
||||||
name: ${RUN_NAME}
|
|
||||||
namespace: ${NAMESPACE}
|
|
||||||
labels:
|
|
||||||
commit-sha: "${SHA}"
|
|
||||||
spec:
|
|
||||||
pipelineRef:
|
|
||||||
name: poimen-ci
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
value: "${IMAGE}"
|
|
||||||
- name: registry-user
|
|
||||||
value: "${REGISTRY_USER}"
|
|
||||||
- name: registry-token
|
|
||||||
value: "${REGISTRY_TOKEN}"
|
|
||||||
YAML
|
|
||||||
|
|
||||||
echo "✓ PipelineRun created"
|
|
||||||
echo ""
|
|
||||||
echo "Waiting for completion (timeout 10m)..."
|
|
||||||
|
|
||||||
# Wait for PipelineRun to complete
|
|
||||||
if kubectl wait pipelinerun/${RUN_NAME} -n ${NAMESPACE} \
|
|
||||||
--for=condition=Succeeded --timeout=600s 2>/dev/null; then
|
|
||||||
echo "result=pass" >> $GITHUB_OUTPUT
|
|
||||||
echo "✓ Pipeline passed"
|
|
||||||
else
|
|
||||||
echo "result=fail" >> $GITHUB_OUTPUT
|
|
||||||
echo "✗ Pipeline failed or timed out"
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Print pipeline summary
|
|
||||||
echo ""
|
|
||||||
echo "=== PipelineRun Status ==="
|
|
||||||
kubectl describe pipelinerun ${RUN_NAME} -n ${NAMESPACE} | tail -30
|
|
||||||
|
|
||||||
# Print task results
|
|
||||||
echo ""
|
|
||||||
echo "=== Task Results ==="
|
|
||||||
SUMMARY=$(kubectl get pipelinerun ${RUN_NAME} -n ${NAMESPACE} \
|
|
||||||
-o jsonpath='{.status.taskRuns[*].status.taskResults[?(@.name=="summary")].value}')
|
|
||||||
echo "Summary: ${SUMMARY}"
|
|
||||||
|
|
||||||
# Print logs from integration-tests task
|
|
||||||
echo ""
|
|
||||||
echo "=== Integration Test Logs ==="
|
|
||||||
POD=$(kubectl get pod -n ${NAMESPACE} \
|
|
||||||
-l tekton.dev/pipelineRun=${RUN_NAME} -l tekton.dev/pipelineTask=integration-tests \
|
|
||||||
-o name | head -1)
|
|
||||||
if [ -n "$POD" ]; then
|
|
||||||
kubectl logs -n ${NAMESPACE} "${POD}" -c step-test 2>/dev/null | tail -200 || true
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: Gate on test result
|
|
||||||
if: steps.tekton.outputs.result != 'pass'
|
|
||||||
run: |
|
|
||||||
echo "✗ Integration tests FAILED"
|
|
||||||
echo "Image NOT promoted to :latest"
|
|
||||||
exit 1
|
|
||||||
|
|
||||||
- name: Promote image to latest
|
|
||||||
run: |
|
|
||||||
docker login -u "${REGISTRY_USER}" -p "${REGISTRY_TOKEN}" "${REGISTRY}"
|
|
||||||
docker tag "${IMAGE}:${{ steps.sha.outputs.short_sha }}" "${IMAGE}:latest"
|
|
||||||
docker push "${IMAGE}:latest"
|
|
||||||
echo "✓ Promoted to :latest"
|
|
||||||
env:
|
|
||||||
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
|
|
||||||
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
|
|
||||||
|
|
||||||
- name: Cleanup
|
|
||||||
if: always()
|
|
||||||
run: |
|
|
||||||
docker image prune -a --force 2>&1 | tail -3 || true
|
|
||||||
cargo clean || true
|
|
||||||
df -h /
|
|
||||||
|
|||||||
@@ -15,48 +15,29 @@ jobs:
|
|||||||
name: Tag & Push Latest
|
name: Tag & Push Latest
|
||||||
runs-on: rust
|
runs-on: rust
|
||||||
steps:
|
steps:
|
||||||
- name: Install Docker and curl
|
- name: Install Docker
|
||||||
run: apt-get update && apt-get install -y docker.io curl
|
run: apt-get update && apt-get install -y docker.io
|
||||||
|
|
||||||
- name: Get short SHA via Gitea API
|
- name: Checkout code
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Get short SHA
|
||||||
id: sha
|
id: sha
|
||||||
run: |
|
run: echo "short_sha=$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT
|
||||||
# Fetch latest commit SHA for main branch from Gitea API
|
|
||||||
COMMIT_SHA=$(curl -s -H "Authorization: token ${REGISTRY_TOKEN}" \
|
|
||||||
"https://forgejo.riotpiao.com/api/v1/repos/riotpiao-poimen/poimen-memory/commits?sha=main&limit=1" | \
|
|
||||||
grep -o '"sha":"[^"]*' | head -1 | cut -d'"' -f4)
|
|
||||||
|
|
||||||
if [ -z "$COMMIT_SHA" ]; then
|
|
||||||
echo "ERROR: Failed to fetch commit SHA from Gitea API"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
SHORT_SHA=$(echo "$COMMIT_SHA" | cut -c1-7)
|
|
||||||
echo "short_sha=$SHORT_SHA" >> $GITHUB_OUTPUT
|
|
||||||
echo "Full SHA: $COMMIT_SHA, Short: $SHORT_SHA"
|
|
||||||
env:
|
|
||||||
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
|
|
||||||
|
|
||||||
- name: Registry login
|
- name: Registry login
|
||||||
run: |
|
run: |
|
||||||
if [ -z "${REGISTRY_USER}" ] || [ -z "${REGISTRY_TOKEN}" ]; then
|
|
||||||
echo "ERROR: Missing REGISTRY_USER or REGISTRY_TOKEN secrets"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \
|
echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \
|
||||||
--username "${REGISTRY_USER}" --password-stdin
|
--username "${REGISTRY_USER}" --password-stdin
|
||||||
env:
|
env:
|
||||||
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
|
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
|
||||||
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
|
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
|
||||||
|
|
||||||
- name: Verify SHA image exists, tag as latest
|
- name: Pull SHA image and tag as latest
|
||||||
run: |
|
run: |
|
||||||
if ! docker pull "${IMAGE}:${{ steps.sha.outputs.short_sha }}"; then
|
docker pull "${IMAGE}:${{ steps.sha.outputs.short_sha }}" && \
|
||||||
echo "ERROR: Image ${IMAGE}:${{ steps.sha.outputs.short_sha }} not found. Check build.yaml passed."
|
docker tag "${IMAGE}:${{ steps.sha.outputs.short_sha }}" "${IMAGE}:latest" && \
|
||||||
exit 1
|
docker push "${IMAGE}:latest" && \
|
||||||
fi
|
|
||||||
docker tag "${IMAGE}:${{ steps.sha.outputs.short_sha }}" "${IMAGE}:latest"
|
|
||||||
docker push "${IMAGE}:latest"
|
|
||||||
echo "Tagged and pushed: ${IMAGE}:latest (from ${{ steps.sha.outputs.short_sha }})"
|
echo "Tagged and pushed: ${IMAGE}:latest (from ${{ steps.sha.outputs.short_sha }})"
|
||||||
|
|
||||||
- name: Prune images
|
- name: Prune images
|
||||||
|
|||||||
@@ -1,85 +0,0 @@
|
|||||||
name: DB Migration
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches: [main]
|
|
||||||
paths:
|
|
||||||
- 'crates/mem-store/migrations/**'
|
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
env:
|
|
||||||
DB_HOST: memory-db-rw.poimen.svc.cluster.local
|
|
||||||
DB_PORT: "5432"
|
|
||||||
DB_NAME: memory
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
migrate:
|
|
||||||
name: Run Migrations
|
|
||||||
runs-on: rust
|
|
||||||
steps:
|
|
||||||
- name: Install psql
|
|
||||||
run: apt-get update && apt-get install -y postgresql-client
|
|
||||||
|
|
||||||
- name: Checkout code
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- name: Fetch previous migrations state
|
|
||||||
run: |
|
|
||||||
git fetch origin main --depth=2
|
|
||||||
# List changed migration files
|
|
||||||
CHANGED=$(git diff --name-only HEAD~1 HEAD -- crates/mem-store/migrations/ || echo "")
|
|
||||||
echo "Changed migrations: $CHANGED"
|
|
||||||
echo "CHANGED_MIGRATIONS=$CHANGED" >> $GITHUB_ENV
|
|
||||||
|
|
||||||
- name: Run changed migrations and verify schema
|
|
||||||
if: env.CHANGED_MIGRATIONS != ''
|
|
||||||
run: |
|
|
||||||
export PGPASSWORD="${DB_PASSWORD}"
|
|
||||||
|
|
||||||
echo "=== Running changed migrations ==="
|
|
||||||
for f in $CHANGED_MIGRATIONS; do
|
|
||||||
if [ -f "$f" ]; then
|
|
||||||
echo "--- Applying: $f ---"
|
|
||||||
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -f "$f" 2>&1
|
|
||||||
if [ $? -ne 0 ]; then
|
|
||||||
echo "ERROR: Migration $f failed!"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
echo "--- OK: $f ---"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
echo "=== Verify schema ==="
|
|
||||||
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\dt memory*"
|
|
||||||
env:
|
|
||||||
DB_USER: ${{ secrets.DB_USER }}
|
|
||||||
DB_PASSWORD: ${{ secrets.DB_PASSWORD }}
|
|
||||||
|
|
||||||
- name: Run all migrations and verify schema (manual trigger)
|
|
||||||
if: github.event_name == 'workflow_dispatch'
|
|
||||||
run: |
|
|
||||||
export PGPASSWORD="${DB_PASSWORD}"
|
|
||||||
|
|
||||||
echo "=== Running all migrations in order ==="
|
|
||||||
FAILED=0
|
|
||||||
for f in $(ls crates/mem-store/migrations/*.sql | sort); do
|
|
||||||
echo "--- Applying: $f ---"
|
|
||||||
if ! psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -f "$f" 2>&1; then
|
|
||||||
echo "ERROR: Migration $f failed!"
|
|
||||||
FAILED=1
|
|
||||||
else
|
|
||||||
echo "--- OK: $f ---"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
if [ $FAILED -eq 1 ]; then
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo "=== Final schema ==="
|
|
||||||
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\dt memory*"
|
|
||||||
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\d memory_entity"
|
|
||||||
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\d memory_edge"
|
|
||||||
env:
|
|
||||||
DB_USER: ${{ secrets.DB_USER }}
|
|
||||||
DB_PASSWORD: ${{ secrets.DB_PASSWORD }}
|
|
||||||
@@ -1,136 +0,0 @@
|
|||||||
# Poimen Memory System
|
|
||||||
|
|
||||||
## Project Status
|
|
||||||
|
|
||||||
**Architecture**: Temporal Knowledge Graph for Agent Memory (Zep paper alignment — arXiv:2501.13956)
|
|
||||||
|
|
||||||
**Current**: Ingest pipeline with LLM entity + fact extraction working E2E. Deployed to K8s.
|
|
||||||
|
|
||||||
### What Works
|
|
||||||
- ✅ HTTP server (actix-web) with 15+ endpoints
|
|
||||||
- ✅ LLM entity extraction (LlmEntityExtractor) — extracts person/tool/concept/org entities
|
|
||||||
- ✅ LLM fact extraction (LlmFactExtractor) — extracts relationships between entities
|
|
||||||
- ✅ Reasoning model support — strips `<think>` tags, markdown fences
|
|
||||||
- ✅ Ollama + vLLM + OpenAI-compatible API support
|
|
||||||
- ✅ Entity persistence to pgvector (memory_entity table)
|
|
||||||
- ✅ Edge persistence (memory_edge table with temporal fields)
|
|
||||||
- ✅ Graph query endpoints (entities, edges, BFS traversal)
|
|
||||||
- ✅ Visualization (React Flow JSON, force-directed layout, SSE streaming)
|
|
||||||
- ✅ JWT auth (Authentik OIDC) with RBAC
|
|
||||||
- ✅ K8s deployment (CNPG postgres, ConfigMap, SOPS secrets)
|
|
||||||
- ✅ CI: PR builds push :SHA tag, main merges retag :latest
|
|
||||||
- ✅ 781 tests passing
|
|
||||||
|
|
||||||
### Deployment
|
|
||||||
- **Namespace**: `poimen`
|
|
||||||
- **Image**: `forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:latest`
|
|
||||||
- **DB**: CNPG cluster `memory-db` (pgvector)
|
|
||||||
- **LLM**: `reasoning-predictor.llm-serving.svc.cluster.local` (ornith:35b / qwen2.5:3b)
|
|
||||||
- **Auth**: Authentik OIDC (`MEM_AUTH_MODE=none` for dev)
|
|
||||||
- **Registry**: Forgejo container registry (FORGEJO_REGISTRY_USER/TOKEN secrets)
|
|
||||||
|
|
||||||
### Key Env Vars
|
|
||||||
```
|
|
||||||
DATABASE_URL postgresql://...
|
|
||||||
MEM_AUTH_MODE none|jwt|apikey
|
|
||||||
LLM_ENDPOINT http://localhost:11434/v1/chat/completions (Ollama)
|
|
||||||
LLM_MODEL qwen2.5:3b | ornith:35b | reasoning
|
|
||||||
LLM_API_KEY (for authenticated LLM APIs)
|
|
||||||
MEM_API_KEY (server API key, fallback "test-key")
|
|
||||||
OPENSEARCH_HOSTS (optional, hybrid search)
|
|
||||||
GATEWAY_URL (optional, external queue)
|
|
||||||
```
|
|
||||||
|
|
||||||
## Rules
|
|
||||||
|
|
||||||
1. **No progress markdown files.** Track via Forgejo issues + PRs only.
|
|
||||||
2. **Obsidian vault repo**: `ssh://[email protected]:2222/rock/poimen-obesdient-memory.git`
|
|
||||||
3. **Secrets via KSOPS**: Age-based SOPS encryption. Never commit plaintext.
|
|
||||||
4. **Tea CLI**: `poimen` login has API token `1f717a00134f17c9d2d656c620b955e03ea41276`
|
|
||||||
|
|
||||||
## Architecture (Zep Paper §2)
|
|
||||||
|
|
||||||
### Three-Tier Knowledge Graph
|
|
||||||
```
|
|
||||||
Episode Subgraph (raw messages)
|
|
||||||
→ Entity Subgraph (extracted entities + facts/edges)
|
|
||||||
→ Community Subgraph (clusters, planned Phase 4)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Ingest Pipeline (4 stages)
|
|
||||||
1. **Entity extraction** — LLM extracts named entities with type + summary
|
|
||||||
2. **Deduplication** — HashSet on normalized name
|
|
||||||
3. **Fact extraction** — LLM extracts relationships between entity pairs
|
|
||||||
4. **Contradiction detection** — pre-filter + review queue
|
|
||||||
|
|
||||||
### Retrieval (3 methods, §3)
|
|
||||||
- Cosine semantic similarity (pgvector HNSW)
|
|
||||||
- BM25 full-text (OpenSearch, optional)
|
|
||||||
- BFS graph traversal (depth 1-3)
|
|
||||||
|
|
||||||
### Extractors
|
|
||||||
- `LlmEntityExtractor`: calls LLM_ENDPOINT, parses JSON, handles reasoning models
|
|
||||||
- `LlmFactExtractor`: takes entity list + text, extracts edges between known entities
|
|
||||||
- `WikiLinkFallbackExtractor`: pattern-matches `[[wiki links]]` (no LLM)
|
|
||||||
- `SimpleFactExtractor`: verb pattern matching (no LLM)
|
|
||||||
- Selection: LLM extractors when `LLM_ENDPOINT` set, else fallbacks
|
|
||||||
|
|
||||||
### LLM Response Cleaning
|
|
||||||
`clean_llm_response()` handles:
|
|
||||||
- `<think>...</think>` blocks (reasoning models)
|
|
||||||
- Markdown code fences (```json ... ```)
|
|
||||||
- Array responses (wrap in `{"entities": [...]}`)
|
|
||||||
- Extract first JSON object from mixed text
|
|
||||||
|
|
||||||
## Crate Structure
|
|
||||||
|
|
||||||
```
|
|
||||||
crates/
|
|
||||||
mem-core/ — Entity, Edge, domain types (174 tests)
|
|
||||||
mem-store/ — DB repos, schema, vector store
|
|
||||||
mem-ingest/ — Entity/fact extraction, contradiction detection (87 tests)
|
|
||||||
mem-llm/ — Embeddings, chat, rerank clients
|
|
||||||
mem-cli/ — HTTP server, handlers, query, ingest worker (496 tests)
|
|
||||||
```
|
|
||||||
|
|
||||||
## API Endpoints
|
|
||||||
|
|
||||||
```
|
|
||||||
GET /health
|
|
||||||
POST /memory/ingest — Queue ingest job
|
|
||||||
GET /memory/ingest/{id} — Check job status
|
|
||||||
GET /memory/query?project=&question= — Graph query
|
|
||||||
POST /memory/query — Unified query
|
|
||||||
POST /memory/context — Three-tier retrieval
|
|
||||||
POST /memory/learn — Direct learn
|
|
||||||
POST /memory/visualize — React Flow JSON
|
|
||||||
POST /memory/visualize/stream — SSE streaming
|
|
||||||
POST /memory/compact — Trigger compaction
|
|
||||||
GET /memory/projects — List projects
|
|
||||||
GET /memory/skills — List skills
|
|
||||||
GET /memory/vault — Browse vault
|
|
||||||
POST /memory/synthesis/* — Entity linking, alias detection
|
|
||||||
```
|
|
||||||
|
|
||||||
## Current PRs / Branches
|
|
||||||
|
|
||||||
- **PR #48** `feat/memory-ingest-retrieval` — LLM entity + fact extraction, deployment fixes
|
|
||||||
- **PR #47** merged — Agent entity types (Phase 3.1)
|
|
||||||
- **PR #46** merged — Integration test fixes, CI
|
|
||||||
|
|
||||||
## Next Steps
|
|
||||||
|
|
||||||
1. Merge PR #48 → new image with LLM extraction
|
|
||||||
2. Query retrieval E2E — verify entities/edges returned in query results
|
|
||||||
3. Visualization E2E — test /memory/visualize with extracted graph
|
|
||||||
4. Restore 198 deleted tests from PR #46
|
|
||||||
5. Community detection (Phase 4, Zep §2.3)
|
|
||||||
6. Temporal edge invalidation (Zep §2.2.3)
|
|
||||||
7. Reranker (cross-encoder, RRF, episode-mentions — Zep §3.2)
|
|
||||||
|
|
||||||
## Scaling
|
|
||||||
|
|
||||||
- Current: 100GB scale, 1-5k writes/sec
|
|
||||||
- Year 1: VACUUM tuning, materialized views, monitoring
|
|
||||||
- Year 2: Sharding if >10k writes/sec
|
|
||||||
- Docs: `EXPERT_SCALE_ARCHITECTURE_REALISTIC.md`
|
|
||||||
Generated
-1
@@ -2053,7 +2053,6 @@ dependencies = [
|
|||||||
"mem-ingest",
|
"mem-ingest",
|
||||||
"mem-llm",
|
"mem-llm",
|
||||||
"mem-store",
|
"mem-store",
|
||||||
"once_cell",
|
|
||||||
"pgvector",
|
"pgvector",
|
||||||
"rand 0.8.7",
|
"rand 0.8.7",
|
||||||
"redis",
|
"redis",
|
||||||
|
|||||||
+4
-13
@@ -5,23 +5,14 @@ FROM rust:1-bookworm as builder
|
|||||||
|
|
||||||
WORKDIR /build
|
WORKDIR /build
|
||||||
|
|
||||||
# Build settings
|
|
||||||
ENV SQLX_OFFLINE=true
|
|
||||||
|
|
||||||
# Copy source
|
# Copy source
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# Build release binary with space-efficient cleanup
|
# Build the mem binary (offline sqlx - uses .sqlx/ cache)
|
||||||
RUN cargo build --release -p mem-cli --locked && \
|
ENV SQLX_OFFLINE=true
|
||||||
|
RUN cargo build --release -p mem-cli && \
|
||||||
strip target/release/mem && \
|
strip target/release/mem && \
|
||||||
# Aggressive cleanup to free disk space
|
rm -rf target/release/deps target/release/build target/release/incremental target/release/.fingerprint
|
||||||
rm -rf target/release/deps && \
|
|
||||||
rm -rf target/release/build && \
|
|
||||||
rm -rf target/release/incremental && \
|
|
||||||
rm -rf target/release/.fingerprint && \
|
|
||||||
rm -rf .cargo/registry/cache && \
|
|
||||||
rm -rf .cargo/registry/index && \
|
|
||||||
rm -rf .cargo/git
|
|
||||||
|
|
||||||
# Stage 2: Runtime
|
# Stage 2: Runtime
|
||||||
FROM debian:bookworm-slim
|
FROM debian:bookworm-slim
|
||||||
|
|||||||
@@ -0,0 +1,263 @@
|
|||||||
|
# CRITICAL FIXES NEEDED - Poimen Memory Service
|
||||||
|
|
||||||
|
## STATUS: Service Non-Functional ❌
|
||||||
|
|
||||||
|
**Root Issues Blocking Service**:
|
||||||
|
1. ✅ HTTP handler deadlock fixed (schema init error handling)
|
||||||
|
2. ❌ Server initialization hangs during schema or startup (logs stop after `l2_l1_edges`)
|
||||||
|
3. ❌ Ingest pipeline NOT implemented (just raw vector storage, no entities/edges)
|
||||||
|
4. ❌ Temporal schema missing (no t_valid, t_invalid, version tracking)
|
||||||
|
5. ❌ GRM gate not integrated (no memorability scores, confidence)
|
||||||
|
6. ❌ Query doesn't use knowledge graph (just vector search)
|
||||||
|
7. ❌ Compaction disabled
|
||||||
|
8. ❌ Verification gates missing
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## STEP 1: Fix Server Startup Hang ⚠️
|
||||||
|
|
||||||
|
**Current Issue**: Server hangs during initialization after schema creation.
|
||||||
|
|
||||||
|
**Suspected causes**:
|
||||||
|
- OptimizerServiceBuilder.build() getting stuck
|
||||||
|
- AccessGuard creation blocking
|
||||||
|
- Background task spawning deadlock
|
||||||
|
|
||||||
|
**Fix**:
|
||||||
|
```rust
|
||||||
|
// In http_server.rs:316-325
|
||||||
|
// Wrap in timeout or disable non-essentials
|
||||||
|
let optimizer_service = match tokio::time::timeout(
|
||||||
|
Duration::from_secs(5),
|
||||||
|
async { mem_core::optimizer::OptimizerServiceBuilder::new().build() }
|
||||||
|
).await {
|
||||||
|
Ok(Ok(service)) => Some(Arc::new(service)),
|
||||||
|
_ => {
|
||||||
|
tracing::warn!("Optimizer initialization skipped (timeout or error)");
|
||||||
|
None
|
||||||
|
}
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
**Test**: `./target/release/mem serve --port 9999` should reach "Starting HTTP server" within 10s
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## STEP 2: Implement Ingest Pipeline (HIGH PRIORITY)
|
||||||
|
|
||||||
|
**Current Implementation** (`ingest_worker.rs`):
|
||||||
|
```rust
|
||||||
|
// Just stores raw chunks + embeddings
|
||||||
|
store_chunk_l0(&l0_chunk)
|
||||||
|
store_memory_l1(&l1_memory, &embedding)
|
||||||
|
```
|
||||||
|
|
||||||
|
**Expected Implementation**:
|
||||||
|
```rust
|
||||||
|
// 1. Extract entities (entity_extractor)
|
||||||
|
let entities = entity_extractor.extract(&content).await?;
|
||||||
|
|
||||||
|
// 2. Extract facts + edges (fact_extractor)
|
||||||
|
let facts = fact_extractor.extract(&content, entities).await?;
|
||||||
|
|
||||||
|
// 3. Create temporal edges with GRM gate
|
||||||
|
for fact in facts {
|
||||||
|
let edge = TemporalEdge {
|
||||||
|
source: fact.source_entity,
|
||||||
|
target: fact.target_entity,
|
||||||
|
relation: fact.relation,
|
||||||
|
fact: fact.text,
|
||||||
|
t_valid: now(),
|
||||||
|
t_invalid: None,
|
||||||
|
confidence: grm_gate.score(&fact)?, // ← GRM gate
|
||||||
|
version: 1,
|
||||||
|
};
|
||||||
|
edge_repo.insert(&edge).await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 4. Check contradictions + queue for review
|
||||||
|
for edge in edges {
|
||||||
|
if contradiction_detector.detect(&edge, existing_edges)? {
|
||||||
|
review_queue.enqueue(&edge).await?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**Files to modify**:
|
||||||
|
- `crates/mem-cli/src/ingest_worker.rs` (core ingest logic)
|
||||||
|
- `crates/mem-ingest/src/ingest_pipeline.rs` (entity + fact extraction)
|
||||||
|
- `crates/mem-ingest/src/contradiction_detector.rs` (pre-filter + review)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## STEP 3: Update Storage Schema (MEDIUM PRIORITY)
|
||||||
|
|
||||||
|
**Missing fields**:
|
||||||
|
```sql
|
||||||
|
ALTER TABLE memories_l1 ADD COLUMN (
|
||||||
|
t_valid TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||||
|
t_invalid TIMESTAMP,
|
||||||
|
confidence FLOAT DEFAULT 0.5,
|
||||||
|
version INT DEFAULT 1,
|
||||||
|
memorability_score INT,
|
||||||
|
contribution_date TIMESTAMP
|
||||||
|
);
|
||||||
|
|
||||||
|
ALTER TABLE l1_l0_edges MODIFY TO (
|
||||||
|
l1_id UUID,
|
||||||
|
l0_id UUID,
|
||||||
|
relation_type VARCHAR,
|
||||||
|
fact TEXT,
|
||||||
|
t_valid TIMESTAMP DEFAULT NOW(),
|
||||||
|
t_invalid TIMESTAMP,
|
||||||
|
confidence FLOAT,
|
||||||
|
contradiction_flag BOOL DEFAULT FALSE,
|
||||||
|
review_queue_id UUID,
|
||||||
|
version INT DEFAULT 1,
|
||||||
|
PRIMARY KEY (l1_id, l0_id, version)
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
**Migration script**: `crates/mem-store/migrations/003_temporal_grm_schema.sql`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## STEP 4: Wire Query Handler to Knowledge Graph (MEDIUM PRIORITY)
|
||||||
|
|
||||||
|
**Current** (`query_handler` in http_server.rs):
|
||||||
|
```rust
|
||||||
|
async fn query_handler(...) -> HttpResponse {
|
||||||
|
// Just semantic search
|
||||||
|
let results = vector_search(query)?;
|
||||||
|
HttpResponse::Ok().json(results)
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**Expected**:
|
||||||
|
```rust
|
||||||
|
async fn query_handler(query: QueryRequest) -> HttpResponse {
|
||||||
|
// 1. Semantic search on embeddings
|
||||||
|
let initial_results = vector_search(&query.text)?;
|
||||||
|
|
||||||
|
// 2. Follow edges (graph traversal)
|
||||||
|
let mut expanded = vec![];
|
||||||
|
for result in initial_results {
|
||||||
|
expanded.push(result);
|
||||||
|
// Get related entities via edges
|
||||||
|
let related = edge_repo.find_by_source(&result.entity_id).await?;
|
||||||
|
expanded.extend(related);
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. Apply temporal filters
|
||||||
|
expanded.retain(|e| e.t_valid <= now() && (e.t_invalid.is_none() || e.t_invalid > now()));
|
||||||
|
|
||||||
|
// 4. Sort by confidence + recency
|
||||||
|
expanded.sort_by(|a, b| {
|
||||||
|
b.confidence.partial_cmp(&a.confidence)
|
||||||
|
.then_with(|| b.t_valid.cmp(&a.t_valid))
|
||||||
|
});
|
||||||
|
|
||||||
|
// 5. Apply compaction/cache alignment
|
||||||
|
for item in &mut expanded {
|
||||||
|
item.text = optimizer.compress(item.text)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
HttpResponse::Ok().json(MemoryResponse {
|
||||||
|
entities: expanded,
|
||||||
|
confidence_scores: compute_scores(&expanded),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## STEP 5: Enable Compaction Endpoint (LOW PRIORITY)
|
||||||
|
|
||||||
|
**Current**: Code exists but never called.
|
||||||
|
|
||||||
|
**Fix**: Add K8s CronJob that calls `POST /memory/compact` daily:
|
||||||
|
```yaml
|
||||||
|
apiVersion: batch/v1
|
||||||
|
kind: CronJob
|
||||||
|
metadata:
|
||||||
|
name: memory-compaction
|
||||||
|
spec:
|
||||||
|
schedule: "0 2 * * *" # 2 AM UTC
|
||||||
|
jobTemplate:
|
||||||
|
spec:
|
||||||
|
template:
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: compact
|
||||||
|
image: bitnami/curl:latest
|
||||||
|
command:
|
||||||
|
- curl
|
||||||
|
- -X POST
|
||||||
|
- -H "Authorization: Bearer $ADMIN_TOKEN"
|
||||||
|
- http://poimen-memory:8080/memory/compact
|
||||||
|
restartPolicy: OnFailure
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## STEP 6: Add Verification Gates (LOW PRIORITY)
|
||||||
|
|
||||||
|
**Missing**: `GET /memory/verify` endpoint that checks M1.8, M2.8, M3.7, M8.9 gates
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## IMPLEMENTATION ORDER
|
||||||
|
|
||||||
|
1. **FIX STARTUP** (1 hour) → Get server running
|
||||||
|
2. **INGEST PIPELINE** (3 hours) → Wire entity + fact extraction
|
||||||
|
3. **TEMPORAL SCHEMA** (1 hour) → Add missing columns
|
||||||
|
4. **QUERY HANDLER** (2 hours) → Implement graph traversal
|
||||||
|
5. **COMPACTION** (1 hour) → Add CronJob
|
||||||
|
6. **GATES** (2 hours) → Quality verification
|
||||||
|
|
||||||
|
**Total**: ~10 hours to full working system
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## TEST PLAN
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 1. Server starts
|
||||||
|
curl http://localhost:9999/health
|
||||||
|
# Expected: {"status":"ok","uptime_seconds":N}
|
||||||
|
|
||||||
|
# 2. Ingest works
|
||||||
|
curl -X POST http://localhost:9999/memory/ingest \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{"project":"test","source":"test://1","ingest_id":"i1","records":[{"role":"user","text":"Hello world","timestamp":"2026-01-08T16:00:00Z","source_position":0}]}'
|
||||||
|
# Expected: {"ingest_id":"i1","status":"pending",...}
|
||||||
|
|
||||||
|
# 3. Query returns entities with edges
|
||||||
|
curl -X POST http://localhost:9999/memory/query \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{"project":"test","query":"hello"}'
|
||||||
|
# Expected: {"results":[{"type":"entity","name":"...","edges":[...]}]}
|
||||||
|
|
||||||
|
# 4. Temporal filtering works
|
||||||
|
curl http://localhost:9999/memory/query?project=test&temporal_floor=2026-01-01
|
||||||
|
|
||||||
|
# 5. Compaction works
|
||||||
|
curl -X POST http://localhost:9999/memory/compact
|
||||||
|
# Expected: {"phase":"completed","records_deduplicated":N}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## FILES MODIFIED SO FAR
|
||||||
|
|
||||||
|
✅ `crates/mem-cli/src/http_server.rs` - Added error handling for schema init
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## NEXT SESSION TODO
|
||||||
|
|
||||||
|
- [ ] Fix server startup hang (debug OptimizerService)
|
||||||
|
- [ ] Implement ingest_worker to call entity_extractor + fact_extractor
|
||||||
|
- [ ] Add temporal columns to schema
|
||||||
|
- [ ] Update query_handler to traverse edges
|
||||||
|
- [ ] Test end-to-end with sample data
|
||||||
@@ -1,84 +0,0 @@
|
|||||||
# Local Development Setup
|
|
||||||
|
|
||||||
Running poimen-memory locally for development.
|
|
||||||
|
|
||||||
## Quick Start
|
|
||||||
|
|
||||||
1. **Copy env template**:
|
|
||||||
```bash
|
|
||||||
cp .env.example .env
|
|
||||||
```
|
|
||||||
|
|
||||||
2. **Edit `.env`** with your local endpoints:
|
|
||||||
```bash
|
|
||||||
# Edit .env with your local/dev service URLs
|
|
||||||
# Example: LLM service on localhost:11434, OpenSearch on localhost:9200
|
|
||||||
```
|
|
||||||
|
|
||||||
3. **Run the service**:
|
|
||||||
```bash
|
|
||||||
cargo run --release -- serve --port 8080
|
|
||||||
```
|
|
||||||
|
|
||||||
The application loads configuration from `.env` (via `dotenvy` or similar).
|
|
||||||
|
|
||||||
## `.env` File
|
|
||||||
|
|
||||||
**Location**: Project root (`.env`)
|
|
||||||
**Status**: Gitignored - never committed
|
|
||||||
**Template**: `.env.example` (included in repo, shows all available variables)
|
|
||||||
|
|
||||||
### Key Variables
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# Database
|
|
||||||
DATABASE_URL=postgresql://user:pass@localhost:5432/memory
|
|
||||||
|
|
||||||
# LLM (point to your local LLM service)
|
|
||||||
LLM_ENDPOINT=http://localhost:11434/v1/chat/completions
|
|
||||||
LLM_MODEL=qwen:7b
|
|
||||||
|
|
||||||
# OpenSearch (local vector store)
|
|
||||||
OPENSEARCH_HOST=localhost:9200
|
|
||||||
|
|
||||||
# Auth (disabled for local dev)
|
|
||||||
MEM_AUTH_MODE=none
|
|
||||||
|
|
||||||
# API Key (test key for local dev)
|
|
||||||
MEM_API_KEY=test-key
|
|
||||||
```
|
|
||||||
|
|
||||||
## Local Service Stack (Example)
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# Terminal 1: OpenSearch
|
|
||||||
docker run -d -p 9200:9200 -e OPENSEARCH_JAVA_OPTS="-Xms512m -Xmx512m" \
|
|
||||||
opensearchproject/opensearch:latest
|
|
||||||
|
|
||||||
# Terminal 2: Ollama (LLM)
|
|
||||||
ollama serve
|
|
||||||
|
|
||||||
# Terminal 3: poimen-memory
|
|
||||||
cargo run --release -- serve --port 8080
|
|
||||||
```
|
|
||||||
|
|
||||||
## Production vs Local
|
|
||||||
|
|
||||||
| Aspect | Production (K8s) | Local Dev |
|
|
||||||
|--------|-----------------|-----------|
|
|
||||||
| **Config** | `k8s/app/config.yaml` (SOPS-encrypted) | `.env` (gitignored) |
|
|
||||||
| **Injection** | ConfigMap via `envFrom:` | dotenv via `dotenvy` crate |
|
|
||||||
| **Services** | Cluster-internal DNS | localhost/127.0.0.1 |
|
|
||||||
| **Auth** | JWT (Authentik) | None (disabled) |
|
|
||||||
| **Commit?** | Yes (encrypted) | No (gitignored) |
|
|
||||||
|
|
||||||
## Switching to Production Config
|
|
||||||
|
|
||||||
To run against production services (not recommended locally):
|
|
||||||
1. Edit `.env` with production URLs
|
|
||||||
2. Set credentials appropriately
|
|
||||||
3. Ensure network access to production services
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
See `.env.example` for all available environment variables.
|
|
||||||
@@ -0,0 +1,217 @@
|
|||||||
|
# Monitoring Agent: Implementation Tasks
|
||||||
|
|
||||||
|
**Milestone**: `monitoring-agent`
|
||||||
|
**Status**: 🔧 Not started
|
||||||
|
**Duration**: 4-6 weeks
|
||||||
|
**Effort**: ~1,500 LOC
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 1: Temporal Setup (3-5 days)
|
||||||
|
|
||||||
|
### Task 1.1: Deploy Temporal Server in K8s
|
||||||
|
- [ ] StatefulSet configuration (persistence)
|
||||||
|
- [ ] PostgreSQL event log backend
|
||||||
|
- [ ] ElasticSearch for visibility
|
||||||
|
- [ ] K8s manifests in `k8s/temporal/`
|
||||||
|
- [ ] Health checks + readiness probes
|
||||||
|
- **Effort**: 150 LOC | **Time**: 2 days
|
||||||
|
- **Dependencies**: None
|
||||||
|
- **Blocks**: Phase 2
|
||||||
|
|
||||||
|
### Task 1.2: Add Temporal SDK to Rust Project
|
||||||
|
- [ ] Add `temporal-rust-sdk` to `Cargo.toml`
|
||||||
|
- [ ] Create `crates/mem-temporal/` workspace crate
|
||||||
|
- [ ] Worker registration + gRPC connection
|
||||||
|
- [ ] Activity executor setup
|
||||||
|
- [ ] Workflow executor setup
|
||||||
|
- **Effort**: 200 LOC | **Time**: 1 day
|
||||||
|
- **Dependencies**: 1.1
|
||||||
|
- **Blocks**: Phase 2
|
||||||
|
|
||||||
|
### Task 1.3: Temporal Configuration + Secrets
|
||||||
|
- [ ] Environment variables (TEMPORAL_HOST, TEMPORAL_NAMESPACE)
|
||||||
|
- [ ] Worker identity configuration
|
||||||
|
- [ ] Task queue setup (synthesis-queue, compaction-queue)
|
||||||
|
- **Effort**: 50 LOC | **Time**: 4 hours
|
||||||
|
- **Dependencies**: 1.1, 1.2
|
||||||
|
- **Blocks**: Phase 2
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 2: Agent Workflows (1-2 weeks)
|
||||||
|
|
||||||
|
### Task 2.1: Synthesis Workflow Definition
|
||||||
|
- [ ] `crates/mem-temporal/src/workflows/synthesis_workflow.rs`
|
||||||
|
- [ ] Workflow orchestration logic
|
||||||
|
- [ ] Activity composition (health check → synthesis → logging → metrics)
|
||||||
|
- [ ] Retry policies (exponential backoff, max 5 retries)
|
||||||
|
- [ ] Heartbeat configuration (every 10s)
|
||||||
|
- **Effort**: 200 LOC | **Time**: 3 days
|
||||||
|
- **Dependencies**: 1.2, 1.3
|
||||||
|
- **Blocks**: 2.3, 2.4
|
||||||
|
|
||||||
|
### Task 2.2: Synthesis Activities (5 activities)
|
||||||
|
- [ ] `MonitorMemoryHealth` activity
|
||||||
|
- GET /health check
|
||||||
|
- Latency measurement
|
||||||
|
- Failure detection
|
||||||
|
|
||||||
|
- [ ] `ExecuteSynthesis` activity
|
||||||
|
- POST /memory/synthesize call
|
||||||
|
- LLM integration
|
||||||
|
- Heartbeat emission
|
||||||
|
|
||||||
|
- [ ] `LogSynthesisResult` activity
|
||||||
|
- POST /memory/ingest (audit)
|
||||||
|
- Temporal audit trail
|
||||||
|
|
||||||
|
- [ ] `UpdateCacheMetrics` activity
|
||||||
|
- Metric recording
|
||||||
|
- Performance tracking
|
||||||
|
|
||||||
|
- [ ] `CoordinateCompaction` activity
|
||||||
|
- Signal to compaction agent
|
||||||
|
- Readiness check
|
||||||
|
|
||||||
|
- **Effort**: 250 LOC | **Time**: 4 days
|
||||||
|
- **Dependencies**: 2.1
|
||||||
|
- **Blocks**: 2.3
|
||||||
|
|
||||||
|
### Task 2.3: Compaction Workflow Definition
|
||||||
|
- [ ] `crates/mem-temporal/src/workflows/compaction_workflow.rs`
|
||||||
|
- [ ] 4-stage orchestration (identify → dedup → gc → invalidate)
|
||||||
|
- [ ] Failure handling + rollback strategy
|
||||||
|
- **Effort**: 150 LOC | **Time**: 2 days
|
||||||
|
- **Dependencies**: 1.2, 1.3
|
||||||
|
- **Blocks**: 2.4
|
||||||
|
|
||||||
|
### Task 2.4: Compaction Activities (4 activities)
|
||||||
|
- [ ] `IdentifyDuplicates` activity
|
||||||
|
- [ ] `DeduplicateEdges` activity
|
||||||
|
- [ ] `GarbageCollection` activity
|
||||||
|
- [ ] `InvalidateCache` activity
|
||||||
|
- **Effort**: 200 LOC | **Time**: 3 days
|
||||||
|
- **Dependencies**: 2.3
|
||||||
|
- **Blocks**: Integration tests
|
||||||
|
|
||||||
|
### Task 2.5: Worker + Task Queue Registration
|
||||||
|
- [ ] Activity worker setup
|
||||||
|
- [ ] Workflow worker setup
|
||||||
|
- [ ] Task queue polling
|
||||||
|
- [ ] Namespace configuration
|
||||||
|
- **Effort**: 100 LOC | **Time**: 1 day
|
||||||
|
- **Dependencies**: 2.1-2.4
|
||||||
|
- **Blocks**: Phase 3
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 3: Agent Self-Awareness (2-3 weeks)
|
||||||
|
|
||||||
|
### Task 3.1: AGENT_PROMPT Entity Type
|
||||||
|
- [ ] Schema: New entity type in memory_entity
|
||||||
|
- [ ] Repository: `synthesis_cache_repo.rs` (get_agent_prompt)
|
||||||
|
- [ ] Migration: Add to entity type enum
|
||||||
|
- [ ] Activity: Load prompt on agent startup
|
||||||
|
- **Effort**: 100 LOC | **Time**: 1 day
|
||||||
|
- **Dependencies**: Memory service
|
||||||
|
- **Blocks**: 3.2
|
||||||
|
|
||||||
|
### Task 3.2: AGENT_SKILL Linking
|
||||||
|
- [ ] Edge type: agent → skill relationships
|
||||||
|
- [ ] Repository methods: link_agent_to_skill, get_agent_skills
|
||||||
|
- [ ] Confidence tracking per skill
|
||||||
|
- [ ] Success rate calculation
|
||||||
|
- **Effort**: 80 LOC | **Time**: 1 day
|
||||||
|
- **Dependencies**: 3.1
|
||||||
|
- **Blocks**: 3.4
|
||||||
|
|
||||||
|
### Task 3.3: AGENT_PERFORMANCE Metrics
|
||||||
|
- [ ] Entity type: Temporal metrics
|
||||||
|
- [ ] Repository: Store + query metrics
|
||||||
|
- [ ] Activity: Log performance data post-execution
|
||||||
|
- [ ] Time window filtering (last_7_days, last_30_days)
|
||||||
|
- **Effort**: 120 LOC | **Time**: 2 days
|
||||||
|
- **Dependencies**: 3.1
|
||||||
|
- **Blocks**: 3.4
|
||||||
|
|
||||||
|
### Task 3.4: Agent Decision Tracking + Learning
|
||||||
|
- [ ] Edge type: agent_decision_outcome
|
||||||
|
- [ ] Decision logging (parameter, value, confidence before)
|
||||||
|
- [ ] Outcome recording (result, metric)
|
||||||
|
- [ ] Confidence evolution (update after outcome)
|
||||||
|
- [ ] Learning loop in agent code
|
||||||
|
- **Effort**: 200 LOC | **Time**: 3 days
|
||||||
|
- **Dependencies**: 3.1-3.3
|
||||||
|
- **Blocks**: 3.5
|
||||||
|
|
||||||
|
### Task 3.5: Agent Audit Trail Integration
|
||||||
|
- [ ] Dual audit: Temporal history + Memory entities
|
||||||
|
- [ ] Query interface for reviewers
|
||||||
|
- [ ] Temporal CLI integration
|
||||||
|
- [ ] Retention policy (365 days)
|
||||||
|
- **Effort**: 100 LOC | **Time**: 1 day
|
||||||
|
- **Dependencies**: 3.1-3.4
|
||||||
|
- **Blocks**: Testing
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Testing & Documentation
|
||||||
|
|
||||||
|
### Task 4.1: Integration Tests
|
||||||
|
- [ ] Workflow execution end-to-end
|
||||||
|
- [ ] Activity retry behavior
|
||||||
|
- [ ] Heartbeat detection
|
||||||
|
- [ ] Failure recovery
|
||||||
|
- [ ] State replay on restart
|
||||||
|
- **Effort**: 300 LOC | **Time**: 3 days
|
||||||
|
- **Dependencies**: Phase 2 complete
|
||||||
|
- **Blocks**: Integration
|
||||||
|
|
||||||
|
### Task 4.2: Monitoring & Observability
|
||||||
|
- [ ] Temporal UI setup (temporal.riotpiao.com)
|
||||||
|
- [ ] Prometheus metrics export
|
||||||
|
- [ ] Alerting rules (workflow timeout, activity failure)
|
||||||
|
- [ ] Grafana dashboards
|
||||||
|
- **Effort**: 150 LOC | **Time**: 2 days
|
||||||
|
- **Dependencies**: Phase 1 complete
|
||||||
|
- **Blocks**: Production
|
||||||
|
|
||||||
|
### Task 4.3: Documentation
|
||||||
|
- [ ] Agent architecture diagram
|
||||||
|
- [ ] Workflow execution flow
|
||||||
|
- [ ] Operational runbook
|
||||||
|
- [ ] Troubleshooting guide
|
||||||
|
- **Effort**: 50 LOC | **Time**: 1 day
|
||||||
|
- **Dependencies**: All phases
|
||||||
|
- **Blocks**: Release
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Credentials Status
|
||||||
|
|
||||||
|
✅ **SOPS Encrypted**: `k8s/app/memory-agent-secrets.enc.yaml`
|
||||||
|
- CLIENT_ID: `memory-agent`
|
||||||
|
- CLIENT_SECRET: Encrypted
|
||||||
|
- TOKEN_URL: `https://authentik.riotpiao.com/application/o/token/`
|
||||||
|
- AUTHENTIK_ISSUER: `https://authentik.riotpiao.com/application/o/memory-agent/`
|
||||||
|
|
||||||
|
✅ **JWT Auth Verified**: `memory-agent` credentials working
|
||||||
|
- Test result: Token obtained successfully
|
||||||
|
- Expiry: 1 hour (3600s)
|
||||||
|
- Scopes: Default (sufficient for LLM operations)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Timeline
|
||||||
|
|
||||||
|
```
|
||||||
|
Week 1 (Phase 1): Temporal setup
|
||||||
|
Week 2-3 (Phase 2): Agent workflows
|
||||||
|
Week 4-5 (Phase 3): Self-awareness
|
||||||
|
Week 6 (Testing + Docs): Integration + release
|
||||||
|
```
|
||||||
|
|
||||||
|
**Start Date**: TBD
|
||||||
|
**Target End Date**: TBD (+4-6 weeks)
|
||||||
|
|
||||||
@@ -0,0 +1,191 @@
|
|||||||
|
# Current Status - Poimen Memory Service (2026-01-08)
|
||||||
|
|
||||||
|
## ✅ COMPLETED THIS SESSION
|
||||||
|
|
||||||
|
### 1. Removed AccessGuard RBAC (Blocker Issue #1)
|
||||||
|
- ❌ ~~AccessGuard initialization~~ REMOVED
|
||||||
|
- ❌ ~~RBAC checks in handlers~~ REMOVED
|
||||||
|
- ❌ ~~Permission-based access control~~ DEFERRED
|
||||||
|
- ✅ Code now compiles with `cargo build --release`
|
||||||
|
- ✅ Binary created: `target/release/mem`
|
||||||
|
|
||||||
|
### 2. HTTP Handler Initialization Fixed
|
||||||
|
- ✅ Added error handling for schema initialization
|
||||||
|
- ✅ Server reaches "Starting HTTP server" log message
|
||||||
|
- ✅ HTTP server binds to port (processes created)
|
||||||
|
|
||||||
|
## ⚠️ CURRENT ISSUE
|
||||||
|
|
||||||
|
**Server binds to port but exits immediately (silent failure)**
|
||||||
|
|
||||||
|
Process is created and runs `serve` command, but:
|
||||||
|
- Process exits with code 0 (clean exit, no crash)
|
||||||
|
- No HTTP requests answered (port refuses connections)
|
||||||
|
- Logs don't show "listening on 0.0.0.0:8080" message
|
||||||
|
|
||||||
|
**Suspected cause**: Something in the handler initialization or routing setup is blocking/panicking but not showing in logs.
|
||||||
|
|
||||||
|
## 🔧 DEBUGGING STEPS NEEDED
|
||||||
|
|
||||||
|
1. Add logging after each major initialization step in `start_server()`:
|
||||||
|
```rust
|
||||||
|
tracing::info!("About to create AppState");
|
||||||
|
let state = web::Data::new(AppState { ... });
|
||||||
|
tracing::info!("AppState created");
|
||||||
|
|
||||||
|
tracing::info!("About to create HttpServer");
|
||||||
|
HttpServer::new(move || { ... })
|
||||||
|
tracing::info!("HttpServer created, about to bind");
|
||||||
|
|
||||||
|
.bind(("0.0.0.0", port))?
|
||||||
|
tracing::info!("Bound to port {}", port);
|
||||||
|
|
||||||
|
.run()
|
||||||
|
tracing::info!("About to run()");
|
||||||
|
.await?;
|
||||||
|
tracing::info!("Server running");
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Run with `RUST_BACKTRACE=1` to see panics
|
||||||
|
3. Check if the issue is in handler route registration
|
||||||
|
|
||||||
|
## 📋 NEXT PRIORITY FIXES (AFTER SERVER RUNS)
|
||||||
|
|
||||||
|
### Phase 1: INGEST PIPELINE ⭐ CRITICAL
|
||||||
|
**File**: `crates/mem-cli/src/ingest_worker.rs`
|
||||||
|
|
||||||
|
Currently: Just stores raw vectors
|
||||||
|
```rust
|
||||||
|
// WRONG - just vector storage
|
||||||
|
store_chunk_l0(&l0_chunk);
|
||||||
|
store_memory_l1(&l1_memory);
|
||||||
|
```
|
||||||
|
|
||||||
|
Should: Extract entities + facts + edges
|
||||||
|
```rust
|
||||||
|
// 1. Extract entities
|
||||||
|
let entities = entity_extractor.extract(&content).await?;
|
||||||
|
|
||||||
|
// 2. Extract facts/relationships
|
||||||
|
let facts = fact_extractor.extract(&content, &entities).await?;
|
||||||
|
|
||||||
|
// 3. Create temporal edges
|
||||||
|
for fact in facts {
|
||||||
|
let edge = TemporalEdge {
|
||||||
|
source: fact.source_entity,
|
||||||
|
target: fact.target_entity,
|
||||||
|
relation: fact.relation,
|
||||||
|
fact: fact.text,
|
||||||
|
t_valid: now(),
|
||||||
|
t_invalid: None,
|
||||||
|
confidence: 0.8, // GRM gate score
|
||||||
|
version: 1,
|
||||||
|
};
|
||||||
|
edge_repo.insert(&edge).await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 4. Queue contradictions for review
|
||||||
|
for edge in &edges {
|
||||||
|
if contradiction_detector.detect(edge, existing_edges)? {
|
||||||
|
review_queue.enqueue(edge).await?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Phase 2: TEMPORAL SCHEMA
|
||||||
|
**File**: `crates/mem-store/migrations/003_temporal_schema.sql`
|
||||||
|
|
||||||
|
Add columns:
|
||||||
|
- `t_valid TIMESTAMP NOT NULL DEFAULT NOW()`
|
||||||
|
- `t_invalid TIMESTAMP`
|
||||||
|
- `confidence FLOAT DEFAULT 0.8`
|
||||||
|
- `version INT DEFAULT 1`
|
||||||
|
- `update_reason VARCHAR`
|
||||||
|
|
||||||
|
Create edge table:
|
||||||
|
```sql
|
||||||
|
CREATE TABLE memory_edge (
|
||||||
|
source_id UUID NOT NULL,
|
||||||
|
target_id UUID NOT NULL,
|
||||||
|
relation VARCHAR NOT NULL,
|
||||||
|
fact TEXT NOT NULL,
|
||||||
|
t_valid TIMESTAMP DEFAULT NOW(),
|
||||||
|
t_invalid TIMESTAMP,
|
||||||
|
confidence FLOAT,
|
||||||
|
version INT,
|
||||||
|
PRIMARY KEY (source_id, target_id, relation, version)
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### Phase 3: QUERY HANDLER
|
||||||
|
**File**: `crates/mem-cli/src/http_server.rs`
|
||||||
|
|
||||||
|
Change `query_handler()` from vector-only to graph-aware:
|
||||||
|
```rust
|
||||||
|
// 1. Vector search
|
||||||
|
let results = semantic_search(query)?;
|
||||||
|
|
||||||
|
// 2. Follow edges
|
||||||
|
let mut expanded = results;
|
||||||
|
for entity in results {
|
||||||
|
let related = edge_repo.find_by_source(&entity.id).await?;
|
||||||
|
expanded.extend(related);
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. Apply temporal filter
|
||||||
|
expanded.retain(|e| is_valid_at_time(e, now()));
|
||||||
|
|
||||||
|
// 4. Sort by confidence + recency
|
||||||
|
expanded.sort_by_key(|e| (-e.confidence, -e.t_valid));
|
||||||
|
|
||||||
|
// 5. Return
|
||||||
|
HttpResponse::Ok().json(expanded)
|
||||||
|
```
|
||||||
|
|
||||||
|
### Phase 4: END-TO-END TESTING
|
||||||
|
```bash
|
||||||
|
# 1. Ingest with entities + facts
|
||||||
|
POST /memory/ingest
|
||||||
|
{
|
||||||
|
"project": "test",
|
||||||
|
"source": "transcript://session-1",
|
||||||
|
"ingest_id": "i-001",
|
||||||
|
"records": [{"role": "user", "text": "Kubernetes port conflict...", ...}]
|
||||||
|
}
|
||||||
|
# Expected: {"ingest_id":"i-001","status":"pending"}
|
||||||
|
|
||||||
|
# 2. Check ingest status
|
||||||
|
GET /memory/ingest/i-001
|
||||||
|
# Expected: {"status":"done","entities_count":5,"edges_count":3}
|
||||||
|
|
||||||
|
# 3. Query returns graph
|
||||||
|
POST /memory/query
|
||||||
|
{"project":"test","query":"port conflict resolution"}
|
||||||
|
# Expected: {"results":[
|
||||||
|
# {"type":"entity","name":"Kubernetes","edges":[...]},
|
||||||
|
# {"type":"entity","name":"Port","edges":[...]},
|
||||||
|
# {"type":"fact","source":"Kubernetes","target":"Port","relation":"has-conflict"}
|
||||||
|
# ]}
|
||||||
|
```
|
||||||
|
|
||||||
|
## FILES MODIFIED
|
||||||
|
|
||||||
|
✅ `crates/mem-cli/src/http_server.rs` - Removed RBAC, added error handling
|
||||||
|
✅ Created `STATUS_CURRENT.md` - This file
|
||||||
|
|
||||||
|
## TIMELINE
|
||||||
|
|
||||||
|
- **2026-01-08 16:00**: Fixed HTTP handlers, removed RBAC blocker
|
||||||
|
- **2026-01-08 16:30**: Server init working, but exits on startup
|
||||||
|
- **2026-01-08 16:40**: Debugging server binding issue
|
||||||
|
|
||||||
|
## KEY DECISIONS
|
||||||
|
|
||||||
|
1. **RBAC deferred**: MVP focuses on core ingest/query, auth added later
|
||||||
|
2. **Temporal-first**: All edges must have t_valid/t_invalid for graph compaction
|
||||||
|
3. **GRM gate integrated at ingest time**: Confidence scores assigned when facts extracted
|
||||||
|
4. **No queue worker** in MVP: Enable it after core working
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Next action**: Add detailed logging to `start_server()` to see where process exits.
|
||||||
@@ -46,4 +46,3 @@ futures-util = "0.3"
|
|||||||
async-stream = "0.3"
|
async-stream = "0.3"
|
||||||
rand = "0.8"
|
rand = "0.8"
|
||||||
lru = "0.12"
|
lru = "0.12"
|
||||||
once_cell = { workspace = true }
|
|
||||||
|
|||||||
@@ -224,20 +224,9 @@ impl KvCacheAligner {
|
|||||||
|
|
||||||
/// Pre-load hot chunks into cache
|
/// Pre-load hot chunks into cache
|
||||||
pub fn preload_hot_chunks(&self, hot_chunks: Vec<(&str, &str)>) -> Result<()> {
|
pub fn preload_hot_chunks(&self, hot_chunks: Vec<(&str, &str)>) -> Result<()> {
|
||||||
let count = hot_chunks.len();
|
|
||||||
for (chunk_id, text) in hot_chunks {
|
for (chunk_id, text) in hot_chunks {
|
||||||
self.cache.put(chunk_id, text);
|
self.cache.put(chunk_id, text);
|
||||||
}
|
}
|
||||||
let metrics = self.cache.metrics();
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "cache_preload",
|
|
||||||
preloaded = count,
|
|
||||||
cache_hits = metrics.hits,
|
|
||||||
cache_misses = metrics.misses,
|
|
||||||
hit_ratio = format!("{:.2}", metrics.hit_ratio()),
|
|
||||||
"Cache preload complete"
|
|
||||||
);
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -211,31 +211,16 @@ impl ChunkOptimizer {
|
|||||||
|
|
||||||
/// End-to-end optimization pipeline
|
/// End-to-end optimization pipeline
|
||||||
pub fn optimize(&self, chunks: Vec<OptimizableChunk>) -> (Vec<OptimizableChunk>, SelectionMetrics) {
|
pub fn optimize(&self, chunks: Vec<OptimizableChunk>) -> (Vec<OptimizableChunk>, SelectionMetrics) {
|
||||||
let input_count = chunks.len();
|
|
||||||
|
|
||||||
// Step 1: Filter by threshold
|
// Step 1: Filter by threshold
|
||||||
let filtered = self.threshold_filter.filter(chunks.clone());
|
let filtered = self.threshold_filter.filter(chunks.clone());
|
||||||
let after_filter = filtered.len();
|
|
||||||
|
|
||||||
// Step 2: Deduplicate
|
// Step 2: Deduplicate
|
||||||
let (deduplicated, dedup_removed) = self.deduplicator.deduplicate(filtered);
|
let (deduplicated, dedup_removed) = self.deduplicator.deduplicate(filtered);
|
||||||
let after_dedup = deduplicated.len();
|
|
||||||
|
|
||||||
// Step 3: Select within budget
|
// Step 3: Select within budget
|
||||||
let (selected, mut metrics) = self.budget_selector.select(deduplicated);
|
let (selected, mut metrics) = self.budget_selector.select(deduplicated);
|
||||||
metrics.dedup_removed = dedup_removed;
|
|
||||||
|
|
||||||
tracing::info!(
|
metrics.dedup_removed = dedup_removed;
|
||||||
target: "observability",
|
|
||||||
event = "chunk_optimize",
|
|
||||||
input = input_count,
|
|
||||||
after_threshold_filter = after_filter,
|
|
||||||
after_dedup = after_dedup,
|
|
||||||
dedup_removed = dedup_removed,
|
|
||||||
selected = selected.len(),
|
|
||||||
budget_bytes = metrics.total_bytes,
|
|
||||||
"Chunk optimization complete"
|
|
||||||
);
|
|
||||||
|
|
||||||
(selected, metrics)
|
(selected, metrics)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -346,19 +346,7 @@ pub async fn compact_memory(
|
|||||||
}
|
}
|
||||||
|
|
||||||
total_stats.duration_ms = start.elapsed().as_millis() as u64;
|
total_stats.duration_ms = start.elapsed().as_millis() as u64;
|
||||||
info!(
|
info!("Compaction complete in {}ms: {:?}", total_stats.duration_ms, total_stats);
|
||||||
target: "observability",
|
|
||||||
event = "compaction_complete",
|
|
||||||
mode = ?mode,
|
|
||||||
duration_ms = total_stats.duration_ms,
|
|
||||||
duplicate_edges_deleted = total_stats.duplicate_edges_deleted,
|
|
||||||
stale_facts_deleted = total_stats.stale_facts_deleted,
|
|
||||||
semantic_merged = total_stats.semantic_merged,
|
|
||||||
llm_calls = total_stats.llm_calls,
|
|
||||||
bytes_freed = total_stats.bytes_freed,
|
|
||||||
human_reviews_queued = total_stats.human_reviews_queued,
|
|
||||||
"Compaction complete"
|
|
||||||
);
|
|
||||||
|
|
||||||
Ok(total_stats)
|
Ok(total_stats)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -344,22 +344,6 @@ impl FullPipeline {
|
|||||||
|
|
||||||
metrics.total_latency_ms = start.elapsed().as_millis() as u64;
|
metrics.total_latency_ms = start.elapsed().as_millis() as u64;
|
||||||
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "full_pipeline_complete",
|
|
||||||
query = query,
|
|
||||||
candidates = metrics.wiki_scope_docs,
|
|
||||||
prefiltered = metrics.prefilter_candidates,
|
|
||||||
optimized = metrics.post_optimization_count,
|
|
||||||
dedup_removed = metrics.dedup_removed,
|
|
||||||
boosts_applied = metrics.metadata_boosts_applied,
|
|
||||||
cache_hit_ratio = format!("{:.2}", metrics.cache_hit_ratio),
|
|
||||||
budget_bytes = metrics.budget_used_bytes,
|
|
||||||
total_ms = metrics.total_latency_ms,
|
|
||||||
"Full query pipeline complete"
|
|
||||||
);
|
|
||||||
|
|
||||||
|
|
||||||
Ok(PipelineResult {
|
Ok(PipelineResult {
|
||||||
query: query.to_string(),
|
query: query.to_string(),
|
||||||
query_intent,
|
query_intent,
|
||||||
@@ -483,22 +467,6 @@ impl FullPipeline {
|
|||||||
|
|
||||||
metrics.total_latency_ms = start.elapsed().as_millis() as u64;
|
metrics.total_latency_ms = start.elapsed().as_millis() as u64;
|
||||||
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "full_pipeline_complete",
|
|
||||||
query = query,
|
|
||||||
candidates = metrics.wiki_scope_docs,
|
|
||||||
prefiltered = metrics.prefilter_candidates,
|
|
||||||
optimized = metrics.post_optimization_count,
|
|
||||||
dedup_removed = metrics.dedup_removed,
|
|
||||||
boosts_applied = metrics.metadata_boosts_applied,
|
|
||||||
cache_hit_ratio = format!("{:.2}", metrics.cache_hit_ratio),
|
|
||||||
budget_bytes = metrics.budget_used_bytes,
|
|
||||||
total_ms = metrics.total_latency_ms,
|
|
||||||
"Full query pipeline complete"
|
|
||||||
);
|
|
||||||
|
|
||||||
|
|
||||||
Ok(PipelineResult {
|
Ok(PipelineResult {
|
||||||
query: query.to_string(),
|
query: query.to_string(),
|
||||||
query_intent,
|
query_intent,
|
||||||
|
|||||||
@@ -1,17 +1,11 @@
|
|||||||
//! Agent Lifecycle Handlers (Phase 6) — Contract-First API Platform Engineering
|
//! Agent Lifecycle Handlers (Phase 6)
|
||||||
//!
|
|
||||||
//! Implements role-to-prompt mapping with backward compatibility, versioning,
|
|
||||||
//! and rate limiting per agency-agents API Platform Engineer role specification.
|
|
||||||
|
|
||||||
use actix_web::{web, HttpRequest, HttpResponse};
|
use actix_web::{web, HttpRequest, HttpResponse};
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use uuid::Uuid;
|
|
||||||
use chrono::Utc;
|
|
||||||
use crate::agent::{Agent, AgentConfig, AgentCapability, DefaultAgent};
|
use crate::agent::{Agent, AgentConfig, AgentCapability, DefaultAgent};
|
||||||
use crate::agent::client_sdk::SynthesisClient;
|
use crate::agent::client_sdk::SynthesisClient;
|
||||||
use crate::handlers::response_builder;
|
use crate::handlers::response_builder;
|
||||||
use mem_store::agent_repo::{AgentRepository, AgentPrompt, AgentSkill, AgentDecision, RolePromptMapping};
|
|
||||||
use tracing::{debug, info, error, warn};
|
use tracing::{debug, info, error, warn};
|
||||||
|
|
||||||
/// Register agent request
|
/// Register agent request
|
||||||
@@ -86,50 +80,7 @@ pub async fn register_agent_handler(
|
|||||||
metadata: std::collections::HashMap::new(),
|
metadata: std::collections::HashMap::new(),
|
||||||
};
|
};
|
||||||
|
|
||||||
// Persist agent config to database via agent_registry table
|
// Store agent config (stub: would persist to DB)
|
||||||
let agent_repo = AgentRepository::new(state.pool.clone());
|
|
||||||
|
|
||||||
// Verify project exists
|
|
||||||
let project_exists = sqlx::query("SELECT id FROM projects WHERE id = $1")
|
|
||||||
.bind(&body.project_id)
|
|
||||||
.fetch_optional(&state.pool)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
if let Err(e) = project_exists {
|
|
||||||
error!("Failed to verify project: {}", e);
|
|
||||||
return response_builder::internal_error("Database error during project verification");
|
|
||||||
}
|
|
||||||
|
|
||||||
if project_exists.unwrap().is_none() {
|
|
||||||
return response_builder::bad_request(&format!("Project not found: {}", body.project_id));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Insert agent registry record
|
|
||||||
let agent_insert = sqlx::query(
|
|
||||||
r#"
|
|
||||||
INSERT INTO agent_registry
|
|
||||||
(project_id, agent_id, capabilities, webhook_url, rate_limit, status)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, 'active')
|
|
||||||
ON CONFLICT (project_id, agent_id) DO UPDATE SET
|
|
||||||
capabilities = $3,
|
|
||||||
webhook_url = $4,
|
|
||||||
rate_limit = $5,
|
|
||||||
updated_at = NOW()
|
|
||||||
"#
|
|
||||||
)
|
|
||||||
.bind(&body.project_id)
|
|
||||||
.bind(&body.agent_id)
|
|
||||||
.bind(&body.capabilities)
|
|
||||||
.bind(&body.webhook_url)
|
|
||||||
.bind(body.rate_limit.unwrap_or(1000))
|
|
||||||
.execute(&state.pool)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
if let Err(e) = agent_insert {
|
|
||||||
error!("Failed to insert agent registry: {}", e);
|
|
||||||
return response_builder::internal_error("Failed to register agent");
|
|
||||||
}
|
|
||||||
|
|
||||||
let agent = DefaultAgent::new(config);
|
let agent = DefaultAgent::new(config);
|
||||||
|
|
||||||
// Extract JWT from request for agent reasoning calls
|
// Extract JWT from request for agent reasoning calls
|
||||||
@@ -139,7 +90,7 @@ pub async fn register_agent_handler(
|
|||||||
warn!("Agent registered without JWT token");
|
warn!("Agent registered without JWT token");
|
||||||
}
|
}
|
||||||
|
|
||||||
info!("Agent registered and persisted: {}", agent.config().agent_id);
|
info!("Agent registered: {}", agent.config().agent_id);
|
||||||
|
|
||||||
// Wire Temporal workflow (via api.riotpiao.com/workflow)
|
// Wire Temporal workflow (via api.riotpiao.com/workflow)
|
||||||
// Temporal activities will:
|
// Temporal activities will:
|
||||||
@@ -181,6 +132,8 @@ pub async fn register_agent_handler(
|
|||||||
let workflow_id = data.get("workflow_id").and_then(|v| v.as_str()).unwrap_or("unknown");
|
let workflow_id = data.get("workflow_id").and_then(|v| v.as_str()).unwrap_or("unknown");
|
||||||
let run_id = data.get("run_id").and_then(|v| v.as_str()).unwrap_or("unknown");
|
let run_id = data.get("run_id").and_then(|v| v.as_str()).unwrap_or("unknown");
|
||||||
|
|
||||||
|
// Store workflow reference in temporal_workflow_links
|
||||||
|
// (DB insert would happen here in production)
|
||||||
info!("Agent workflow started: workflow_id={}, run_id={}", workflow_id, run_id);
|
info!("Agent workflow started: workflow_id={}, run_id={}", workflow_id, run_id);
|
||||||
debug!("Temporal activity will persist agent state + reasoning traces");
|
debug!("Temporal activity will persist agent state + reasoning traces");
|
||||||
}
|
}
|
||||||
@@ -198,7 +151,7 @@ pub async fn register_agent_handler(
|
|||||||
capabilities: body.capabilities.clone(),
|
capabilities: body.capabilities.clone(),
|
||||||
webhook_url: body.webhook_url.clone(),
|
webhook_url: body.webhook_url.clone(),
|
||||||
rate_limit: agent.config().rate_limit,
|
rate_limit: agent.config().rate_limit,
|
||||||
created_at: Utc::now().to_rfc3339(),
|
created_at: chrono::Utc::now().to_rfc3339(),
|
||||||
status: "active".to_string(),
|
status: "active".to_string(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -364,247 +317,3 @@ pub async fn delete_agent_handler(
|
|||||||
}))
|
}))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
// Role-to-Prompt Mapping Handlers (API Platform Engineer role support)
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
|
||||||
pub struct CreatePromptRequest {
|
|
||||||
pub name: String,
|
|
||||||
pub template: String,
|
|
||||||
pub target_model: Option<String>,
|
|
||||||
pub task_category: String,
|
|
||||||
pub tags: Option<Vec<String>>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize)]
|
|
||||||
pub struct PromptResponse {
|
|
||||||
pub id: String,
|
|
||||||
pub name: String,
|
|
||||||
pub template: String,
|
|
||||||
pub target_model: Option<String>,
|
|
||||||
pub task_category: String,
|
|
||||||
pub tags: Vec<String>,
|
|
||||||
pub usage_count: i64,
|
|
||||||
pub avg_quality: f32,
|
|
||||||
pub version: i32,
|
|
||||||
pub created_at: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// POST /memory/agents/{project_id}/prompts - Create agent prompt
|
|
||||||
pub async fn create_prompt_handler(
|
|
||||||
req: HttpRequest,
|
|
||||||
path: web::Path<String>,
|
|
||||||
body: web::Json<CreatePromptRequest>,
|
|
||||||
state: web::Data<crate::AppState>,
|
|
||||||
) -> HttpResponse {
|
|
||||||
let project_id = path.into_inner();
|
|
||||||
|
|
||||||
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
|
|
||||||
&req, &state, "prompt", 100
|
|
||||||
) {
|
|
||||||
return response;
|
|
||||||
}
|
|
||||||
|
|
||||||
if body.name.is_empty() || body.template.is_empty() {
|
|
||||||
return response_builder::bad_request("name and template required");
|
|
||||||
}
|
|
||||||
|
|
||||||
debug!("Creating prompt for project: {} with name: {}", project_id, body.name);
|
|
||||||
|
|
||||||
let prompt_id = Uuid::new_v4();
|
|
||||||
let now = Utc::now();
|
|
||||||
let tags = body.tags.clone().unwrap_or_default();
|
|
||||||
|
|
||||||
let prompt_insert = sqlx::query(
|
|
||||||
r#"
|
|
||||||
INSERT INTO agent_prompt
|
|
||||||
(id, project_id, name, template, target_model, task_category, tags, version, active)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, $6, $7, 1, true)
|
|
||||||
"#
|
|
||||||
)
|
|
||||||
.bind(prompt_id)
|
|
||||||
.bind(&project_id)
|
|
||||||
.bind(&body.name)
|
|
||||||
.bind(&body.template)
|
|
||||||
.bind(&body.target_model)
|
|
||||||
.bind(&body.task_category)
|
|
||||||
.bind(&tags)
|
|
||||||
.execute(&state.pool)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
match prompt_insert {
|
|
||||||
Ok(_) => {
|
|
||||||
info!("Prompt created: {} in project {}", body.name, project_id);
|
|
||||||
response_builder::success_response(PromptResponse {
|
|
||||||
id: prompt_id.to_string(),
|
|
||||||
name: body.name.clone(),
|
|
||||||
template: body.template.clone(),
|
|
||||||
target_model: body.target_model.clone(),
|
|
||||||
task_category: body.task_category.clone(),
|
|
||||||
tags,
|
|
||||||
usage_count: 0,
|
|
||||||
avg_quality: 0.0,
|
|
||||||
version: 1,
|
|
||||||
created_at: now.to_rfc3339(),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
error!("Failed to create prompt: {}", e);
|
|
||||||
response_builder::internal_error("Failed to create prompt")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
|
||||||
pub struct MapRoleToPromptRequest {
|
|
||||||
pub role_name: String,
|
|
||||||
pub prompt_id: String,
|
|
||||||
pub priority: Option<i32>,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// POST /memory/agents/{project_id}/roles - Map role to prompt
|
|
||||||
pub async fn map_role_to_prompt_handler(
|
|
||||||
req: HttpRequest,
|
|
||||||
path: web::Path<String>,
|
|
||||||
body: web::Json<MapRoleToPromptRequest>,
|
|
||||||
state: web::Data<crate::AppState>,
|
|
||||||
) -> HttpResponse {
|
|
||||||
let project_id = path.into_inner();
|
|
||||||
|
|
||||||
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
|
|
||||||
&req, &state, "role-mapping", 100
|
|
||||||
) {
|
|
||||||
return response;
|
|
||||||
}
|
|
||||||
|
|
||||||
if body.role_name.is_empty() || body.prompt_id.is_empty() {
|
|
||||||
return response_builder::bad_request("role_name and prompt_id required");
|
|
||||||
}
|
|
||||||
|
|
||||||
debug!("Mapping role {} to prompt {} in project {}", body.role_name, body.prompt_id, project_id);
|
|
||||||
|
|
||||||
let prompt_uuid = match Uuid::parse_str(&body.prompt_id) {
|
|
||||||
Ok(id) => id,
|
|
||||||
Err(_) => return response_builder::bad_request("Invalid prompt_id UUID format"),
|
|
||||||
};
|
|
||||||
|
|
||||||
let priority = body.priority.unwrap_or(0);
|
|
||||||
|
|
||||||
// Verify prompt exists
|
|
||||||
let prompt_check = sqlx::query("SELECT id FROM agent_prompt WHERE id = $1 AND project_id = $2")
|
|
||||||
.bind(prompt_uuid)
|
|
||||||
.bind(&project_id)
|
|
||||||
.fetch_optional(&state.pool)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
match prompt_check {
|
|
||||||
Ok(Some(_)) => {
|
|
||||||
// Create mapping
|
|
||||||
let mapping_insert = sqlx::query(
|
|
||||||
r#"
|
|
||||||
INSERT INTO role_prompt_mapping
|
|
||||||
(project_id, role_name, prompt_id, priority, active)
|
|
||||||
VALUES ($1, $2, $3, $4, true)
|
|
||||||
ON CONFLICT (project_id, role_name, prompt_id) DO UPDATE SET
|
|
||||||
priority = $4, active = true, updated_at = NOW()
|
|
||||||
"#
|
|
||||||
)
|
|
||||||
.bind(&project_id)
|
|
||||||
.bind(&body.role_name)
|
|
||||||
.bind(prompt_uuid)
|
|
||||||
.bind(priority)
|
|
||||||
.execute(&state.pool)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
match mapping_insert {
|
|
||||||
Ok(_) => {
|
|
||||||
info!("Mapped role {} to prompt {} (priority: {})", body.role_name, body.prompt_id, priority);
|
|
||||||
response_builder::success_response(serde_json::json!({
|
|
||||||
"role_name": body.role_name,
|
|
||||||
"prompt_id": body.prompt_id,
|
|
||||||
"priority": priority,
|
|
||||||
"status": "mapped"
|
|
||||||
}))
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
error!("Failed to create role mapping: {}", e);
|
|
||||||
response_builder::internal_error("Failed to map role to prompt")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Ok(None) => {
|
|
||||||
response_builder::not_found(&format!("Prompt not found: {}", body.prompt_id))
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
error!("Database error checking prompt: {}", e);
|
|
||||||
response_builder::internal_error("Database error")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize)]
|
|
||||||
pub struct RolePromptsResponse {
|
|
||||||
pub role_name: String,
|
|
||||||
pub prompts: Vec<PromptResponse>,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// GET /memory/agents/{project_id}/roles/{role_name}/prompts - Get prompts for role
|
|
||||||
pub async fn get_role_prompts_handler(
|
|
||||||
req: HttpRequest,
|
|
||||||
path: web::Path<(String, String)>,
|
|
||||||
state: web::Data<crate::AppState>,
|
|
||||||
) -> HttpResponse {
|
|
||||||
let (project_id, role_name) = path.into_inner();
|
|
||||||
|
|
||||||
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
|
|
||||||
&req, &state, "role-query", 200
|
|
||||||
) {
|
|
||||||
return response;
|
|
||||||
}
|
|
||||||
|
|
||||||
debug!("Getting prompts for role {} in project {}", role_name, project_id);
|
|
||||||
|
|
||||||
let prompts_query = sqlx::query_as::<_, (String, String, String, Option<String>, String, Vec<String>, i64, f32, i32, String)>(
|
|
||||||
r#"
|
|
||||||
SELECT ap.id, ap.name, ap.template, ap.target_model, ap.task_category,
|
|
||||||
ap.tags, ap.usage_count, ap.avg_quality, ap.version, ap.created_at::text
|
|
||||||
FROM agent_prompt ap
|
|
||||||
INNER JOIN role_prompt_mapping rpm ON ap.id = rpm.prompt_id
|
|
||||||
WHERE rpm.project_id = $1 AND rpm.role_name = $2 AND rpm.active = true
|
|
||||||
ORDER BY rpm.priority DESC, ap.created_at DESC
|
|
||||||
"#
|
|
||||||
)
|
|
||||||
.bind(&project_id)
|
|
||||||
.bind(&role_name)
|
|
||||||
.fetch_all(&state.pool)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
match prompts_query {
|
|
||||||
Ok(rows) => {
|
|
||||||
let prompts: Vec<PromptResponse> = rows.into_iter().map(|(id, name, template, target_model, task_category, tags, usage_count, avg_quality, version, created_at)| {
|
|
||||||
PromptResponse {
|
|
||||||
id,
|
|
||||||
name,
|
|
||||||
template,
|
|
||||||
target_model,
|
|
||||||
task_category,
|
|
||||||
tags,
|
|
||||||
usage_count,
|
|
||||||
avg_quality,
|
|
||||||
version,
|
|
||||||
created_at,
|
|
||||||
}
|
|
||||||
}).collect();
|
|
||||||
|
|
||||||
info!("Retrieved {} prompts for role {}", prompts.len(), role_name);
|
|
||||||
response_builder::success_response(RolePromptsResponse {
|
|
||||||
role_name,
|
|
||||||
prompts,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
error!("Failed to fetch role prompts: {}", e);
|
|
||||||
response_builder::internal_error("Failed to fetch role prompts")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -61,49 +61,6 @@ pub fn validate_and_rate_limit(
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract user identity from JWT claims (sub field)
|
|
||||||
///
|
|
||||||
/// Tries to decode JWT from Authorization header to get `sub` claim.
|
|
||||||
/// Falls back to "anonymous" if auth is disabled or header missing.
|
|
||||||
/// Used by metrics to track errors/requests per user.
|
|
||||||
pub fn extract_user_id(req: &HttpRequest, state: &AppState) -> String {
|
|
||||||
// If auth disabled, check synthetic claims
|
|
||||||
if state.jwt_validator.is_none() {
|
|
||||||
return "anonymous".to_string();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Try to extract sub from JWT
|
|
||||||
let token = req.headers()
|
|
||||||
.get("Authorization")
|
|
||||||
.and_then(|h| h.to_str().ok())
|
|
||||||
.and_then(|h| h.strip_prefix("Bearer "))
|
|
||||||
.unwrap_or("");
|
|
||||||
|
|
||||||
if token.is_empty() {
|
|
||||||
return "anonymous".to_string();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Decode JWT payload without validation (already validated by validate_and_rate_limit)
|
|
||||||
// JWT format: header.payload.signature
|
|
||||||
let parts: Vec<&str> = token.split('.').collect();
|
|
||||||
if parts.len() != 3 {
|
|
||||||
return "anonymous".to_string();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Decode base64 payload
|
|
||||||
use base64::Engine;
|
|
||||||
let engine = base64::engine::general_purpose::URL_SAFE_NO_PAD;
|
|
||||||
if let Ok(payload_bytes) = engine.decode(parts[1]) {
|
|
||||||
if let Ok(payload) = serde_json::from_slice::<serde_json::Value>(&payload_bytes) {
|
|
||||||
if let Some(sub) = payload.get("sub").and_then(|s| s.as_str()) {
|
|
||||||
return sub.to_string();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
"anonymous".to_string()
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|||||||
@@ -118,28 +118,17 @@ pub async fn unified_query_handler(
|
|||||||
body: web::Json<UnifiedQueryRequest>,
|
body: web::Json<UnifiedQueryRequest>,
|
||||||
state: web::Data<AppState>,
|
state: web::Data<AppState>,
|
||||||
) -> HttpResponse {
|
) -> HttpResponse {
|
||||||
use crate::metrics::*;
|
|
||||||
QUERY_REQUESTS_TOTAL.inc();
|
|
||||||
QUERY_IN_FLIGHT.inc();
|
|
||||||
let _timer = Timer::new(&QUERY_DURATION);
|
|
||||||
let start_time = std::time::Instant::now();
|
let start_time = std::time::Instant::now();
|
||||||
|
|
||||||
// 1. Validate JWT + rate limit
|
// 1. Validate JWT + rate limit
|
||||||
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
|
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
|
||||||
&req, &state, "query", 500
|
&req, &state, "query", 500
|
||||||
) {
|
) {
|
||||||
QUERY_AUTH_FAILURES.inc();
|
|
||||||
QUERY_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_AUTH_FAILURE_QUERY.inc();
|
|
||||||
QUERY_IN_FLIGHT.dec();
|
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. Validate input
|
// 2. Validate input
|
||||||
if let Err(response) = validate_unified_request(&body) {
|
if let Err(response) = validate_unified_request(&body) {
|
||||||
QUERY_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_BAD_REQUEST_QUERY.inc();
|
|
||||||
QUERY_IN_FLIGHT.dec();
|
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -147,17 +136,9 @@ pub async fn unified_query_handler(
|
|||||||
body.search_type, body.query, body.entity_type, body.relation_type);
|
body.search_type, body.query, body.entity_type, body.relation_type);
|
||||||
|
|
||||||
// 3. Embed query once (reused for all search types)
|
// 3. Embed query once (reused for all search types)
|
||||||
let embed_start = std::time::Instant::now();
|
|
||||||
let query_embedding = match state.embeddings.embed_one(&body.query).await {
|
let query_embedding = match state.embeddings.embed_one(&body.query).await {
|
||||||
Ok(emb) => {
|
Ok(emb) => emb.to_vec(),
|
||||||
QUERY_EMBEDDING_DURATION.observe(embed_start.elapsed().as_secs_f64());
|
|
||||||
emb.to_vec()
|
|
||||||
}
|
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
QUERY_EMBEDDING_FAILURES.inc();
|
|
||||||
QUERY_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_EMBEDDING_FAILURE_QUERY.inc();
|
|
||||||
QUERY_IN_FLIGHT.dec();
|
|
||||||
error!("Embedding failed: {}", e);
|
error!("Embedding failed: {}", e);
|
||||||
return crate::handlers::response_builder::internal_error(
|
return crate::handlers::response_builder::internal_error(
|
||||||
"Failed to embed query"
|
"Failed to embed query"
|
||||||
@@ -171,15 +152,12 @@ pub async fn unified_query_handler(
|
|||||||
"edges" => search_edges(&body, &state, &query_embedding, start_time).await,
|
"edges" => search_edges(&body, &state, &query_embedding, start_time).await,
|
||||||
"hybrid" => search_hybrid(&body, &state, &query_embedding, start_time).await,
|
"hybrid" => search_hybrid(&body, &state, &query_embedding, start_time).await,
|
||||||
_ => {
|
_ => {
|
||||||
QUERY_ERRORS_TOTAL.inc();
|
|
||||||
QUERY_IN_FLIGHT.dec();
|
|
||||||
return crate::handlers::response_builder::bad_request(
|
return crate::handlers::response_builder::bad_request(
|
||||||
"search_type must be 'entities', 'edges', or 'hybrid'"
|
"search_type must be 'entities', 'edges', or 'hybrid'"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
QUERY_IN_FLIGHT.dec();
|
|
||||||
response
|
response
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -203,9 +181,7 @@ async fn search_entities(
|
|||||||
).await {
|
).await {
|
||||||
Ok(r) => r,
|
Ok(r) => r,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
|
error!("Entity search failed: {}", e);
|
||||||
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
|
|
||||||
error!("Unexpected error: entity search failed: {}", e);
|
|
||||||
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
|
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -271,10 +247,6 @@ async fn search_entities(
|
|||||||
|
|
||||||
info!("Unified query (entities): {} results in {}ms", count, elapsed);
|
info!("Unified query (entities): {} results in {}ms", count, elapsed);
|
||||||
|
|
||||||
// O2: Track result counts
|
|
||||||
crate::metrics::QUERY_RESULTS_TOTAL.inc_by(count as u64);
|
|
||||||
if count == 0 { crate::metrics::QUERY_EMPTY_RESULTS.inc(); }
|
|
||||||
|
|
||||||
let response = UnifiedQueryResponse {
|
let response = UnifiedQueryResponse {
|
||||||
query: req.query.clone(),
|
query: req.query.clone(),
|
||||||
search_type: "entities".to_string(),
|
search_type: "entities".to_string(),
|
||||||
@@ -307,9 +279,7 @@ async fn search_edges(
|
|||||||
).await {
|
).await {
|
||||||
Ok(r) => r,
|
Ok(r) => r,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
|
error!("Edge search failed: {}", e);
|
||||||
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
|
|
||||||
error!("Unexpected error: edge search failed: {}", e);
|
|
||||||
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
|
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -335,9 +305,6 @@ async fn search_edges(
|
|||||||
|
|
||||||
info!("Unified query (edges): {} results in {}ms", count, elapsed);
|
info!("Unified query (edges): {} results in {}ms", count, elapsed);
|
||||||
|
|
||||||
crate::metrics::QUERY_RESULTS_TOTAL.inc_by(count as u64);
|
|
||||||
if count == 0 { crate::metrics::QUERY_EMPTY_RESULTS.inc(); }
|
|
||||||
|
|
||||||
let response = UnifiedQueryResponse {
|
let response = UnifiedQueryResponse {
|
||||||
query: req.query.clone(),
|
query: req.query.clone(),
|
||||||
search_type: "edges".to_string(),
|
search_type: "edges".to_string(),
|
||||||
@@ -371,9 +338,7 @@ async fn search_hybrid(
|
|||||||
).await {
|
).await {
|
||||||
Ok(r) => r,
|
Ok(r) => r,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
|
error!("Hybrid search failed: {}", e);
|
||||||
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
|
|
||||||
error!("Unexpected error: hybrid search failed: {}", e);
|
|
||||||
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
|
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -385,9 +350,6 @@ async fn search_hybrid(
|
|||||||
|
|
||||||
info!("Unified query (hybrid): {} results in {}ms", count, elapsed);
|
info!("Unified query (hybrid): {} results in {}ms", count, elapsed);
|
||||||
|
|
||||||
crate::metrics::QUERY_RESULTS_TOTAL.inc_by(count as u64);
|
|
||||||
if count == 0 { crate::metrics::QUERY_EMPTY_RESULTS.inc(); }
|
|
||||||
|
|
||||||
let response = UnifiedQueryResponse {
|
let response = UnifiedQueryResponse {
|
||||||
query: req.query.clone(),
|
query: req.query.clone(),
|
||||||
search_type: "hybrid".to_string(),
|
search_type: "hybrid".to_string(),
|
||||||
|
|||||||
@@ -374,30 +374,6 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
|
|||||||
});
|
});
|
||||||
|
|
||||||
tracing::info!("Starting HTTP server on port {}", port);
|
tracing::info!("Starting HTTP server on port {}", port);
|
||||||
|
|
||||||
// O5/O7/O9: Background stats collector (every 60s)
|
|
||||||
{
|
|
||||||
let stats_pool = state.get_ref().pool.clone();
|
|
||||||
tokio::spawn(async move {
|
|
||||||
let mut interval = tokio::time::interval(std::time::Duration::from_secs(60));
|
|
||||||
loop {
|
|
||||||
interval.tick().await;
|
|
||||||
// O5: Table row counts
|
|
||||||
if let Ok(row) = sqlx::query_as::<_, (i64,)>("SELECT COUNT(*) FROM memory_entity")
|
|
||||||
.fetch_one(&stats_pool).await {
|
|
||||||
crate::metrics::DB_TABLE_ENTITY_ROWS.set(row.0 as u64);
|
|
||||||
}
|
|
||||||
if let Ok(row) = sqlx::query_as::<_, (i64,)>("SELECT COUNT(*) FROM memory_edge")
|
|
||||||
.fetch_one(&stats_pool).await {
|
|
||||||
crate::metrics::DB_TABLE_EDGE_ROWS.set(row.0 as u64);
|
|
||||||
}
|
|
||||||
// O9: Pool stats
|
|
||||||
crate::metrics::DB_POOL_SIZE.set(stats_pool.size() as u64);
|
|
||||||
crate::metrics::DB_POOL_IDLE.set(stats_pool.num_idle() as u64);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
tracing::info!("Creating HttpServer instance...");
|
tracing::info!("Creating HttpServer instance...");
|
||||||
|
|
||||||
let server = HttpServer::new(move || {
|
let server = HttpServer::new(move || {
|
||||||
@@ -406,7 +382,6 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
|
|||||||
.app_data(state.clone())
|
.app_data(state.clone())
|
||||||
.wrap(Logger::default())
|
.wrap(Logger::default())
|
||||||
.route("/health", web::get().to(health_check))
|
.route("/health", web::get().to(health_check))
|
||||||
.route("/metrics", web::get().to(crate::metrics::metrics_handler))
|
|
||||||
.route("/memory/ingest", web::post().to(ingest_handler))
|
.route("/memory/ingest", web::post().to(ingest_handler))
|
||||||
.route("/memory/ingest/{ingest_id}", web::get().to(ingest_status))
|
.route("/memory/ingest/{ingest_id}", web::get().to(ingest_status))
|
||||||
.route("/memory/query", web::get().to(query_handler))
|
.route("/memory/query", web::get().to(query_handler))
|
||||||
@@ -440,9 +415,6 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
|
|||||||
.route("/agents/{id}", web::put().to(crate::handlers::agent_handler::update_agent_handler))
|
.route("/agents/{id}", web::put().to(crate::handlers::agent_handler::update_agent_handler))
|
||||||
.route("/agents/{id}", web::delete().to(crate::handlers::agent_handler::delete_agent_handler))
|
.route("/agents/{id}", web::delete().to(crate::handlers::agent_handler::delete_agent_handler))
|
||||||
.route("/agents/{id}/metrics", web::get().to(crate::handlers::agent_handler::get_agent_metrics_handler))
|
.route("/agents/{id}/metrics", web::get().to(crate::handlers::agent_handler::get_agent_metrics_handler))
|
||||||
.route("/memory/agents/{project_id}/prompts", web::post().to(crate::handlers::agent_handler::create_prompt_handler))
|
|
||||||
.route("/memory/agents/{project_id}/roles", web::post().to(crate::handlers::agent_handler::map_role_to_prompt_handler))
|
|
||||||
.route("/memory/agents/{project_id}/roles/{role_name}/prompts", web::get().to(crate::handlers::agent_handler::get_role_prompts_handler))
|
|
||||||
});
|
});
|
||||||
|
|
||||||
tracing::info!("HttpServer instance created, binding to 0.0.0.0:{}", port);
|
tracing::info!("HttpServer instance created, binding to 0.0.0.0:{}", port);
|
||||||
@@ -456,24 +428,7 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
|
|||||||
|
|
||||||
/// Health check (no auth)
|
/// Health check (no auth)
|
||||||
pub async fn health_check(state: web::Data<AppState>) -> HttpResponse {
|
pub async fn health_check(state: web::Data<AppState>) -> HttpResponse {
|
||||||
use crate::metrics::*;
|
|
||||||
HEALTH_CHECKS_TOTAL.inc();
|
|
||||||
let uptime = state.start_time.elapsed().as_secs();
|
let uptime = state.start_time.elapsed().as_secs();
|
||||||
APP_UPTIME_SECONDS.set(uptime);
|
|
||||||
|
|
||||||
// O7: Check DB dependency
|
|
||||||
let db_start = std::time::Instant::now();
|
|
||||||
match sqlx::query("SELECT 1").execute(&state.pool).await {
|
|
||||||
Ok(_) => {
|
|
||||||
DEP_DB_UP.set(1);
|
|
||||||
DEP_DB_LATENCY.observe(db_start.elapsed().as_secs_f64());
|
|
||||||
}
|
|
||||||
Err(_) => {
|
|
||||||
DEP_DB_UP.set(0);
|
|
||||||
HEALTH_CHECK_FAILURES.inc();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
HttpResponse::Ok().json(json!({"status": "ok", "uptime_seconds": uptime}))
|
HttpResponse::Ok().json(json!({"status": "ok", "uptime_seconds": uptime}))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -483,75 +438,35 @@ pub async fn ingest_handler(
|
|||||||
body: web::Json<IngestRequest>,
|
body: web::Json<IngestRequest>,
|
||||||
state: web::Data<AppState>,
|
state: web::Data<AppState>,
|
||||||
) -> HttpResponse {
|
) -> HttpResponse {
|
||||||
use crate::metrics::*;
|
|
||||||
INGEST_REQUESTS_TOTAL.inc();
|
|
||||||
INGEST_IN_FLIGHT.inc();
|
|
||||||
let _timer = Timer::new(&INGEST_DURATION);
|
|
||||||
|
|
||||||
// Auth + capability check
|
// Auth + capability check
|
||||||
let (claims, _token) = match validate_auth(&req, &state).await {
|
let (claims, _token) = match validate_auth(&req, &state).await {
|
||||||
Ok(c) => c,
|
Ok(c) => c,
|
||||||
Err(e) => {
|
Err(e) => return e,
|
||||||
INGEST_AUTH_FAILURES.inc();
|
|
||||||
INGEST_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_AUTH_FAILURE_INGEST.inc();
|
|
||||||
INGEST_IN_FLIGHT.dec();
|
|
||||||
return e;
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
|
|
||||||
let user_id = &claims.sub;
|
|
||||||
if !has_capability(&claims, "memory:write") {
|
if !has_capability(&claims, "memory:write") {
|
||||||
INGEST_AUTH_FAILURES.inc();
|
|
||||||
INGEST_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_FORBIDDEN_INGEST.inc();
|
|
||||||
INGEST_IN_FLIGHT.dec();
|
|
||||||
return HttpResponse::Forbidden().json(json!({
|
return HttpResponse::Forbidden().json(json!({
|
||||||
"error": "forbidden",
|
"error": "forbidden",
|
||||||
"reason": "missing capability: memory:write"
|
"reason": "missing capability: memory:write"
|
||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
if let Err(e) = check_rate_limit(&claims, &state, "/memory/ingest") {
|
if let Err(e) = check_rate_limit(&claims, &state, "/memory/ingest") {
|
||||||
INGEST_RATE_LIMITED.inc();
|
|
||||||
ERROR_RATE_LIMITED_INGEST.inc();
|
|
||||||
INGEST_IN_FLIGHT.dec();
|
|
||||||
return e;
|
return e;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Check idempotency
|
// Check idempotency
|
||||||
if let Some(cached) = state.idempotency_store.get(&body.ingest_id) {
|
if let Some(cached) = state.idempotency_store.get(&body.ingest_id) {
|
||||||
tracing::info!("Returning cached response for ingest_id: {}", body.ingest_id);
|
tracing::info!("Returning cached response for ingest_id: {}", body.ingest_id);
|
||||||
INGEST_DUPLICATES_TOTAL.inc();
|
|
||||||
INGEST_IN_FLIGHT.dec();
|
|
||||||
return HttpResponse::Accepted().json(cached);
|
return HttpResponse::Accepted().json(cached);
|
||||||
}
|
}
|
||||||
|
|
||||||
let byte_count: usize = body.records.iter().map(|r| r.text.len()).sum();
|
|
||||||
INGEST_BYTES_TOTAL.inc_by(byte_count as u64);
|
|
||||||
INGEST_RECORDS_TOTAL.inc_by(body.records.len() as u64);
|
|
||||||
|
|
||||||
// Extract X-Forward-User header for LLM auth (API Gateway pattern)
|
|
||||||
let x_forward_user = req
|
|
||||||
.headers()
|
|
||||||
.get("X-Forward-User")
|
|
||||||
.and_then(|h| h.to_str().ok())
|
|
||||||
.map(|s| s.to_string());
|
|
||||||
|
|
||||||
if let Some(ref user) = x_forward_user {
|
|
||||||
tracing::info!("Ingest request with X-Forward-User: {}", user);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Execute ingest
|
// Execute ingest
|
||||||
let resp = execute_ingest(&state, &body, x_forward_user).await;
|
execute_ingest(&state, &body).await
|
||||||
INGEST_IN_FLIGHT.dec();
|
|
||||||
resp
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Execute ingest job creation and spawn worker
|
/// Execute ingest job creation and spawn worker
|
||||||
async fn execute_ingest(
|
async fn execute_ingest(
|
||||||
state: &web::Data<AppState>,
|
state: &web::Data<AppState>,
|
||||||
body: &IngestRequest,
|
body: &IngestRequest,
|
||||||
x_forward_user: Option<String>,
|
|
||||||
) -> HttpResponse {
|
) -> HttpResponse {
|
||||||
let records: Vec<(String, String)> = body.records
|
let records: Vec<(String, String)> = body.records
|
||||||
.iter()
|
.iter()
|
||||||
@@ -582,9 +497,8 @@ async fn execute_ingest(
|
|||||||
let worker = state.ingest_worker.clone();
|
let worker = state.ingest_worker.clone();
|
||||||
let project = body.project.clone();
|
let project = body.project.clone();
|
||||||
let ingest_id = body.ingest_id.clone();
|
let ingest_id = body.ingest_id.clone();
|
||||||
let x_fwd = x_forward_user.clone();
|
|
||||||
tokio::spawn(async move {
|
tokio::spawn(async move {
|
||||||
if let Err(e) = worker.process_ingest_with_auth(&project, &ingest_id, records, x_fwd).await {
|
if let Err(e) = worker.process_ingest(&project, &ingest_id, records).await {
|
||||||
tracing::error!("Ingest failed: {}", e);
|
tracing::error!("Ingest failed: {}", e);
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
@@ -597,9 +511,7 @@ async fn execute_ingest(
|
|||||||
HttpResponse::Accepted().json(response)
|
HttpResponse::Accepted().json(response)
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
crate::metrics::ERROR_UNEXPECTED_INGEST.inc();
|
tracing::error!("DB error: {}", e);
|
||||||
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
|
|
||||||
tracing::error!(user_id = body.project.as_str(), "Unexpected DB error during ingest: {}", e);
|
|
||||||
HttpResponse::InternalServerError().json(json!({"error": "database_error"}))
|
HttpResponse::InternalServerError().json(json!({"error": "database_error"}))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -856,13 +768,8 @@ async fn store_compacted_memory(
|
|||||||
.await;
|
.await;
|
||||||
|
|
||||||
match result {
|
match result {
|
||||||
Ok(_) => {
|
Ok(_) => true,
|
||||||
crate::metrics::WRITE_CHUNKS_TOTAL.inc();
|
|
||||||
crate::metrics::WRITE_BYTES_TOTAL.inc_by(memory.len() as u64);
|
|
||||||
true
|
|
||||||
}
|
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
crate::metrics::WRITE_ERRORS_TOTAL.inc();
|
|
||||||
tracing::error!("Failed to store compacted memory: {}", e);
|
tracing::error!("Failed to store compacted memory: {}", e);
|
||||||
false
|
false
|
||||||
}
|
}
|
||||||
@@ -923,9 +830,7 @@ pub async fn query_handler(
|
|||||||
match query_temporal_graph(&state, ¶ms).await {
|
match query_temporal_graph(&state, ¶ms).await {
|
||||||
Ok(response) => HttpResponse::Ok().json(response),
|
Ok(response) => HttpResponse::Ok().json(response),
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
|
tracing::error!("Temporal graph query failed: {}", e);
|
||||||
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
|
|
||||||
tracing::error!(user_id = claims.sub.as_str(), "Unexpected error: temporal graph query failed: {}", e);
|
|
||||||
HttpResponse::InternalServerError().json(json!({"error": "query_failed", "reason": e.to_string()}))
|
HttpResponse::InternalServerError().json(json!({"error": "query_failed", "reason": e.to_string()}))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1061,23 +966,13 @@ pub async fn context_handler(
|
|||||||
body: web::Json<crate::context_endpoint::ContextRequest>,
|
body: web::Json<crate::context_endpoint::ContextRequest>,
|
||||||
state: web::Data<AppState>,
|
state: web::Data<AppState>,
|
||||||
) -> HttpResponse {
|
) -> HttpResponse {
|
||||||
use crate::metrics::*;
|
|
||||||
CONTEXT_REQUESTS_TOTAL.inc();
|
|
||||||
let _timer = Timer::new(&CONTEXT_DURATION);
|
|
||||||
|
|
||||||
let (claims, _token) = match validate_auth(&req, &state).await {
|
let (claims, _token) = match validate_auth(&req, &state).await {
|
||||||
Ok(c) => c,
|
Ok(c) => c,
|
||||||
Err(e) => {
|
Err(e) => return e,
|
||||||
CONTEXT_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_AUTH_FAILURE_CONTEXT.inc();
|
|
||||||
return e;
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
|
|
||||||
let user_id = &claims.sub;
|
// Check read capability
|
||||||
if !has_capability(&claims, "memory:read") {
|
if !has_capability(&claims, "memory:read") {
|
||||||
CONTEXT_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_FORBIDDEN_CONTEXT.inc();
|
|
||||||
return HttpResponse::Forbidden().json(json!({
|
return HttpResponse::Forbidden().json(json!({
|
||||||
"error": "forbidden",
|
"error": "forbidden",
|
||||||
"reason": "missing capability: memory:read"
|
"reason": "missing capability: memory:read"
|
||||||
@@ -1102,14 +997,9 @@ pub async fn context_handler(
|
|||||||
skills = response.skills.len(),
|
skills = response.skills.len(),
|
||||||
"context lookup successful"
|
"context lookup successful"
|
||||||
);
|
);
|
||||||
// O3: Track tier hits
|
|
||||||
let total = response.lessons.len() + response.skills.len();
|
|
||||||
if total == 0 { CONTEXT_EMPTY_RESULTS.inc(); }
|
|
||||||
HttpResponse::Ok().json(response)
|
HttpResponse::Ok().json(response)
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
CONTEXT_ERRORS_TOTAL.inc();
|
|
||||||
ERROR_LOOKUP_FAILURE_CONTEXT.inc();
|
|
||||||
tracing::error!("context lookup error: {}", e);
|
tracing::error!("context lookup error: {}", e);
|
||||||
HttpResponse::BadRequest().json(json!({
|
HttpResponse::BadRequest().json(json!({
|
||||||
"error": "lookup_failed",
|
"error": "lookup_failed",
|
||||||
|
|||||||
@@ -2,15 +2,14 @@ use anyhow::Result;
|
|||||||
use mem_store::{MemoryL1, VectorStore, ChunkL0, EntityRepoOps, EdgeRepoOps};
|
use mem_store::{MemoryL1, VectorStore, ChunkL0, EntityRepoOps, EdgeRepoOps};
|
||||||
use mem_llm::EmbeddingsClient;
|
use mem_llm::EmbeddingsClient;
|
||||||
use mem_ingest::ingest_pipeline::{IngestPipeline, Episode};
|
use mem_ingest::ingest_pipeline::{IngestPipeline, Episode};
|
||||||
use mem_ingest::entity_extractor::{WikiLinkFallbackExtractor, LlmEntityExtractor};
|
use mem_ingest::entity_extractor::WikiLinkFallbackExtractor;
|
||||||
use mem_ingest::fact_extractor::{SimpleFactExtractor, LlmFactExtractor};
|
use mem_ingest::fact_extractor::SimpleFactExtractor;
|
||||||
use mem_ingest::contradiction_detector::ContradictionHandler;
|
use mem_ingest::contradiction_detector::ContradictionHandler;
|
||||||
use sqlx::PgPool;
|
use sqlx::PgPool;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use pgvector::Vector;
|
use pgvector::Vector;
|
||||||
|
|
||||||
|
|
||||||
/// Ingest worker — processes queued records through entity/fact extraction pipeline
|
/// Ingest worker — processes queued records through entity/fact extraction pipeline
|
||||||
pub struct IngestWorker {
|
pub struct IngestWorker {
|
||||||
pool: PgPool,
|
pool: PgPool,
|
||||||
@@ -27,25 +26,11 @@ impl IngestWorker {
|
|||||||
) -> Self {
|
) -> Self {
|
||||||
let vector_store = Arc::new(VectorStore::new(pool.clone()));
|
let vector_store = Arc::new(VectorStore::new(pool.clone()));
|
||||||
|
|
||||||
// Initialize extraction pipeline — use LLM if LLM_ENDPOINT is set, else fallback to wiki links
|
// Initialize extraction pipeline
|
||||||
let entity_extractor: Arc<dyn mem_ingest::entity_extractor::EntityExtractor> =
|
let entity_extractor: Arc<dyn mem_ingest::entity_extractor::EntityExtractor> =
|
||||||
if std::env::var("LLM_ENDPOINT").is_ok() {
|
Arc::new(WikiLinkFallbackExtractor);
|
||||||
let model = std::env::var("LLM_MODEL").unwrap_or_else(|_| "qwen2.5:3b-instruct".to_string());
|
|
||||||
tracing::info!("Using LLM entity extractor: model={}", model);
|
|
||||||
Arc::new(LlmEntityExtractor::new(&model))
|
|
||||||
} else {
|
|
||||||
tracing::info!("LLM_ENDPOINT not set, using WikiLink fallback extractor");
|
|
||||||
Arc::new(WikiLinkFallbackExtractor)
|
|
||||||
};
|
|
||||||
let fact_extractor: Arc<dyn mem_ingest::fact_extractor::FactExtractor> =
|
let fact_extractor: Arc<dyn mem_ingest::fact_extractor::FactExtractor> =
|
||||||
if std::env::var("LLM_ENDPOINT").is_ok() {
|
Arc::new(SimpleFactExtractor);
|
||||||
let model = std::env::var("LLM_MODEL").unwrap_or_else(|_| "qwen2.5:3b-instruct".to_string());
|
|
||||||
tracing::info!("Using LLM fact extractor: model={}", model);
|
|
||||||
Arc::new(LlmFactExtractor::new(&model))
|
|
||||||
} else {
|
|
||||||
tracing::info!("LLM_ENDPOINT not set, using simple pattern fact extractor");
|
|
||||||
Arc::new(SimpleFactExtractor)
|
|
||||||
};
|
|
||||||
let contradiction_detector = Arc::new(ContradictionHandler::default());
|
let contradiction_detector = Arc::new(ContradictionHandler::default());
|
||||||
let pipeline = Arc::new(IngestPipeline::new(
|
let pipeline = Arc::new(IngestPipeline::new(
|
||||||
entity_extractor,
|
entity_extractor,
|
||||||
@@ -68,201 +53,77 @@ impl IngestWorker {
|
|||||||
ingest_id: &str,
|
ingest_id: &str,
|
||||||
records: Vec<(String, String)>, // (content, source)
|
records: Vec<(String, String)>, // (content, source)
|
||||||
) -> Result<()> {
|
) -> Result<()> {
|
||||||
self.process_ingest_with_auth(project, ingest_id, records, None).await
|
tracing::info!("Processing ingest: project={}, id={}, records={}", project, ingest_id, records.len());
|
||||||
}
|
|
||||||
|
|
||||||
/// Process ingest with optional X-Forward-User auth header (API Gateway pattern)
|
|
||||||
pub async fn process_ingest_with_auth(
|
|
||||||
&self,
|
|
||||||
project: &str,
|
|
||||||
ingest_id: &str,
|
|
||||||
records: Vec<(String, String)>, // (content, source)
|
|
||||||
x_forward_user: Option<String>,
|
|
||||||
) -> Result<()> {
|
|
||||||
tracing::info!(
|
|
||||||
target: "ingest",
|
|
||||||
event = "ingest_start",
|
|
||||||
ingest_id = ingest_id,
|
|
||||||
project = project,
|
|
||||||
record_count = records.len(),
|
|
||||||
"Starting ingest job"
|
|
||||||
);
|
|
||||||
|
|
||||||
// Update job status to processing
|
// Update job status to processing
|
||||||
if let Err(e) = sqlx::query("UPDATE ingest_jobs SET status=$1, started_at=NOW() WHERE ingest_id=$2")
|
sqlx::query("UPDATE ingest_jobs SET status=$1, started_at=NOW() WHERE ingest_id=$2")
|
||||||
.bind("processing")
|
.bind("processing")
|
||||||
.bind(ingest_id)
|
.bind(ingest_id)
|
||||||
.execute(&self.pool)
|
.execute(&self.pool)
|
||||||
.await
|
.await?;
|
||||||
{
|
|
||||||
tracing::error!(
|
|
||||||
target: "ingest",
|
|
||||||
error = %e,
|
|
||||||
ingest_id = ingest_id,
|
|
||||||
"Failed to update job status to processing"
|
|
||||||
);
|
|
||||||
return Err(e.into());
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut total_entities = 0;
|
let mut total_entities = 0;
|
||||||
let mut total_edges = 0;
|
let mut total_edges = 0;
|
||||||
let mut total_reviews = 0;
|
let mut total_reviews = 0;
|
||||||
let mut extraction_errors = Vec::new();
|
|
||||||
let mut save_errors = Vec::new();
|
|
||||||
|
|
||||||
// Process each record through the ingest pipeline
|
// Process each record through the ingest pipeline
|
||||||
for (idx, (content, source)) in records.iter().enumerate() {
|
for (idx, (content, source)) in records.iter().enumerate() {
|
||||||
let record_id = format!("{}-{}", ingest_id, idx);
|
|
||||||
tracing::debug!(
|
|
||||||
target: "ingest",
|
|
||||||
record_id = %record_id,
|
|
||||||
source = source,
|
|
||||||
content_len = content.len(),
|
|
||||||
"Processing record"
|
|
||||||
);
|
|
||||||
|
|
||||||
// Create episode from record
|
// Create episode from record
|
||||||
let episode = Episode {
|
let episode = Episode {
|
||||||
id: record_id.clone(),
|
id: format!("{}-{}", ingest_id, idx),
|
||||||
project_id: project.to_string(),
|
project_id: project.to_string(),
|
||||||
text: content.clone(),
|
text: content.clone(),
|
||||||
wiki_links: extract_wiki_links(content),
|
wiki_links: extract_wiki_links(content),
|
||||||
};
|
};
|
||||||
|
|
||||||
// Run extraction pipeline (entity + fact extraction + contradiction detection)
|
// Run extraction pipeline (entity + fact extraction + contradiction detection)
|
||||||
let x_forward_user_ref = x_forward_user.as_deref();
|
match self.pipeline.ingest(&episode).await {
|
||||||
match self.pipeline.ingest_with_auth(&episode, x_forward_user_ref).await {
|
|
||||||
Ok(result) => {
|
Ok(result) => {
|
||||||
tracing::debug!(
|
tracing::debug!(
|
||||||
target: "ingest",
|
"Pipeline extracted {} entities, {} edges for episode {}",
|
||||||
record_id = %record_id,
|
result.entities.len(),
|
||||||
entity_count = result.entities.len(),
|
result.edges.len(),
|
||||||
edge_count = result.edges.len(),
|
episode.id
|
||||||
review_count = result.reviews.len(),
|
|
||||||
"Pipeline extraction successful"
|
|
||||||
);
|
);
|
||||||
|
|
||||||
// Save entities to database (normally via EntityRepo, using direct SQL for now)
|
// Save entities to database (normally via EntityRepo, using direct SQL for now)
|
||||||
for entity in &result.entities {
|
for entity in &result.entities {
|
||||||
match save_entity_to_db(&self.pool, entity).await {
|
if let Err(e) = save_entity_to_db(&self.pool, entity).await {
|
||||||
Ok(_) => {
|
tracing::warn!("Failed to save entity {}: {}", entity.name, e);
|
||||||
tracing::debug!(
|
} else {
|
||||||
target: "ingest",
|
total_entities += 1;
|
||||||
record_id = %record_id,
|
|
||||||
entity_name = &entity.name,
|
|
||||||
entity_type = entity.entity_type.as_str(),
|
|
||||||
"Saved entity"
|
|
||||||
);
|
|
||||||
total_entities += 1;
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
let msg = format!("Failed to save entity '{}': {}", entity.name, e);
|
|
||||||
tracing::warn!(
|
|
||||||
target: "ingest",
|
|
||||||
error = %e,
|
|
||||||
record_id = %record_id,
|
|
||||||
entity_name = &entity.name,
|
|
||||||
"Entity save failed"
|
|
||||||
);
|
|
||||||
save_errors.push(msg);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Save edges to database (normally via EdgeRepo, using direct SQL for now)
|
// Save edges to database (normally via EdgeRepo, using direct SQL for now)
|
||||||
for edge in &result.edges {
|
for edge in &result.edges {
|
||||||
match save_edge_to_db(&self.pool, edge).await {
|
if let Err(e) = save_edge_to_db(&self.pool, edge).await {
|
||||||
Ok(_) => {
|
tracing::warn!("Failed to save edge: {}", e);
|
||||||
tracing::debug!(
|
} else {
|
||||||
target: "ingest",
|
total_edges += 1;
|
||||||
record_id = %record_id,
|
|
||||||
relation_type = &edge.relation_type,
|
|
||||||
"Saved edge"
|
|
||||||
);
|
|
||||||
total_edges += 1;
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
let msg = format!("Failed to save edge: {}", e);
|
|
||||||
tracing::warn!(
|
|
||||||
target: "ingest",
|
|
||||||
error = %e,
|
|
||||||
record_id = %record_id,
|
|
||||||
"Edge save failed"
|
|
||||||
);
|
|
||||||
save_errors.push(msg);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
total_reviews += result.reviews.len();
|
total_reviews += result.reviews.len();
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
let msg = format!("Record {}: {}", record_id, e);
|
tracing::error!("Pipeline failed for episode {}: {}", episode.id, e);
|
||||||
tracing::error!(
|
|
||||||
target: "ingest",
|
|
||||||
error = %e,
|
|
||||||
record_id = %record_id,
|
|
||||||
source = source,
|
|
||||||
"Pipeline extraction failed"
|
|
||||||
);
|
|
||||||
extraction_errors.push(msg);
|
|
||||||
// Continue processing other records
|
// Continue processing other records
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Mark job complete
|
// Mark job complete
|
||||||
let final_status = if extraction_errors.is_empty() && save_errors.is_empty() {
|
sqlx::query("UPDATE ingest_jobs SET status=$1, completed_at=NOW() WHERE ingest_id=$2")
|
||||||
"done"
|
.bind("done")
|
||||||
} else {
|
|
||||||
"done_with_errors"
|
|
||||||
};
|
|
||||||
|
|
||||||
if let Err(e) = sqlx::query("UPDATE ingest_jobs SET status=$1, completed_at=NOW() WHERE ingest_id=$2")
|
|
||||||
.bind(final_status)
|
|
||||||
.bind(ingest_id)
|
.bind(ingest_id)
|
||||||
.execute(&self.pool)
|
.execute(&self.pool)
|
||||||
.await
|
.await?;
|
||||||
{
|
|
||||||
tracing::error!(
|
|
||||||
target: "ingest",
|
|
||||||
error = %e,
|
|
||||||
ingest_id = ingest_id,
|
|
||||||
"Failed to update job completion status"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
tracing::info!(
|
tracing::info!(
|
||||||
target: "ingest",
|
"Ingest completed: {} (entities={}, edges={}, reviews={})",
|
||||||
event = "ingest_complete",
|
ingest_id, total_entities, total_edges, total_reviews
|
||||||
ingest_id = ingest_id,
|
|
||||||
project = project,
|
|
||||||
entities = total_entities,
|
|
||||||
edges = total_edges,
|
|
||||||
reviews = total_reviews,
|
|
||||||
extraction_errors = extraction_errors.len(),
|
|
||||||
save_errors = save_errors.len(),
|
|
||||||
status = final_status,
|
|
||||||
"Ingest job completed"
|
|
||||||
);
|
);
|
||||||
|
|
||||||
if !extraction_errors.is_empty() {
|
|
||||||
tracing::warn!(
|
|
||||||
target: "ingest",
|
|
||||||
errors = ?extraction_errors,
|
|
||||||
ingest_id = ingest_id,
|
|
||||||
"Extraction errors occurred during ingest"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if !save_errors.is_empty() {
|
|
||||||
tracing::warn!(
|
|
||||||
target: "ingest",
|
|
||||||
errors = ?save_errors,
|
|
||||||
ingest_id = ingest_id,
|
|
||||||
"Save errors occurred during ingest"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -312,12 +173,7 @@ async fn save_entity_to_db(pool: &PgPool, entity: &mem_core::entity::Entity) ->
|
|||||||
sqlx::query(
|
sqlx::query(
|
||||||
"INSERT INTO memory_entity (id, project_id, name, entity_type, description, t_created, t_updated, confidence)
|
"INSERT INTO memory_entity (id, project_id, name, entity_type, description, t_created, t_updated, confidence)
|
||||||
VALUES ($1, $2, $3, $4, $5, $6::TIMESTAMPTZ, $7::TIMESTAMPTZ, $8)
|
VALUES ($1, $2, $3, $4, $5, $6::TIMESTAMPTZ, $7::TIMESTAMPTZ, $8)
|
||||||
ON CONFLICT (project_id, name) DO UPDATE SET
|
ON CONFLICT (id) DO NOTHING"
|
||||||
entity_type = EXCLUDED.entity_type,
|
|
||||||
description = COALESCE(NULLIF(EXCLUDED.description, ''), memory_entity.description),
|
|
||||||
t_updated = NOW(),
|
|
||||||
confidence = GREATEST(memory_entity.confidence, EXCLUDED.confidence),
|
|
||||||
source_count = memory_entity.source_count + 1"
|
|
||||||
)
|
)
|
||||||
.bind(&entity.id)
|
.bind(&entity.id)
|
||||||
.bind(&entity.project_id)
|
.bind(&entity.project_id)
|
||||||
@@ -337,7 +193,7 @@ async fn save_entity_to_db(pool: &PgPool, entity: &mem_core::entity::Entity) ->
|
|||||||
async fn save_edge_to_db(pool: &PgPool, edge: &mem_core::edge::Edge) -> Result<()> {
|
async fn save_edge_to_db(pool: &PgPool, edge: &mem_core::edge::Edge) -> Result<()> {
|
||||||
// Try temporal schema first (id, project_id, source_entity_id, etc)
|
// Try temporal schema first (id, project_id, source_entity_id, etc)
|
||||||
let result = sqlx::query(
|
let result = sqlx::query(
|
||||||
"INSERT INTO memory_edge (id, project_id, source_id, target_id, relation_type, fact, t_valid, t_invalid, t_created, confidence)
|
"INSERT INTO memory_edge (id, project_id, source_entity_id, target_entity_id, relation_type, fact, t_valid, t_invalid, t_created, confidence)
|
||||||
VALUES ($1, $2, $3, $4, $5, $6, $7::TIMESTAMPTZ, $8::TIMESTAMPTZ, $9::TIMESTAMPTZ, $10)
|
VALUES ($1, $2, $3, $4, $5, $6, $7::TIMESTAMPTZ, $8::TIMESTAMPTZ, $9::TIMESTAMPTZ, $10)
|
||||||
ON CONFLICT (id) DO NOTHING"
|
ON CONFLICT (id) DO NOTHING"
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -1,9 +1,6 @@
|
|||||||
pub mod endpoints;
|
pub mod endpoints;
|
||||||
pub mod handlers;
|
pub mod handlers;
|
||||||
pub mod http_server;
|
pub mod http_server;
|
||||||
pub mod metrics;
|
|
||||||
pub mod metrics_snapshot;
|
|
||||||
pub mod relevance_judge;
|
|
||||||
pub mod query;
|
pub mod query;
|
||||||
pub mod auth;
|
pub mod auth;
|
||||||
pub mod ingest_worker;
|
pub mod ingest_worker;
|
||||||
|
|||||||
@@ -1,686 +0,0 @@
|
|||||||
//! Prometheus metrics module (O10)
|
|
||||||
//!
|
|
||||||
//! Centralized metrics registry for poimen-memory observability.
|
|
||||||
//! All handlers instrument via these shared metrics.
|
|
||||||
//! Exposed at GET /metrics in Prometheus text format.
|
|
||||||
|
|
||||||
use once_cell::sync::Lazy;
|
|
||||||
use std::sync::atomic::{AtomicU64, Ordering};
|
|
||||||
use std::collections::HashMap;
|
|
||||||
use std::sync::Mutex;
|
|
||||||
use std::time::Instant;
|
|
||||||
|
|
||||||
// ─── Metric Types ───────────────────────────────────────────
|
|
||||||
|
|
||||||
/// Simple counter (monotonically increasing)
|
|
||||||
pub struct Counter {
|
|
||||||
value: AtomicU64,
|
|
||||||
name: &'static str,
|
|
||||||
help: &'static str,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Counter {
|
|
||||||
pub const fn new(name: &'static str, help: &'static str) -> Self {
|
|
||||||
Self { value: AtomicU64::new(0), name, help }
|
|
||||||
}
|
|
||||||
pub fn inc(&self) { self.value.fetch_add(1, Ordering::Relaxed); }
|
|
||||||
pub fn inc_by(&self, n: u64) { self.value.fetch_add(n, Ordering::Relaxed); }
|
|
||||||
pub fn get(&self) -> u64 { self.value.load(Ordering::Relaxed) }
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Gauge (can go up and down)
|
|
||||||
pub struct Gauge {
|
|
||||||
value: AtomicU64,
|
|
||||||
name: &'static str,
|
|
||||||
help: &'static str,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Gauge {
|
|
||||||
pub const fn new(name: &'static str, help: &'static str) -> Self {
|
|
||||||
Self { value: AtomicU64::new(0), name, help }
|
|
||||||
}
|
|
||||||
pub fn set(&self, v: u64) { self.value.store(v, Ordering::Relaxed); }
|
|
||||||
pub fn inc(&self) { self.value.fetch_add(1, Ordering::Relaxed); }
|
|
||||||
pub fn dec(&self) { self.value.fetch_sub(1, Ordering::Relaxed); }
|
|
||||||
pub fn get(&self) -> u64 { self.value.load(Ordering::Relaxed) }
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Gauge for f64 values (stored as bits)
|
|
||||||
pub struct GaugeF64 {
|
|
||||||
bits: AtomicU64,
|
|
||||||
name: &'static str,
|
|
||||||
help: &'static str,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl GaugeF64 {
|
|
||||||
pub const fn new(name: &'static str, help: &'static str) -> Self {
|
|
||||||
Self { bits: AtomicU64::new(0), name, help }
|
|
||||||
}
|
|
||||||
pub fn set(&self, v: f64) { self.bits.store(v.to_bits(), Ordering::Relaxed); }
|
|
||||||
pub fn get(&self) -> f64 { f64::from_bits(self.bits.load(Ordering::Relaxed)) }
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Histogram with fixed buckets for latency tracking
|
|
||||||
pub struct Histogram {
|
|
||||||
pub buckets: &'static [f64],
|
|
||||||
pub counts: Vec<AtomicU64>,
|
|
||||||
pub sum: AtomicU64, // stored as f64 bits
|
|
||||||
pub count: AtomicU64,
|
|
||||||
pub name: &'static str,
|
|
||||||
pub help: &'static str,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Histogram {
|
|
||||||
pub fn new(name: &'static str, help: &'static str, buckets: &'static [f64]) -> Self {
|
|
||||||
let counts = (0..buckets.len() + 1).map(|_| AtomicU64::new(0)).collect();
|
|
||||||
Self {
|
|
||||||
buckets, counts, name, help,
|
|
||||||
sum: AtomicU64::new(0f64.to_bits()),
|
|
||||||
count: AtomicU64::new(0),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn observe(&self, value: f64) {
|
|
||||||
self.count.fetch_add(1, Ordering::Relaxed);
|
|
||||||
// Add to sum (CAS loop for f64)
|
|
||||||
loop {
|
|
||||||
let old_bits = self.sum.load(Ordering::Relaxed);
|
|
||||||
let old = f64::from_bits(old_bits);
|
|
||||||
let new = old + value;
|
|
||||||
if self.sum.compare_exchange(old_bits, new.to_bits(), Ordering::Relaxed, Ordering::Relaxed).is_ok() {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Increment bucket counters
|
|
||||||
for (i, &bound) in self.buckets.iter().enumerate() {
|
|
||||||
if value <= bound {
|
|
||||||
self.counts[i].fetch_add(1, Ordering::Relaxed);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// +Inf bucket
|
|
||||||
self.counts[self.buckets.len()].fetch_add(1, Ordering::Relaxed);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Labeled counter (key = label combination string)
|
|
||||||
pub struct LabeledCounter {
|
|
||||||
values: Mutex<HashMap<String, u64>>,
|
|
||||||
name: &'static str,
|
|
||||||
help: &'static str,
|
|
||||||
label_names: &'static [&'static str],
|
|
||||||
}
|
|
||||||
|
|
||||||
impl LabeledCounter {
|
|
||||||
pub fn new(name: &'static str, help: &'static str, label_names: &'static [&'static str]) -> Self {
|
|
||||||
Self { values: Mutex::new(HashMap::new()), name, help, label_names }
|
|
||||||
}
|
|
||||||
pub fn inc(&self, labels: &[&str]) {
|
|
||||||
let key = labels.join(",");
|
|
||||||
let mut map = self.values.lock().unwrap();
|
|
||||||
*map.entry(key).or_insert(0) += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ─── Timer helper ───────────────────────────────────────────
|
|
||||||
|
|
||||||
/// RAII timer: observes duration on drop
|
|
||||||
pub struct Timer<'a> {
|
|
||||||
histogram: &'a Histogram,
|
|
||||||
start: Instant,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl<'a> Timer<'a> {
|
|
||||||
pub fn new(histogram: &'a Histogram) -> Self {
|
|
||||||
Self { histogram, start: Instant::now() }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl<'a> Drop for Timer<'a> {
|
|
||||||
fn drop(&mut self) {
|
|
||||||
let elapsed = self.start.elapsed().as_secs_f64();
|
|
||||||
self.histogram.observe(elapsed);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ─── Default buckets ────────────────────────────────────────
|
|
||||||
|
|
||||||
/// Latency buckets for HTTP handlers (seconds)
|
|
||||||
pub static HTTP_BUCKETS: &[f64] = &[0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0, 10.0];
|
|
||||||
/// Latency buckets for LLM calls (seconds)
|
|
||||||
pub static LLM_BUCKETS: &[f64] = &[0.1, 0.25, 0.5, 1.0, 2.5, 5.0, 10.0, 30.0, 60.0];
|
|
||||||
/// Latency buckets for DB queries (seconds)
|
|
||||||
pub static DB_BUCKETS: &[f64] = &[0.001, 0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0];
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O1: Ingest handler metrics (I1-I12)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static INGEST_REQUESTS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_requests_total", "Total ingest requests received");
|
|
||||||
pub static INGEST_ERRORS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_errors_total", "Total ingest request errors");
|
|
||||||
pub static INGEST_RECORDS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_records_total", "Total records ingested");
|
|
||||||
pub static INGEST_ENTITIES_EXTRACTED: Counter = Counter::new(
|
|
||||||
"memory_ingest_entities_extracted_total", "Total entities extracted during ingest");
|
|
||||||
pub static INGEST_EDGES_EXTRACTED: Counter = Counter::new(
|
|
||||||
"memory_ingest_edges_extracted_total", "Total edges extracted during ingest");
|
|
||||||
pub static INGEST_IN_FLIGHT: Gauge = Gauge::new(
|
|
||||||
"memory_ingest_in_flight", "Currently processing ingest jobs");
|
|
||||||
pub static INGEST_QUEUE_SIZE: Gauge = Gauge::new(
|
|
||||||
"memory_ingest_queue_size", "Number of jobs waiting in ingest queue");
|
|
||||||
pub static INGEST_DUPLICATES_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_duplicates_total", "Total duplicate ingest requests (idempotency)");
|
|
||||||
pub static INGEST_BYTES_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_bytes_total", "Total bytes ingested");
|
|
||||||
pub static INGEST_AUTH_FAILURES: Counter = Counter::new(
|
|
||||||
"memory_ingest_auth_failures_total", "Total auth failures on ingest endpoint");
|
|
||||||
pub static INGEST_RATE_LIMITED: Counter = Counter::new(
|
|
||||||
"memory_ingest_rate_limited_total", "Total rate-limited ingest requests");
|
|
||||||
|
|
||||||
pub static INGEST_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_ingest_duration_seconds", "Ingest request duration", HTTP_BUCKETS));
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O2: Query handler metrics (Q1-Q12)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static QUERY_REQUESTS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_query_requests_total", "Total query requests received");
|
|
||||||
pub static QUERY_ERRORS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_query_errors_total", "Total query request errors");
|
|
||||||
pub static QUERY_RESULTS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_query_results_total", "Total results returned across all queries");
|
|
||||||
pub static QUERY_EMPTY_RESULTS: Counter = Counter::new(
|
|
||||||
"memory_query_empty_results_total", "Queries returning zero results");
|
|
||||||
pub static QUERY_EMBEDDING_FAILURES: Counter = Counter::new(
|
|
||||||
"memory_query_embedding_failures_total", "Total embedding failures during query");
|
|
||||||
pub static QUERY_IN_FLIGHT: Gauge = Gauge::new(
|
|
||||||
"memory_query_in_flight", "Currently processing queries");
|
|
||||||
pub static QUERY_AUTH_FAILURES: Counter = Counter::new(
|
|
||||||
"memory_query_auth_failures_total", "Total auth failures on query endpoint");
|
|
||||||
pub static QUERY_RATE_LIMITED: Counter = Counter::new(
|
|
||||||
"memory_query_rate_limited_total", "Total rate-limited query requests");
|
|
||||||
pub static QUERY_CACHE_HITS: Counter = Counter::new(
|
|
||||||
"memory_query_cache_hits_total", "Total query cache hits");
|
|
||||||
pub static QUERY_CACHE_MISSES: Counter = Counter::new(
|
|
||||||
"memory_query_cache_misses_total", "Total query cache misses");
|
|
||||||
|
|
||||||
pub static QUERY_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_query_duration_seconds", "Query request duration", HTTP_BUCKETS));
|
|
||||||
pub static QUERY_EMBEDDING_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_query_embedding_duration_seconds", "Embedding call duration during query", LLM_BUCKETS));
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O3: Context endpoint metrics (C1-C8)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static CONTEXT_REQUESTS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_context_requests_total", "Total context retrieval requests");
|
|
||||||
pub static CONTEXT_ERRORS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_context_errors_total", "Total context retrieval errors");
|
|
||||||
pub static CONTEXT_SEMANTIC_HITS: Counter = Counter::new(
|
|
||||||
"memory_context_semantic_hits_total", "Results from semantic (cosine) tier");
|
|
||||||
pub static CONTEXT_BM25_HITS: Counter = Counter::new(
|
|
||||||
"memory_context_bm25_hits_total", "Results from BM25 (lexical) tier");
|
|
||||||
pub static CONTEXT_GRAPH_HITS: Counter = Counter::new(
|
|
||||||
"memory_context_graph_hits_total", "Results from graph traversal tier");
|
|
||||||
pub static CONTEXT_EMPTY_RESULTS: Counter = Counter::new(
|
|
||||||
"memory_context_empty_results_total", "Context requests returning zero results");
|
|
||||||
|
|
||||||
pub static CONTEXT_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_context_duration_seconds", "Context retrieval duration", HTTP_BUCKETS));
|
|
||||||
pub static CONTEXT_TIER_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_context_tier_duration_seconds", "Per-tier retrieval duration", DB_BUCKETS));
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O4: Relevance judge metrics (R1-R9)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static RELEVANCE_EVALS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_relevance_evals_total", "Total relevance evaluations performed");
|
|
||||||
pub static RELEVANCE_ERRORS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_relevance_errors_total", "Total relevance evaluation errors");
|
|
||||||
pub static RELEVANCE_RELEVANT_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_relevance_relevant_total", "Results judged relevant");
|
|
||||||
pub static RELEVANCE_IRRELEVANT_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_relevance_irrelevant_total", "Results judged irrelevant");
|
|
||||||
|
|
||||||
pub static RELEVANCE_SCORE: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_relevance_score", "Distribution of relevance scores",
|
|
||||||
&[0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0]));
|
|
||||||
pub static RELEVANCE_PRECISION: GaugeF64 = GaugeF64::new(
|
|
||||||
"memory_relevance_precision", "Current precision (relevant/retrieved)");
|
|
||||||
pub static RELEVANCE_RECALL: GaugeF64 = GaugeF64::new(
|
|
||||||
"memory_relevance_recall", "Current recall (relevant/total_relevant)");
|
|
||||||
pub static RELEVANCE_F1: GaugeF64 = GaugeF64::new(
|
|
||||||
"memory_relevance_f1_score", "Current F1 score");
|
|
||||||
pub static RELEVANCE_EVAL_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_relevance_eval_duration_seconds", "Relevance evaluation duration", LLM_BUCKETS));
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O5: Write volume and storage metrics (W1-W12)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static WRITE_ENTITIES_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_write_entities_total", "Total entities written to DB");
|
|
||||||
pub static WRITE_EDGES_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_write_edges_total", "Total edges written to DB");
|
|
||||||
pub static WRITE_CHUNKS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_write_chunks_total", "Total chunks written to DB");
|
|
||||||
pub static WRITE_ERRORS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_write_errors_total", "Total write errors");
|
|
||||||
pub static WRITE_BYTES_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_write_bytes_total", "Total bytes written to storage");
|
|
||||||
|
|
||||||
pub static DB_ENTITY_COUNT: Gauge = Gauge::new(
|
|
||||||
"memory_db_entity_count", "Current entity count in memory_entity table");
|
|
||||||
pub static DB_EDGE_COUNT: Gauge = Gauge::new(
|
|
||||||
"memory_db_edge_count", "Current edge count in memory_edge table");
|
|
||||||
pub static DB_CHUNK_COUNT: Gauge = Gauge::new(
|
|
||||||
"memory_db_chunk_count", "Current chunk count in memory_chunks table");
|
|
||||||
|
|
||||||
pub static WRITE_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_write_duration_seconds", "Write operation duration", DB_BUCKETS));
|
|
||||||
pub static WRITE_BATCH_SIZE: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_write_batch_size", "Write batch sizes",
|
|
||||||
&[1.0, 5.0, 10.0, 25.0, 50.0, 100.0, 250.0, 500.0]));
|
|
||||||
|
|
||||||
// Storage gauges (updated periodically)
|
|
||||||
pub static DB_SIZE_BYTES: Gauge = Gauge::new(
|
|
||||||
"memory_db_size_bytes", "Total database size in bytes");
|
|
||||||
pub static DB_INDEX_SIZE_BYTES: Gauge = Gauge::new(
|
|
||||||
"memory_db_index_size_bytes", "Total index size in bytes");
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O6: Pod resource observability (P1-P13)
|
|
||||||
// (Most collected by node-exporter/cAdvisor, but we track app-level)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static APP_UPTIME_SECONDS: Gauge = Gauge::new(
|
|
||||||
"memory_app_uptime_seconds", "Application uptime in seconds");
|
|
||||||
pub static APP_ACTIVE_CONNECTIONS: Gauge = Gauge::new(
|
|
||||||
"memory_app_active_connections", "Active HTTP connections");
|
|
||||||
pub static APP_GOROUTINES: Gauge = Gauge::new(
|
|
||||||
"memory_app_tokio_tasks", "Active tokio tasks (approximate)");
|
|
||||||
pub static APP_HEAP_BYTES: Gauge = Gauge::new(
|
|
||||||
"memory_app_heap_bytes", "Approximate heap memory usage");
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O7: Availability metrics and dependency health (A1-A10)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static HEALTH_CHECKS_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_health_checks_total", "Total health check requests");
|
|
||||||
pub static HEALTH_CHECK_FAILURES: Counter = Counter::new(
|
|
||||||
"memory_health_check_failures_total", "Total health check failures");
|
|
||||||
|
|
||||||
pub static DEP_DB_UP: Gauge = Gauge::new(
|
|
||||||
"memory_dependency_db_up", "Database dependency health (1=up, 0=down)");
|
|
||||||
pub static DEP_EMBEDDING_UP: Gauge = Gauge::new(
|
|
||||||
"memory_dependency_embedding_up", "Embedding service health (1=up, 0=down)");
|
|
||||||
pub static DEP_OPENSEARCH_UP: Gauge = Gauge::new(
|
|
||||||
"memory_dependency_opensearch_up", "OpenSearch dependency health (1=up, 0=down)");
|
|
||||||
pub static DEP_LLM_UP: Gauge = Gauge::new(
|
|
||||||
"memory_dependency_llm_up", "LLM service health (1=up, 0=down)");
|
|
||||||
|
|
||||||
pub static DEP_DB_LATENCY: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_dependency_db_latency_seconds", "DB health check latency", DB_BUCKETS));
|
|
||||||
pub static DEP_EMBEDDING_LATENCY: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_dependency_embedding_latency_seconds", "Embedding health check latency", LLM_BUCKETS));
|
|
||||||
|
|
||||||
pub static REQUEST_ERRORS_BY_STATUS: Lazy<LabeledCounter> = Lazy::new(||
|
|
||||||
LabeledCounter::new(
|
|
||||||
"memory_request_errors_by_status", "Request errors by HTTP status code",
|
|
||||||
&["status", "endpoint"]));
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// Named error counters (per error type, per endpoint)
|
|
||||||
// Format: memory_error_{ERROR_NAME}_{ENDPOINT}_total
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
// Ingest errors
|
|
||||||
pub static ERROR_AUTH_FAILURE_INGEST: Counter = Counter::new(
|
|
||||||
"memory_error_auth_failure_ingest_total", "Auth failures on ingest endpoint");
|
|
||||||
pub static ERROR_FORBIDDEN_INGEST: Counter = Counter::new(
|
|
||||||
"memory_error_forbidden_ingest_total", "Forbidden (missing capability) on ingest");
|
|
||||||
pub static ERROR_RATE_LIMITED_INGEST: Counter = Counter::new(
|
|
||||||
"memory_error_rate_limited_ingest_total", "Rate limited on ingest");
|
|
||||||
pub static ERROR_BAD_REQUEST_INGEST: Counter = Counter::new(
|
|
||||||
"memory_error_bad_request_ingest_total", "Bad request on ingest");
|
|
||||||
pub static ERROR_DB_ERROR_INGEST: Counter = Counter::new(
|
|
||||||
"memory_error_db_error_ingest_total", "Database error during ingest");
|
|
||||||
|
|
||||||
// Query errors
|
|
||||||
pub static ERROR_AUTH_FAILURE_QUERY: Counter = Counter::new(
|
|
||||||
"memory_error_auth_failure_query_total", "Auth failures on query endpoint");
|
|
||||||
pub static ERROR_FORBIDDEN_QUERY: Counter = Counter::new(
|
|
||||||
"memory_error_forbidden_query_total", "Forbidden (missing capability) on query");
|
|
||||||
pub static ERROR_BAD_REQUEST_QUERY: Counter = Counter::new(
|
|
||||||
"memory_error_bad_request_query_total", "Bad request on query");
|
|
||||||
pub static ERROR_EMBEDDING_FAILURE_QUERY: Counter = Counter::new(
|
|
||||||
"memory_error_embedding_failure_query_total", "Embedding service failure during query");
|
|
||||||
pub static ERROR_SEARCH_FAILURE_QUERY: Counter = Counter::new(
|
|
||||||
"memory_error_search_failure_query_total", "Search execution failure during query");
|
|
||||||
|
|
||||||
// Context errors
|
|
||||||
pub static ERROR_AUTH_FAILURE_CONTEXT: Counter = Counter::new(
|
|
||||||
"memory_error_auth_failure_context_total", "Auth failures on context endpoint");
|
|
||||||
pub static ERROR_FORBIDDEN_CONTEXT: Counter = Counter::new(
|
|
||||||
"memory_error_forbidden_context_total", "Forbidden (missing capability) on context");
|
|
||||||
pub static ERROR_LOOKUP_FAILURE_CONTEXT: Counter = Counter::new(
|
|
||||||
"memory_error_lookup_failure_context_total", "Context lookup failure");
|
|
||||||
|
|
||||||
// Unexpected errors (unhandled 500s, panics, unknown failures)
|
|
||||||
pub static ERROR_UNEXPECTED_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_error_unexpected_total", "Total unexpected/unhandled errors (500s)");
|
|
||||||
pub static ERROR_UNEXPECTED_INGEST: Counter = Counter::new(
|
|
||||||
"memory_error_unexpected_ingest_total", "Unexpected errors during ingest");
|
|
||||||
pub static ERROR_UNEXPECTED_QUERY: Counter = Counter::new(
|
|
||||||
"memory_error_unexpected_query_total", "Unexpected errors during query");
|
|
||||||
pub static ERROR_UNEXPECTED_CONTEXT: Counter = Counter::new(
|
|
||||||
"memory_error_unexpected_context_total", "Unexpected errors during context");
|
|
||||||
|
|
||||||
// Last error info (most recent error for debugging)
|
|
||||||
pub static LAST_ERROR_TIMESTAMP: Gauge = Gauge::new(
|
|
||||||
"memory_last_error_timestamp_seconds", "Unix timestamp of most recent error");
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O8: Ingest rate pattern tracking (IR1-IR10)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static INGEST_RATE_1M: GaugeF64 = GaugeF64::new(
|
|
||||||
"memory_ingest_rate_1m", "Ingest rate per second (1-minute window)");
|
|
||||||
pub static INGEST_RATE_5M: GaugeF64 = GaugeF64::new(
|
|
||||||
"memory_ingest_rate_5m", "Ingest rate per second (5-minute window)");
|
|
||||||
pub static INGEST_LLM_EXTRACT_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_ingest_llm_extract_duration_seconds", "LLM entity extraction duration", LLM_BUCKETS));
|
|
||||||
pub static INGEST_FACT_EXTRACT_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_ingest_fact_extract_duration_seconds", "LLM fact extraction duration", LLM_BUCKETS));
|
|
||||||
pub static INGEST_DEDUP_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_dedup_total", "Total entities deduplicated");
|
|
||||||
pub static INGEST_CONTRADICTION_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_ingest_contradiction_total", "Total contradictions detected");
|
|
||||||
pub static INGEST_PROJECTS: Gauge = Gauge::new(
|
|
||||||
"memory_ingest_active_projects", "Number of active projects with ingested data");
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// O9: Postgres internal observability (PG1-PG33)
|
|
||||||
// (Most collected by pg_exporter, we expose app-visible DB stats)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
pub static DB_POOL_SIZE: Gauge = Gauge::new(
|
|
||||||
"memory_db_pool_size", "Current connection pool size");
|
|
||||||
pub static DB_POOL_IDLE: Gauge = Gauge::new(
|
|
||||||
"memory_db_pool_idle", "Idle connections in pool");
|
|
||||||
pub static DB_POOL_ACTIVE: Gauge = Gauge::new(
|
|
||||||
"memory_db_pool_active", "Active connections in pool");
|
|
||||||
pub static DB_QUERY_TOTAL: Counter = Counter::new(
|
|
||||||
"memory_db_queries_total", "Total DB queries executed");
|
|
||||||
pub static DB_QUERY_ERRORS: Counter = Counter::new(
|
|
||||||
"memory_db_query_errors_total", "Total DB query errors");
|
|
||||||
pub static DB_QUERY_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_db_query_duration_seconds", "DB query duration", DB_BUCKETS));
|
|
||||||
pub static DB_TRANSACTION_DURATION: Lazy<Histogram> = Lazy::new(||
|
|
||||||
Histogram::new("memory_db_transaction_duration_seconds", "DB transaction duration", DB_BUCKETS));
|
|
||||||
|
|
||||||
// Table-specific row counts (updated periodically)
|
|
||||||
pub static DB_TABLE_ENTITY_ROWS: Gauge = Gauge::new(
|
|
||||||
"memory_db_table_entity_rows", "Rows in memory_entity table");
|
|
||||||
pub static DB_TABLE_EDGE_ROWS: Gauge = Gauge::new(
|
|
||||||
"memory_db_table_edge_rows", "Rows in memory_edge table");
|
|
||||||
pub static DB_TABLE_CHUNK_ROWS: Gauge = Gauge::new(
|
|
||||||
"memory_db_table_chunk_rows", "Rows in memory_chunks table");
|
|
||||||
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
// Metrics export (Prometheus text format)
|
|
||||||
// ═══════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
/// Render all metrics in Prometheus text exposition format
|
|
||||||
pub fn render_metrics() -> String {
|
|
||||||
let mut out = String::with_capacity(8192);
|
|
||||||
|
|
||||||
// Helper macros
|
|
||||||
macro_rules! counter {
|
|
||||||
($c:expr) => {
|
|
||||||
out.push_str(&format!("# HELP {} {}\n# TYPE {} counter\n{} {}\n",
|
|
||||||
$c.name, $c.help, $c.name, $c.name, $c.get()));
|
|
||||||
};
|
|
||||||
}
|
|
||||||
macro_rules! gauge {
|
|
||||||
($g:expr) => {
|
|
||||||
out.push_str(&format!("# HELP {} {}\n# TYPE {} gauge\n{} {}\n",
|
|
||||||
$g.name, $g.help, $g.name, $g.name, $g.get()));
|
|
||||||
};
|
|
||||||
}
|
|
||||||
macro_rules! gauge_f64 {
|
|
||||||
($g:expr) => {
|
|
||||||
out.push_str(&format!("# HELP {} {}\n# TYPE {} gauge\n{} {:.6}\n",
|
|
||||||
$g.name, $g.help, $g.name, $g.name, $g.get()));
|
|
||||||
};
|
|
||||||
}
|
|
||||||
macro_rules! histogram {
|
|
||||||
($h:expr) => {
|
|
||||||
out.push_str(&format!("# HELP {} {}\n# TYPE {} histogram\n", $h.name, $h.help, $h.name));
|
|
||||||
for (i, &bound) in $h.buckets.iter().enumerate() {
|
|
||||||
out.push_str(&format!("{}_bucket{{le=\"{}\"}} {}\n",
|
|
||||||
$h.name, bound, $h.counts[i].load(Ordering::Relaxed)));
|
|
||||||
}
|
|
||||||
out.push_str(&format!("{}_bucket{{le=\"+Inf\"}} {}\n",
|
|
||||||
$h.name, $h.counts[$h.buckets.len()].load(Ordering::Relaxed)));
|
|
||||||
out.push_str(&format!("{}_sum {:.6}\n", $h.name,
|
|
||||||
f64::from_bits($h.sum.load(Ordering::Relaxed))));
|
|
||||||
out.push_str(&format!("{}_count {}\n", $h.name,
|
|
||||||
$h.count.load(Ordering::Relaxed)));
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// O1: Ingest
|
|
||||||
counter!(INGEST_REQUESTS_TOTAL);
|
|
||||||
counter!(INGEST_ERRORS_TOTAL);
|
|
||||||
counter!(INGEST_RECORDS_TOTAL);
|
|
||||||
counter!(INGEST_ENTITIES_EXTRACTED);
|
|
||||||
counter!(INGEST_EDGES_EXTRACTED);
|
|
||||||
gauge!(INGEST_IN_FLIGHT);
|
|
||||||
gauge!(INGEST_QUEUE_SIZE);
|
|
||||||
counter!(INGEST_DUPLICATES_TOTAL);
|
|
||||||
counter!(INGEST_BYTES_TOTAL);
|
|
||||||
counter!(INGEST_AUTH_FAILURES);
|
|
||||||
counter!(INGEST_RATE_LIMITED);
|
|
||||||
histogram!(INGEST_DURATION);
|
|
||||||
|
|
||||||
// O2: Query
|
|
||||||
counter!(QUERY_REQUESTS_TOTAL);
|
|
||||||
counter!(QUERY_ERRORS_TOTAL);
|
|
||||||
counter!(QUERY_RESULTS_TOTAL);
|
|
||||||
counter!(QUERY_EMPTY_RESULTS);
|
|
||||||
counter!(QUERY_EMBEDDING_FAILURES);
|
|
||||||
gauge!(QUERY_IN_FLIGHT);
|
|
||||||
counter!(QUERY_AUTH_FAILURES);
|
|
||||||
counter!(QUERY_RATE_LIMITED);
|
|
||||||
counter!(QUERY_CACHE_HITS);
|
|
||||||
counter!(QUERY_CACHE_MISSES);
|
|
||||||
histogram!(QUERY_DURATION);
|
|
||||||
histogram!(QUERY_EMBEDDING_DURATION);
|
|
||||||
|
|
||||||
// O3: Context
|
|
||||||
counter!(CONTEXT_REQUESTS_TOTAL);
|
|
||||||
counter!(CONTEXT_ERRORS_TOTAL);
|
|
||||||
counter!(CONTEXT_SEMANTIC_HITS);
|
|
||||||
counter!(CONTEXT_BM25_HITS);
|
|
||||||
counter!(CONTEXT_GRAPH_HITS);
|
|
||||||
counter!(CONTEXT_EMPTY_RESULTS);
|
|
||||||
histogram!(CONTEXT_DURATION);
|
|
||||||
histogram!(CONTEXT_TIER_DURATION);
|
|
||||||
|
|
||||||
// O4: Relevance
|
|
||||||
counter!(RELEVANCE_EVALS_TOTAL);
|
|
||||||
counter!(RELEVANCE_ERRORS_TOTAL);
|
|
||||||
counter!(RELEVANCE_RELEVANT_TOTAL);
|
|
||||||
counter!(RELEVANCE_IRRELEVANT_TOTAL);
|
|
||||||
histogram!(RELEVANCE_SCORE);
|
|
||||||
gauge_f64!(RELEVANCE_PRECISION);
|
|
||||||
gauge_f64!(RELEVANCE_RECALL);
|
|
||||||
gauge_f64!(RELEVANCE_F1);
|
|
||||||
histogram!(RELEVANCE_EVAL_DURATION);
|
|
||||||
|
|
||||||
// O5: Write volume
|
|
||||||
counter!(WRITE_ENTITIES_TOTAL);
|
|
||||||
counter!(WRITE_EDGES_TOTAL);
|
|
||||||
counter!(WRITE_CHUNKS_TOTAL);
|
|
||||||
counter!(WRITE_ERRORS_TOTAL);
|
|
||||||
counter!(WRITE_BYTES_TOTAL);
|
|
||||||
gauge!(DB_ENTITY_COUNT);
|
|
||||||
gauge!(DB_EDGE_COUNT);
|
|
||||||
gauge!(DB_CHUNK_COUNT);
|
|
||||||
histogram!(WRITE_DURATION);
|
|
||||||
histogram!(WRITE_BATCH_SIZE);
|
|
||||||
gauge!(DB_SIZE_BYTES);
|
|
||||||
gauge!(DB_INDEX_SIZE_BYTES);
|
|
||||||
|
|
||||||
// O6: Pod resources
|
|
||||||
gauge!(APP_UPTIME_SECONDS);
|
|
||||||
gauge!(APP_ACTIVE_CONNECTIONS);
|
|
||||||
gauge!(APP_GOROUTINES);
|
|
||||||
gauge!(APP_HEAP_BYTES);
|
|
||||||
|
|
||||||
// O7: Availability
|
|
||||||
counter!(HEALTH_CHECKS_TOTAL);
|
|
||||||
counter!(HEALTH_CHECK_FAILURES);
|
|
||||||
gauge!(DEP_DB_UP);
|
|
||||||
gauge!(DEP_EMBEDDING_UP);
|
|
||||||
gauge!(DEP_OPENSEARCH_UP);
|
|
||||||
gauge!(DEP_LLM_UP);
|
|
||||||
histogram!(DEP_DB_LATENCY);
|
|
||||||
histogram!(DEP_EMBEDDING_LATENCY);
|
|
||||||
|
|
||||||
// O8: Ingest rate
|
|
||||||
gauge_f64!(INGEST_RATE_1M);
|
|
||||||
gauge_f64!(INGEST_RATE_5M);
|
|
||||||
histogram!(INGEST_LLM_EXTRACT_DURATION);
|
|
||||||
histogram!(INGEST_FACT_EXTRACT_DURATION);
|
|
||||||
counter!(INGEST_DEDUP_TOTAL);
|
|
||||||
counter!(INGEST_CONTRADICTION_TOTAL);
|
|
||||||
gauge!(INGEST_PROJECTS);
|
|
||||||
|
|
||||||
// O9: Postgres
|
|
||||||
gauge!(DB_POOL_SIZE);
|
|
||||||
gauge!(DB_POOL_IDLE);
|
|
||||||
gauge!(DB_POOL_ACTIVE);
|
|
||||||
counter!(DB_QUERY_TOTAL);
|
|
||||||
counter!(DB_QUERY_ERRORS);
|
|
||||||
histogram!(DB_QUERY_DURATION);
|
|
||||||
histogram!(DB_TRANSACTION_DURATION);
|
|
||||||
gauge!(DB_TABLE_ENTITY_ROWS);
|
|
||||||
gauge!(DB_TABLE_EDGE_ROWS);
|
|
||||||
gauge!(DB_TABLE_CHUNK_ROWS);
|
|
||||||
|
|
||||||
// Named error counters
|
|
||||||
counter!(ERROR_AUTH_FAILURE_INGEST);
|
|
||||||
counter!(ERROR_FORBIDDEN_INGEST);
|
|
||||||
counter!(ERROR_RATE_LIMITED_INGEST);
|
|
||||||
counter!(ERROR_BAD_REQUEST_INGEST);
|
|
||||||
counter!(ERROR_DB_ERROR_INGEST);
|
|
||||||
counter!(ERROR_AUTH_FAILURE_QUERY);
|
|
||||||
counter!(ERROR_FORBIDDEN_QUERY);
|
|
||||||
counter!(ERROR_BAD_REQUEST_QUERY);
|
|
||||||
counter!(ERROR_EMBEDDING_FAILURE_QUERY);
|
|
||||||
counter!(ERROR_SEARCH_FAILURE_QUERY);
|
|
||||||
counter!(ERROR_AUTH_FAILURE_CONTEXT);
|
|
||||||
counter!(ERROR_FORBIDDEN_CONTEXT);
|
|
||||||
counter!(ERROR_LOOKUP_FAILURE_CONTEXT);
|
|
||||||
counter!(ERROR_UNEXPECTED_TOTAL);
|
|
||||||
counter!(ERROR_UNEXPECTED_INGEST);
|
|
||||||
counter!(ERROR_UNEXPECTED_QUERY);
|
|
||||||
counter!(ERROR_UNEXPECTED_CONTEXT);
|
|
||||||
gauge!(LAST_ERROR_TIMESTAMP);
|
|
||||||
|
|
||||||
out
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Render a labeled counter in Prometheus format
|
|
||||||
fn render_labeled_counter(out: &mut String, lc: &LabeledCounter) {
|
|
||||||
let map = lc.values.lock().unwrap();
|
|
||||||
if map.is_empty() { return; }
|
|
||||||
out.push_str(&format!("# HELP {} {}\n# TYPE {} counter\n", lc.name, lc.help, lc.name));
|
|
||||||
for (key, val) in map.iter() {
|
|
||||||
let parts: Vec<&str> = key.split(',').collect();
|
|
||||||
let labels: Vec<String> = lc.label_names.iter().zip(parts.iter())
|
|
||||||
.map(|(name, val)| format!("{}=\"{}\"", name, val))
|
|
||||||
.collect();
|
|
||||||
out.push_str(&format!("{}{{{}}} {}\n", lc.name, labels.join(","), val));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// GET /metrics handler
|
|
||||||
pub async fn metrics_handler() -> actix_web::HttpResponse {
|
|
||||||
actix_web::HttpResponse::Ok()
|
|
||||||
.content_type("text/plain; version=0.0.4; charset=utf-8")
|
|
||||||
.body(render_metrics())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use super::*;
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_counter() {
|
|
||||||
let c = Counter::new("test_counter", "test");
|
|
||||||
assert_eq!(c.get(), 0);
|
|
||||||
c.inc();
|
|
||||||
assert_eq!(c.get(), 1);
|
|
||||||
c.inc_by(5);
|
|
||||||
assert_eq!(c.get(), 6);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_gauge() {
|
|
||||||
let g = Gauge::new("test_gauge", "test");
|
|
||||||
assert_eq!(g.get(), 0);
|
|
||||||
g.set(42);
|
|
||||||
assert_eq!(g.get(), 42);
|
|
||||||
g.inc();
|
|
||||||
assert_eq!(g.get(), 43);
|
|
||||||
g.dec();
|
|
||||||
assert_eq!(g.get(), 42);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_gauge_f64() {
|
|
||||||
let g = GaugeF64::new("test_gauge_f64", "test");
|
|
||||||
assert_eq!(g.get(), 0.0);
|
|
||||||
g.set(3.14);
|
|
||||||
assert!((g.get() - 3.14).abs() < 0.001);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_histogram() {
|
|
||||||
let h = Histogram::new("test_hist", "test", &[0.1, 0.5, 1.0]);
|
|
||||||
h.observe(0.05);
|
|
||||||
h.observe(0.3);
|
|
||||||
h.observe(0.8);
|
|
||||||
h.observe(2.0);
|
|
||||||
assert_eq!(h.count.load(Ordering::Relaxed), 4);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_render_metrics_not_empty() {
|
|
||||||
INGEST_REQUESTS_TOTAL.inc();
|
|
||||||
QUERY_REQUESTS_TOTAL.inc();
|
|
||||||
let output = render_metrics();
|
|
||||||
assert!(output.contains("memory_ingest_requests_total"));
|
|
||||||
assert!(output.contains("memory_query_requests_total"));
|
|
||||||
assert!(output.contains("# HELP"));
|
|
||||||
assert!(output.contains("# TYPE"));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_timer_observes_on_drop() {
|
|
||||||
let h = Histogram::new("timer_test", "test", HTTP_BUCKETS);
|
|
||||||
{
|
|
||||||
let _t = Timer::new(&h);
|
|
||||||
std::thread::sleep(std::time::Duration::from_millis(1));
|
|
||||||
}
|
|
||||||
assert_eq!(h.count.load(Ordering::Relaxed), 1);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,418 +0,0 @@
|
|||||||
//! Metrics Snapshot & Assertion (Test Harness)
|
|
||||||
//!
|
|
||||||
//! Captures metric state before/after a test scenario,
|
|
||||||
//! then asserts expected deltas per metric.
|
|
||||||
//!
|
|
||||||
//! Usage:
|
|
||||||
//! ```rust
|
|
||||||
//! let snap = MetricsSnapshot::capture();
|
|
||||||
//! // ... run handler / scenario ...
|
|
||||||
//! snap.assert_counter_inc("memory_ingest_requests_total", 1);
|
|
||||||
//! snap.assert_counter_inc("memory_ingest_errors_total", 0);
|
|
||||||
//! snap.assert_gauge_eq("memory_ingest_in_flight", 0);
|
|
||||||
//! snap.assert_histogram_count_inc("memory_ingest_duration_seconds", 1);
|
|
||||||
//! ```
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
|
||||||
use std::sync::atomic::Ordering;
|
|
||||||
|
|
||||||
use crate::metrics;
|
|
||||||
|
|
||||||
/// Snapshot of all metric values at a point in time
|
|
||||||
#[derive(Debug, Clone)]
|
|
||||||
pub struct MetricsSnapshot {
|
|
||||||
counters: HashMap<&'static str, u64>,
|
|
||||||
gauges: HashMap<&'static str, u64>,
|
|
||||||
gauges_f64: HashMap<&'static str, f64>,
|
|
||||||
histogram_counts: HashMap<&'static str, u64>,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl MetricsSnapshot {
|
|
||||||
/// Capture current state of all metrics
|
|
||||||
pub fn capture() -> Self {
|
|
||||||
let mut counters = HashMap::new();
|
|
||||||
let mut gauges = HashMap::new();
|
|
||||||
let mut gauges_f64 = HashMap::new();
|
|
||||||
let mut histogram_counts = HashMap::new();
|
|
||||||
|
|
||||||
// O1: Ingest counters
|
|
||||||
counters.insert("memory_ingest_requests_total", metrics::INGEST_REQUESTS_TOTAL.get());
|
|
||||||
counters.insert("memory_ingest_errors_total", metrics::INGEST_ERRORS_TOTAL.get());
|
|
||||||
counters.insert("memory_ingest_records_total", metrics::INGEST_RECORDS_TOTAL.get());
|
|
||||||
counters.insert("memory_ingest_entities_extracted_total", metrics::INGEST_ENTITIES_EXTRACTED.get());
|
|
||||||
counters.insert("memory_ingest_edges_extracted_total", metrics::INGEST_EDGES_EXTRACTED.get());
|
|
||||||
counters.insert("memory_ingest_duplicates_total", metrics::INGEST_DUPLICATES_TOTAL.get());
|
|
||||||
counters.insert("memory_ingest_bytes_total", metrics::INGEST_BYTES_TOTAL.get());
|
|
||||||
counters.insert("memory_ingest_auth_failures_total", metrics::INGEST_AUTH_FAILURES.get());
|
|
||||||
counters.insert("memory_ingest_rate_limited_total", metrics::INGEST_RATE_LIMITED.get());
|
|
||||||
|
|
||||||
// O1: Ingest gauges
|
|
||||||
gauges.insert("memory_ingest_in_flight", metrics::INGEST_IN_FLIGHT.get());
|
|
||||||
gauges.insert("memory_ingest_queue_size", metrics::INGEST_QUEUE_SIZE.get());
|
|
||||||
|
|
||||||
// O1: Ingest histogram (force Lazy init)
|
|
||||||
histogram_counts.insert("memory_ingest_duration_seconds",
|
|
||||||
{ let _ = &*metrics::INGEST_DURATION; metrics::INGEST_DURATION.count.load(Ordering::Relaxed) });
|
|
||||||
|
|
||||||
// O2: Query counters
|
|
||||||
counters.insert("memory_query_requests_total", metrics::QUERY_REQUESTS_TOTAL.get());
|
|
||||||
counters.insert("memory_query_errors_total", metrics::QUERY_ERRORS_TOTAL.get());
|
|
||||||
counters.insert("memory_query_results_total", metrics::QUERY_RESULTS_TOTAL.get());
|
|
||||||
counters.insert("memory_query_empty_results_total", metrics::QUERY_EMPTY_RESULTS.get());
|
|
||||||
counters.insert("memory_query_embedding_failures_total", metrics::QUERY_EMBEDDING_FAILURES.get());
|
|
||||||
counters.insert("memory_query_auth_failures_total", metrics::QUERY_AUTH_FAILURES.get());
|
|
||||||
counters.insert("memory_query_rate_limited_total", metrics::QUERY_RATE_LIMITED.get());
|
|
||||||
counters.insert("memory_query_cache_hits_total", metrics::QUERY_CACHE_HITS.get());
|
|
||||||
counters.insert("memory_query_cache_misses_total", metrics::QUERY_CACHE_MISSES.get());
|
|
||||||
|
|
||||||
// O2: Query gauges
|
|
||||||
gauges.insert("memory_query_in_flight", metrics::QUERY_IN_FLIGHT.get());
|
|
||||||
|
|
||||||
// O2: Query histograms
|
|
||||||
histogram_counts.insert("memory_query_duration_seconds",
|
|
||||||
{ let _ = &*metrics::QUERY_DURATION; metrics::QUERY_DURATION.count.load(Ordering::Relaxed) });
|
|
||||||
histogram_counts.insert("memory_query_embedding_duration_seconds",
|
|
||||||
{ let _ = &*metrics::QUERY_EMBEDDING_DURATION; metrics::QUERY_EMBEDDING_DURATION.count.load(Ordering::Relaxed) });
|
|
||||||
|
|
||||||
// O3: Context
|
|
||||||
counters.insert("memory_context_requests_total", metrics::CONTEXT_REQUESTS_TOTAL.get());
|
|
||||||
counters.insert("memory_context_errors_total", metrics::CONTEXT_ERRORS_TOTAL.get());
|
|
||||||
counters.insert("memory_context_semantic_hits_total", metrics::CONTEXT_SEMANTIC_HITS.get());
|
|
||||||
counters.insert("memory_context_bm25_hits_total", metrics::CONTEXT_BM25_HITS.get());
|
|
||||||
counters.insert("memory_context_graph_hits_total", metrics::CONTEXT_GRAPH_HITS.get());
|
|
||||||
counters.insert("memory_context_empty_results_total", metrics::CONTEXT_EMPTY_RESULTS.get());
|
|
||||||
histogram_counts.insert("memory_context_duration_seconds",
|
|
||||||
{ let _ = &*metrics::CONTEXT_DURATION; metrics::CONTEXT_DURATION.count.load(Ordering::Relaxed) });
|
|
||||||
|
|
||||||
// O4: Relevance histograms
|
|
||||||
histogram_counts.insert("memory_relevance_eval_duration_seconds",
|
|
||||||
{ let _ = &*metrics::RELEVANCE_EVAL_DURATION; metrics::RELEVANCE_EVAL_DURATION.count.load(Ordering::Relaxed) });
|
|
||||||
|
|
||||||
// O5: Write histogram
|
|
||||||
histogram_counts.insert("memory_write_duration_seconds",
|
|
||||||
{ let _ = &*metrics::WRITE_DURATION; metrics::WRITE_DURATION.count.load(Ordering::Relaxed) });
|
|
||||||
|
|
||||||
// O7: Dependency latency
|
|
||||||
histogram_counts.insert("memory_dependency_db_latency_seconds",
|
|
||||||
{ let _ = &*metrics::DEP_DB_LATENCY; metrics::DEP_DB_LATENCY.count.load(Ordering::Relaxed) });
|
|
||||||
|
|
||||||
// O4: Relevance
|
|
||||||
counters.insert("memory_relevance_evals_total", metrics::RELEVANCE_EVALS_TOTAL.get());
|
|
||||||
counters.insert("memory_relevance_errors_total", metrics::RELEVANCE_ERRORS_TOTAL.get());
|
|
||||||
counters.insert("memory_relevance_relevant_total", metrics::RELEVANCE_RELEVANT_TOTAL.get());
|
|
||||||
counters.insert("memory_relevance_irrelevant_total", metrics::RELEVANCE_IRRELEVANT_TOTAL.get());
|
|
||||||
gauges_f64.insert("memory_relevance_precision", metrics::RELEVANCE_PRECISION.get());
|
|
||||||
gauges_f64.insert("memory_relevance_recall", metrics::RELEVANCE_RECALL.get());
|
|
||||||
gauges_f64.insert("memory_relevance_f1_score", metrics::RELEVANCE_F1.get());
|
|
||||||
|
|
||||||
// O5: Write
|
|
||||||
counters.insert("memory_write_entities_total", metrics::WRITE_ENTITIES_TOTAL.get());
|
|
||||||
counters.insert("memory_write_edges_total", metrics::WRITE_EDGES_TOTAL.get());
|
|
||||||
counters.insert("memory_write_chunks_total", metrics::WRITE_CHUNKS_TOTAL.get());
|
|
||||||
counters.insert("memory_write_errors_total", metrics::WRITE_ERRORS_TOTAL.get());
|
|
||||||
counters.insert("memory_write_bytes_total", metrics::WRITE_BYTES_TOTAL.get());
|
|
||||||
|
|
||||||
// O7: Health
|
|
||||||
counters.insert("memory_health_checks_total", metrics::HEALTH_CHECKS_TOTAL.get());
|
|
||||||
counters.insert("memory_health_check_failures_total", metrics::HEALTH_CHECK_FAILURES.get());
|
|
||||||
gauges.insert("memory_dependency_db_up", metrics::DEP_DB_UP.get());
|
|
||||||
gauges.insert("memory_dependency_embedding_up", metrics::DEP_EMBEDDING_UP.get());
|
|
||||||
|
|
||||||
// O8: Ingest rate
|
|
||||||
counters.insert("memory_ingest_dedup_total", metrics::INGEST_DEDUP_TOTAL.get());
|
|
||||||
counters.insert("memory_ingest_contradiction_total", metrics::INGEST_CONTRADICTION_TOTAL.get());
|
|
||||||
|
|
||||||
// O9: DB
|
|
||||||
counters.insert("memory_db_queries_total", metrics::DB_QUERY_TOTAL.get());
|
|
||||||
counters.insert("memory_db_query_errors_total", metrics::DB_QUERY_ERRORS.get());
|
|
||||||
|
|
||||||
Self { counters, gauges, gauges_f64, histogram_counts }
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Assert a counter increased by exactly `expected` since snapshot
|
|
||||||
pub fn assert_counter_inc(&self, name: &str, expected: u64) {
|
|
||||||
let before = self.counters.get(name)
|
|
||||||
.unwrap_or_else(|| panic!("Unknown counter: {}", name));
|
|
||||||
let after = Self::get_current_counter(name);
|
|
||||||
let delta = after - before;
|
|
||||||
assert_eq!(delta, expected,
|
|
||||||
"Counter {} expected +{} but got +{} (before={}, after={})",
|
|
||||||
name, expected, delta, before, after);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Assert a counter increased by at least `min` since snapshot
|
|
||||||
pub fn assert_counter_inc_at_least(&self, name: &str, min: u64) {
|
|
||||||
let before = self.counters.get(name)
|
|
||||||
.unwrap_or_else(|| panic!("Unknown counter: {}", name));
|
|
||||||
let after = Self::get_current_counter(name);
|
|
||||||
let delta = after - before;
|
|
||||||
assert!(delta >= min,
|
|
||||||
"Counter {} expected at least +{} but got +{} (before={}, after={})",
|
|
||||||
name, min, delta, before, after);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Assert a gauge equals exactly `expected`
|
|
||||||
pub fn assert_gauge_eq(&self, name: &str, expected: u64) {
|
|
||||||
let current = Self::get_current_gauge(name);
|
|
||||||
assert_eq!(current, expected,
|
|
||||||
"Gauge {} expected {} but got {}", name, expected, current);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Assert a histogram observation count increased by `expected`
|
|
||||||
pub fn assert_histogram_count_inc(&self, name: &str, expected: u64) {
|
|
||||||
let before = self.histogram_counts.get(name)
|
|
||||||
.unwrap_or_else(|| panic!("Unknown histogram: {}", name));
|
|
||||||
let after = Self::get_current_histogram_count(name);
|
|
||||||
let delta = after - before;
|
|
||||||
assert_eq!(delta, expected,
|
|
||||||
"Histogram {} count expected +{} but got +{} (before={}, after={})",
|
|
||||||
name, expected, delta, before, after);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Assert a f64 gauge is within tolerance
|
|
||||||
pub fn assert_gauge_f64_approx(&self, name: &str, expected: f64, tolerance: f64) {
|
|
||||||
let current = Self::get_current_gauge_f64(name);
|
|
||||||
assert!((current - expected).abs() <= tolerance,
|
|
||||||
"Gauge {} expected {:.4} (±{}) but got {:.4}",
|
|
||||||
name, expected, tolerance, current);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Get delta for a counter since snapshot
|
|
||||||
pub fn counter_delta(&self, name: &str) -> u64 {
|
|
||||||
let before = self.counters.get(name).copied().unwrap_or(0);
|
|
||||||
let after = Self::get_current_counter(name);
|
|
||||||
after - before
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Print all deltas since snapshot (for debugging)
|
|
||||||
pub fn print_deltas(&self) {
|
|
||||||
println!("=== Metrics Deltas ===");
|
|
||||||
for (name, before) in &self.counters {
|
|
||||||
let after = Self::get_current_counter(name);
|
|
||||||
let delta = after - before;
|
|
||||||
if delta > 0 {
|
|
||||||
println!(" {} +{} ({} -> {})", name, delta, before, after);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (name, before) in &self.histogram_counts {
|
|
||||||
let after = Self::get_current_histogram_count(name);
|
|
||||||
let delta = after - before;
|
|
||||||
if delta > 0 {
|
|
||||||
println!(" {} count +{}", name, delta);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ─── Internal helpers ───────────────────────────────────
|
|
||||||
|
|
||||||
fn get_current_counter(name: &str) -> u64 {
|
|
||||||
match name {
|
|
||||||
"memory_ingest_requests_total" => metrics::INGEST_REQUESTS_TOTAL.get(),
|
|
||||||
"memory_ingest_errors_total" => metrics::INGEST_ERRORS_TOTAL.get(),
|
|
||||||
"memory_ingest_records_total" => metrics::INGEST_RECORDS_TOTAL.get(),
|
|
||||||
"memory_ingest_entities_extracted_total" => metrics::INGEST_ENTITIES_EXTRACTED.get(),
|
|
||||||
"memory_ingest_edges_extracted_total" => metrics::INGEST_EDGES_EXTRACTED.get(),
|
|
||||||
"memory_ingest_duplicates_total" => metrics::INGEST_DUPLICATES_TOTAL.get(),
|
|
||||||
"memory_ingest_bytes_total" => metrics::INGEST_BYTES_TOTAL.get(),
|
|
||||||
"memory_ingest_auth_failures_total" => metrics::INGEST_AUTH_FAILURES.get(),
|
|
||||||
"memory_ingest_rate_limited_total" => metrics::INGEST_RATE_LIMITED.get(),
|
|
||||||
"memory_query_requests_total" => metrics::QUERY_REQUESTS_TOTAL.get(),
|
|
||||||
"memory_query_errors_total" => metrics::QUERY_ERRORS_TOTAL.get(),
|
|
||||||
"memory_query_results_total" => metrics::QUERY_RESULTS_TOTAL.get(),
|
|
||||||
"memory_query_empty_results_total" => metrics::QUERY_EMPTY_RESULTS.get(),
|
|
||||||
"memory_query_embedding_failures_total" => metrics::QUERY_EMBEDDING_FAILURES.get(),
|
|
||||||
"memory_query_auth_failures_total" => metrics::QUERY_AUTH_FAILURES.get(),
|
|
||||||
"memory_query_rate_limited_total" => metrics::QUERY_RATE_LIMITED.get(),
|
|
||||||
"memory_query_cache_hits_total" => metrics::QUERY_CACHE_HITS.get(),
|
|
||||||
"memory_query_cache_misses_total" => metrics::QUERY_CACHE_MISSES.get(),
|
|
||||||
"memory_context_requests_total" => metrics::CONTEXT_REQUESTS_TOTAL.get(),
|
|
||||||
"memory_context_errors_total" => metrics::CONTEXT_ERRORS_TOTAL.get(),
|
|
||||||
"memory_context_semantic_hits_total" => metrics::CONTEXT_SEMANTIC_HITS.get(),
|
|
||||||
"memory_context_bm25_hits_total" => metrics::CONTEXT_BM25_HITS.get(),
|
|
||||||
"memory_context_graph_hits_total" => metrics::CONTEXT_GRAPH_HITS.get(),
|
|
||||||
"memory_context_empty_results_total" => metrics::CONTEXT_EMPTY_RESULTS.get(),
|
|
||||||
"memory_relevance_evals_total" => metrics::RELEVANCE_EVALS_TOTAL.get(),
|
|
||||||
"memory_relevance_errors_total" => metrics::RELEVANCE_ERRORS_TOTAL.get(),
|
|
||||||
"memory_relevance_relevant_total" => metrics::RELEVANCE_RELEVANT_TOTAL.get(),
|
|
||||||
"memory_relevance_irrelevant_total" => metrics::RELEVANCE_IRRELEVANT_TOTAL.get(),
|
|
||||||
"memory_write_entities_total" => metrics::WRITE_ENTITIES_TOTAL.get(),
|
|
||||||
"memory_write_edges_total" => metrics::WRITE_EDGES_TOTAL.get(),
|
|
||||||
"memory_write_chunks_total" => metrics::WRITE_CHUNKS_TOTAL.get(),
|
|
||||||
"memory_write_errors_total" => metrics::WRITE_ERRORS_TOTAL.get(),
|
|
||||||
"memory_write_bytes_total" => metrics::WRITE_BYTES_TOTAL.get(),
|
|
||||||
"memory_health_checks_total" => metrics::HEALTH_CHECKS_TOTAL.get(),
|
|
||||||
"memory_health_check_failures_total" => metrics::HEALTH_CHECK_FAILURES.get(),
|
|
||||||
"memory_ingest_dedup_total" => metrics::INGEST_DEDUP_TOTAL.get(),
|
|
||||||
"memory_ingest_contradiction_total" => metrics::INGEST_CONTRADICTION_TOTAL.get(),
|
|
||||||
"memory_db_queries_total" => metrics::DB_QUERY_TOTAL.get(),
|
|
||||||
"memory_db_query_errors_total" => metrics::DB_QUERY_ERRORS.get(),
|
|
||||||
_ => panic!("Unknown counter: {}", name),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn get_current_gauge(name: &str) -> u64 {
|
|
||||||
match name {
|
|
||||||
"memory_ingest_in_flight" => metrics::INGEST_IN_FLIGHT.get(),
|
|
||||||
"memory_ingest_queue_size" => metrics::INGEST_QUEUE_SIZE.get(),
|
|
||||||
"memory_query_in_flight" => metrics::QUERY_IN_FLIGHT.get(),
|
|
||||||
"memory_dependency_db_up" => metrics::DEP_DB_UP.get(),
|
|
||||||
"memory_dependency_embedding_up" => metrics::DEP_EMBEDDING_UP.get(),
|
|
||||||
"memory_dependency_opensearch_up" => metrics::DEP_OPENSEARCH_UP.get(),
|
|
||||||
"memory_dependency_llm_up" => metrics::DEP_LLM_UP.get(),
|
|
||||||
"memory_app_uptime_seconds" => metrics::APP_UPTIME_SECONDS.get(),
|
|
||||||
"memory_db_pool_size" => metrics::DB_POOL_SIZE.get(),
|
|
||||||
"memory_db_pool_idle" => metrics::DB_POOL_IDLE.get(),
|
|
||||||
"memory_db_table_entity_rows" => metrics::DB_TABLE_ENTITY_ROWS.get(),
|
|
||||||
"memory_db_table_edge_rows" => metrics::DB_TABLE_EDGE_ROWS.get(),
|
|
||||||
"memory_db_table_chunk_rows" => metrics::DB_TABLE_CHUNK_ROWS.get(),
|
|
||||||
_ => panic!("Unknown gauge: {}", name),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn get_current_gauge_f64(name: &str) -> f64 {
|
|
||||||
match name {
|
|
||||||
"memory_relevance_precision" => metrics::RELEVANCE_PRECISION.get(),
|
|
||||||
"memory_relevance_recall" => metrics::RELEVANCE_RECALL.get(),
|
|
||||||
"memory_relevance_f1_score" => metrics::RELEVANCE_F1.get(),
|
|
||||||
"memory_ingest_rate_1m" => metrics::INGEST_RATE_1M.get(),
|
|
||||||
"memory_ingest_rate_5m" => metrics::INGEST_RATE_5M.get(),
|
|
||||||
_ => panic!("Unknown gauge_f64: {}", name),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn get_current_histogram_count(name: &str) -> u64 {
|
|
||||||
match name {
|
|
||||||
"memory_ingest_duration_seconds" =>
|
|
||||||
metrics::INGEST_DURATION.count.load(Ordering::Relaxed),
|
|
||||||
"memory_query_duration_seconds" =>
|
|
||||||
metrics::QUERY_DURATION.count.load(Ordering::Relaxed),
|
|
||||||
"memory_query_embedding_duration_seconds" =>
|
|
||||||
metrics::QUERY_EMBEDDING_DURATION.count.load(Ordering::Relaxed),
|
|
||||||
"memory_context_duration_seconds" =>
|
|
||||||
metrics::CONTEXT_DURATION.count.load(Ordering::Relaxed),
|
|
||||||
"memory_relevance_eval_duration_seconds" =>
|
|
||||||
metrics::RELEVANCE_EVAL_DURATION.count.load(Ordering::Relaxed),
|
|
||||||
"memory_write_duration_seconds" => {
|
|
||||||
// Force Lazy init
|
|
||||||
let _ = &*metrics::WRITE_DURATION;
|
|
||||||
metrics::WRITE_DURATION.count.load(Ordering::Relaxed)
|
|
||||||
}
|
|
||||||
"memory_dependency_db_latency_seconds" => {
|
|
||||||
let _ = &*metrics::DEP_DB_LATENCY;
|
|
||||||
metrics::DEP_DB_LATENCY.count.load(Ordering::Relaxed)
|
|
||||||
}
|
|
||||||
_ => panic!("Unknown histogram: {}", name),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use super::*;
|
|
||||||
use crate::relevance_judge::RelevanceJudge;
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_snapshot_captures_state() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
assert!(snap.counters.contains_key("memory_ingest_requests_total"));
|
|
||||||
assert!(snap.counters.contains_key("memory_query_requests_total"));
|
|
||||||
assert!(snap.gauges.contains_key("memory_ingest_in_flight"));
|
|
||||||
assert!(snap.histogram_counts.contains_key("memory_ingest_duration_seconds"));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_counter_delta_zero_when_no_change() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
snap.assert_counter_inc("memory_write_entities_total", 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_counter_tracks_increment() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
metrics::WRITE_ENTITIES_TOTAL.inc_by(3);
|
|
||||||
snap.assert_counter_inc("memory_write_entities_total", 3);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_counter_delta_method() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
metrics::WRITE_EDGES_TOTAL.inc_by(7);
|
|
||||||
assert_eq!(snap.counter_delta("memory_write_edges_total"), 7);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_histogram_count_tracks() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
metrics::WRITE_DURATION.observe(0.05);
|
|
||||||
metrics::WRITE_DURATION.observe(0.10);
|
|
||||||
snap.assert_histogram_count_inc("memory_write_duration_seconds", 2);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_relevance_scenario_metrics() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
|
|
||||||
let judge = RelevanceJudge::new(0.5);
|
|
||||||
let results = vec![
|
|
||||||
("good result".to_string(), 0.9),
|
|
||||||
("bad result".to_string(), 0.1),
|
|
||||||
("ok result".to_string(), 0.6),
|
|
||||||
];
|
|
||||||
let summary = judge.evaluate_batch("test query", &results);
|
|
||||||
|
|
||||||
// Verify metrics match scenario
|
|
||||||
snap.assert_counter_inc("memory_relevance_evals_total", 3);
|
|
||||||
snap.assert_counter_inc("memory_relevance_relevant_total", 2); // 0.9 + 0.6
|
|
||||||
snap.assert_counter_inc("memory_relevance_irrelevant_total", 1); // 0.1
|
|
||||||
|
|
||||||
// Verify precision gauge
|
|
||||||
snap.assert_gauge_f64_approx("memory_relevance_precision", summary.precision, 0.01);
|
|
||||||
|
|
||||||
assert_eq!(summary.total, 3);
|
|
||||||
assert_eq!(summary.relevant, 2);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_ingest_counter_scenario() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
|
|
||||||
// Simulate ingest scenario
|
|
||||||
metrics::INGEST_REQUESTS_TOTAL.inc();
|
|
||||||
metrics::INGEST_RECORDS_TOTAL.inc_by(5);
|
|
||||||
metrics::INGEST_BYTES_TOTAL.inc_by(1024);
|
|
||||||
metrics::INGEST_ENTITIES_EXTRACTED.inc_by(3);
|
|
||||||
metrics::INGEST_EDGES_EXTRACTED.inc_by(2);
|
|
||||||
|
|
||||||
snap.assert_counter_inc("memory_ingest_requests_total", 1);
|
|
||||||
snap.assert_counter_inc("memory_ingest_records_total", 5);
|
|
||||||
snap.assert_counter_inc("memory_ingest_bytes_total", 1024);
|
|
||||||
snap.assert_counter_inc("memory_ingest_entities_extracted_total", 3);
|
|
||||||
snap.assert_counter_inc("memory_ingest_edges_extracted_total", 2);
|
|
||||||
snap.assert_counter_inc("memory_ingest_errors_total", 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_query_error_scenario() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
|
|
||||||
// Simulate query that fails at embedding
|
|
||||||
metrics::QUERY_REQUESTS_TOTAL.inc();
|
|
||||||
metrics::QUERY_IN_FLIGHT.inc();
|
|
||||||
metrics::QUERY_EMBEDDING_FAILURES.inc();
|
|
||||||
metrics::QUERY_ERRORS_TOTAL.inc();
|
|
||||||
metrics::QUERY_IN_FLIGHT.dec();
|
|
||||||
|
|
||||||
snap.assert_counter_inc("memory_query_requests_total", 1);
|
|
||||||
snap.assert_counter_inc("memory_query_embedding_failures_total", 1);
|
|
||||||
snap.assert_counter_inc("memory_query_errors_total", 1);
|
|
||||||
snap.assert_counter_inc("memory_query_results_total", 0);
|
|
||||||
snap.assert_gauge_eq("memory_query_in_flight", 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_print_deltas_works() {
|
|
||||||
let snap = MetricsSnapshot::capture();
|
|
||||||
metrics::HEALTH_CHECKS_TOTAL.inc();
|
|
||||||
snap.print_deltas(); // Should not panic
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -238,17 +238,6 @@ impl QueryRouter {
|
|||||||
|
|
||||||
let latency_ms = start.elapsed().as_millis() as u64;
|
let latency_ms = start.elapsed().as_millis() as u64;
|
||||||
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "query_route",
|
|
||||||
route = "direct",
|
|
||||||
candidates = all_candidates.len(),
|
|
||||||
prefiltered = prefilter_size,
|
|
||||||
selected = selected_chunks.len(),
|
|
||||||
latency_ms = latency_ms,
|
|
||||||
"Query routing complete"
|
|
||||||
);
|
|
||||||
|
|
||||||
Ok(RoutedResult {
|
Ok(RoutedResult {
|
||||||
selected_chunks,
|
selected_chunks,
|
||||||
route,
|
route,
|
||||||
|
|||||||
@@ -1,161 +0,0 @@
|
|||||||
//! Relevance Judge (O4)
|
|
||||||
//!
|
|
||||||
//! Evaluates retrieval quality by scoring query-result relevance.
|
|
||||||
//! Uses LLM (Qwen-7B or similar) to judge if retrieved results are relevant.
|
|
||||||
//! Tracks precision, recall, F1 via Prometheus metrics.
|
|
||||||
|
|
||||||
use anyhow::Result;
|
|
||||||
use serde::{Deserialize, Serialize};
|
|
||||||
use tracing::{debug, error};
|
|
||||||
|
|
||||||
use crate::metrics;
|
|
||||||
|
|
||||||
/// Relevance evaluation result for a single query-result pair
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
|
||||||
pub struct RelevanceResult {
|
|
||||||
pub query: String,
|
|
||||||
pub result_text: String,
|
|
||||||
pub score: f64,
|
|
||||||
pub relevant: bool,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Batch evaluation summary
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
|
||||||
pub struct RelevanceSummary {
|
|
||||||
pub total: usize,
|
|
||||||
pub relevant: usize,
|
|
||||||
pub irrelevant: usize,
|
|
||||||
pub precision: f64,
|
|
||||||
pub recall: f64,
|
|
||||||
pub f1: f64,
|
|
||||||
pub avg_score: f64,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Simple relevance judge using cosine similarity threshold
|
|
||||||
/// (LLM-based judge can be plugged in later via trait)
|
|
||||||
pub struct RelevanceJudge {
|
|
||||||
threshold: f64,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl RelevanceJudge {
|
|
||||||
pub fn new(threshold: f64) -> Self {
|
|
||||||
Self { threshold }
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate a single query-result pair using similarity score
|
|
||||||
pub fn evaluate(&self, query: &str, result_text: &str, similarity: f64) -> RelevanceResult {
|
|
||||||
let start = std::time::Instant::now();
|
|
||||||
|
|
||||||
metrics::RELEVANCE_EVALS_TOTAL.inc();
|
|
||||||
|
|
||||||
let relevant = similarity >= self.threshold;
|
|
||||||
|
|
||||||
if relevant {
|
|
||||||
metrics::RELEVANCE_RELEVANT_TOTAL.inc();
|
|
||||||
} else {
|
|
||||||
metrics::RELEVANCE_IRRELEVANT_TOTAL.inc();
|
|
||||||
}
|
|
||||||
|
|
||||||
metrics::RELEVANCE_SCORE.observe(similarity);
|
|
||||||
metrics::RELEVANCE_EVAL_DURATION.observe(start.elapsed().as_secs_f64());
|
|
||||||
|
|
||||||
debug!("Relevance eval: query='{}', score={:.3}, relevant={}",
|
|
||||||
&query[..query.len().min(50)], similarity, relevant);
|
|
||||||
|
|
||||||
RelevanceResult {
|
|
||||||
query: query.to_string(),
|
|
||||||
result_text: result_text.to_string(),
|
|
||||||
score: similarity,
|
|
||||||
relevant,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate a batch of results and compute summary metrics
|
|
||||||
pub fn evaluate_batch(
|
|
||||||
&self,
|
|
||||||
query: &str,
|
|
||||||
results: &[(String, f64)], // (result_text, similarity_score)
|
|
||||||
) -> RelevanceSummary {
|
|
||||||
let mut relevant_count = 0;
|
|
||||||
let mut total_score = 0.0;
|
|
||||||
|
|
||||||
for (text, score) in results {
|
|
||||||
let result = self.evaluate(query, text, *score);
|
|
||||||
if result.relevant {
|
|
||||||
relevant_count += 1;
|
|
||||||
}
|
|
||||||
total_score += score;
|
|
||||||
}
|
|
||||||
|
|
||||||
let total = results.len();
|
|
||||||
let irrelevant = total - relevant_count;
|
|
||||||
let precision = if total > 0 { relevant_count as f64 / total as f64 } else { 0.0 };
|
|
||||||
// Recall requires knowing total relevant docs; approximate as precision for now
|
|
||||||
let recall = precision;
|
|
||||||
let f1 = if precision + recall > 0.0 {
|
|
||||||
2.0 * precision * recall / (precision + recall)
|
|
||||||
} else {
|
|
||||||
0.0
|
|
||||||
};
|
|
||||||
let avg_score = if total > 0 { total_score / total as f64 } else { 0.0 };
|
|
||||||
|
|
||||||
// Update gauge metrics
|
|
||||||
metrics::RELEVANCE_PRECISION.set(precision);
|
|
||||||
metrics::RELEVANCE_RECALL.set(recall);
|
|
||||||
metrics::RELEVANCE_F1.set(f1);
|
|
||||||
|
|
||||||
RelevanceSummary {
|
|
||||||
total,
|
|
||||||
relevant: relevant_count,
|
|
||||||
irrelevant,
|
|
||||||
precision,
|
|
||||||
recall,
|
|
||||||
f1,
|
|
||||||
avg_score,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use super::*;
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_relevance_judge_above_threshold() {
|
|
||||||
let judge = RelevanceJudge::new(0.5);
|
|
||||||
let result = judge.evaluate("test query", "test result", 0.8);
|
|
||||||
assert!(result.relevant);
|
|
||||||
assert!((result.score - 0.8).abs() < 0.001);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_relevance_judge_below_threshold() {
|
|
||||||
let judge = RelevanceJudge::new(0.5);
|
|
||||||
let result = judge.evaluate("test query", "test result", 0.3);
|
|
||||||
assert!(!result.relevant);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_relevance_batch() {
|
|
||||||
let judge = RelevanceJudge::new(0.5);
|
|
||||||
let results = vec![
|
|
||||||
("relevant result".to_string(), 0.8),
|
|
||||||
("somewhat relevant".to_string(), 0.6),
|
|
||||||
("irrelevant".to_string(), 0.2),
|
|
||||||
];
|
|
||||||
let summary = judge.evaluate_batch("test", &results);
|
|
||||||
assert_eq!(summary.total, 3);
|
|
||||||
assert_eq!(summary.relevant, 2);
|
|
||||||
assert_eq!(summary.irrelevant, 1);
|
|
||||||
assert!((summary.precision - 0.6667).abs() < 0.01);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_relevance_empty_batch() {
|
|
||||||
let judge = RelevanceJudge::new(0.5);
|
|
||||||
let summary = judge.evaluate_batch("test", &[]);
|
|
||||||
assert_eq!(summary.total, 0);
|
|
||||||
assert_eq!(summary.precision, 0.0);
|
|
||||||
assert_eq!(summary.f1, 0.0);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -235,18 +235,6 @@ impl BudgetCompressor {
|
|||||||
let strategy = self.select_strategy(estimated);
|
let strategy = self.select_strategy(estimated);
|
||||||
let compressed = self.compressor.compress_batch(results, strategy);
|
let compressed = self.compressor.compress_batch(results, strategy);
|
||||||
|
|
||||||
let compressed_size: usize = compressed.iter().map(|c| c.text.as_ref().map_or(0, |t| t.len())).sum();
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "result_compress",
|
|
||||||
input_count = compressed.len(),
|
|
||||||
estimated_bytes = estimated,
|
|
||||||
compressed_bytes = compressed_size,
|
|
||||||
budget_bytes = self.max_budget_bytes,
|
|
||||||
strategy = ?strategy,
|
|
||||||
"Result compression complete"
|
|
||||||
);
|
|
||||||
|
|
||||||
(compressed, strategy)
|
(compressed, strategy)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,7 +2,6 @@
|
|||||||
///
|
///
|
||||||
/// These structures attach to Entity via entity_type discriminator.
|
/// These structures attach to Entity via entity_type discriminator.
|
||||||
/// AgentPrompt, AgentSkill, AgentDecision each carry domain-specific
|
/// AgentPrompt, AgentSkill, AgentDecision each carry domain-specific
|
||||||
#[allow(clippy::empty_line_after_doc_comments)]
|
|
||||||
/// fields that enable the agent to learn from its own behavior.
|
/// fields that enable the agent to learn from its own behavior.
|
||||||
|
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
/// Community domain model for temporal graph-RAG.
|
/// Community domain model for temporal graph-RAG.
|
||||||
/// Single Responsibility: Community (cluster) storage and metadata.
|
/// Single Responsibility: Community (cluster) storage and metadata.
|
||||||
#[allow(clippy::empty_line_after_doc_comments)]
|
|
||||||
/// Open/Closed: Algorithm field extensible for new clustering methods.
|
/// Open/Closed: Algorithm field extensible for new clustering methods.
|
||||||
|
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
/// Edge domain model for temporal graph-RAG.
|
/// Edge domain model for temporal graph-RAG.
|
||||||
/// Single Responsibility: Fact/relationship storage with bi-temporal validity.
|
/// Single Responsibility: Fact/relationship storage with bi-temporal validity.
|
||||||
#[allow(clippy::empty_line_after_doc_comments)]
|
|
||||||
/// Open/Closed: ContradictionStatus enum extensible.
|
/// Open/Closed: ContradictionStatus enum extensible.
|
||||||
|
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
@@ -30,7 +29,6 @@ impl ContradictionStatus {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(clippy::should_implement_trait)]
|
|
||||||
pub fn from_str(s: &str) -> Self {
|
pub fn from_str(s: &str) -> Self {
|
||||||
match s.to_lowercase().as_str() {
|
match s.to_lowercase().as_str() {
|
||||||
"active" => Self::Active,
|
"active" => Self::Active,
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
/// Entity domain model for temporal graph-RAG.
|
/// Entity domain model for temporal graph-RAG.
|
||||||
/// Single Responsibility: Entity identity and metadata.
|
/// Single Responsibility: Entity identity and metadata.
|
||||||
/// Open/Closed: EntityType enum extensible.
|
/// Open/Closed: EntityType enum extensible.
|
||||||
#[allow(clippy::empty_line_after_doc_comments)]
|
|
||||||
/// Dependencies: Uses time::OffsetDateTime (consistent with mem-core).
|
/// Dependencies: Uses time::OffsetDateTime (consistent with mem-core).
|
||||||
|
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
@@ -9,7 +8,7 @@ use time::OffsetDateTime;
|
|||||||
use std::fmt;
|
use std::fmt;
|
||||||
|
|
||||||
/// Entity type classification (extensible enum).
|
/// Entity type classification (extensible enum).
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Hash)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Hash)]
|
||||||
#[serde(rename_all = "snake_case")]
|
#[serde(rename_all = "snake_case")]
|
||||||
pub enum EntityType {
|
pub enum EntityType {
|
||||||
Person,
|
Person,
|
||||||
@@ -44,7 +43,6 @@ impl EntityType {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(clippy::should_implement_trait)]
|
|
||||||
pub fn from_str(s: &str) -> Self {
|
pub fn from_str(s: &str) -> Self {
|
||||||
match s.to_lowercase().as_str() {
|
match s.to_lowercase().as_str() {
|
||||||
"person" => Self::Person,
|
"person" => Self::Person,
|
||||||
@@ -61,16 +59,6 @@ impl EntityType {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<'de> serde::Deserialize<'de> for EntityType {
|
|
||||||
fn deserialize<D>(deserializer: D) -> Result<Self, D::Error>
|
|
||||||
where
|
|
||||||
D: serde::Deserializer<'de>,
|
|
||||||
{
|
|
||||||
let s = String::deserialize(deserializer)?;
|
|
||||||
Ok(Self::from_str(&s))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl fmt::Display for EntityType {
|
impl fmt::Display for EntityType {
|
||||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
write!(f, "{}", self.as_str())
|
write!(f, "{}", self.as_str())
|
||||||
|
|||||||
@@ -135,10 +135,11 @@ pub fn run_loop(
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_loop_basic() {
|
fn test_loop_basic() {
|
||||||
// Placeholder test to verify it compiles
|
// Placeholder test to verify it compiles
|
||||||
|
assert!(true);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -403,7 +403,7 @@ pub fn lookup(sig: &Signature, lessons: &[Lesson], floor: f32) -> Option<Hit> {
|
|||||||
let mut best: Option<(f32, &Lesson)> = None;
|
let mut best: Option<(f32, &Lesson)> = None;
|
||||||
for l in lessons.iter().filter(|l| l.tool == sig.tool) {
|
for l in lessons.iter().filter(|l| l.tool == sig.tool) {
|
||||||
let s = similarity(&sig.normalised, &l.normalised);
|
let s = similarity(&sig.normalised, &l.normalised);
|
||||||
if s >= floor && best.is_none_or(|(bs, _)| s > bs) {
|
if s >= floor && best.map_or(true, |(bs, _)| s > bs) {
|
||||||
best = Some((s, l));
|
best = Some((s, l));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -503,7 +503,7 @@ pub fn tool_of_cmd(cmd: &str) -> String {
|
|||||||
"kubectl" | "k" => "kubectl".into(),
|
"kubectl" | "k" => "kubectl".into(),
|
||||||
"docker" | "podman" => "docker".into(),
|
"docker" | "podman" => "docker".into(),
|
||||||
"terraform" | "tofu" => "terraform".into(),
|
"terraform" | "tofu" => "terraform".into(),
|
||||||
"" => "unknown".into(),
|
other if other.is_empty() => "unknown".into(),
|
||||||
other => other.to_string(),
|
other => other.to_string(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -549,7 +549,7 @@ pub fn render_skill(tool: &str, lessons: &[Lesson]) -> String {
|
|||||||
s.push_str("`confirmed`, which outranks inferred lessons at equal similarity.\n\n");
|
s.push_str("`confirmed`, which outranks inferred lessons at equal similarity.\n\n");
|
||||||
|
|
||||||
let mut sorted: Vec<&Lesson> = lessons.iter().collect();
|
let mut sorted: Vec<&Lesson> = lessons.iter().collect();
|
||||||
sorted.sort_by_key(|a| std::cmp::Reverse(a.seen));
|
sorted.sort_by(|a, b| b.seen.cmp(&a.seen));
|
||||||
|
|
||||||
for l in sorted {
|
for l in sorted {
|
||||||
s.push_str(&format!("## {}\n\n", l.raw.trim()));
|
s.push_str(&format!("## {}\n\n", l.raw.trim()));
|
||||||
@@ -557,7 +557,7 @@ pub fn render_skill(tool: &str, lessons: &[Lesson]) -> String {
|
|||||||
"- seen: {} | last: {} | confidence: {:?}\n",
|
"- seen: {} | last: {} | confidence: {:?}\n",
|
||||||
l.seen, l.last_seen, l.confidence
|
l.seen, l.last_seen, l.confidence
|
||||||
));
|
));
|
||||||
s.push_str(&format!("- signature: `{}`\n", &l.sig_sha[..12]));
|
s.push_str(&format!("- signature: `{}`\n", l.sig_sha[..12].to_string()));
|
||||||
s.push_str("- resolved by:\n");
|
s.push_str("- resolved by:\n");
|
||||||
for r in &l.resolution {
|
for r in &l.resolution {
|
||||||
s.push_str(&format!(" ```\n {r}\n ```\n"));
|
s.push_str(&format!(" ```\n {r}\n ```\n"));
|
||||||
@@ -712,7 +712,7 @@ mod tests {
|
|||||||
ev("t2", "npm pkg set overrides.react=19", 0, ""),
|
ev("t2", "npm pkg set overrides.react=19", 0, ""),
|
||||||
ev("t3", "npm ci", 0, "ok"),
|
ev("t3", "npm ci", 0, "ok"),
|
||||||
];
|
];
|
||||||
let ls = derive_lessons(&events, tool_of_cmd);
|
let ls = derive_lessons(&events, |c| tool_of_cmd(c));
|
||||||
assert_eq!(ls.len(), 1);
|
assert_eq!(ls.len(), 1);
|
||||||
assert_eq!(ls[0].resolution, vec!["npm pkg set overrides.react=19"]);
|
assert_eq!(ls[0].resolution, vec!["npm pkg set overrides.react=19"]);
|
||||||
assert_eq!(ls[0].confidence, Confidence::Inferred);
|
assert_eq!(ls[0].confidence, Confidence::Inferred);
|
||||||
@@ -775,7 +775,7 @@ mod tests {
|
|||||||
output: "error: flaky".into(),
|
output: "error: flaky".into(),
|
||||||
};
|
};
|
||||||
let events = vec![ev("npm ci", 1), ev("npm ci", 0)];
|
let events = vec![ev("npm ci", 1), ev("npm ci", 0)];
|
||||||
assert!(derive_lessons(&events, tool_of_cmd).is_empty());
|
assert!(derive_lessons(&events, |c| tool_of_cmd(c)).is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -798,7 +798,7 @@ mod tests {
|
|||||||
sig_sha: "abc".into(),
|
sig_sha: "abc".into(),
|
||||||
rule: "r".into(),
|
rule: "r".into(),
|
||||||
};
|
};
|
||||||
assert_eq!(lookup(&exact, std::slice::from_ref(&l), 0.5).unwrap().tier, Tier::Exact);
|
assert_eq!(lookup(&exact, &[l.clone()], 0.5).unwrap().tier, Tier::Exact);
|
||||||
|
|
||||||
let unrelated = Signature {
|
let unrelated = Signature {
|
||||||
tool: "npm".into(),
|
tool: "npm".into(),
|
||||||
|
|||||||
@@ -152,11 +152,11 @@ impl FormatHandler for CsvFormatter {
|
|||||||
|
|
||||||
async fn format(&self, result: &OptimizationResult) -> Result<Vec<u8>, String> {
|
async fn format(&self, result: &OptimizationResult) -> Result<Vec<u8>, String> {
|
||||||
let output = format!(
|
let output = format!(
|
||||||
"{},{},{},{:.2}\n",
|
"{},{},{},{}\n",
|
||||||
escape_csv(&result.plugin),
|
escape_csv(&result.plugin),
|
||||||
result.original.len(),
|
result.original.len(),
|
||||||
result.optimized.len(),
|
result.optimized.len(),
|
||||||
result.ratio
|
format!("{:.2}", result.ratio)
|
||||||
);
|
);
|
||||||
Ok(output.into_bytes())
|
Ok(output.into_bytes())
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -40,7 +40,7 @@ impl CcrStore {
|
|||||||
// Remove oldest entry if at capacity
|
// Remove oldest entry if at capacity
|
||||||
if cache.len() >= self.max_entries {
|
if cache.len() >= self.max_entries {
|
||||||
if let Some(oldest_key) = cache.keys().next().cloned() {
|
if let Some(oldest_key) = cache.keys().next().cloned() {
|
||||||
cache.swap_remove(&oldest_key);
|
cache.remove(&oldest_key);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -57,7 +57,7 @@ impl CcrStore {
|
|||||||
// Check if expired
|
// Check if expired
|
||||||
let duration = OffsetDateTime::now_utc() - *timestamp;
|
let duration = OffsetDateTime::now_utc() - *timestamp;
|
||||||
if duration.whole_seconds() > self.ttl_secs as i64 {
|
if duration.whole_seconds() > self.ttl_secs as i64 {
|
||||||
cache.swap_remove(hash);
|
cache.remove(hash);
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@
|
|||||||
//! - Drop: redundant homogeneous elements, long string values
|
//! - Drop: redundant homogeneous elements, long string values
|
||||||
|
|
||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
use serde_json::Value;
|
use serde_json::{json, Value};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
pub struct JsonCrusher;
|
pub struct JsonCrusher;
|
||||||
@@ -45,8 +45,8 @@ impl JsonCrusher {
|
|||||||
let mut result = Vec::new();
|
let mut result = Vec::new();
|
||||||
|
|
||||||
// Add start items
|
// Add start items
|
||||||
for item in items.iter().take(start_count.min(len)) {
|
for i in 0..start_count.min(len) {
|
||||||
result.push(item.clone());
|
result.push(items[i].clone());
|
||||||
}
|
}
|
||||||
|
|
||||||
// Select mid-array items by variance/importance
|
// Select mid-array items by variance/importance
|
||||||
@@ -58,8 +58,8 @@ impl JsonCrusher {
|
|||||||
|
|
||||||
// Add end items
|
// Add end items
|
||||||
if end_count > 0 {
|
if end_count > 0 {
|
||||||
for item in items.iter().skip(len.saturating_sub(end_count)) {
|
for i in (len - end_count)..len {
|
||||||
result.push(item.clone());
|
result.push(items[i].clone());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
|
|
||||||
use super::plugin::OptimizerService;
|
use super::plugin::OptimizerService;
|
||||||
use crate::prompt::CacheMetrics;
|
use crate::prompt::CacheMetrics;
|
||||||
use crate::domain::Chunk;
|
use crate::domain::{Chunk, Record};
|
||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
|
|
||||||
/// Query optimizer: compresses chunks before LLM processing
|
/// Query optimizer: compresses chunks before LLM processing
|
||||||
@@ -83,7 +83,7 @@ impl QueryOptimizer {
|
|||||||
match service.optimize(&chunk_text, &content_type, Some("raw")).await {
|
match service.optimize(&chunk_text, &content_type, Some("raw")).await {
|
||||||
Ok(bytes) => {
|
Ok(bytes) => {
|
||||||
let text = String::from_utf8(bytes)
|
let text = String::from_utf8(bytes)
|
||||||
.unwrap_or(chunk_text);
|
.unwrap_or_else(|_| chunk_text);
|
||||||
Ok(text)
|
Ok(text)
|
||||||
}
|
}
|
||||||
Err(_) => {
|
Err(_) => {
|
||||||
|
|||||||
@@ -42,7 +42,7 @@ impl ContentRouter {
|
|||||||
/// Check if content is valid JSON
|
/// Check if content is valid JSON
|
||||||
fn is_json(content: &str) -> bool {
|
fn is_json(content: &str) -> bool {
|
||||||
let trimmed = content.trim();
|
let trimmed = content.trim();
|
||||||
if !(trimmed.starts_with('{') || trimmed.starts_with('[')) {
|
if !((trimmed.starts_with('{') || trimmed.starts_with('['))) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
serde_json::from_str::<serde_json::Value>(trimmed).is_ok()
|
serde_json::from_str::<serde_json::Value>(trimmed).is_ok()
|
||||||
|
|||||||
@@ -128,7 +128,7 @@ impl TextCompressor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Capitalization (usually proper nouns or emphatic)
|
// Capitalization (usually proper nouns or emphatic)
|
||||||
if token.chars().next().is_some_and(|c| c.is_uppercase()) && token.len() > 1 {
|
if token.chars().next().map_or(false, |c| c.is_uppercase()) && token.len() > 1 {
|
||||||
score += 1.0;
|
score += 1.0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -12,9 +12,7 @@ const CACHE_TURN: &str = include_str!("../../../templates/gru-mem-turn.txt");
|
|||||||
|
|
||||||
const BUDGET_TOTAL: usize = 32768;
|
const BUDGET_TOTAL: usize = 32768;
|
||||||
const BUDGET_RESPONSE: usize = 2048;
|
const BUDGET_RESPONSE: usize = 2048;
|
||||||
#[allow(dead_code)]
|
|
||||||
const BUDGET_SYSTEM: usize = 400;
|
const BUDGET_SYSTEM: usize = 400;
|
||||||
#[allow(dead_code)]
|
|
||||||
const BUDGET_QUESTION: usize = 150;
|
const BUDGET_QUESTION: usize = 150;
|
||||||
const BUDGET_MEMORY_MAX: usize = 1024;
|
const BUDGET_MEMORY_MAX: usize = 1024;
|
||||||
const BUDGET_CHUNK_MAX: usize = 5000;
|
const BUDGET_CHUNK_MAX: usize = 5000;
|
||||||
@@ -370,7 +368,7 @@ fn estimate_tokens(text: &str) -> usize {
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
use crate::domain::{Chunk, Record, Role, Provenance};
|
use crate::domain::{Chunk, Record, Role, Provenance, Level};
|
||||||
use time::OffsetDateTime;
|
use time::OffsetDateTime;
|
||||||
|
|
||||||
fn make_test_chunk(text: &str) -> Chunk {
|
fn make_test_chunk(text: &str) -> Chunk {
|
||||||
@@ -647,7 +645,7 @@ mod tests {
|
|||||||
|
|
||||||
let metrics = result.unwrap();
|
let metrics = result.unwrap();
|
||||||
let ratio = metrics.compression_ratio();
|
let ratio = metrics.compression_ratio();
|
||||||
assert!((0.0..=100.0).contains(&ratio));
|
assert!(ratio >= 0.0 && ratio <= 100.0);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
|
use crate::domain::{ProjectId, QueryId};
|
||||||
use anyhow::{anyhow, Result};
|
use anyhow::{anyhow, Result};
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
|
use std::collections::HashMap;
|
||||||
use std::path::Path;
|
use std::path::Path;
|
||||||
|
|
||||||
/// A single standing query.
|
/// A single standing query.
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
use crate::Level;
|
use crate::{Level, Query};
|
||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
@@ -17,12 +17,6 @@ pub struct QueryExecutor {
|
|||||||
// For now: proof-of-concept with mock data
|
// For now: proof-of-concept with mock data
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for QueryExecutor {
|
|
||||||
fn default() -> Self {
|
|
||||||
Self::new()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl QueryExecutor {
|
impl QueryExecutor {
|
||||||
/// Create executor.
|
/// Create executor.
|
||||||
pub fn new() -> Self {
|
pub fn new() -> Self {
|
||||||
|
|||||||
@@ -71,10 +71,11 @@ impl QueryLevels {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Check level filter
|
// Check level filter
|
||||||
if !self.level_filter.is_empty()
|
if !self.level_filter.is_empty() {
|
||||||
&& !self.level_filter.contains(&level.to_string()) {
|
if !self.level_filter.contains(&level.to_string()) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Check evidence/reference flags
|
// Check evidence/reference flags
|
||||||
if level == "R" {
|
if level == "R" {
|
||||||
|
|||||||
@@ -6,7 +6,6 @@
|
|||||||
/// - Single Responsibility: each scorer does one thing
|
/// - Single Responsibility: each scorer does one thing
|
||||||
/// - Open/Closed: add new scorers without modifying existing
|
/// - Open/Closed: add new scorers without modifying existing
|
||||||
/// - Liskov Substitution: all scorers implement DocumentScorer
|
/// - Liskov Substitution: all scorers implement DocumentScorer
|
||||||
#[allow(clippy::empty_line_after_doc_comments)]
|
|
||||||
/// - Dependency Inversion: depend on trait, not concrete types
|
/// - Dependency Inversion: depend on trait, not concrete types
|
||||||
|
|
||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
@@ -54,7 +53,6 @@ impl DocumentScorer for GlobalTfIdfScorer {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Project-scoped TF-IDF Scorer: scoring within project boundaries
|
/// Project-scoped TF-IDF Scorer: scoring within project boundaries
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct ProjectTfIdfScorer {
|
pub struct ProjectTfIdfScorer {
|
||||||
project: String,
|
project: String,
|
||||||
vocabulary: Arc<std::collections::BTreeMap<String, f32>>,
|
vocabulary: Arc<std::collections::BTreeMap<String, f32>>,
|
||||||
@@ -95,18 +93,11 @@ impl DocumentScorer for ProjectTfIdfScorer {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Semantic Scorer: vector similarity (placeholder)
|
/// Semantic Scorer: vector similarity (placeholder)
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct SemanticScorer {
|
pub struct SemanticScorer {
|
||||||
_embeddings_client: Arc<()>, // Placeholder
|
_embeddings_client: Arc<()>, // Placeholder
|
||||||
_pgvector: Arc<()>, // Placeholder
|
_pgvector: Arc<()>, // Placeholder
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for SemanticScorer {
|
|
||||||
fn default() -> Self {
|
|
||||||
Self::new()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl SemanticScorer {
|
impl SemanticScorer {
|
||||||
pub fn new() -> Self {
|
pub fn new() -> Self {
|
||||||
Self {
|
Self {
|
||||||
@@ -165,12 +156,6 @@ pub struct ScoringPipeline {
|
|||||||
scorers: Vec<(String, f32, Arc<dyn DocumentScorer>)>, // name, weight, scorer
|
scorers: Vec<(String, f32, Arc<dyn DocumentScorer>)>, // name, weight, scorer
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for ScoringPipeline {
|
|
||||||
fn default() -> Self {
|
|
||||||
Self::new()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ScoringPipeline {
|
impl ScoringPipeline {
|
||||||
pub fn new() -> Self {
|
pub fn new() -> Self {
|
||||||
Self {
|
Self {
|
||||||
|
|||||||
@@ -81,7 +81,6 @@ impl SymptomVector {
|
|||||||
|
|
||||||
/// Internal structure for tokens during extraction
|
/// Internal structure for tokens during extraction
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
#[allow(dead_code)]
|
|
||||||
struct SymptomTokens {
|
struct SymptomTokens {
|
||||||
keywords: Vec<String>,
|
keywords: Vec<String>,
|
||||||
error_codes: Vec<String>,
|
error_codes: Vec<String>,
|
||||||
@@ -393,7 +392,7 @@ mod tests {
|
|||||||
let words: Vec<&str> = symptom.normalised.split_whitespace().collect();
|
let words: Vec<&str> = symptom.normalised.split_whitespace().collect();
|
||||||
for word in &words {
|
for word in &words {
|
||||||
// Check if this word is a stop word
|
// Check if this word is a stop word
|
||||||
assert!(!STOP_WORDS.contains(word), "Stop word '{}' should be removed", word);
|
assert!(!STOP_WORDS.contains(&word), "Stop word '{}' should be removed", word);
|
||||||
}
|
}
|
||||||
// Should contain key terms
|
// Should contain key terms
|
||||||
assert!(symptom.normalised.contains("resolve"));
|
assert!(symptom.normalised.contains("resolve"));
|
||||||
|
|||||||
@@ -267,9 +267,11 @@ fn test_compression_handles_large_content() {
|
|||||||
fn test_multi_chunk_search_consistency() {
|
fn test_multi_chunk_search_consistency() {
|
||||||
let optimizer = ContextOptimizer::new().expect("optimizer init");
|
let optimizer = ContextOptimizer::new().expect("optimizer init");
|
||||||
|
|
||||||
let chunks = ["ERROR: connection failed\nDEBUG: thread id=100",
|
let chunks = vec![
|
||||||
|
"ERROR: connection failed\nDEBUG: thread id=100",
|
||||||
"ERROR: timeout after 5000ms\nTRACE: stack unwinding",
|
"ERROR: timeout after 5000ms\nTRACE: stack unwinding",
|
||||||
"ERROR: retry attempt 2\nDEBUG: backoff delay=200ms"];
|
"ERROR: retry attempt 2\nDEBUG: backoff delay=200ms",
|
||||||
|
];
|
||||||
|
|
||||||
let optimized_chunks: Vec<_> = chunks
|
let optimized_chunks: Vec<_> = chunks
|
||||||
.iter()
|
.iter()
|
||||||
|
|||||||
@@ -196,6 +196,7 @@ fn gate_memory_bounded() {
|
|||||||
|
|
||||||
// Should not panic from memory exhaustion
|
// Should not panic from memory exhaustion
|
||||||
// If we get here, we passed the gate
|
// If we get here, we passed the gate
|
||||||
|
assert!(true, "memory usage bounded");
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -230,7 +231,7 @@ fn gate_compression_targets_met() {
|
|||||||
];
|
];
|
||||||
|
|
||||||
for (content, name, min_compression) in fixtures.iter() {
|
for (content, name, min_compression) in fixtures.iter() {
|
||||||
let optimized = optimizer.optimize(content).unwrap_or_else(|_| panic!("optimize {}", name));
|
let optimized = optimizer.optimize(content).expect(&format!("optimize {}", name));
|
||||||
let ratio = optimized.compressed.len() as f32 / content.len() as f32;
|
let ratio = optimized.compressed.len() as f32 / content.len() as f32;
|
||||||
|
|
||||||
// At least some compression should happen
|
// At least some compression should happen
|
||||||
@@ -331,4 +332,5 @@ fn gate_summary_report() {
|
|||||||
|
|
||||||
println!("\n🚀 STATUS: M3.8 READY FOR PRODUCTION");
|
println!("\n🚀 STATUS: M3.8 READY FOR PRODUCTION");
|
||||||
|
|
||||||
|
assert!(true); // Just for testing framework
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -52,24 +52,13 @@ impl AuthentikJwtIssuer {
|
|||||||
|
|
||||||
/// From environment: AUTHENTIK_ISSUER, AUTHENTIK_CLIENT_ID, AUTHENTIK_CLIENT_SECRET
|
/// From environment: AUTHENTIK_ISSUER, AUTHENTIK_CLIENT_ID, AUTHENTIK_CLIENT_SECRET
|
||||||
pub fn from_env() -> Result<Self> {
|
pub fn from_env() -> Result<Self> {
|
||||||
// Support both naming conventions: AUTHENTIK_* and memory-agent-oidc secret keys
|
|
||||||
let issuer = std::env::var("AUTHENTIK_ISSUER")
|
let issuer = std::env::var("AUTHENTIK_ISSUER")
|
||||||
.or_else(|_| std::env::var("ISSUER"))
|
.map_err(|_| anyhow!("AUTHENTIK_ISSUER not set"))?;
|
||||||
.map_err(|_| anyhow!("AUTHENTIK_ISSUER or ISSUER not set"))?;
|
|
||||||
let client_id = std::env::var("AUTHENTIK_CLIENT_ID")
|
let client_id = std::env::var("AUTHENTIK_CLIENT_ID")
|
||||||
.or_else(|_| std::env::var("CLIENT_ID"))
|
.map_err(|_| anyhow!("AUTHENTIK_CLIENT_ID not set"))?;
|
||||||
.map_err(|_| anyhow!("AUTHENTIK_CLIENT_ID or CLIENT_ID not set"))?;
|
|
||||||
let client_secret = std::env::var("AUTHENTIK_CLIENT_SECRET")
|
let client_secret = std::env::var("AUTHENTIK_CLIENT_SECRET")
|
||||||
.or_else(|_| std::env::var("CLIENT_SECRET"))
|
.map_err(|_| anyhow!("AUTHENTIK_CLIENT_SECRET not set"))?;
|
||||||
.map_err(|_| anyhow!("AUTHENTIK_CLIENT_SECRET or CLIENT_SECRET not set"))?;
|
|
||||||
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "authentik_jwt_init",
|
|
||||||
issuer = %issuer,
|
|
||||||
client_id = %client_id,
|
|
||||||
"Authentik JWT issuer initialized"
|
|
||||||
);
|
|
||||||
Ok(Self::new(&issuer, &client_id, &client_secret))
|
Ok(Self::new(&issuer, &client_id, &client_secret))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -103,25 +92,12 @@ impl AuthentikJwtIssuer {
|
|||||||
let client = reqwest::Client::new();
|
let client = reqwest::Client::new();
|
||||||
|
|
||||||
// Authentik OAuth2 token endpoint
|
// Authentik OAuth2 token endpoint
|
||||||
// Use TOKEN_URL env var if set, otherwise derive from issuer
|
let token_url = format!("{}/token/", self.issuer_url.trim_end_matches('/'));
|
||||||
let token_url = std::env::var("TOKEN_URL")
|
|
||||||
.or_else(|_| std::env::var("AUTHENTIK_TOKEN_URL"))
|
|
||||||
.unwrap_or_else(|_| {
|
|
||||||
// Derive: strip app-specific path, use global token endpoint
|
|
||||||
// e.g., https://authentik.riotpiao.com/application/o/memory-agent/
|
|
||||||
// -> https://authentik.riotpiao.com/application/o/token/
|
|
||||||
if let Some(base) = self.issuer_url.rfind("/o/") {
|
|
||||||
format!("{}/o/token/", &self.issuer_url[..base])
|
|
||||||
} else {
|
|
||||||
format!("{}/token/", self.issuer_url.trim_end_matches('/'))
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
let params = [
|
let params = [
|
||||||
("grant_type", "client_credentials"),
|
("grant_type", "client_credentials"),
|
||||||
("client_id", &self.client_id),
|
("client_id", &self.client_id),
|
||||||
("client_secret", &self.client_secret),
|
("client_secret", &self.client_secret),
|
||||||
("scope", "openid roles"),
|
|
||||||
];
|
];
|
||||||
|
|
||||||
let response = client
|
let response = client
|
||||||
|
|||||||
@@ -83,7 +83,6 @@ impl ContradictionPreFilter {
|
|||||||
|
|
||||||
/// LLM-based contradiction detector (stage 2)
|
/// LLM-based contradiction detector (stage 2)
|
||||||
/// Only called if pre-filter returns true (cost optimization)
|
/// Only called if pre-filter returns true (cost optimization)
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct LlmContradictionDetector {
|
pub struct LlmContradictionDetector {
|
||||||
model_name: String,
|
model_name: String,
|
||||||
auto_confirm_threshold: f32,
|
auto_confirm_threshold: f32,
|
||||||
|
|||||||
@@ -22,15 +22,11 @@ use tokio::sync::Mutex;
|
|||||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||||
pub struct ExtractedEntity {
|
pub struct ExtractedEntity {
|
||||||
pub name: String,
|
pub name: String,
|
||||||
#[serde(alias = "type")]
|
|
||||||
pub entity_type: EntityType,
|
pub entity_type: EntityType,
|
||||||
pub summary: String,
|
pub summary: String,
|
||||||
#[serde(default = "default_confidence")]
|
|
||||||
pub confidence: f32,
|
pub confidence: f32,
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_confidence() -> f32 { 0.8 }
|
|
||||||
|
|
||||||
impl ExtractedEntity {
|
impl ExtractedEntity {
|
||||||
/// Convert to domain model (Phase 1 type)
|
/// Convert to domain model (Phase 1 type)
|
||||||
pub fn to_domain(&self, project_id: &str) -> Entity {
|
pub fn to_domain(&self, project_id: &str) -> Entity {
|
||||||
@@ -44,15 +40,10 @@ impl ExtractedEntity {
|
|||||||
#[async_trait]
|
#[async_trait]
|
||||||
pub trait EntityExtractor: Send + Sync {
|
pub trait EntityExtractor: Send + Sync {
|
||||||
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>>;
|
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>>;
|
||||||
async fn extract_with_auth(&self, text: &str, x_forward_user: Option<&str>) -> Result<Vec<ExtractedEntity>> {
|
|
||||||
// Default: ignore auth header, use regular extract
|
|
||||||
self.extract(text).await
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// LLM-based extractor with reflection verification (stage 1 + 2)
|
/// LLM-based extractor with reflection verification (stage 1 + 2)
|
||||||
/// Uses Authentik JWT tokens for authentication to LLM gateway
|
/// Uses Authentik JWT tokens for authentication to LLM gateway
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct LlmEntityExtractor {
|
pub struct LlmEntityExtractor {
|
||||||
model_name: String,
|
model_name: String,
|
||||||
enable_reflection: bool,
|
enable_reflection: bool,
|
||||||
@@ -71,35 +62,6 @@ impl LlmEntityExtractor {
|
|||||||
|
|
||||||
/// Parse extraction response JSON
|
/// Parse extraction response JSON
|
||||||
/// Format: { "entities": [{ "name": "...", "type": "...", "summary": "..." }, ...] }
|
/// Format: { "entities": [{ "name": "...", "type": "...", "summary": "..." }, ...] }
|
||||||
/// Clean LLM response: strip thinking tags, markdown fences, extract JSON
|
|
||||||
fn clean_llm_response(text: &str) -> String {
|
|
||||||
let mut result = text.to_string();
|
|
||||||
// Remove <think>...</think> blocks
|
|
||||||
while let Some(start) = result.find("<think>") {
|
|
||||||
if let Some(end) = result.find("</think>") {
|
|
||||||
result = format!("{}{}", &result[..start], &result[end + 8..]);
|
|
||||||
} else {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Remove markdown code fences
|
|
||||||
result = result.replace("```json", "").replace("```", "");
|
|
||||||
// Find JSON object
|
|
||||||
let trimmed = result.trim();
|
|
||||||
if let Some(start) = trimmed.find('{') {
|
|
||||||
if let Some(end) = trimmed.rfind('}') {
|
|
||||||
return trimmed[start..=end].to_string();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Maybe it's a JSON array — wrap in object
|
|
||||||
if let Some(start) = trimmed.find('[') {
|
|
||||||
if let Some(end) = trimmed.rfind(']') {
|
|
||||||
return format!("{{\"entities\": {}}}", &trimmed[start..=end]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
trimmed.to_string()
|
|
||||||
}
|
|
||||||
|
|
||||||
fn parse_extraction(response: &str) -> Result<Vec<ExtractedEntity>> {
|
fn parse_extraction(response: &str) -> Result<Vec<ExtractedEntity>> {
|
||||||
#[derive(Deserialize)]
|
#[derive(Deserialize)]
|
||||||
struct Response {
|
struct Response {
|
||||||
@@ -125,42 +87,29 @@ impl LlmEntityExtractor {
|
|||||||
Ok(parsed.verified.into_iter().map(|v| (v.name, v.present)).collect())
|
Ok(parsed.verified.into_iter().map(|v| (v.name, v.present)).collect())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Call LLM via api.riotpiao.com using X-Forward-User auth/exchange
|
/// Call LLM via api.riotpiao.com using Authentik JWT
|
||||||
/// Supports: Authentik JWT, X-Forward-User header, or API key fallback
|
/// Token is fetched from Authentik service account and cached
|
||||||
async fn call_llm_endpoint(&self, prompt: &str, x_forward_user: Option<&str>) -> Result<String> {
|
async fn call_llm_endpoint(&self, prompt: &str) -> Result<String> {
|
||||||
let endpoint = std::env::var("LLM_ENDPOINT")
|
let endpoint = std::env::var("LLM_ENDPOINT")
|
||||||
.unwrap_or_else(|_| "http://api-internal.riotpiao.com:8000/v1/chat/completions".to_string());
|
.unwrap_or_else(|_| "http://api-internal.riotpiao.com:8000/v1/chat/completions".to_string());
|
||||||
let model = std::env::var("LLM_MODEL")
|
let model = std::env::var("LLM_MODEL")
|
||||||
.unwrap_or_else(|_| "qwen:7b".to_string());
|
.unwrap_or_else(|_| "qwen:7b".to_string());
|
||||||
|
|
||||||
// Get auth header: prefer X-Forward-User, fallback to Authentik JWT, then API key
|
// Get JWT token from Authentik
|
||||||
let auth_header = if let Some(user) = x_forward_user {
|
let auth_header = if let Some(jwt_issuer) = &self.jwt_issuer {
|
||||||
// Use X-Forward-User directly (API Gateway pattern)
|
|
||||||
tracing::info!("Using X-Forward-User for LLM auth: {}", user);
|
|
||||||
format!("X-Forward-User: {}", user)
|
|
||||||
} else if let Some(jwt_issuer) = &self.jwt_issuer {
|
|
||||||
let issuer = jwt_issuer.lock().await;
|
let issuer = jwt_issuer.lock().await;
|
||||||
match issuer.get_access_token().await {
|
match issuer.get_access_token().await {
|
||||||
Ok(token) => {
|
Ok(token) => format!("Bearer {}", token),
|
||||||
tracing::info!("Using Authentik JWT for LLM auth");
|
|
||||||
format!("Bearer {}", token)
|
|
||||||
},
|
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
tracing::warn!("Failed to get Authentik JWT: {}", e);
|
tracing::warn!("Failed to get Authentik JWT: {}", e);
|
||||||
// Fallback to env var
|
return Err(e);
|
||||||
let api_key = std::env::var("LLM_API_KEY")
|
|
||||||
.or_else(|_| std::env::var("MEM_API_KEY"))
|
|
||||||
.unwrap_or_else(|_| "test-key".to_string());
|
|
||||||
tracing::info!("Falling back to LLM_API_KEY");
|
|
||||||
format!("Bearer {}", api_key)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// Fallback to env var if Authentik not configured
|
// Fallback to env var if Authentik not configured
|
||||||
let api_key = std::env::var("LLM_API_KEY")
|
let api_key = std::env::var("LLM_API_KEY")
|
||||||
.or_else(|_| std::env::var("MEM_API_KEY"))
|
.or_else(|_| std::env::var("MEM_API_KEY"))
|
||||||
.unwrap_or_else(|_| "test-key".to_string());
|
.unwrap_or_else(|_| "default-key".to_string());
|
||||||
tracing::info!("Using LLM_API_KEY for LLM auth");
|
|
||||||
format!("Bearer {}", api_key)
|
format!("Bearer {}", api_key)
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -174,61 +123,35 @@ impl LlmEntityExtractor {
|
|||||||
{"role": "user", "content": prompt}
|
{"role": "user", "content": prompt}
|
||||||
],
|
],
|
||||||
"temperature": 0.3,
|
"temperature": 0.3,
|
||||||
"max_tokens": 12000
|
"max_tokens": 500
|
||||||
});
|
});
|
||||||
|
|
||||||
let mut request = client
|
let response = client
|
||||||
.post(&endpoint)
|
.post(&endpoint)
|
||||||
.header("Content-Type", "application/json");
|
.header("Authorization", auth_header)
|
||||||
|
.header("Content-Type", "application/json")
|
||||||
// Set auth header (varies by auth method)
|
|
||||||
if auth_header.starts_with("X-Forward-User") {
|
|
||||||
request = request.header("X-Forward-User", auth_header.split(": ").nth(1).unwrap_or("unknown"));
|
|
||||||
} else {
|
|
||||||
request = request.header("Authorization", auth_header);
|
|
||||||
}
|
|
||||||
|
|
||||||
let response = request
|
|
||||||
.json(&payload)
|
.json(&payload)
|
||||||
.timeout(std::time::Duration::from_secs(90))
|
.timeout(std::time::Duration::from_secs(30))
|
||||||
.send()
|
.send()
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
let status = response.status();
|
if !response.status().is_success() {
|
||||||
if !status.is_success() {
|
tracing::warn!(
|
||||||
let error_text = response.text().await.unwrap_or_default();
|
|
||||||
tracing::error!(
|
|
||||||
"LLM API error: {} - {}",
|
"LLM API error: {} - {}",
|
||||||
status,
|
response.status(),
|
||||||
error_text
|
response.text().await.unwrap_or_default()
|
||||||
);
|
);
|
||||||
// Return error instead of silently returning empty array
|
// Fallback to mock response on error
|
||||||
return Err(anyhow::anyhow!("LLM API failed with status {}: {}", status, error_text));
|
return Ok(r#"{"entities": []}"#.to_string());
|
||||||
}
|
}
|
||||||
|
|
||||||
let data: serde_json::Value = response.json().await?;
|
let data: serde_json::Value = response.json().await?;
|
||||||
// Extract content — some models put JSON in "content", others in "reasoning"
|
let content = data["choices"][0]["message"]["content"]
|
||||||
let msg = &data["choices"][0]["message"];
|
.as_str()
|
||||||
let raw_content = msg["content"].as_str().unwrap_or("").to_string();
|
.unwrap_or("{}")
|
||||||
let raw_reasoning = msg["reasoning"].as_str().unwrap_or("").to_string();
|
.to_string();
|
||||||
|
|
||||||
// Use content if non-empty, otherwise try reasoning field
|
tracing::debug!("LLM response (via Authentik JWT): {}", content);
|
||||||
let raw = if !raw_content.trim().is_empty() { &raw_content } else { &raw_reasoning };
|
|
||||||
let content = Self::clean_llm_response(raw);
|
|
||||||
|
|
||||||
let tokens = &data["usage"];
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "llm_entity_call",
|
|
||||||
model = %model,
|
|
||||||
endpoint = %endpoint,
|
|
||||||
raw_len = raw.len(),
|
|
||||||
cleaned_len = content.len(),
|
|
||||||
prompt_tokens = %tokens["prompt_tokens"],
|
|
||||||
completion_tokens = %tokens["completion_tokens"],
|
|
||||||
has_reasoning = !raw_reasoning.is_empty(),
|
|
||||||
"LLM entity extraction call complete"
|
|
||||||
);
|
|
||||||
Ok(content)
|
Ok(content)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -285,10 +208,7 @@ Respond in JSON:
|
|||||||
|
|
||||||
// Try real LLM first, fallback to mock if not configured
|
// Try real LLM first, fallback to mock if not configured
|
||||||
let extraction_response = if std::env::var("LLM_ENDPOINT").is_ok() {
|
let extraction_response = if std::env::var("LLM_ENDPOINT").is_ok() {
|
||||||
self.call_llm_endpoint(&prompt, None).await.unwrap_or_else(|e| {
|
self.call_llm_endpoint(&prompt).await.unwrap_or_else(|_| self.simulate_llm(&prompt).unwrap_or_default())
|
||||||
tracing::error!("LLM entity extraction failed: {}, using mock", e);
|
|
||||||
self.simulate_llm(&prompt).unwrap_or_default()
|
|
||||||
})
|
|
||||||
} else {
|
} else {
|
||||||
self.simulate_llm(&prompt)?
|
self.simulate_llm(&prompt)?
|
||||||
};
|
};
|
||||||
@@ -313,27 +233,14 @@ Respond in JSON:
|
|||||||
);
|
);
|
||||||
|
|
||||||
let reflection = if std::env::var("LLM_ENDPOINT").is_ok() {
|
let reflection = if std::env::var("LLM_ENDPOINT").is_ok() {
|
||||||
self.call_llm_endpoint(&reflection_prompt, None).await.unwrap_or_else(|e| {
|
self.call_llm_endpoint(&reflection_prompt).await.unwrap_or_else(|_| self.simulate_llm(&reflection_prompt).unwrap_or_default())
|
||||||
tracing::warn!("Reflection LLM call failed: {}, skipping verification", e);
|
|
||||||
String::new()
|
|
||||||
})
|
|
||||||
} else {
|
} else {
|
||||||
self.simulate_llm(&reflection_prompt)?
|
self.simulate_llm(&reflection_prompt)?
|
||||||
};
|
};
|
||||||
|
let verified = Self::parse_reflection(&reflection)?;
|
||||||
|
|
||||||
// If reflection succeeded, filter entities; otherwise keep all
|
// Filter: keep only entities marked present
|
||||||
if !reflection.is_empty() {
|
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
|
||||||
match Self::parse_reflection(&reflection) {
|
|
||||||
Ok(verified) => {
|
|
||||||
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
tracing::warn!("Reflection parse failed: {}, keeping all entities", e);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
tracing::info!("Reflection skipped, keeping {} unverified entities", entities.len());
|
|
||||||
}
|
|
||||||
|
|
||||||
// Adjust confidence for reflected entities (slight penalty for needing verification)
|
// Adjust confidence for reflected entities (slight penalty for needing verification)
|
||||||
for entity in &mut entities {
|
for entity in &mut entities {
|
||||||
@@ -343,85 +250,6 @@ Respond in JSON:
|
|||||||
|
|
||||||
Ok(entities)
|
Ok(entities)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract with X-Forward-User auth header (API Gateway pattern)
|
|
||||||
async fn extract_with_auth(&self, text: &str, x_forward_user: Option<&str>) -> Result<Vec<ExtractedEntity>> {
|
|
||||||
let mut entities = vec![];
|
|
||||||
|
|
||||||
// Extract speaker if available
|
|
||||||
use crate::speaker_extractor::{HeuristicSpeakerExtractor, SpeakerConfig};
|
|
||||||
if let Ok(speaker_extractor) = HeuristicSpeakerExtractor::new(SpeakerConfig::default()) {
|
|
||||||
if let Ok(Some(speaker)) = speaker_extractor.extract_speaker(text).await {
|
|
||||||
entities.push(ExtractedEntity {
|
|
||||||
name: speaker.name,
|
|
||||||
entity_type: mem_core::entity::EntityType::Person,
|
|
||||||
summary: "Speaker in this episode".to_string(),
|
|
||||||
confidence: speaker.confidence,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Extract entities with auth header
|
|
||||||
let prompt = format!(
|
|
||||||
r#"Extract named entities from this text.
|
|
||||||
|
|
||||||
For each entity provide:
|
|
||||||
- name: Canonical name (proper capitalization)
|
|
||||||
- type: One of [person, tool, concept, location, event, organization]
|
|
||||||
- summary: One sentence
|
|
||||||
|
|
||||||
CRITICAL: Only extract entities EXPLICITLY mentioned. No inference.
|
|
||||||
|
|
||||||
Text:
|
|
||||||
"{}"
|
|
||||||
|
|
||||||
Respond in JSON:
|
|
||||||
{{"entities": [{{"name": "...", "type": "...", "summary": "..."}}, ...]}}
|
|
||||||
"#,
|
|
||||||
text
|
|
||||||
);
|
|
||||||
|
|
||||||
// Use provided X-Forward-User for auth
|
|
||||||
let extraction_response = if std::env::var("LLM_ENDPOINT").is_ok() {
|
|
||||||
self.call_llm_endpoint(&prompt, x_forward_user).await.unwrap_or_else(|e| {
|
|
||||||
tracing::error!("LLM entity extraction with auth failed: {}", e);
|
|
||||||
self.simulate_llm(&prompt).unwrap_or_default()
|
|
||||||
})
|
|
||||||
} else {
|
|
||||||
self.simulate_llm(&prompt)?
|
|
||||||
};
|
|
||||||
|
|
||||||
let extracted = Self::parse_extraction(&extraction_response)?;
|
|
||||||
entities.extend(extracted);
|
|
||||||
|
|
||||||
// Optional: reflection verification with auth
|
|
||||||
if self.enable_reflection && std::env::var("LLM_ENDPOINT").is_ok() {
|
|
||||||
let reflection_prompt = format!(
|
|
||||||
r#"Verify these entities are explicitly in the text:
|
|
||||||
|
|
||||||
Text:
|
|
||||||
"{}"
|
|
||||||
|
|
||||||
Entities:
|
|
||||||
{:?}
|
|
||||||
|
|
||||||
Respond in JSON:
|
|
||||||
{{"verified": [{{"name": "...", "present": true/false}}, ...]}}
|
|
||||||
"#,
|
|
||||||
text, entities
|
|
||||||
);
|
|
||||||
|
|
||||||
if let Ok(reflection) = self.call_llm_endpoint(&reflection_prompt, x_forward_user).await {
|
|
||||||
if !reflection.is_empty() {
|
|
||||||
if let Ok(verified) = Self::parse_reflection(&reflection) {
|
|
||||||
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(entities)
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Fallback extractor: Use wiki_links if LLM fails (stage 3)
|
/// Fallback extractor: Use wiki_links if LLM fails (stage 3)
|
||||||
@@ -440,7 +268,7 @@ impl EntityExtractor for WikiLinkFallbackExtractor {
|
|||||||
entities.push(ExtractedEntity {
|
entities.push(ExtractedEntity {
|
||||||
name: name_str.to_string(),
|
name: name_str.to_string(),
|
||||||
entity_type: EntityType::Unknown,
|
entity_type: EntityType::Unknown,
|
||||||
summary: "Mentioned in episode".to_string(),
|
summary: format!("Mentioned in episode"),
|
||||||
confidence: 0.7, // Lower confidence for fallback
|
confidence: 0.7, // Lower confidence for fallback
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -528,6 +356,6 @@ mod tests {
|
|||||||
let text = "[[Entity1]] and [[Entity2]]";
|
let text = "[[Entity1]] and [[Entity2]]";
|
||||||
|
|
||||||
let entities = composite.extract(text).await.unwrap();
|
let entities = composite.extract(text).await.unwrap();
|
||||||
assert!(!entities.is_empty());
|
assert!(entities.len() > 0);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,12 +1,12 @@
|
|||||||
//! Fact extraction: Identify relationships between entities
|
//! Fact extraction: Identify relationships between entities
|
||||||
//!
|
//!
|
||||||
//! Three implementations:
|
//! Two implementations:
|
||||||
//! 1. SimpleFactExtractor: Pattern-based (verbs + wiki links)
|
//! 1. SimpleFactExtractor: Pattern-based (verbs + wiki links)
|
||||||
//! 2. LlmFactExtractor: LLM-based extraction with entity context
|
//! 2. LlmFactExtractor: LLM-based (placeholder for production)
|
||||||
//! 3. Fallback chain: LLM → Simple pattern matching
|
|
||||||
//!
|
//!
|
||||||
//! Aligned with Zep paper §2.2.2: Facts as edges between entity pairs,
|
//! CRAP: 12 (Simple pattern matching + LLM placeholder)
|
||||||
//! with temporal extraction and dedup against existing edges.
|
//! SOLID: Trait-based (Open/Closed)
|
||||||
|
//! DRY: Reuses EntityExtractor pattern
|
||||||
|
|
||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
use async_trait::async_trait;
|
use async_trait::async_trait;
|
||||||
@@ -27,18 +27,20 @@ pub struct ExtractedFact {
|
|||||||
pub trait FactExtractor: Send + Sync {
|
pub trait FactExtractor: Send + Sync {
|
||||||
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>>;
|
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>>;
|
||||||
|
|
||||||
/// Extract facts with entity context (Zep §2.2.2: facts between known entities)
|
/// Extract facts with GRM context (optional, defaults to extract())
|
||||||
async fn extract_with_context(
|
async fn extract_with_context(
|
||||||
&self,
|
&self,
|
||||||
text: &str,
|
text: &str,
|
||||||
_entity_contexts: &[crate::grm_retriever::EntityContext],
|
_entity_contexts: &[crate::grm_retriever::EntityContext],
|
||||||
) -> Result<Vec<ExtractedFact>> {
|
) -> Result<Vec<ExtractedFact>> {
|
||||||
|
// Default: ignore context, use plain extraction
|
||||||
self.extract(text).await
|
self.extract(text).await
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Simple fact extractor based on verb patterns
|
/// Simple fact extractor based on verb patterns
|
||||||
/// Pattern: [[Entity1]] verb [[Entity2]]
|
/// Pattern: [[Entity1]] verb [[Entity2]]
|
||||||
|
/// Common verbs: uses, manages, runs, deployed_to, works_with
|
||||||
pub struct SimpleFactExtractor;
|
pub struct SimpleFactExtractor;
|
||||||
|
|
||||||
#[async_trait]
|
#[async_trait]
|
||||||
@@ -46,15 +48,17 @@ impl FactExtractor for SimpleFactExtractor {
|
|||||||
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>> {
|
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>> {
|
||||||
let mut facts = vec![];
|
let mut facts = vec![];
|
||||||
|
|
||||||
|
// Extract [[Entity]] patterns
|
||||||
let entity_pattern = Regex::new(r"\[\[([^\]]+)\]\]")?;
|
let entity_pattern = Regex::new(r"\[\[([^\]]+)\]\]")?;
|
||||||
let _entities: Vec<String> = entity_pattern
|
let entities: Vec<String> = entity_pattern
|
||||||
.captures_iter(text)
|
.captures_iter(text)
|
||||||
.filter_map(|cap| cap.get(1).map(|m| m.as_str().to_string()))
|
.filter_map(|cap| cap.get(1).map(|m| m.as_str().to_string()))
|
||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
let verbs = ["uses", "manages", "runs", "deployed_to", "works_with",
|
// Common relationship verbs
|
||||||
"depends_on", "contains", "extends", "implements", "connects_to"];
|
let verbs = ["uses", "manages", "runs", "deployed_to", "works_with"];
|
||||||
|
|
||||||
|
// Simple heuristic: if two entities appear close together with a verb between them
|
||||||
for verb in &verbs {
|
for verb in &verbs {
|
||||||
let pattern = format!(
|
let pattern = format!(
|
||||||
r"\[\[([^\]]+)\]\].*?{}.*?\[\[([^\]]+)\]\]",
|
r"\[\[([^\]]+)\]\].*?{}.*?\[\[([^\]]+)\]\]",
|
||||||
@@ -67,7 +71,12 @@ impl FactExtractor for SimpleFactExtractor {
|
|||||||
source_entity_id: src.as_str().to_string(),
|
source_entity_id: src.as_str().to_string(),
|
||||||
target_entity_id: tgt.as_str().to_string(),
|
target_entity_id: tgt.as_str().to_string(),
|
||||||
relation_type: verb.to_uppercase(),
|
relation_type: verb.to_uppercase(),
|
||||||
fact: format!("{} {} {}", src.as_str(), verb, tgt.as_str()),
|
fact: format!(
|
||||||
|
"{} {} {}",
|
||||||
|
src.as_str(),
|
||||||
|
verb,
|
||||||
|
tgt.as_str()
|
||||||
|
),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -78,251 +87,18 @@ impl FactExtractor for SimpleFactExtractor {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// LLM-based fact extractor (Zep §2.2.2 alignment)
|
/// LLM-based fact extractor (placeholder for production)
|
||||||
/// Extracts relationships between entity pairs using LLM
|
/// TODO (Phase 2.6): Implement with real LLM API
|
||||||
pub struct LlmFactExtractor {
|
/// TODO (Phase 2.6): Support complex relationships (3-way, temporal, conditional)
|
||||||
model_name: String,
|
pub struct LlmFactExtractor;
|
||||||
jwt_issuer: Option<std::sync::Arc<tokio::sync::Mutex<crate::authentik_jwt::AuthentikJwtIssuer>>>,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl LlmFactExtractor {
|
|
||||||
pub fn new(model_name: &str) -> Self {
|
|
||||||
let jwt_issuer = crate::authentik_jwt::AuthentikJwtIssuer::from_env().ok();
|
|
||||||
Self {
|
|
||||||
model_name: model_name.to_string(),
|
|
||||||
jwt_issuer: jwt_issuer.map(|iss| std::sync::Arc::new(tokio::sync::Mutex::new(iss))),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Clean LLM response: strip thinking tags, markdown fences, extract JSON
|
|
||||||
fn clean_llm_response(text: &str) -> String {
|
|
||||||
let mut result = text.to_string();
|
|
||||||
while let Some(start) = result.find("<think>") {
|
|
||||||
if let Some(end) = result.find("</think>") {
|
|
||||||
result = format!("{}{}", &result[..start], &result[end + 8..]);
|
|
||||||
} else { break; }
|
|
||||||
}
|
|
||||||
result = result.replace("```json", "").replace("```", "");
|
|
||||||
let trimmed = result.trim();
|
|
||||||
if let Some(start) = trimmed.find('{') {
|
|
||||||
if let Some(end) = trimmed.rfind('}') {
|
|
||||||
return trimmed[start..=end].to_string();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if let Some(start) = trimmed.find('[') {
|
|
||||||
if let Some(end) = trimmed.rfind(']') {
|
|
||||||
return format!("{{\"facts\": {}}}", &trimmed[start..=end]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
trimmed.to_string()
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn call_llm(&self, prompt: &str) -> Result<String> {
|
|
||||||
let endpoint = std::env::var("LLM_ENDPOINT")
|
|
||||||
.unwrap_or_else(|_| "http://localhost:11434/v1/chat/completions".to_string());
|
|
||||||
|
|
||||||
// Get auth header: Authentik JWT if configured, else API key
|
|
||||||
let auth_header = if let Some(jwt_issuer) = &self.jwt_issuer {
|
|
||||||
let issuer = jwt_issuer.lock().await;
|
|
||||||
match issuer.get_access_token().await {
|
|
||||||
Ok(token) => format!("Bearer {}", token),
|
|
||||||
Err(e) => {
|
|
||||||
tracing::warn!(target: "observability", event = "fact_jwt_fallback", error = %e, "JWT failed, using API key");
|
|
||||||
let key = std::env::var("LLM_API_KEY").unwrap_or_else(|_| "default-key".to_string());
|
|
||||||
format!("Bearer {}", key)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
let key = std::env::var("LLM_API_KEY")
|
|
||||||
.or_else(|_| std::env::var("MEM_API_KEY"))
|
|
||||||
.unwrap_or_else(|_| "default-key".to_string());
|
|
||||||
format!("Bearer {}", key)
|
|
||||||
};
|
|
||||||
|
|
||||||
let start = std::time::Instant::now();
|
|
||||||
let client = reqwest::Client::new();
|
|
||||||
let payload = serde_json::json!({
|
|
||||||
"model": self.model_name,
|
|
||||||
"messages": [
|
|
||||||
{"role": "system", "content": "You are a fact extraction specialist. Extract relationships between entities from text. Output ONLY valid JSON."},
|
|
||||||
{"role": "user", "content": prompt}
|
|
||||||
],
|
|
||||||
"max_tokens": 12000,
|
|
||||||
"temperature": 0.1
|
|
||||||
});
|
|
||||||
|
|
||||||
let response = client
|
|
||||||
.post(&endpoint)
|
|
||||||
.header("Authorization", &auth_header)
|
|
||||||
.header("Content-Type", "application/json")
|
|
||||||
.json(&payload)
|
|
||||||
.timeout(std::time::Duration::from_secs(120))
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let status = response.status();
|
|
||||||
if !status.is_success() {
|
|
||||||
let body = response.text().await.unwrap_or_default();
|
|
||||||
tracing::warn!(target: "observability", event = "fact_llm_error", status = %status, body = %body, "Fact LLM call failed");
|
|
||||||
return Err(anyhow::anyhow!("LLM API error: {}", status));
|
|
||||||
}
|
|
||||||
|
|
||||||
let elapsed = start.elapsed();
|
|
||||||
let data: serde_json::Value = response.json().await?;
|
|
||||||
|
|
||||||
// Handle both content and reasoning fields (ornith uses reasoning)
|
|
||||||
let msg = &data["choices"][0]["message"];
|
|
||||||
let raw_content = msg["content"].as_str().unwrap_or("").to_string();
|
|
||||||
let raw_reasoning = msg["reasoning"].as_str().unwrap_or("").to_string();
|
|
||||||
let raw = if !raw_content.trim().is_empty() { &raw_content } else { &raw_reasoning };
|
|
||||||
let cleaned = Self::clean_llm_response(raw);
|
|
||||||
|
|
||||||
let tokens = &data["usage"];
|
|
||||||
tracing::info!(
|
|
||||||
target: "observability",
|
|
||||||
event = "llm_fact_call",
|
|
||||||
model = %self.model_name,
|
|
||||||
endpoint = %endpoint,
|
|
||||||
raw_len = raw.len(),
|
|
||||||
cleaned_len = cleaned.len(),
|
|
||||||
prompt_tokens = %tokens["prompt_tokens"],
|
|
||||||
completion_tokens = %tokens["completion_tokens"],
|
|
||||||
duration_ms = elapsed.as_millis() as u64,
|
|
||||||
has_reasoning = !raw_reasoning.is_empty(),
|
|
||||||
"LLM fact extraction call complete"
|
|
||||||
);
|
|
||||||
Ok(cleaned)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[async_trait]
|
#[async_trait]
|
||||||
impl FactExtractor for LlmFactExtractor {
|
impl FactExtractor for LlmFactExtractor {
|
||||||
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>> {
|
async fn extract(&self, _text: &str) -> Result<Vec<ExtractedFact>> {
|
||||||
self.extract_with_context(text, &[]).await
|
// TODO (Phase 2.6): Implement LLM-based extraction
|
||||||
}
|
// Pattern: Send text to api.riotpiao.com with prompt
|
||||||
|
// Parse response for [source, relation, target] tuples
|
||||||
async fn extract_with_context(
|
Ok(vec![])
|
||||||
&self,
|
|
||||||
text: &str,
|
|
||||||
entity_contexts: &[crate::grm_retriever::EntityContext],
|
|
||||||
) -> Result<Vec<ExtractedFact>> {
|
|
||||||
// Build entity list for prompt
|
|
||||||
let entity_names: Vec<&str> = entity_contexts
|
|
||||||
.iter()
|
|
||||||
.map(|e| e.entity_name.as_str())
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
if entity_names.is_empty() {
|
|
||||||
tracing::debug!("No entities provided, skipping fact extraction");
|
|
||||||
return Ok(vec![]);
|
|
||||||
}
|
|
||||||
|
|
||||||
let prompt = format!(
|
|
||||||
r#"Extract relationships (facts) between these entities from the text.
|
|
||||||
|
|
||||||
Entities: {:?}
|
|
||||||
|
|
||||||
Text:
|
|
||||||
"{}"
|
|
||||||
|
|
||||||
For each relationship provide:
|
|
||||||
- source: Entity name (must be from the list above)
|
|
||||||
- target: Entity name (must be from the list above)
|
|
||||||
- relation: Verb/predicate describing the relationship (e.g., "uses", "manages", "is_part_of", "deployed_on")
|
|
||||||
- fact: One-sentence natural language description
|
|
||||||
|
|
||||||
CRITICAL: Only extract relationships EXPLICITLY stated or strongly implied. Source and target must both be from the entity list.
|
|
||||||
|
|
||||||
Respond in JSON:
|
|
||||||
{{"facts": [{{"source": "...", "target": "...", "relation": "...", "fact": "..."}}, ...]}}
|
|
||||||
"#,
|
|
||||||
entity_names, text
|
|
||||||
);
|
|
||||||
|
|
||||||
let llm_ok = std::env::var("LLM_ENDPOINT").is_ok();
|
|
||||||
let response = if llm_ok {
|
|
||||||
match self.call_llm(&prompt).await {
|
|
||||||
Ok(r) => r,
|
|
||||||
Err(e) => {
|
|
||||||
tracing::warn!("Fact extraction LLM failed: {}, returning empty", e);
|
|
||||||
return Ok(vec![]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
tracing::debug!("LLM_ENDPOINT not set, skipping LLM fact extraction");
|
|
||||||
return Ok(vec![]);
|
|
||||||
};
|
|
||||||
|
|
||||||
// Parse response
|
|
||||||
#[derive(Deserialize)]
|
|
||||||
struct FactResponse {
|
|
||||||
facts: Vec<RawFact>,
|
|
||||||
}
|
|
||||||
#[derive(Deserialize)]
|
|
||||||
struct RawFact {
|
|
||||||
source: String,
|
|
||||||
target: String,
|
|
||||||
relation: String,
|
|
||||||
fact: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
// Try parsing, if trailing chars error try trimming to valid JSON
|
|
||||||
let parsed = match serde_json::from_str::<FactResponse>(&response) {
|
|
||||||
Ok(r) => Ok(r),
|
|
||||||
Err(e) if e.to_string().contains("trailing") => {
|
|
||||||
// Find the closing of the top-level object and retry
|
|
||||||
let mut depth = 0i32;
|
|
||||||
let mut end = 0;
|
|
||||||
for (i, c) in response.char_indices() {
|
|
||||||
match c {
|
|
||||||
'{' | '[' => depth += 1,
|
|
||||||
'}' | ']' => { depth -= 1; if depth == 0 { end = i + 1; break; } },
|
|
||||||
_ => {}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if end > 0 {
|
|
||||||
serde_json::from_str::<FactResponse>(&response[..end])
|
|
||||||
} else {
|
|
||||||
Err(e)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(e) => Err(e),
|
|
||||||
};
|
|
||||||
match parsed {
|
|
||||||
Ok(parsed) => {
|
|
||||||
let facts: Vec<ExtractedFact> = parsed.facts
|
|
||||||
.into_iter()
|
|
||||||
.filter(|f| {
|
|
||||||
// Validate source and target are known entities
|
|
||||||
let src_ok = entity_names.iter().any(|e| e.eq_ignore_ascii_case(&f.source));
|
|
||||||
let tgt_ok = entity_names.iter().any(|e| e.eq_ignore_ascii_case(&f.target));
|
|
||||||
if !src_ok || !tgt_ok {
|
|
||||||
tracing::debug!(
|
|
||||||
"Dropping fact with unknown entity: {} -> {}",
|
|
||||||
f.source, f.target
|
|
||||||
);
|
|
||||||
}
|
|
||||||
src_ok && tgt_ok && f.source != f.target
|
|
||||||
})
|
|
||||||
.map(|f| ExtractedFact {
|
|
||||||
source_entity_id: f.source,
|
|
||||||
target_entity_id: f.target,
|
|
||||||
relation_type: f.relation.to_uppercase(),
|
|
||||||
fact: f.fact,
|
|
||||||
})
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
tracing::info!(
|
|
||||||
"LLM fact extraction: {} facts from {} entities",
|
|
||||||
facts.len(), entity_names.len()
|
|
||||||
);
|
|
||||||
Ok(facts)
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
tracing::warn!("Fact extraction JSON parse failed: {}", e);
|
|
||||||
Ok(vec![])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -334,38 +110,9 @@ mod tests {
|
|||||||
async fn test_simple_fact_extraction() {
|
async fn test_simple_fact_extraction() {
|
||||||
let extractor = SimpleFactExtractor;
|
let extractor = SimpleFactExtractor;
|
||||||
let text = "[[Rock]] uses [[Kubernetes]] and [[ArgoCD]]";
|
let text = "[[Rock]] uses [[Kubernetes]] and [[ArgoCD]]";
|
||||||
|
|
||||||
let facts = extractor.extract(text).await.unwrap();
|
let facts = extractor.extract(text).await.unwrap();
|
||||||
assert!(!facts.is_empty());
|
assert!(facts.len() > 0);
|
||||||
assert!(facts.iter().any(|f| f.relation_type == "USES"));
|
assert!(facts.iter().any(|f| f.relation_type == "USES"));
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn test_simple_no_wiki_links() {
|
|
||||||
let extractor = SimpleFactExtractor;
|
|
||||||
let text = "Kubernetes uses etcd for storage";
|
|
||||||
let facts = extractor.extract(text).await.unwrap();
|
|
||||||
assert!(facts.is_empty()); // No [[wiki links]]
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_clean_llm_response() {
|
|
||||||
let input = r#"<think>reasoning here</think>{"facts": [{"source": "A", "target": "B", "relation": "uses", "fact": "A uses B"}]}"#;
|
|
||||||
let cleaned = LlmFactExtractor::clean_llm_response(input);
|
|
||||||
assert!(cleaned.starts_with("{"));
|
|
||||||
assert!(cleaned.contains("facts"));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_strip_thinking_no_tags() {
|
|
||||||
let input = r#"{"facts": []}"#;
|
|
||||||
let cleaned = LlmFactExtractor::clean_llm_response(input);
|
|
||||||
assert_eq!(cleaned, input);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn test_llm_fact_no_entities_returns_empty() {
|
|
||||||
let extractor = LlmFactExtractor::new("test");
|
|
||||||
let facts = extractor.extract_with_context("some text", &[]).await.unwrap();
|
|
||||||
assert!(facts.is_empty());
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,7 +10,10 @@
|
|||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
use async_trait::async_trait;
|
use async_trait::async_trait;
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
use tracing::debug;
|
use std::collections::HashMap;
|
||||||
|
use tracing::{debug, info};
|
||||||
|
use mem_core::entity::Entity;
|
||||||
|
use mem_core::edge::Edge;
|
||||||
|
|
||||||
/// Memorability decision for entity or fact
|
/// Memorability decision for entity or fact
|
||||||
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq)]
|
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq)]
|
||||||
|
|||||||
@@ -59,15 +59,10 @@ impl IngestPipeline {
|
|||||||
/// Execute extraction pipeline for episode
|
/// Execute extraction pipeline for episode
|
||||||
/// CRAP: 14 (Low: orchestration only, delegates to stages)
|
/// CRAP: 14 (Low: orchestration only, delegates to stages)
|
||||||
pub async fn ingest(&self, episode: &Episode) -> Result<ExtractionResult> {
|
pub async fn ingest(&self, episode: &Episode) -> Result<ExtractionResult> {
|
||||||
self.ingest_with_auth(episode, None).await
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Ingest with optional X-Forward-User auth header
|
|
||||||
pub async fn ingest_with_auth(&self, episode: &Episode, x_forward_user: Option<&str>) -> Result<ExtractionResult> {
|
|
||||||
debug!("Starting ingest for episode: {}", episode.id);
|
debug!("Starting ingest for episode: {}", episode.id);
|
||||||
|
|
||||||
// Stage 1: Extract entities (with optional auth header)
|
// Stage 1: Extract entities
|
||||||
let extracted_entities = self.entity_extractor.extract_with_auth(&episode.text, x_forward_user).await?;
|
let extracted_entities = self.entity_extractor.extract(&episode.text).await?;
|
||||||
debug!("Extracted {} entities", extracted_entities.len());
|
debug!("Extracted {} entities", extracted_entities.len());
|
||||||
|
|
||||||
// Convert to domain entities
|
// Convert to domain entities
|
||||||
@@ -149,7 +144,6 @@ impl IngestPipeline {
|
|||||||
|
|
||||||
/// Async queue worker: Process episodes from queue
|
/// Async queue worker: Process episodes from queue
|
||||||
/// CRAP: 12 (Async loop, straightforward)
|
/// CRAP: 12 (Async loop, straightforward)
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct QueueWorker {
|
pub struct QueueWorker {
|
||||||
pipeline: Arc<IngestPipeline>,
|
pipeline: Arc<IngestPipeline>,
|
||||||
batch_size: usize,
|
batch_size: usize,
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ use tracing::{debug, info};
|
|||||||
use crate::grm_retriever::{
|
use crate::grm_retriever::{
|
||||||
EntityContext, FactContext, GraphContextRetriever, MemorabilityDecision, GrmConfig, MockGrmRetriever,
|
EntityContext, FactContext, GraphContextRetriever, MemorabilityDecision, GrmConfig, MockGrmRetriever,
|
||||||
};
|
};
|
||||||
use mem_core::entity::Entity;
|
use mem_core::entity::{Entity, EntityType};
|
||||||
use mem_core::edge::Edge;
|
use mem_core::edge::Edge;
|
||||||
|
|
||||||
/// Entity filtering result
|
/// Entity filtering result
|
||||||
@@ -88,7 +88,7 @@ impl MemorabilityGate {
|
|||||||
let (filtered, reason) = match context.decision {
|
let (filtered, reason) = match context.decision {
|
||||||
MemorabilityDecision::Keep => {
|
MemorabilityDecision::Keep => {
|
||||||
if context.matched_entity_id.is_some() {
|
if context.matched_entity_id.is_some() {
|
||||||
(true, "Existing entity (merge required)".to_string())
|
(true, format!("Existing entity (merge required)"))
|
||||||
} else {
|
} else {
|
||||||
(false, format!("New entity (score: {:.2})", context.memorability_score))
|
(false, format!("New entity (score: {:.2})", context.memorability_score))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -20,7 +20,6 @@ pub struct RefMetadata {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Obsidian REST API client
|
/// Obsidian REST API client
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct ObsidianClient {
|
pub struct ObsidianClient {
|
||||||
base_url: String,
|
base_url: String,
|
||||||
}
|
}
|
||||||
@@ -48,7 +47,6 @@ impl ObsidianClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// ObsidianRefSource: Fetches & chunks reference documents from Obsidian vault
|
/// ObsidianRefSource: Fetches & chunks reference documents from Obsidian vault
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct ObsidianRefSource {
|
pub struct ObsidianRefSource {
|
||||||
client: ObsidianClient,
|
client: ObsidianClient,
|
||||||
project: String,
|
project: String,
|
||||||
@@ -70,13 +68,11 @@ impl ObsidianRefSource {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Check if a file path is allowed (matches configured prefixes)
|
/// Check if a file path is allowed (matches configured prefixes)
|
||||||
#[allow(dead_code)]
|
|
||||||
fn is_allowed_path(&self, path: &str) -> bool {
|
fn is_allowed_path(&self, path: &str) -> bool {
|
||||||
self.allowed_paths.iter().any(|prefix| path.starts_with(prefix))
|
self.allowed_paths.iter().any(|prefix| path.starts_with(prefix))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Chunk reference document via heading-boundary logic
|
/// Chunk reference document via heading-boundary logic
|
||||||
#[allow(dead_code)]
|
|
||||||
fn chunk_document(&self, path: &str, content: &str) -> Vec<Record> {
|
fn chunk_document(&self, path: &str, content: &str) -> Vec<Record> {
|
||||||
// M3.6.1 heading-boundary chunking
|
// M3.6.1 heading-boundary chunking
|
||||||
// - Split by headings
|
// - Split by headings
|
||||||
@@ -207,7 +203,7 @@ mod tests {
|
|||||||
let chunks = source.chunk_document("docs/test.md", content);
|
let chunks = source.chunk_document("docs/test.md", content);
|
||||||
|
|
||||||
// Should split by headings
|
// Should split by headings
|
||||||
assert!(!chunks.is_empty());
|
assert!(chunks.len() > 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -60,7 +60,8 @@ impl MetricsCollector {
|
|||||||
self.by_project
|
self.by_project
|
||||||
.lock()
|
.lock()
|
||||||
.unwrap()
|
.unwrap()
|
||||||
.get(project).cloned()
|
.get(project)
|
||||||
|
.map(|m| m.clone())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get all project metrics.
|
/// Get all project metrics.
|
||||||
|
|||||||
@@ -306,7 +306,7 @@ impl QueryMetricsRepository {
|
|||||||
let mut repo = self.metrics.lock().unwrap();
|
let mut repo = self.metrics.lock().unwrap();
|
||||||
repo.get_mut(query_id)
|
repo.get_mut(query_id)
|
||||||
.ok_or_else(|| format!("Query {} not found", query_id))
|
.ok_or_else(|| format!("Query {} not found", query_id))
|
||||||
.map(f)
|
.map(|metrics| f(metrics))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get progress for a query
|
/// Get progress for a query
|
||||||
|
|||||||
@@ -4,10 +4,9 @@
|
|||||||
///
|
///
|
||||||
/// Used to scope queries to project namespaces and enable graph traversal.
|
/// Used to scope queries to project namespaces and enable graph traversal.
|
||||||
/// For example: poimen/tools/kubectl.md [[debugging.md]] creates an edge
|
/// For example: poimen/tools/kubectl.md [[debugging.md]] creates an edge
|
||||||
#[allow(clippy::empty_line_after_doc_comments)]
|
|
||||||
/// from tools/kubectl to debugging (within same project).
|
/// from tools/kubectl to debugging (within same project).
|
||||||
|
|
||||||
use anyhow::Result;
|
use anyhow::{anyhow, Result};
|
||||||
use regex::Regex;
|
use regex::Regex;
|
||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::path::{Path, PathBuf};
|
use std::path::{Path, PathBuf};
|
||||||
@@ -80,7 +79,6 @@ impl WikiLinkParser {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Graph Index: Stores and queries wiki-link relationships
|
/// Graph Index: Stores and queries wiki-link relationships
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct WikiLinkGraph {
|
pub struct WikiLinkGraph {
|
||||||
/// Forward links: source -> [targets]
|
/// Forward links: source -> [targets]
|
||||||
forward_links: HashMap<String, Vec<String>>,
|
forward_links: HashMap<String, Vec<String>>,
|
||||||
@@ -102,11 +100,11 @@ impl WikiLinkGraph {
|
|||||||
/// Add a wiki-link edge
|
/// Add a wiki-link edge
|
||||||
pub fn add_link(&mut self, source: &str, target: &str) {
|
pub fn add_link(&mut self, source: &str, target: &str) {
|
||||||
self.forward_links.entry(source.to_string())
|
self.forward_links.entry(source.to_string())
|
||||||
.or_default()
|
.or_insert_with(Vec::new)
|
||||||
.push(target.to_string());
|
.push(target.to_string());
|
||||||
|
|
||||||
self.backward_links.entry(target.to_string())
|
self.backward_links.entry(target.to_string())
|
||||||
.or_default()
|
.or_insert_with(Vec::new)
|
||||||
.push(source.to_string());
|
.push(source.to_string());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ pub enum AuthMode {
|
|||||||
|
|
||||||
impl AuthMode {
|
impl AuthMode {
|
||||||
/// Detect from base URL or explicit env var.
|
/// Detect from base URL or explicit env var.
|
||||||
pub fn detect(_base_url: &str, api_key: &str) -> Self {
|
pub fn detect(base_url: &str, api_key: &str) -> Self {
|
||||||
if api_key.is_empty() {
|
if api_key.is_empty() {
|
||||||
return Self::None;
|
return Self::None;
|
||||||
}
|
}
|
||||||
@@ -87,7 +87,6 @@ struct Choice {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize)]
|
||||||
#[allow(dead_code)]
|
|
||||||
struct MessageResponse {
|
struct MessageResponse {
|
||||||
role: String,
|
role: String,
|
||||||
content: String,
|
content: String,
|
||||||
@@ -209,11 +208,12 @@ impl ChatClient {
|
|||||||
Ok(r) => r,
|
Ok(r) => r,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
last_error = Some(anyhow!("Request failed: {}", e));
|
last_error = Some(anyhow!("Request failed: {}", e));
|
||||||
if (e.is_timeout() || e.is_status())
|
if e.is_timeout() || e.is_status() {
|
||||||
&& attempt < self.max_retries - 1 {
|
if attempt < self.max_retries - 1 {
|
||||||
tokio::time::sleep(Duration::from_millis(100 * 2_u64.pow(attempt))).await;
|
tokio::time::sleep(Duration::from_millis(100 * 2_u64.pow(attempt))).await;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
}
|
||||||
return Err(last_error.unwrap());
|
return Err(last_error.unwrap());
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -27,7 +27,6 @@ struct EmbeddingRequest {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize)]
|
||||||
#[allow(dead_code)]
|
|
||||||
#[serde(untagged)]
|
#[serde(untagged)]
|
||||||
enum EmbeddingResponse {
|
enum EmbeddingResponse {
|
||||||
Success {
|
Success {
|
||||||
@@ -43,7 +42,6 @@ enum EmbeddingResponse {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize)]
|
||||||
#[allow(dead_code)]
|
|
||||||
struct EmbeddingData {
|
struct EmbeddingData {
|
||||||
embedding: Vec<f32>,
|
embedding: Vec<f32>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
@@ -122,10 +120,10 @@ impl EmbeddingsClient {
|
|||||||
/// Embed a single text string, returning a 768-dim vector
|
/// Embed a single text string, returning a 768-dim vector
|
||||||
pub async fn embed_one(&self, text: &str) -> Result<Vector> {
|
pub async fn embed_one(&self, text: &str) -> Result<Vector> {
|
||||||
let embeddings = self.embed(&[text.to_string()]).await?;
|
let embeddings = self.embed(&[text.to_string()]).await?;
|
||||||
embeddings
|
Ok(embeddings
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.next()
|
.next()
|
||||||
.ok_or_else(|| anyhow!("empty embedding response"))
|
.ok_or_else(|| anyhow!("empty embedding response"))?)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Embed multiple texts, batched at ≤32 per request, preserving input order
|
/// Embed multiple texts, batched at ≤32 per request, preserving input order
|
||||||
@@ -168,18 +166,8 @@ impl EmbeddingsClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
let resp = builder.json(&req).send().await?;
|
let resp = builder.json(&req).send().await?;
|
||||||
let status = resp.status();
|
let _status = resp.status();
|
||||||
let raw_body = resp.text().await?;
|
let body: EmbeddingResponse = resp.json().await?;
|
||||||
|
|
||||||
if !status.is_success() {
|
|
||||||
tracing::error!("Embedding API returned {}: {}", status, &raw_body[..raw_body.len().min(500)]);
|
|
||||||
return Err(anyhow!("Embedding API returned {}: {}", status, &raw_body[..raw_body.len().min(200)]));
|
|
||||||
}
|
|
||||||
|
|
||||||
let body: EmbeddingResponse = serde_json::from_str(&raw_body).map_err(|e| {
|
|
||||||
tracing::error!("Failed to parse embedding response: {}. Raw body: {}", e, &raw_body[..raw_body.len().min(500)]);
|
|
||||||
anyhow!("Failed to parse embedding response: {}. Raw: {}", e, &raw_body[..raw_body.len().min(200)])
|
|
||||||
})?;
|
|
||||||
|
|
||||||
match body {
|
match body {
|
||||||
EmbeddingResponse::Error { error } => {
|
EmbeddingResponse::Error { error } => {
|
||||||
@@ -214,73 +202,4 @@ mod tests {
|
|||||||
assert_eq!(BATCH_SIZE, 32);
|
assert_eq!(BATCH_SIZE, 32);
|
||||||
assert_eq!(EMBEDDINGS_DIM, 768);
|
assert_eq!(EMBEDDINGS_DIM, 768);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_parse_real_embedding_response() {
|
|
||||||
// Exact format returned by embeddings-predictor service
|
|
||||||
let raw = r#"{"object":"list","data":[{"object":"embedding","embedding":[0.1,0.2,0.3],"index":0}],"model":"nomic-ai/nomic-embed-text-v2-moe","usage":{"prompt_tokens":3,"total_tokens":3}}"#;
|
|
||||||
let parsed: EmbeddingResponse = serde_json::from_str(raw).expect("should parse");
|
|
||||||
match parsed {
|
|
||||||
EmbeddingResponse::Success { data, .. } => {
|
|
||||||
assert_eq!(data.len(), 1);
|
|
||||||
assert_eq!(data[0].embedding.len(), 3);
|
|
||||||
assert_eq!(data[0].index, 0);
|
|
||||||
}
|
|
||||||
EmbeddingResponse::Error { error } => panic!("parsed as error: {:?}", error),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_parse_embedding_error_response() {
|
|
||||||
let raw = r#"{"error":"model not found"}"#;
|
|
||||||
let parsed: EmbeddingResponse = serde_json::from_str(raw).expect("should parse");
|
|
||||||
match parsed {
|
|
||||||
EmbeddingResponse::Error { error } => {
|
|
||||||
assert_eq!(error.as_str().unwrap(), "model not found");
|
|
||||||
}
|
|
||||||
EmbeddingResponse::Success { .. } => panic!("should be error"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_parse_768_dim_response() {
|
|
||||||
// 768 floats
|
|
||||||
let embedding: Vec<f32> = (0..768).map(|i| i as f32 * 0.001).collect();
|
|
||||||
let raw = format!(
|
|
||||||
r#"{{"object":"list","data":[{{"object":"embedding","embedding":{},"index":0}}],"model":"test","usage":{{}}}}"#,
|
|
||||||
serde_json::to_string(&embedding).unwrap()
|
|
||||||
);
|
|
||||||
let parsed: EmbeddingResponse = serde_json::from_str(&raw).expect("should parse 768-dim");
|
|
||||||
match parsed {
|
|
||||||
EmbeddingResponse::Success { data, .. } => {
|
|
||||||
assert_eq!(data[0].embedding.len(), 768);
|
|
||||||
}
|
|
||||||
_ => panic!("should be success"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_parse_html_fails_gracefully() {
|
|
||||||
// Simulates gateway returning HTML error page
|
|
||||||
let raw = "<html><body>502 Bad Gateway</body></html>";
|
|
||||||
let result: Result<EmbeddingResponse, _> = serde_json::from_str(raw);
|
|
||||||
assert!(result.is_err(), "HTML should fail to parse as JSON");
|
|
||||||
let err_msg = result.unwrap_err().to_string();
|
|
||||||
assert!(err_msg.contains("expected"), "Error should mention parsing: {}", err_msg);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_parse_multi_input_response() {
|
|
||||||
// Array input returns multiple embeddings
|
|
||||||
let raw = r#"{"object":"list","data":[{"object":"embedding","embedding":[0.1,0.2,0.3],"index":0},{"object":"embedding","embedding":[0.4,0.5,0.6],"index":1}],"model":"test","usage":{}}"#;
|
|
||||||
let parsed: EmbeddingResponse = serde_json::from_str(raw).expect("should parse");
|
|
||||||
match parsed {
|
|
||||||
EmbeddingResponse::Success { data, .. } => {
|
|
||||||
assert_eq!(data.len(), 2);
|
|
||||||
assert_eq!(data[0].index, 0);
|
|
||||||
assert_eq!(data[1].index, 1);
|
|
||||||
}
|
|
||||||
_ => panic!("should be success"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,67 +0,0 @@
|
|||||||
-- Migration 009: Temporal edge schema (Zep paper §2.2.2)
|
|
||||||
-- Replaces old memory_edge (child_sha/parent_sha node graph)
|
|
||||||
-- with temporal edge schema supporting relation types, facts, and validity periods.
|
|
||||||
-- Idempotent: safe to run multiple times.
|
|
||||||
|
|
||||||
-- Rename old table if it still exists (skip if already migrated)
|
|
||||||
DO $$
|
|
||||||
BEGIN
|
|
||||||
IF EXISTS (SELECT 1 FROM information_schema.tables WHERE table_name = 'memory_edge'
|
|
||||||
AND EXISTS (SELECT 1 FROM information_schema.columns
|
|
||||||
WHERE table_name = 'memory_edge' AND column_name = 'child_sha'))
|
|
||||||
THEN
|
|
||||||
ALTER TABLE memory_edge RENAME TO memory_edge_legacy;
|
|
||||||
END IF;
|
|
||||||
END $$;
|
|
||||||
|
|
||||||
-- Create temporal edge table
|
|
||||||
CREATE TABLE IF NOT EXISTS memory_edge (
|
|
||||||
id TEXT PRIMARY KEY,
|
|
||||||
project_id TEXT NOT NULL DEFAULT 'default',
|
|
||||||
source_id TEXT NOT NULL,
|
|
||||||
target_id TEXT NOT NULL,
|
|
||||||
relation_type TEXT NOT NULL DEFAULT '',
|
|
||||||
fact TEXT NOT NULL DEFAULT '',
|
|
||||||
weight REAL NOT NULL DEFAULT 1.0,
|
|
||||||
strength REAL DEFAULT 1.0,
|
|
||||||
confidence REAL DEFAULT 0.8,
|
|
||||||
t_valid TIMESTAMPTZ,
|
|
||||||
t_invalid TIMESTAMPTZ,
|
|
||||||
t_created TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
||||||
t_expired TIMESTAMPTZ,
|
|
||||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
||||||
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
||||||
episode_id TEXT,
|
|
||||||
deleted_at TIMESTAMPTZ
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Ensure app user owns the table
|
|
||||||
DO $$ BEGIN
|
|
||||||
IF EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'app') THEN
|
|
||||||
ALTER TABLE memory_edge OWNER TO app;
|
|
||||||
END IF;
|
|
||||||
END $$;
|
|
||||||
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_memory_edge_source ON memory_edge(source_id);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_memory_edge_target ON memory_edge(target_id);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_memory_edge_project ON memory_edge(project_id);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_memory_edge_relation ON memory_edge(relation_type);
|
|
||||||
|
|
||||||
-- Ensure memory_entity has all columns code expects
|
|
||||||
ALTER TABLE memory_entity ADD COLUMN IF NOT EXISTS deleted_at TIMESTAMPTZ;
|
|
||||||
ALTER TABLE memory_entity ADD COLUMN IF NOT EXISTS source_count INTEGER DEFAULT 1;
|
|
||||||
|
|
||||||
-- Unique constraint for entity upsert dedup
|
|
||||||
DO $$
|
|
||||||
BEGIN
|
|
||||||
-- Dedup existing rows before creating unique index
|
|
||||||
DELETE FROM memory_entity a USING memory_entity b
|
|
||||||
WHERE a.project_id = b.project_id AND a.name = b.name
|
|
||||||
AND a.t_created < b.t_created;
|
|
||||||
EXCEPTION WHEN OTHERS THEN NULL;
|
|
||||||
END $$;
|
|
||||||
CREATE UNIQUE INDEX IF NOT EXISTS idx_memory_entity_project_name ON memory_entity(project_id, name);
|
|
||||||
|
|
||||||
-- ROLLBACK instructions:
|
|
||||||
-- DROP TABLE IF EXISTS memory_edge;
|
|
||||||
-- ALTER TABLE IF EXISTS memory_edge_legacy RENAME TO memory_edge;
|
|
||||||
@@ -1,358 +0,0 @@
|
|||||||
use anyhow::Result;
|
|
||||||
use sqlx::{PgPool, FromRow};
|
|
||||||
use uuid::Uuid;
|
|
||||||
use serde::{Deserialize, Serialize};
|
|
||||||
use chrono::{DateTime, Utc};
|
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
|
|
||||||
pub struct AgentPrompt {
|
|
||||||
pub id: Uuid,
|
|
||||||
pub project_id: String,
|
|
||||||
pub name: String,
|
|
||||||
pub template: String,
|
|
||||||
pub target_model: Option<String>,
|
|
||||||
pub task_category: String,
|
|
||||||
pub usage_count: i64,
|
|
||||||
pub avg_quality: f32,
|
|
||||||
pub last_used: Option<DateTime<Utc>>,
|
|
||||||
pub active: bool,
|
|
||||||
pub version: i32,
|
|
||||||
pub tags: Vec<String>,
|
|
||||||
pub created_at: DateTime<Utc>,
|
|
||||||
pub updated_at: DateTime<Utc>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
|
|
||||||
pub struct AgentSkill {
|
|
||||||
pub id: Uuid,
|
|
||||||
pub project_id: String,
|
|
||||||
pub agent_id: String,
|
|
||||||
pub name: String,
|
|
||||||
pub description: String,
|
|
||||||
pub trigger_patterns: Vec<String>,
|
|
||||||
pub success_rate: f32,
|
|
||||||
pub invocation_count: i64,
|
|
||||||
pub avg_latency_ms: i64,
|
|
||||||
pub linked_prompts: Vec<Uuid>,
|
|
||||||
pub enabled: bool,
|
|
||||||
pub created_at: DateTime<Utc>,
|
|
||||||
pub updated_at: DateTime<Utc>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
|
|
||||||
pub struct AgentDecision {
|
|
||||||
pub id: Uuid,
|
|
||||||
pub project_id: String,
|
|
||||||
pub agent_id: String,
|
|
||||||
pub action: String,
|
|
||||||
pub reasoning: String,
|
|
||||||
pub alternatives: Vec<String>,
|
|
||||||
pub confidence: f32,
|
|
||||||
pub context_entities: Vec<Uuid>,
|
|
||||||
pub tool: Option<String>,
|
|
||||||
pub task: Option<String>,
|
|
||||||
pub outcome_success: Option<bool>,
|
|
||||||
pub outcome_quality: Option<f32>,
|
|
||||||
pub outcome_feedback: Option<String>,
|
|
||||||
pub outcome_recorded_at: Option<DateTime<Utc>>,
|
|
||||||
pub created_at: DateTime<Utc>,
|
|
||||||
pub updated_at: DateTime<Utc>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
|
|
||||||
pub struct RolePromptMapping {
|
|
||||||
pub id: Uuid,
|
|
||||||
pub project_id: String,
|
|
||||||
pub role_name: String,
|
|
||||||
pub prompt_id: Uuid,
|
|
||||||
pub priority: i32,
|
|
||||||
pub active: bool,
|
|
||||||
pub created_at: DateTime<Utc>,
|
|
||||||
pub updated_at: DateTime<Utc>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
|
|
||||||
pub struct AgentMetrics {
|
|
||||||
pub id: Uuid,
|
|
||||||
pub project_id: String,
|
|
||||||
pub agent_id: String,
|
|
||||||
pub requests_total: i64,
|
|
||||||
pub requests_success: i64,
|
|
||||||
pub requests_failed: i64,
|
|
||||||
pub average_latency_ms: f32,
|
|
||||||
pub p95_latency_ms: f32,
|
|
||||||
pub p99_latency_ms: f32,
|
|
||||||
pub error_rate: f32,
|
|
||||||
pub recorded_at: DateTime<Utc>,
|
|
||||||
}
|
|
||||||
|
|
||||||
pub struct AgentRepository {
|
|
||||||
pool: PgPool,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl AgentRepository {
|
|
||||||
pub fn new(pool: PgPool) -> Self {
|
|
||||||
AgentRepository { pool }
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_prompt(&self, prompt: AgentPrompt) -> Result<AgentPrompt> {
|
|
||||||
let result = sqlx::query_as::<_, AgentPrompt>(
|
|
||||||
r#"
|
|
||||||
INSERT INTO agent_prompt
|
|
||||||
(project_id, name, template, target_model, task_category, active, version, tags)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
|
||||||
RETURNING *
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(&prompt.project_id)
|
|
||||||
.bind(&prompt.name)
|
|
||||||
.bind(&prompt.template)
|
|
||||||
.bind(&prompt.target_model)
|
|
||||||
.bind(&prompt.task_category)
|
|
||||||
.bind(prompt.active)
|
|
||||||
.bind(prompt.version)
|
|
||||||
.bind(&prompt.tags)
|
|
||||||
.fetch_one(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_prompt(&self, id: Uuid) -> Result<Option<AgentPrompt>> {
|
|
||||||
let result = sqlx::query_as::<_, AgentPrompt>(
|
|
||||||
"SELECT * FROM agent_prompt WHERE id = $1"
|
|
||||||
)
|
|
||||||
.bind(id)
|
|
||||||
.fetch_optional(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn list_prompts(&self, project_id: &str) -> Result<Vec<AgentPrompt>> {
|
|
||||||
let results = sqlx::query_as::<_, AgentPrompt>(
|
|
||||||
"SELECT * FROM agent_prompt WHERE project_id = $1 AND active = true ORDER BY created_at DESC"
|
|
||||||
)
|
|
||||||
.bind(project_id)
|
|
||||||
.fetch_all(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(results)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn update_prompt_usage(&self, id: Uuid, quality_score: f32) -> Result<()> {
|
|
||||||
sqlx::query(
|
|
||||||
r#"
|
|
||||||
UPDATE agent_prompt
|
|
||||||
SET usage_count = usage_count + 1,
|
|
||||||
avg_quality = (avg_quality * (usage_count) + $2) / (usage_count + 1),
|
|
||||||
last_used = NOW(),
|
|
||||||
updated_at = NOW()
|
|
||||||
WHERE id = $1
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(id)
|
|
||||||
.bind(quality_score)
|
|
||||||
.execute(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_skill(&self, skill: AgentSkill) -> Result<AgentSkill> {
|
|
||||||
let result = sqlx::query_as::<_, AgentSkill>(
|
|
||||||
r#"
|
|
||||||
INSERT INTO agent_skill
|
|
||||||
(project_id, agent_id, name, description, enabled)
|
|
||||||
VALUES ($1, $2, $3, $4, $5)
|
|
||||||
RETURNING *
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(&skill.project_id)
|
|
||||||
.bind(&skill.agent_id)
|
|
||||||
.bind(&skill.name)
|
|
||||||
.bind(&skill.description)
|
|
||||||
.bind(skill.enabled)
|
|
||||||
.fetch_one(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_skill(&self, id: Uuid) -> Result<Option<AgentSkill>> {
|
|
||||||
let result = sqlx::query_as::<_, AgentSkill>(
|
|
||||||
"SELECT * FROM agent_skill WHERE id = $1"
|
|
||||||
)
|
|
||||||
.bind(id)
|
|
||||||
.fetch_optional(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn list_skills(&self, project_id: &str, agent_id: &str) -> Result<Vec<AgentSkill>> {
|
|
||||||
let results = sqlx::query_as::<_, AgentSkill>(
|
|
||||||
"SELECT * FROM agent_skill WHERE project_id = $1 AND agent_id = $2 AND enabled = true ORDER BY created_at DESC"
|
|
||||||
)
|
|
||||||
.bind(project_id)
|
|
||||||
.bind(agent_id)
|
|
||||||
.fetch_all(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(results)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_decision(&self, decision: AgentDecision) -> Result<AgentDecision> {
|
|
||||||
let result = sqlx::query_as::<_, AgentDecision>(
|
|
||||||
r#"
|
|
||||||
INSERT INTO agent_decision
|
|
||||||
(project_id, agent_id, action, reasoning, confidence, tool, task)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, $6, $7)
|
|
||||||
RETURNING *
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(&decision.project_id)
|
|
||||||
.bind(&decision.agent_id)
|
|
||||||
.bind(&decision.action)
|
|
||||||
.bind(&decision.reasoning)
|
|
||||||
.bind(decision.confidence)
|
|
||||||
.bind(&decision.tool)
|
|
||||||
.bind(&decision.task)
|
|
||||||
.fetch_one(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn record_decision_outcome(
|
|
||||||
&self,
|
|
||||||
id: Uuid,
|
|
||||||
success: bool,
|
|
||||||
quality: f32,
|
|
||||||
feedback: Option<&str>,
|
|
||||||
) -> Result<()> {
|
|
||||||
sqlx::query(
|
|
||||||
r#"
|
|
||||||
UPDATE agent_decision
|
|
||||||
SET outcome_success = $2,
|
|
||||||
outcome_quality = $3,
|
|
||||||
outcome_feedback = $4,
|
|
||||||
outcome_recorded_at = NOW(),
|
|
||||||
updated_at = NOW()
|
|
||||||
WHERE id = $1
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(id)
|
|
||||||
.bind(success)
|
|
||||||
.bind(quality)
|
|
||||||
.bind(feedback)
|
|
||||||
.execute(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_role_mapping(&self, mapping: RolePromptMapping) -> Result<RolePromptMapping> {
|
|
||||||
let result = sqlx::query_as::<_, RolePromptMapping>(
|
|
||||||
r#"
|
|
||||||
INSERT INTO role_prompt_mapping
|
|
||||||
(project_id, role_name, prompt_id, priority, active)
|
|
||||||
VALUES ($1, $2, $3, $4, $5)
|
|
||||||
RETURNING *
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(&mapping.project_id)
|
|
||||||
.bind(&mapping.role_name)
|
|
||||||
.bind(mapping.prompt_id)
|
|
||||||
.bind(mapping.priority)
|
|
||||||
.bind(mapping.active)
|
|
||||||
.fetch_one(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_prompts_for_role(&self, project_id: &str, role_name: &str) -> Result<Vec<AgentPrompt>> {
|
|
||||||
let results = sqlx::query_as::<_, AgentPrompt>(
|
|
||||||
r#"
|
|
||||||
SELECT ap.* FROM agent_prompt ap
|
|
||||||
INNER JOIN role_prompt_mapping rpm ON ap.id = rpm.prompt_id
|
|
||||||
WHERE rpm.project_id = $1 AND rpm.role_name = $2 AND rpm.active = true
|
|
||||||
ORDER BY rpm.priority DESC, ap.created_at DESC
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(project_id)
|
|
||||||
.bind(role_name)
|
|
||||||
.fetch_all(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(results)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn save_metrics(&self, metrics: AgentMetrics) -> Result<()> {
|
|
||||||
sqlx::query(
|
|
||||||
r#"
|
|
||||||
INSERT INTO agent_metrics
|
|
||||||
(project_id, agent_id, requests_total, requests_success, requests_failed,
|
|
||||||
average_latency_ms, p95_latency_ms, p99_latency_ms, error_rate)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
|
||||||
ON CONFLICT (project_id, agent_id, DATE(recorded_at)) DO UPDATE SET
|
|
||||||
requests_total = EXCLUDED.requests_total,
|
|
||||||
requests_success = EXCLUDED.requests_success,
|
|
||||||
requests_failed = EXCLUDED.requests_failed,
|
|
||||||
average_latency_ms = EXCLUDED.average_latency_ms,
|
|
||||||
p95_latency_ms = EXCLUDED.p95_latency_ms,
|
|
||||||
p99_latency_ms = EXCLUDED.p99_latency_ms,
|
|
||||||
error_rate = EXCLUDED.error_rate
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(&metrics.project_id)
|
|
||||||
.bind(&metrics.agent_id)
|
|
||||||
.bind(metrics.requests_total)
|
|
||||||
.bind(metrics.requests_success)
|
|
||||||
.bind(metrics.requests_failed)
|
|
||||||
.bind(metrics.average_latency_ms)
|
|
||||||
.bind(metrics.p95_latency_ms)
|
|
||||||
.bind(metrics.p99_latency_ms)
|
|
||||||
.bind(metrics.error_rate)
|
|
||||||
.execute(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn log_prompt_usage(
|
|
||||||
&self,
|
|
||||||
project_id: &str,
|
|
||||||
prompt_id: Uuid,
|
|
||||||
agent_id: Option<&str>,
|
|
||||||
model: Option<&str>,
|
|
||||||
input_tokens: Option<i32>,
|
|
||||||
output_tokens: Option<i32>,
|
|
||||||
quality_score: Option<f32>,
|
|
||||||
duration_ms: i64,
|
|
||||||
error_message: Option<&str>,
|
|
||||||
) -> Result<()> {
|
|
||||||
sqlx::query(
|
|
||||||
r#"
|
|
||||||
INSERT INTO prompt_usage_log
|
|
||||||
(project_id, prompt_id, agent_id, model_used, input_tokens, output_tokens,
|
|
||||||
quality_score, duration_ms, error_message)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
|
||||||
"#,
|
|
||||||
)
|
|
||||||
.bind(project_id)
|
|
||||||
.bind(prompt_id)
|
|
||||||
.bind(agent_id)
|
|
||||||
.bind(model)
|
|
||||||
.bind(input_tokens)
|
|
||||||
.bind(output_tokens)
|
|
||||||
.bind(quality_score)
|
|
||||||
.bind(duration_ms)
|
|
||||||
.bind(error_message)
|
|
||||||
.execute(&self.pool)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,6 +1,7 @@
|
|||||||
use chrono::{DateTime, Utc};
|
use chrono::{DateTime, Utc};
|
||||||
use sqlx::PgPool;
|
use sqlx::PgPool;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
use serde_json::json;
|
||||||
|
|
||||||
/// Minimal audit logger - records version snapshots on mutation
|
/// Minimal audit logger - records version snapshots on mutation
|
||||||
#[derive(Clone)]
|
#[derive(Clone)]
|
||||||
|
|||||||
@@ -8,7 +8,6 @@ pub mod edge_repo;
|
|||||||
pub mod community_repo;
|
pub mod community_repo;
|
||||||
pub mod versioning;
|
pub mod versioning;
|
||||||
pub mod audit_logger;
|
pub mod audit_logger;
|
||||||
pub mod agent_repo;
|
|
||||||
// pub mod db_repo; // TODO: Fix Entity schema integration
|
// pub mod db_repo; // TODO: Fix Entity schema integration
|
||||||
|
|
||||||
pub use event_log::{EventRecord, LogWriter};
|
pub use event_log::{EventRecord, LogWriter};
|
||||||
|
|||||||
@@ -1,157 +0,0 @@
|
|||||||
# Poimen Memory - Environment Configuration Guide
|
|
||||||
|
|
||||||
All downstream service URIs are read from environment variables, sourced from ConfigMap.
|
|
||||||
|
|
||||||
## How It Works
|
|
||||||
|
|
||||||
1. **ConfigMap provides URIs**: `k8s/app/config.yaml` (production, SOPS-encrypted)
|
|
||||||
2. **Deployment injects via envFrom**: `envFrom: configMapRef: poimen-memory-config`
|
|
||||||
3. **Application reads from ENV**: Code parses `LLM_ENDPOINT`, `OPENSEARCH_HOST`, `AUTHENTIK_ISSUER`, etc.
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
# deployment.yaml
|
|
||||||
envFrom:
|
|
||||||
- configMapRef:
|
|
||||||
name: poimen-memory-config # All vars injected as ENV
|
|
||||||
```
|
|
||||||
|
|
||||||
## Environment Variables
|
|
||||||
|
|
||||||
### LLM Service (Entity & Fact Extraction)
|
|
||||||
- `LLM_ENDPOINT` — full URL to chat/completions endpoint
|
|
||||||
- `LLM_API_BASE` — base API URL (used for client initialization)
|
|
||||||
- `LLM_MODEL` — model identifier (ornith:35b, qwen:7b, etc.)
|
|
||||||
- `LLM_TIMEOUT_SECS` — timeout for LLM requests
|
|
||||||
- `ENABLE_LLM_EXTRACTION` — enable/disable LLM extraction (true/false)
|
|
||||||
|
|
||||||
### OpenSearch (Vector Store, BM25)
|
|
||||||
- `OPENSEARCH_HOST` — hostname:port
|
|
||||||
- `OPENSEARCH_SCHEME` — http or https
|
|
||||||
- `OPENSEARCH_VERIFY_CERTS` — SSL certificate verification (true/false)
|
|
||||||
|
|
||||||
### Authentik (OIDC)
|
|
||||||
- `AUTHENTIK_ISSUER` — OIDC issuer URL
|
|
||||||
- `AUTHENTIK_VERIFY_SSL` — SSL certificate verification (true/false)
|
|
||||||
- `MEM_AUTH_MODE` — auth mode: jwt | apikey | none
|
|
||||||
|
|
||||||
### Temporal (Workflow Orchestration - Future)
|
|
||||||
- `TEMPORAL_ENDPOINT` — temporal frontend hostname:port
|
|
||||||
- `TEMPORAL_NAMESPACE` — temporal namespace
|
|
||||||
|
|
||||||
### API Gateway (Route Optimization - Future)
|
|
||||||
- `GATEWAY_URL` — gateway base URL
|
|
||||||
|
|
||||||
### Memory Service Config
|
|
||||||
- `MEM_AUTH_MODE` — jwt | apikey | none
|
|
||||||
- `MEM_RATE_LIMIT_INGEST` — ingest requests per second
|
|
||||||
- `MEM_RATE_LIMIT_QUERY` — query requests per second
|
|
||||||
- `MEM_EMBEDDING_BATCH_SIZE` — batch size for embeddings
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Deployment Scenarios
|
|
||||||
|
|
||||||
### Production (SOPS-Encrypted ConfigMap)
|
|
||||||
|
|
||||||
**File**: `k8s/app/config.yaml`
|
|
||||||
|
|
||||||
Services use cluster-internal DNS:
|
|
||||||
```yaml
|
|
||||||
LLM_ENDPOINT: http://reasoning-predictor.llm-serving.svc.cluster.local:8000/v1/chat/completions
|
|
||||||
OPENSEARCH_HOST: opensearch.poimen.svc.cluster.local:9200
|
|
||||||
AUTHENTIK_ISSUER: https://authentik.auth.svc.cluster.local:9443/application/o/poimen/
|
|
||||||
TEMPORAL_ENDPOINT: temporal-frontend.temporal.svc.cluster.local:7233
|
|
||||||
GATEWAY_URL: http://api-gw.poimen.svc.cluster.local:8080
|
|
||||||
MEM_AUTH_MODE: jwt
|
|
||||||
```
|
|
||||||
|
|
||||||
**Deploy**:
|
|
||||||
```bash
|
|
||||||
# SOPS auto-decrypts based on .sops.yaml age key
|
|
||||||
kubectl apply -f k8s/app/config.yaml -k k8s/app/
|
|
||||||
```
|
|
||||||
|
|
||||||
### Local/Development (Plaintext ConfigMap)
|
|
||||||
|
|
||||||
**File**: `k8s/app/config.local.yaml`
|
|
||||||
|
|
||||||
Services via external URLs (ingress):
|
|
||||||
```yaml
|
|
||||||
LLM_ENDPOINT: https://api.riotpiao.com/v1/chat/completions
|
|
||||||
OPENSEARCH_HOST: opensearch.riotpiao.com:443
|
|
||||||
AUTHENTIK_ISSUER: https://authentik.riotpiao.com/application/o/poimen/
|
|
||||||
TEMPORAL_ENDPOINT: temporal.riotpiao.com:443
|
|
||||||
GATEWAY_URL: https://api.riotpiao.com
|
|
||||||
MEM_AUTH_MODE: none
|
|
||||||
```
|
|
||||||
|
|
||||||
**Deploy** (override production config):
|
|
||||||
```bash
|
|
||||||
# Delete prod config, apply local
|
|
||||||
kubectl delete configmap poimen-memory-config -n poimen
|
|
||||||
kubectl apply -f k8s/app/config.local.yaml
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Encrypting with SOPS
|
|
||||||
|
|
||||||
Production `config.yaml` is encrypted with SOPS (Age-based).
|
|
||||||
|
|
||||||
**Encrypt**:
|
|
||||||
```bash
|
|
||||||
sops -e k8s/app/config.yaml > k8s/app/config.yaml.enc
|
|
||||||
mv k8s/app/config.yaml.enc k8s/app/config.yaml
|
|
||||||
```
|
|
||||||
|
|
||||||
**Decrypt for editing** (SOPS auto-handles with $EDITOR):
|
|
||||||
```bash
|
|
||||||
sops k8s/app/config.yaml
|
|
||||||
```
|
|
||||||
|
|
||||||
**View decrypted** (without editing):
|
|
||||||
```bash
|
|
||||||
sops -d k8s/app/config.yaml
|
|
||||||
```
|
|
||||||
|
|
||||||
**.sops.yaml** defines encryption key:
|
|
||||||
```yaml
|
|
||||||
creation_rules:
|
|
||||||
- path_regex: k8s/app/config.yaml
|
|
||||||
key_groups:
|
|
||||||
- age:
|
|
||||||
- <age-public-key>
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Application Code Pattern
|
|
||||||
|
|
||||||
Example: Application should read URIs from ENV at startup.
|
|
||||||
|
|
||||||
```rust
|
|
||||||
// Pseudocode
|
|
||||||
let llm_endpoint = env::var("LLM_ENDPOINT")
|
|
||||||
.unwrap_or("http://localhost:11434/v1/chat/completions".to_string());
|
|
||||||
let opensearch_host = env::var("OPENSEARCH_HOST")
|
|
||||||
.unwrap_or("localhost:9200".to_string());
|
|
||||||
let auth_mode = env::var("MEM_AUTH_MODE")
|
|
||||||
.unwrap_or("none".to_string());
|
|
||||||
|
|
||||||
// Initialize clients with these URIs
|
|
||||||
let llm_client = LlmClient::new(llm_endpoint)?;
|
|
||||||
let search_client = OpenSearchClient::new(opensearch_host)?;
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Summary
|
|
||||||
|
|
||||||
| Aspect | Production | Local |
|
|
||||||
|--------|-----------|-------|
|
|
||||||
| **Config File** | `config.yaml` | `config.local.yaml` |
|
|
||||||
| **Encryption** | SOPS (Age) | Plaintext |
|
|
||||||
| **Service URIs** | Cluster-internal DNS | External HTTPS |
|
|
||||||
| **Auth Mode** | JWT (Authentik) | None (disabled) |
|
|
||||||
| **Rate Limits** | 100/1000 | 1000/10000 |
|
|
||||||
| **Deploy** | `kubectl apply -k k8s/app/` | `kubectl apply -f config.local.yaml` |
|
|
||||||
@@ -1,47 +0,0 @@
|
|||||||
# Local/Development configuration (plaintext, external URLs via ingress)
|
|
||||||
# Use this instead of config.yaml for local testing
|
|
||||||
# kubectl apply -f config.local.yaml
|
|
||||||
apiVersion: v1
|
|
||||||
kind: ConfigMap
|
|
||||||
metadata:
|
|
||||||
name: poimen-memory-config
|
|
||||||
namespace: poimen
|
|
||||||
labels:
|
|
||||||
app.kubernetes.io/name: poimen-memory
|
|
||||||
app.kubernetes.io/component: config
|
|
||||||
data:
|
|
||||||
# Auth mode: jwt | apikey | none (disabled for local testing)
|
|
||||||
MEM_AUTH_MODE: "none"
|
|
||||||
|
|
||||||
# Rate limiting (higher for testing)
|
|
||||||
MEM_RATE_LIMIT_INGEST: "1000"
|
|
||||||
MEM_RATE_LIMIT_QUERY: "10000"
|
|
||||||
MEM_IDEMPOTENCY_TTL_SECS: "86400"
|
|
||||||
|
|
||||||
# Embeddings
|
|
||||||
MEM_EMBEDDING_BATCH_SIZE: "32"
|
|
||||||
|
|
||||||
# Downstream services - external URLs via ingress
|
|
||||||
|
|
||||||
# LLM Service (via api.riotpiao.com ingress)
|
|
||||||
LLM_ENDPOINT: "https://api.riotpiao.com/v1/chat/completions"
|
|
||||||
LLM_API_BASE: "https://api.riotpiao.com/v1"
|
|
||||||
LLM_MODEL: "qwen:7b"
|
|
||||||
LLM_TIMEOUT_SECS: "60"
|
|
||||||
ENABLE_LLM_EXTRACTION: "true"
|
|
||||||
|
|
||||||
# OpenSearch (via ingress)
|
|
||||||
OPENSEARCH_HOST: "opensearch.riotpiao.com:443"
|
|
||||||
OPENSEARCH_SCHEME: "https"
|
|
||||||
OPENSEARCH_VERIFY_CERTS: "true"
|
|
||||||
|
|
||||||
# Authentik (via ingress - optional for local)
|
|
||||||
AUTHENTIK_ISSUER: "https://authentik.riotpiao.com/application/o/poimen/"
|
|
||||||
AUTHENTIK_VERIFY_SSL: "true"
|
|
||||||
|
|
||||||
# Temporal (via ingress)
|
|
||||||
TEMPORAL_ENDPOINT: "temporal.riotpiao.com:443"
|
|
||||||
TEMPORAL_NAMESPACE: "poimen"
|
|
||||||
|
|
||||||
# API Gateway (via ingress)
|
|
||||||
GATEWAY_URL: "https://api.riotpiao.com"
|
|
||||||
+11
-32
@@ -1,7 +1,5 @@
|
|||||||
# Production environment configuration for poimen-memory
|
# Non-sensitive environment variables for poimen-memory
|
||||||
# All services use cluster-internal DNS names
|
# Change these without redeploying secrets.
|
||||||
# This file is encrypted with SOPS in production
|
|
||||||
# For local dev, use plaintext version with external URLs
|
|
||||||
apiVersion: v1
|
apiVersion: v1
|
||||||
kind: ConfigMap
|
kind: ConfigMap
|
||||||
metadata:
|
metadata:
|
||||||
@@ -11,39 +9,20 @@ metadata:
|
|||||||
app.kubernetes.io/name: poimen-memory
|
app.kubernetes.io/name: poimen-memory
|
||||||
app.kubernetes.io/component: config
|
app.kubernetes.io/component: config
|
||||||
data:
|
data:
|
||||||
# Auth mode: jwt | apikey | none
|
# Auth mode: jwt | apikey
|
||||||
MEM_AUTH_MODE: "jwt"
|
MEM_AUTH_MODE: "none"
|
||||||
|
|
||||||
# Rate limiting
|
# Rate limiting
|
||||||
MEM_RATE_LIMIT_INGEST: "100"
|
MEM_RATE_LIMIT_INGEST: "100"
|
||||||
MEM_RATE_LIMIT_QUERY: "1000"
|
MEM_RATE_LIMIT_QUERY: "1000"
|
||||||
MEM_IDEMPOTENCY_TTL_SECS: "86400"
|
MEM_IDEMPOTENCY_TTL_SECS: "86400"
|
||||||
|
|
||||||
# Embeddings
|
# Embeddings
|
||||||
MEM_EMBEDDING_BATCH_SIZE: "32"
|
MEM_EMBEDDING_BATCH_SIZE: "32"
|
||||||
|
# OpenSearch
|
||||||
# Downstream services - read by application from ENV
|
OPENSEARCH_HOST: "opensearch.poimen.svc.cluster.local:9200"
|
||||||
# Internal cluster DNS (prod) / external URLs (local)
|
# Obsidian
|
||||||
|
OBSIDIAN_URL: "http://obsidian-server.poimen.svc.cluster.local:8080"
|
||||||
# LLM Service (entity extraction, fact extraction)
|
# LLM Configuration (for entity extraction)
|
||||||
LLM_ENDPOINT: "http://reasoning-predictor.llm-serving.svc.cluster.local:8000/v1/chat/completions"
|
LLM_ENDPOINT: "http://api-internal.riotpiao.com:8000/v1/chat/completions"
|
||||||
LLM_API_BASE: "http://reasoning-predictor.llm-serving.svc.cluster.local:8000/v1"
|
LLM_MODEL: "qwen:7b"
|
||||||
LLM_MODEL: "ornith:35b"
|
|
||||||
LLM_TIMEOUT_SECS: "30"
|
LLM_TIMEOUT_SECS: "30"
|
||||||
ENABLE_LLM_EXTRACTION: "true"
|
ENABLE_LLM_EXTRACTION: "true"
|
||||||
|
|
||||||
# OpenSearch (vector store, BM25 retrieval)
|
|
||||||
OPENSEARCH_HOST: "opensearch.poimen.svc.cluster.local:9200"
|
|
||||||
OPENSEARCH_SCHEME: "http"
|
|
||||||
OPENSEARCH_VERIFY_CERTS: "false"
|
|
||||||
|
|
||||||
# Authentik (OIDC provider)
|
|
||||||
AUTHENTIK_ISSUER: "https://authentik.auth.svc.cluster.local:9443/application/o/poimen/"
|
|
||||||
AUTHENTIK_VERIFY_SSL: "false"
|
|
||||||
|
|
||||||
# Temporal (workflow orchestration - future)
|
|
||||||
TEMPORAL_ENDPOINT: "temporal-frontend.temporal.svc.cluster.local:7233"
|
|
||||||
TEMPORAL_NAMESPACE: "poimen"
|
|
||||||
|
|
||||||
# API Gateway (external queue, route optimization - future)
|
|
||||||
GATEWAY_URL: "http://api-gw.poimen.svc.cluster.local:8080"
|
|
||||||
|
|||||||
+9
-28
@@ -1,6 +1,6 @@
|
|||||||
# Poimen Memory API Server
|
# Poimen Memory API Server
|
||||||
# Serves HTTP endpoints for memory ingest, query, visualization.
|
# Serves 7 HTTP endpoints for memory ingest, query, and management.
|
||||||
# Connects to memory-db (pgvector) + api.riotpiao.com (LLM via Authentik JWT).
|
# Connects to memory-db (pgvector) for persistent storage.
|
||||||
apiVersion: apps/v1
|
apiVersion: apps/v1
|
||||||
kind: Deployment
|
kind: Deployment
|
||||||
metadata:
|
metadata:
|
||||||
@@ -60,44 +60,24 @@ spec:
|
|||||||
key: password
|
key: password
|
||||||
- name: DATABASE_URL
|
- name: DATABASE_URL
|
||||||
value: "postgresql://$(DATABASE_USER):$(DATABASE_PASSWORD)@$(DATABASE_HOST):$(DATABASE_PORT)/$(DATABASE_NAME)?sslmode=disable"
|
value: "postgresql://$(DATABASE_USER):$(DATABASE_PASSWORD)@$(DATABASE_HOST):$(DATABASE_PORT)/$(DATABASE_NAME)?sslmode=disable"
|
||||||
|
# LLM Gateway API key
|
||||||
# All downstream service URIs read from ConfigMap
|
|
||||||
# (LLM_ENDPOINT, LLM_API_BASE, LLM_MODEL, OPENSEARCH_HOST, etc.)
|
|
||||||
# These are injected via envFrom below
|
|
||||||
|
|
||||||
# Authentik service account (memory-agent-oidc secret)
|
|
||||||
# Only needed if MEM_AUTH_MODE=jwt in ConfigMap
|
|
||||||
- name: AUTHENTIK_CLIENT_ID
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-agent-oidc
|
|
||||||
key: CLIENT_ID
|
|
||||||
- name: AUTHENTIK_CLIENT_SECRET
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-agent-oidc
|
|
||||||
key: CLIENT_SECRET
|
|
||||||
- name: TOKEN_URL
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-agent-oidc
|
|
||||||
key: TOKEN_URL
|
|
||||||
|
|
||||||
# Server config
|
|
||||||
- name: MEM_API_KEY
|
- name: MEM_API_KEY
|
||||||
valueFrom:
|
valueFrom:
|
||||||
secretKeyRef:
|
secretKeyRef:
|
||||||
name: poimen-memory-secrets
|
name: poimen-memory-secrets
|
||||||
key: llm-api-key
|
key: llm-api-key
|
||||||
|
# Server config (from ConfigMap)
|
||||||
- name: MEM_PORT
|
- name: MEM_PORT
|
||||||
value: "8080"
|
value: "8080"
|
||||||
- name: MEM_HOME
|
- name: MEM_HOME
|
||||||
value: "/tmp"
|
value: "/tmp"
|
||||||
envFrom:
|
envFrom:
|
||||||
# ConfigMap with all service URIs (prod: encrypted, local: plaintext)
|
|
||||||
- configMapRef:
|
- configMapRef:
|
||||||
name: poimen-memory-config
|
name: poimen-memory-config
|
||||||
command: ["/app/mem"]
|
- secretRef:
|
||||||
|
name: poimen-memory-auth
|
||||||
|
- secretRef:
|
||||||
|
name: poimen-memory-secrets
|
||||||
args:
|
args:
|
||||||
- serve
|
- serve
|
||||||
- --port
|
- --port
|
||||||
@@ -130,6 +110,7 @@ spec:
|
|||||||
- name: tmp
|
- name: tmp
|
||||||
emptyDir:
|
emptyDir:
|
||||||
sizeLimit: 64Mi
|
sizeLimit: 64Mi
|
||||||
|
# Tolerate control-plane nodes
|
||||||
tolerations:
|
tolerations:
|
||||||
- key: node-role.kubernetes.io/control-plane
|
- key: node-role.kubernetes.io/control-plane
|
||||||
operator: Exists
|
operator: Exists
|
||||||
|
|||||||
@@ -1,11 +1,13 @@
|
|||||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
kind: Kustomization
|
kind: Kustomization
|
||||||
namespace: poimen
|
namespace: poimen
|
||||||
|
|
||||||
resources:
|
resources:
|
||||||
|
# vault-pvc.yaml removed — memory service uses pgvector, not local storage
|
||||||
- deployment.yaml
|
- deployment.yaml
|
||||||
- service.yaml
|
- service.yaml
|
||||||
- config.yaml # Production config (SOPS-encrypted)
|
- config.yaml
|
||||||
|
- obsidian.yaml
|
||||||
|
# Legacy secret managed separately
|
||||||
|
# - secrets.yaml
|
||||||
generators:
|
generators:
|
||||||
- secret-generator.yaml
|
- secret-generator.yaml
|
||||||
|
|||||||
@@ -0,0 +1,24 @@
|
|||||||
|
apiVersion: ENC[AES256_GCM,data:gSI=,iv:nfXxHTEXSY6eDPOLfQWxQaX/Ge7s08QF6GqQ847cdKg=,tag:szUjeOolHuomGQQdrV7U4A==,type:str]
|
||||||
|
kind: ENC[AES256_GCM,data:HC8zcR8G,iv:wk4XliU5bPi32M0QV6OhJs3tSkirOczWJjR+1MgjxpM=,tag:jcD8R327PRv8x7wYh4Tdrg==,type:str]
|
||||||
|
metadata:
|
||||||
|
name: ENC[AES256_GCM,data:HouUGg1P3iPycnr5doLc9w==,iv:kzODDxNBix4e/kAGrF8io165crqPHewyuG8MCZhr3mM=,tag:hX9peWSY5LwM7/08S+QLuw==,type:str]
|
||||||
|
namespace: ENC[AES256_GCM,data:OCIDOqNz,iv:GhtxD5cXXTnl/7Po1rY3I+jacI9Kz4bXp+Nz2UVTOTE=,tag:F7TuNLkSluKm4TZZ1Q33VQ==,type:str]
|
||||||
|
type: ENC[AES256_GCM,data:ErqH5L3k,iv:JioZqat2ZYSO83vEnl1MY6YiCC3RttfEkGc2OumJHBY=,tag:K68wmCX8sMzcnGUz1aBpWA==,type:str]
|
||||||
|
stringData:
|
||||||
|
id_ed25519: ENC[AES256_GCM,data:YqrAUmZCvEDC2q8c4Ns+WxW2l+oC31arj1MPYwt6AkIv026kZk9ucywkOWT/Ww0VgiUWF/0t/gzkm7K5D5fSk5vP6PyWV5Zqvlo8Zqy1VBLc3V9gbQeCsA4i8+FO3zZ0k2l3HAefEqhJ4Bnj3dKDOTX4bsbE/H/4n8WCojYnOLdU4esqm5r4bOCnFv5wBkbkob6AwfqekdaBZfuPOqW2sstDSRN6km1UZafCYuMY0XQTxCKYM8Izt3px8sfBq37oA6syDpxNuEpAk0uYumHhSBHKRAnsDY61pjfCbR2xy/3IeEPr6EVf5HwV+2ElhtSB/Zfzin5GZvAgX3gu3HuFznHUYH5olLEEFvVtw8fWLh3avwwCsAlUsDKZoTV1t1bzGni1OPYK3ZfsAmQqI1lvdFlRj++e6L3vDBkVG5qowTelbSb6/TWMDpJw/CsX3bgeKEoFUt2vi9IxwdYO/onuExVrT23WeanoSmrnXRaBqr6xIV5yW5CCbBBRmRU7a6jkwtkhe8dHFTKejaqjBpzPdlZOvhzKBHlOy4eDGjeV7CkzrRw=,iv:bSGkeMli13DSDFAu1+4Kg5sqSJ8LbdpLfN5oIwzLyTM=,tag:9DYK17ET7rkfmmpwxjicog==,type:str]
|
||||||
|
known_hosts: ENC[AES256_GCM,data:FaWsLxkot5Zxh7mobbUGDFqKLOdmP5APg09nkDyqGHyDp0v0Pjc19jeizM+tq3b3aH59YGlPe/4xQf7ZXZRZjWQE+m/TcVujtLD4hfCc40wqZh1XtTtdC6Tf1p3JbqxP0uQVy1+EFVwPCimUsZfp7gtcT/Hmu1JAW9biGmvACO9+dDeHGzBH5ZRCw3+dEYmcLgGIRgzpOwJLfHr/hkvdlflhzmEHMliBIl+TpqQ38GFQmw0ia7UEJzj3ghoDj7HjrHqlBa7aBHJpaEBYVwq8cN7JaLnyO1Y4+LIU8ln/CEzeg9wxVJoMO8IBcQCCgXoC+ogEpNFVb+pdUfRl/3Ye97ZJFmdJoorvSHIR02e9n7E2G3Ox9iImwnwI76X3FokuY0zcGIkcIho6JN3/8k3Z4VLDl2qGflo7jK6QP1DEsGUGhGwPRVrpPNkMDxYQeBIqlwFzfRuF/gDZj3ZWadCYB7NwByVgTcZFqiMtQ74z6jYGMiPIpW2OCY4HGu9ecGPR02USEu39CjJUCWH9WbQZTjmK3n4yYy4X4WPMbc0IekSCC2ossBznMoFsu7q7L13arqC99j9ZOv8aJ7KgMpGpOVPoN2AURJTFhgMX8TD93AvbFNVNqA0t+Y+g0Hq5f/py8XPzj4b6l8A4QxK3Awj4gf5BbK2lNjL+Cgo2kgwPsPvNd4hvx3gbamDOPNf+lH6iGLF19QyoJPHlOD/k9hkcMRerAEOLZriTylgpj2joXOaeKVrk8qkAZBunQKNc3e0xXH1i5obqz871DbbOVvrfYhmSNG93IEDG3hvNPbIc2uXYIJBciXxEaNO2PEypaLDgnNczEoVyUn2MXTJ6XMr/fkKvVBTmDuFd7BWk1JM6gWKnKvdN2G7bBHz5D43jGmmv+Z4bcd4Z4ViO8yAMbB3kKMLxMZQiSsrOudcQM1rzip2HBYb7JhK3yQwTbqTXZo64hejYW6+KYZysSB+A4itITvQR1G50lKndLd1XXqRbfV00DSZmNPRt0dKcTRkEX2RdCOkh6Wddo7D41V4eDRoHY+rctirsDKE/IA875ooAxpj5CWsHjAzlVrSUUVB2D0IBdlCPaZ5LVSo8i8+9S4rakEdmwe5dceUTC9mNIuzFtUeAshW4J2U4UzIQGOVt0BolwWfWiO5QvuAMC7aH1kJOr9gUB69IwDASdd+PbyEvjCQigVXEeObcliJu9QZohcoQU09fI1ajmPC5x2g2Rbo8DPakQTdsHuzdNm/zMZ/yUhi7+5/jZmwI3kRdgA9EMmhl+T2qAwNbhBiW2Iim44R0ZFuWMv6SCX2Kl9KxH+HW7dS8H7FJbAG7w9kiivpHP0sBTRfSgN9ddR6veVRiZWXNJ38sNy88GXdwzRM90XJAwQ==,iv:mv3hoMwPcEmOBbsIRoKLUuEsUolotv1VtikLiItwuJg=,tag:7amQiuiaVwowLAQcNoq72A==,type:str]
|
||||||
|
sops:
|
||||||
|
age:
|
||||||
|
- enc: |
|
||||||
|
-----BEGIN AGE ENCRYPTED FILE-----
|
||||||
|
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBINTF2OGRmUWoxTmFEZUdv
|
||||||
|
SWJKRlFhaVkzbUtOaTVnbHFjd1UzR2RFdVFNClpFODhlQVJIQjhlOEJBL2pDVmJa
|
||||||
|
UjlZZmFHKzA4eDJEMk9KTDMyOGZ2VWcKLS0tIFRHREo1dXNRKzJVYkROQSt1WjFV
|
||||||
|
ZXZYVjAwSlZhT0ZMbG1qNDVUWnJyQ2cKfs4t6HsQG5Wiyp6QvFqvm4+/o4NAL3qu
|
||||||
|
6L9vyhl2jufrbxmR+IsEBCxYS7rh6dCbxTUFap3MD2lYIGF9hRnGjQ==
|
||||||
|
-----END AGE ENCRYPTED FILE-----
|
||||||
|
recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
|
||||||
|
lastmodified: "2026-08-28T23:27:09Z"
|
||||||
|
mac: ENC[AES256_GCM,data:wJC6YCHXq6I/bqUjfwFRvpULZ1Yt39PoWFKzzOAq6h/pHsUrWEgrkm+3+dLaPpz663b0B75BiCQjQb4igXWr38O5I+FKonRHsbsH+D+pO+dq++yNYG8T30KGaquVfnsm8ijWGWxOY9nULUXfKcYfqvsR9P7KCV7bdcWuZ5xzZ5o=,iv:w5f3h0hb2ooeNYK1QZactpmpT8mAYa94V8FBewP0MUY=,tag:qlgUutFanhjk1HIBJqmLQg==,type:str]
|
||||||
|
unencrypted_suffix: _unencrypted
|
||||||
|
version: 3.13.2
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
---
|
||||||
|
# Obsidian server deployment
|
||||||
|
# Serves local vault with web UI and API
|
||||||
|
apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
metadata:
|
||||||
|
name: obsidian-server
|
||||||
|
namespace: poimen
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/name: obsidian-server
|
||||||
|
app.kubernetes.io/part-of: poimen-memory
|
||||||
|
spec:
|
||||||
|
replicas: 1
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app.kubernetes.io/name: obsidian-server
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/name: obsidian-server
|
||||||
|
app.kubernetes.io/part-of: poimen-memory
|
||||||
|
spec:
|
||||||
|
serviceAccountName: obsidian-server
|
||||||
|
securityContext:
|
||||||
|
runAsNonRoot: true
|
||||||
|
runAsUser: 1000
|
||||||
|
runAsGroup: 1000
|
||||||
|
fsGroup: 1000
|
||||||
|
seccompProfile:
|
||||||
|
type: RuntimeDefault
|
||||||
|
initContainers:
|
||||||
|
- name: git-sync-init
|
||||||
|
image: alpine/git:latest
|
||||||
|
securityContext:
|
||||||
|
runAsNonRoot: false
|
||||||
|
runAsUser: 0
|
||||||
|
allowPrivilegeEscalation: false
|
||||||
|
capabilities:
|
||||||
|
drop:
|
||||||
|
- ALL
|
||||||
|
add:
|
||||||
|
- CHOWN
|
||||||
|
- DAC_OVERRIDE
|
||||||
|
command:
|
||||||
|
- sh
|
||||||
|
- -c
|
||||||
|
- |
|
||||||
|
export GIT_SSH_COMMAND="ssh -i /root/.ssh/id_ed25519 -o StrictHostKeyChecking=no"
|
||||||
|
git config --global --add safe.directory /vault
|
||||||
|
if [ -d /vault/.git ]; then
|
||||||
|
cd /vault && git pull origin main || true
|
||||||
|
else
|
||||||
|
# Clone into temp, move contents into vault
|
||||||
|
rm -rf /tmp/repo
|
||||||
|
git clone ssh://[email protected]:2222/rock/poimen-obesdient-memory.git /tmp/repo
|
||||||
|
cp -a /tmp/repo/. /vault/
|
||||||
|
rm -rf /tmp/repo
|
||||||
|
fi
|
||||||
|
chown -R 1000:1000 /vault
|
||||||
|
volumeMounts:
|
||||||
|
- name: vault
|
||||||
|
mountPath: /vault
|
||||||
|
- name: ssh-key
|
||||||
|
mountPath: /root/.ssh
|
||||||
|
readOnly: true
|
||||||
|
containers:
|
||||||
|
- name: obsidian-server
|
||||||
|
image: ppatlabs/obsidian:latest
|
||||||
|
imagePullPolicy: IfNotPresent
|
||||||
|
securityContext:
|
||||||
|
allowPrivilegeEscalation: false
|
||||||
|
capabilities:
|
||||||
|
drop:
|
||||||
|
- ALL
|
||||||
|
ports:
|
||||||
|
- name: http
|
||||||
|
containerPort: 27124
|
||||||
|
protocol: TCP
|
||||||
|
env:
|
||||||
|
- name: VAULT_NAME
|
||||||
|
value: poimen-vault
|
||||||
|
- name: VAULT_PATH
|
||||||
|
value: /vault
|
||||||
|
- name: REST_API_ENABLED
|
||||||
|
value: "true"
|
||||||
|
- name: REST_API_PORT
|
||||||
|
value: "8080"
|
||||||
|
volumeMounts:
|
||||||
|
- name: vault
|
||||||
|
mountPath: /vault
|
||||||
|
- name: config
|
||||||
|
mountPath: /config
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
cpu: 100m
|
||||||
|
memory: 256Mi
|
||||||
|
limits:
|
||||||
|
cpu: 500m
|
||||||
|
memory: 512Mi
|
||||||
|
livenessProbe:
|
||||||
|
httpGet:
|
||||||
|
path: /
|
||||||
|
port: http
|
||||||
|
scheme: HTTPS
|
||||||
|
initialDelaySeconds: 30
|
||||||
|
periodSeconds: 10
|
||||||
|
timeoutSeconds: 5
|
||||||
|
readinessProbe:
|
||||||
|
httpGet:
|
||||||
|
path: /
|
||||||
|
port: http
|
||||||
|
scheme: HTTPS
|
||||||
|
initialDelaySeconds: 15
|
||||||
|
periodSeconds: 5
|
||||||
|
timeoutSeconds: 5
|
||||||
|
volumes:
|
||||||
|
- name: vault
|
||||||
|
persistentVolumeClaim:
|
||||||
|
claimName: obsidian-vault
|
||||||
|
- name: config
|
||||||
|
emptyDir: {}
|
||||||
|
- name: ssh-key
|
||||||
|
secret:
|
||||||
|
secretName: obsidian-git-ssh
|
||||||
|
defaultMode: 0400
|
||||||
|
|
||||||
|
# PVC managed by homelab repo (k8s/infra/databases/obsidian-vault-pvc.yaml)
|
||||||
|
|
||||||
|
---
|
||||||
|
# Service for Obsidian server
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Service
|
||||||
|
metadata:
|
||||||
|
name: obsidian-server
|
||||||
|
namespace: poimen
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/name: obsidian-server
|
||||||
|
spec:
|
||||||
|
type: ClusterIP
|
||||||
|
ports:
|
||||||
|
- name: http
|
||||||
|
port: 80
|
||||||
|
targetPort: 27124
|
||||||
|
protocol: TCP
|
||||||
|
selector:
|
||||||
|
app.kubernetes.io/name: obsidian-server
|
||||||
|
|
||||||
|
---
|
||||||
|
# ServiceAccount for Obsidian
|
||||||
|
apiVersion: v1
|
||||||
|
kind: ServiceAccount
|
||||||
|
metadata:
|
||||||
|
name: obsidian-server
|
||||||
|
namespace: poimen
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/name: obsidian-server
|
||||||
|
|
||||||
|
# Ingress managed by homelab repo (obsidian.riotpiao.com)
|
||||||
|
# See: homelab/k8s/bootstrap/ingress/ingress.yaml
|
||||||
@@ -6,4 +6,3 @@ kind: Kustomization
|
|||||||
resources:
|
resources:
|
||||||
- memory-db.yaml
|
- memory-db.yaml
|
||||||
- opensearch.yaml
|
- opensearch.yaml
|
||||||
- opensearch-secrets.enc.yaml
|
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
# Dedicated CNPG Postgres for Poimen Memory (GitOps, wave 2).
|
---
|
||||||
# Matches homelab/k8s/infra/databases/memory-db.yaml — single source of truth.
|
# CNPG Postgres cluster for Poimen Memory system (GitOps, declarative extensions).
|
||||||
# CNPG generates secret `memory-db-app` + service `memory-db-rw` in ns poimen.
|
# 2 instances, pgvector 0.7.0 via spec.extensions (not manual CREATE EXTENSION).
|
||||||
|
# Storage: 10Gi longhorn, consistent with temporal-db.yaml.
|
||||||
|
# No manual psql needed — all via git/ArgoCD.
|
||||||
apiVersion: postgresql.cnpg.io/v1
|
apiVersion: postgresql.cnpg.io/v1
|
||||||
kind: Cluster
|
kind: Cluster
|
||||||
metadata:
|
metadata:
|
||||||
@@ -9,24 +11,15 @@ metadata:
|
|||||||
annotations:
|
annotations:
|
||||||
argocd.argoproj.io/sync-options: SkipDryRunOnMissingResource=true
|
argocd.argoproj.io/sync-options: SkipDryRunOnMissingResource=true
|
||||||
spec:
|
spec:
|
||||||
instances: 3
|
instances: 2
|
||||||
imageName: ghcr.io/cloudnative-pg/postgresql:16.2
|
imageName: ghcr.io/cloudnative-pg/postgresql:16.2
|
||||||
bootstrap:
|
|
||||||
initdb:
|
|
||||||
database: memory
|
|
||||||
owner: app
|
|
||||||
encoding: UTF8
|
|
||||||
localeCollate: C
|
|
||||||
localeCType: C
|
|
||||||
postInitApplicationSQL:
|
|
||||||
- "CREATE EXTENSION vector;"
|
|
||||||
enableSuperuserAccess: false
|
enableSuperuserAccess: false
|
||||||
|
storage:
|
||||||
|
size: 10Gi
|
||||||
|
storageClass: longhorn
|
||||||
resources:
|
resources:
|
||||||
requests: { memory: "512Mi", cpu: "250m" }
|
requests: { memory: "512Mi", cpu: "250m" }
|
||||||
limits: { memory: "2Gi", cpu: "1" }
|
limits: { memory: "2Gi", cpu: "1" }
|
||||||
storage:
|
|
||||||
size: 20Gi
|
|
||||||
storageClass: longhorn
|
|
||||||
affinity:
|
affinity:
|
||||||
podAntiAffinityType: preferred
|
podAntiAffinityType: preferred
|
||||||
topologyKey: kubernetes.io/hostname
|
topologyKey: kubernetes.io/hostname
|
||||||
@@ -34,3 +27,32 @@ spec:
|
|||||||
- key: node-role.kubernetes.io/control-plane
|
- key: node-role.kubernetes.io/control-plane
|
||||||
operator: Exists
|
operator: Exists
|
||||||
effect: NoSchedule
|
effect: NoSchedule
|
||||||
|
bootstrap:
|
||||||
|
initdb:
|
||||||
|
database: memory
|
||||||
|
owner: app
|
||||||
|
encoding: UTF8
|
||||||
|
localeCollate: C
|
||||||
|
localeCType: C
|
||||||
|
monitoring:
|
||||||
|
enabled: true
|
||||||
|
podMonitorTemplate:
|
||||||
|
spec:
|
||||||
|
interval: 30s
|
||||||
|
scrapeTimeout: 10s
|
||||||
|
---
|
||||||
|
# Database resource with pgvector extension (declarative, git-managed).
|
||||||
|
# CNPG 1.30.0+ supports this via spec.extensions on the Database CRD.
|
||||||
|
# Ensures pgvector is installed and available for HNSW indexing.
|
||||||
|
apiVersion: postgresql.cnpg.io/v1
|
||||||
|
kind: Database
|
||||||
|
metadata:
|
||||||
|
name: memory
|
||||||
|
namespace: poimen
|
||||||
|
spec:
|
||||||
|
cluster:
|
||||||
|
name: memory-db
|
||||||
|
owner: app
|
||||||
|
extensions:
|
||||||
|
- name: vector
|
||||||
|
ensure: present
|
||||||
|
|||||||
@@ -1,47 +0,0 @@
|
|||||||
apiVersion: ENC[AES256_GCM,data:qM0=,iv:znTNMu1+efRh38Vn0GWlNZTk/6VjCJfJeaEzbM17N8c=,tag:sPSc9mwoZWYvjD1bzM+uzg==,type:str]
|
|
||||||
kind: ENC[AES256_GCM,data:pHDYbqGy,iv:8kUzizuj3tkgx8FU19FBr8lcz1DFEN2abQTJCFLPL0w=,tag:YqiHCVZ8Pwyx51YkxLSykQ==,type:str]
|
|
||||||
metadata:
|
|
||||||
name: ENC[AES256_GCM,data:yC5ph8jQnEd2Jn60tCNYJQq2,iv:QRAhTVXNt77kcbcLXDJo9Y1X3hRu1EZXADwTS3rPq/g=,tag:X80UnBGCV28GiOWNo3K/bA==,type:str]
|
|
||||||
namespace: ENC[AES256_GCM,data:7N36Xqio,iv:a8yemv8LA1WdXUyNRgTu5teZIB23ClufXh7ovd9m5GU=,tag:ukU63zxVZdD9PwppgAmaEw==,type:str]
|
|
||||||
type: ENC[AES256_GCM,data:9NGNI47z,iv:tiaioFpXheBY4BimysI3sr5OzFOEI1mG68ObCDiqAIU=,tag:wpEln7lSyAPfxpVjcWhpVg==,type:str]
|
|
||||||
stringData:
|
|
||||||
admin-password: ENC[AES256_GCM,data:HeKM7q8662fdrlJbpWh/7VJuhr7h2sRYK6/sN+eBtBo=,iv:4ur6YKAYp6+kvIkmBcx9/0DK2MvK7XDedoZhKl8gjBY=,tag:pct7MAlre1h7bm8Polvutg==,type:str]
|
|
||||||
sops:
|
|
||||||
age:
|
|
||||||
- enc: |
|
|
||||||
-----BEGIN AGE ENCRYPTED FILE-----
|
|
||||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGV2NJQzhDN0ZacXBDeklV
|
|
||||||
aGE1eGlmMkp6b1RDL2ZiblNwSk1PUkJZdFZjCi84dWpXMFNNcFYrLzkwOUFGZDZ4
|
|
||||||
SGM0NG9UMkJTME82dUU0MkxFNjVzcTAKLS0tIE5xNlg2RUdheUxyUytsblI3UTFH
|
|
||||||
UktjaHNGOUlmZGxiSlhoSkJSMW5LMkkKdNAzdge1HaAgBqbE4dCkJgZBlIAP76P+
|
|
||||||
4GOsh7RbuVDDMzUHTS4aNv2zoM5WC5pv+ZKtf8Yu7LIwiOPAp2u/7g==
|
|
||||||
-----END AGE ENCRYPTED FILE-----
|
|
||||||
recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
|
|
||||||
lastmodified: "2026-09-12T14:22:55Z"
|
|
||||||
mac: ENC[AES256_GCM,data:lO+5lWN4ZVIkg4XAG4mz6n2SxqNfU6KdahoZqj9nZ33maX/9OT7aunwl3eIoE8JlN4vN1UU/s0l1ioT0+PxdGtlQfhisZ0ypzA3z8Nxkcw18XzQaMf99A0Icw1OEGRRx/T6Bf8+l0ZI4HIH+KZmlUg2lAfGK+WTxqr5xJefw5XA=,iv:kWNox9QX7Jv9muHjBo6yuwRjBRuhawaKJ+5+O9E57z4=,tag:qZ+5ceNo2C8cPIt0PBtqiw==,type:str]
|
|
||||||
unencrypted_suffix: _unencrypted
|
|
||||||
version: 3.13.2
|
|
||||||
---
|
|
||||||
apiVersion: ENC[AES256_GCM,data:jew=,iv:bzrjT8rJssrSv4xZCn9ihNtyelKteybg/XZVJRUawvo=,tag:qN50XwBiN2lnWH1CS35W/g==,type:str]
|
|
||||||
kind: ENC[AES256_GCM,data:clwtkPLP,iv:Y2sF8dpOJslo4OHeRprK/wcuvUzdOW30Bz3M0Kh8yE4=,tag:TkQo8zZZow/zdAlxViWQlA==,type:str]
|
|
||||||
metadata:
|
|
||||||
name: ENC[AES256_GCM,data:UiOyRh0x7Yor3qudRqgwtrv5bBHkHnFMK0Smxw==,iv:a3hB+wBIMjD0Xj7p3ZIqDf3/la1xlzbCYRRc/LV80ig=,tag:ivw42aZepKgOa+cVm0URZg==,type:str]
|
|
||||||
namespace: ENC[AES256_GCM,data:a37Jp+qA,iv:cDVuBJ/aFo4EcZTC/N9NGk8UdrCROHKiirWBWlrSDMQ=,tag:/LNyMpSaEhgQYIE8PJcXBg==,type:str]
|
|
||||||
type: ENC[AES256_GCM,data:ql+XYM25,iv:OCfk13+9Ft4Vq6Tq3R6v54zK1tz7imzyr/g8ytcpBEk=,tag:PpJBFsUSpMo3ktGynr8AUw==,type:str]
|
|
||||||
stringData:
|
|
||||||
password: ENC[AES256_GCM,data:lT7f3cY+VLqRcfuYnf9lnI5QVqfmmt5pc9tb2EEuRbE=,iv:2Dnwssmj5ddcr4UypVVkm4UydeLc1L/ExsaRsbu/CVw=,tag:6++s/j3MOBIirkKS66Z46Q==,type:str]
|
|
||||||
sops:
|
|
||||||
age:
|
|
||||||
- enc: |
|
|
||||||
-----BEGIN AGE ENCRYPTED FILE-----
|
|
||||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGV2NJQzhDN0ZacXBDeklV
|
|
||||||
aGE1eGlmMkp6b1RDL2ZiblNwSk1PUkJZdFZjCi84dWpXMFNNcFYrLzkwOUFGZDZ4
|
|
||||||
SGM0NG9UMkJTME82dUU0MkxFNjVzcTAKLS0tIE5xNlg2RUdheUxyUytsblI3UTFH
|
|
||||||
UktjaHNGOUlmZGxiSlhoSkJSMW5LMkkKdNAzdge1HaAgBqbE4dCkJgZBlIAP76P+
|
|
||||||
4GOsh7RbuVDDMzUHTS4aNv2zoM5WC5pv+ZKtf8Yu7LIwiOPAp2u/7g==
|
|
||||||
-----END AGE ENCRYPTED FILE-----
|
|
||||||
recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
|
|
||||||
lastmodified: "2026-09-12T14:22:55Z"
|
|
||||||
mac: ENC[AES256_GCM,data:lO+5lWN4ZVIkg4XAG4mz6n2SxqNfU6KdahoZqj9nZ33maX/9OT7aunwl3eIoE8JlN4vN1UU/s0l1ioT0+PxdGtlQfhisZ0ypzA3z8Nxkcw18XzQaMf99A0Icw1OEGRRx/T6Bf8+l0ZI4HIH+KZmlUg2lAfGK+WTxqr5xJefw5XA=,iv:kWNox9QX7Jv9muHjBo6yuwRjBRuhawaKJ+5+O9E57z4=,tag:qZ+5ceNo2C8cPIt0PBtqiw==,type:str]
|
|
||||||
unencrypted_suffix: _unencrypted
|
|
||||||
version: 3.13.2
|
|
||||||
@@ -65,7 +65,8 @@ data:
|
|||||||
# Cluster settings
|
# Cluster settings
|
||||||
cluster.name: poimen-memory
|
cluster.name: poimen-memory
|
||||||
node.name: ${HOSTNAME}
|
node.name: ${HOSTNAME}
|
||||||
discovery.type: single-node
|
cluster.initial_master_nodes: opensearch-0
|
||||||
|
discovery.seed_hosts: opensearch-0.opensearch.poimen.svc.cluster.local
|
||||||
|
|
||||||
# Network
|
# Network
|
||||||
network.host: 0.0.0.0
|
network.host: 0.0.0.0
|
||||||
@@ -126,12 +127,16 @@ spec:
|
|||||||
spec:
|
spec:
|
||||||
serviceAccountName: opensearch
|
serviceAccountName: opensearch
|
||||||
hostNetwork: false
|
hostNetwork: false
|
||||||
securityContext:
|
|
||||||
fsGroup: 1000
|
initContainers:
|
||||||
tolerations:
|
- name: sysctl
|
||||||
- key: node-role.kubernetes.io/control-plane
|
image: busybox:1.28
|
||||||
operator: Exists
|
command:
|
||||||
effect: NoSchedule
|
- sysctl
|
||||||
|
- -w
|
||||||
|
- vm.max_map_count=262144
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
|
||||||
containers:
|
containers:
|
||||||
- name: opensearch
|
- name: opensearch
|
||||||
@@ -395,6 +400,18 @@ spec:
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
# Secret: OpenSearch Dashboards password
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: opensearch-dashboards-secret
|
||||||
|
namespace: poimen
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
password: "admin" # ⚠️ Change in production
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
# ServiceAccount for OpenSearch Dashboards
|
# ServiceAccount for OpenSearch Dashboards
|
||||||
apiVersion: v1
|
apiVersion: v1
|
||||||
kind: ServiceAccount
|
kind: ServiceAccount
|
||||||
@@ -402,4 +419,14 @@ metadata:
|
|||||||
name: opensearch-dashboards
|
name: opensearch-dashboards
|
||||||
namespace: poimen
|
namespace: poimen
|
||||||
|
|
||||||
# Secrets moved to opensearch-secrets.enc.yaml (SOPS-encrypted)
|
---
|
||||||
|
|
||||||
|
# Secret for OpenSearch Admin Password
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: opensearch-secrets
|
||||||
|
namespace: poimen
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
admin-password: "OpenSearch@Admin123!"
|
||||||
|
|||||||
@@ -1,131 +0,0 @@
|
|||||||
{
|
|
||||||
"annotations": { "list": [] },
|
|
||||||
"editable": true,
|
|
||||||
"fiscalYearStartMonth": 0,
|
|
||||||
"graphTooltip": 0,
|
|
||||||
"id": null,
|
|
||||||
"links": [],
|
|
||||||
"panels": [
|
|
||||||
{
|
|
||||||
"title": "Ingest Rate (req/s)",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 0 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "rate(memory_ingest_requests_total[5m])", "legendFormat": "ingest req/s" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Query Rate (req/s)",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 0 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "rate(memory_query_requests_total[5m])", "legendFormat": "query req/s" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Ingest Latency (p50/p95/p99)",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "histogram_quantile(0.5, rate(memory_ingest_duration_seconds_bucket[5m]))", "legendFormat": "p50" },
|
|
||||||
{ "expr": "histogram_quantile(0.95, rate(memory_ingest_duration_seconds_bucket[5m]))", "legendFormat": "p95" },
|
|
||||||
{ "expr": "histogram_quantile(0.99, rate(memory_ingest_duration_seconds_bucket[5m]))", "legendFormat": "p99" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Query Latency (p50/p95/p99)",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "histogram_quantile(0.5, rate(memory_query_duration_seconds_bucket[5m]))", "legendFormat": "p50" },
|
|
||||||
{ "expr": "histogram_quantile(0.95, rate(memory_query_duration_seconds_bucket[5m]))", "legendFormat": "p95" },
|
|
||||||
{ "expr": "histogram_quantile(0.99, rate(memory_query_duration_seconds_bucket[5m]))", "legendFormat": "p99" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Error Rates",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 16 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "rate(memory_ingest_errors_total[5m])", "legendFormat": "ingest errors" },
|
|
||||||
{ "expr": "rate(memory_query_errors_total[5m])", "legendFormat": "query errors" },
|
|
||||||
{ "expr": "rate(memory_query_embedding_failures_total[5m])", "legendFormat": "embedding failures" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Embedding Latency",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 16 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "histogram_quantile(0.5, rate(memory_query_embedding_duration_seconds_bucket[5m]))", "legendFormat": "p50" },
|
|
||||||
{ "expr": "histogram_quantile(0.95, rate(memory_query_embedding_duration_seconds_bucket[5m]))", "legendFormat": "p95" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "DB Row Counts",
|
|
||||||
"type": "stat",
|
|
||||||
"gridPos": { "h": 4, "w": 12, "x": 0, "y": 24 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "memory_db_table_entity_rows", "legendFormat": "entities" },
|
|
||||||
{ "expr": "memory_db_table_edge_rows", "legendFormat": "edges" },
|
|
||||||
{ "expr": "memory_db_table_chunk_rows", "legendFormat": "chunks" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Dependency Health",
|
|
||||||
"type": "stat",
|
|
||||||
"gridPos": { "h": 4, "w": 12, "x": 12, "y": 24 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "memory_dependency_db_up", "legendFormat": "DB" },
|
|
||||||
{ "expr": "memory_dependency_embedding_up", "legendFormat": "Embedding" },
|
|
||||||
{ "expr": "memory_dependency_opensearch_up", "legendFormat": "OpenSearch" },
|
|
||||||
{ "expr": "memory_dependency_llm_up", "legendFormat": "LLM" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "DB Pool Stats",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 28 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "memory_db_pool_size", "legendFormat": "pool size" },
|
|
||||||
{ "expr": "memory_db_pool_idle", "legendFormat": "idle" },
|
|
||||||
{ "expr": "memory_db_pool_active", "legendFormat": "active" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Relevance Metrics",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 28 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "memory_relevance_precision", "legendFormat": "precision" },
|
|
||||||
{ "expr": "memory_relevance_recall", "legendFormat": "recall" },
|
|
||||||
{ "expr": "memory_relevance_f1_score", "legendFormat": "F1" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "In-Flight Operations",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 36 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "memory_ingest_in_flight", "legendFormat": "ingest" },
|
|
||||||
{ "expr": "memory_query_in_flight", "legendFormat": "query" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"title": "Write Volume",
|
|
||||||
"type": "timeseries",
|
|
||||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 36 },
|
|
||||||
"targets": [
|
|
||||||
{ "expr": "rate(memory_write_entities_total[5m])", "legendFormat": "entities/s" },
|
|
||||||
{ "expr": "rate(memory_write_edges_total[5m])", "legendFormat": "edges/s" },
|
|
||||||
{ "expr": "rate(memory_write_chunks_total[5m])", "legendFormat": "chunks/s" }
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"schemaVersion": 39,
|
|
||||||
"tags": ["poimen", "memory", "observability"],
|
|
||||||
"templating": { "list": [] },
|
|
||||||
"time": { "from": "now-1h", "to": "now" },
|
|
||||||
"title": "Poimen Memory Observability",
|
|
||||||
"uid": "poimen-memory-obs"
|
|
||||||
}
|
|
||||||
@@ -1,130 +0,0 @@
|
|||||||
# Prometheus alerting rules for Poimen Memory (O12)
|
|
||||||
# Deploy: kubectl apply -f k8s/infra/prometheus-alerts.yaml
|
|
||||||
apiVersion: monitoring.coreos.com/v1
|
|
||||||
kind: PrometheusRule
|
|
||||||
metadata:
|
|
||||||
name: poimen-memory-alerts
|
|
||||||
namespace: poimen
|
|
||||||
labels:
|
|
||||||
app: poimen-memory
|
|
||||||
prometheus: k8s
|
|
||||||
role: alert-rules
|
|
||||||
spec:
|
|
||||||
groups:
|
|
||||||
- name: poimen-memory.availability
|
|
||||||
rules:
|
|
||||||
- alert: MemoryServiceDown
|
|
||||||
expr: up{job="poimen-memory"} == 0
|
|
||||||
for: 2m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "Poimen memory service is down"
|
|
||||||
description: "Memory service has been unreachable for > 2 minutes"
|
|
||||||
|
|
||||||
- alert: MemoryDBDown
|
|
||||||
expr: memory_dependency_db_up == 0
|
|
||||||
for: 1m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "Memory service cannot reach database"
|
|
||||||
description: "DB dependency health check failing for > 1 minute"
|
|
||||||
|
|
||||||
- alert: MemoryEmbeddingDown
|
|
||||||
expr: memory_dependency_embedding_up == 0
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Embedding service unreachable"
|
|
||||||
description: "Embedding dependency health check failing for > 5 minutes"
|
|
||||||
|
|
||||||
- name: poimen-memory.latency
|
|
||||||
rules:
|
|
||||||
- alert: MemoryIngestLatencyHigh
|
|
||||||
expr: histogram_quantile(0.95, rate(memory_ingest_duration_seconds_bucket[5m])) > 5
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Ingest p95 latency > 5s"
|
|
||||||
description: "95th percentile ingest latency is {{ $value }}s"
|
|
||||||
|
|
||||||
- alert: MemoryQueryLatencyHigh
|
|
||||||
expr: histogram_quantile(0.95, rate(memory_query_duration_seconds_bucket[5m])) > 2
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Query p95 latency > 2s"
|
|
||||||
description: "95th percentile query latency is {{ $value }}s"
|
|
||||||
|
|
||||||
- alert: MemoryEmbeddingLatencyHigh
|
|
||||||
expr: histogram_quantile(0.95, rate(memory_query_embedding_duration_seconds_bucket[5m])) > 10
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Embedding p95 latency > 10s"
|
|
||||||
description: "95th percentile embedding call latency is {{ $value }}s"
|
|
||||||
|
|
||||||
- name: poimen-memory.errors
|
|
||||||
rules:
|
|
||||||
- alert: MemoryIngestErrorRateHigh
|
|
||||||
expr: rate(memory_ingest_errors_total[5m]) / rate(memory_ingest_requests_total[5m]) > 0.1
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Ingest error rate > 10%"
|
|
||||||
description: "{{ $value | humanizePercentage }} of ingest requests are failing"
|
|
||||||
|
|
||||||
- alert: MemoryQueryErrorRateHigh
|
|
||||||
expr: rate(memory_query_errors_total[5m]) / rate(memory_query_requests_total[5m]) > 0.1
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Query error rate > 10%"
|
|
||||||
description: "{{ $value | humanizePercentage }} of query requests are failing"
|
|
||||||
|
|
||||||
- alert: MemoryEmbeddingFailureRate
|
|
||||||
expr: rate(memory_query_embedding_failures_total[5m]) > 0.5
|
|
||||||
for: 3m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "Embedding failures > 0.5/s"
|
|
||||||
description: "Embedding service failing at {{ $value }}/s — queries cannot embed"
|
|
||||||
|
|
||||||
- name: poimen-memory.storage
|
|
||||||
rules:
|
|
||||||
- alert: MemoryDBPoolExhausted
|
|
||||||
expr: memory_db_pool_idle == 0
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "DB connection pool exhausted"
|
|
||||||
description: "No idle DB connections for > 5 minutes"
|
|
||||||
|
|
||||||
- alert: MemoryWriteErrorsHigh
|
|
||||||
expr: rate(memory_write_errors_total[5m]) > 1
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Write errors > 1/s"
|
|
||||||
description: "Database write errors at {{ $value }}/s"
|
|
||||||
|
|
||||||
- name: poimen-memory.quality
|
|
||||||
rules:
|
|
||||||
- alert: MemoryRelevanceLow
|
|
||||||
expr: memory_relevance_precision < 0.3
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Retrieval relevance precision < 30%"
|
|
||||||
description: "Relevance precision is {{ $value | humanizePercentage }}"
|
|
||||||
@@ -1,94 +0,0 @@
|
|||||||
# CronJob for periodic relevance evaluation (O13)
|
|
||||||
# Runs sample queries against memory service and evaluates result relevance
|
|
||||||
# Pushes metrics to Prometheus via pushgateway or direct scrape
|
|
||||||
apiVersion: batch/v1
|
|
||||||
kind: CronJob
|
|
||||||
metadata:
|
|
||||||
name: memory-relevance-eval
|
|
||||||
namespace: poimen
|
|
||||||
labels:
|
|
||||||
app: memory-relevance-eval
|
|
||||||
spec:
|
|
||||||
# Run every 6 hours
|
|
||||||
schedule: "0 */6 * * *"
|
|
||||||
successfulJobsHistoryLimit: 3
|
|
||||||
failedJobsHistoryLimit: 1
|
|
||||||
|
|
||||||
jobTemplate:
|
|
||||||
spec:
|
|
||||||
template:
|
|
||||||
metadata:
|
|
||||||
labels:
|
|
||||||
app: memory-relevance-eval
|
|
||||||
spec:
|
|
||||||
securityContext:
|
|
||||||
runAsNonRoot: true
|
|
||||||
runAsUser: 1000
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
containers:
|
|
||||||
- name: eval
|
|
||||||
image: curlimages/curl:8.13.0
|
|
||||||
securityContext:
|
|
||||||
allowPrivilegeEscalation: false
|
|
||||||
capabilities:
|
|
||||||
drop: ["ALL"]
|
|
||||||
command:
|
|
||||||
- /bin/sh
|
|
||||||
- -c
|
|
||||||
- |
|
|
||||||
MEMORY_URL="http://poimen-memory.poimen.svc.cluster.local:8080"
|
|
||||||
|
|
||||||
echo "=== Relevance evaluation at $(date) ==="
|
|
||||||
|
|
||||||
# Sample queries for evaluation
|
|
||||||
QUERIES='[
|
|
||||||
"kubernetes deployment",
|
|
||||||
"database migration",
|
|
||||||
"LLM entity extraction",
|
|
||||||
"tea cli forgejo",
|
|
||||||
"SOPS encryption secrets"
|
|
||||||
]'
|
|
||||||
|
|
||||||
TOTAL=0
|
|
||||||
RELEVANT=0
|
|
||||||
|
|
||||||
for q in "kubernetes deployment" "database migration" "LLM entity extraction"; do
|
|
||||||
echo "Testing query: $q"
|
|
||||||
RESULT=$(curl -s --max-time 30 -X POST "$MEMORY_URL/memory/query" \
|
|
||||||
-H "Content-Type: application/json" \
|
|
||||||
-d "{\"query\": \"$q\", \"search_type\": \"entities\", \"top_k\": 5}")
|
|
||||||
|
|
||||||
COUNT=$(echo "$RESULT" | grep -o '"total_count":[0-9]*' | cut -d: -f2)
|
|
||||||
TOTAL=$((TOTAL + 1))
|
|
||||||
|
|
||||||
if [ "${COUNT:-0}" -gt 0 ]; then
|
|
||||||
RELEVANT=$((RELEVANT + 1))
|
|
||||||
echo " Result: $COUNT results (relevant)"
|
|
||||||
else
|
|
||||||
echo " Result: 0 results (irrelevant)"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
PRECISION=$(echo "scale=2; $RELEVANT / $TOTAL" | bc 2>/dev/null || echo "0")
|
|
||||||
echo ""
|
|
||||||
echo "=== Summary ==="
|
|
||||||
echo "Total queries: $TOTAL"
|
|
||||||
echo "Queries with results: $RELEVANT"
|
|
||||||
echo "Precision: $PRECISION"
|
|
||||||
echo ""
|
|
||||||
echo "=== Health check ==="
|
|
||||||
curl -s "$MEMORY_URL/health"
|
|
||||||
echo ""
|
|
||||||
echo "=== Metrics snapshot ==="
|
|
||||||
curl -s "$MEMORY_URL/metrics" | grep -E "^memory_(query|relevance|ingest)_" | head -20
|
|
||||||
|
|
||||||
resources:
|
|
||||||
requests:
|
|
||||||
cpu: 10m
|
|
||||||
memory: 16Mi
|
|
||||||
limits:
|
|
||||||
cpu: 50m
|
|
||||||
memory: 32Mi
|
|
||||||
|
|
||||||
restartPolicy: OnFailure
|
|
||||||
@@ -1,121 +0,0 @@
|
|||||||
# CronJob to periodically clean Gitea Actions runner disk space
|
|
||||||
# Prevents "no space left on device" errors during Docker builds
|
|
||||||
# Deploy to: kubectl apply -f k8s/infra/runner-cleanup-cronjob.yaml
|
|
||||||
|
|
||||||
apiVersion: batch/v1
|
|
||||||
kind: CronJob
|
|
||||||
metadata:
|
|
||||||
name: runner-disk-cleanup
|
|
||||||
namespace: ci # Adjust to your runner namespace
|
|
||||||
labels:
|
|
||||||
app: runner-cleanup
|
|
||||||
spec:
|
|
||||||
# Run daily at 2 AM
|
|
||||||
schedule: "0 2 * * *"
|
|
||||||
|
|
||||||
# Keep last 3 successful jobs
|
|
||||||
successfulJobsHistoryLimit: 3
|
|
||||||
failedJobsHistoryLimit: 1
|
|
||||||
|
|
||||||
jobTemplate:
|
|
||||||
spec:
|
|
||||||
template:
|
|
||||||
metadata:
|
|
||||||
labels:
|
|
||||||
app: runner-cleanup
|
|
||||||
spec:
|
|
||||||
serviceAccountName: runner-cleanup
|
|
||||||
|
|
||||||
# Run on node with Gitea Actions runner
|
|
||||||
affinity:
|
|
||||||
nodeAffinity:
|
|
||||||
requiredDuringSchedulingIgnoredDuringExecution:
|
|
||||||
nodeSelectorTerms:
|
|
||||||
- matchExpressions:
|
|
||||||
- key: kubernetes.io/hostname
|
|
||||||
operator: In
|
|
||||||
values:
|
|
||||||
- runner-node # Adjust to your runner node name
|
|
||||||
|
|
||||||
containers:
|
|
||||||
- name: cleanup
|
|
||||||
image: docker:24
|
|
||||||
securityContext:
|
|
||||||
privileged: true # Needed to access Docker daemon
|
|
||||||
command:
|
|
||||||
- /bin/sh
|
|
||||||
- -c
|
|
||||||
- |
|
|
||||||
echo "=== Runner disk cleanup at $(date) ==="
|
|
||||||
|
|
||||||
df -h /
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
echo "Cleaning Docker..."
|
|
||||||
docker system prune -af --volumes 2>&1 | tail -5
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "Cleaning Cargo cache..."
|
|
||||||
rm -rf /root/.cargo/registry/cache 2>/dev/null
|
|
||||||
rm -rf /root/.cargo/registry/index 2>/dev/null
|
|
||||||
rm -rf /root/.cargo/git 2>/dev/null
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "Cleaning /tmp..."
|
|
||||||
rm -rf /tmp/* 2>/dev/null
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "Disk after cleanup:"
|
|
||||||
df -h /
|
|
||||||
|
|
||||||
volumeMounts:
|
|
||||||
- name: docker-sock
|
|
||||||
mountPath: /var/run/docker.sock
|
|
||||||
- name: runner-home
|
|
||||||
mountPath: /root
|
|
||||||
|
|
||||||
volumes:
|
|
||||||
# Access Docker daemon on host
|
|
||||||
- name: docker-sock
|
|
||||||
hostPath:
|
|
||||||
path: /var/run/docker.sock
|
|
||||||
# Access runner home directory
|
|
||||||
- name: runner-home
|
|
||||||
hostPath:
|
|
||||||
path: /home/runner # Adjust to your runner home path
|
|
||||||
|
|
||||||
restartPolicy: OnFailure
|
|
||||||
|
|
||||||
---
|
|
||||||
# ServiceAccount for cleanup job
|
|
||||||
apiVersion: v1
|
|
||||||
kind: ServiceAccount
|
|
||||||
metadata:
|
|
||||||
name: runner-cleanup
|
|
||||||
namespace: ci
|
|
||||||
|
|
||||||
---
|
|
||||||
# Role for cleanup job
|
|
||||||
apiVersion: rbac.authorization.k8s.io/v1
|
|
||||||
kind: ClusterRole
|
|
||||||
metadata:
|
|
||||||
name: runner-cleanup
|
|
||||||
rules:
|
|
||||||
- apiGroups: [""]
|
|
||||||
resources: ["nodes"]
|
|
||||||
verbs: ["get", "list"]
|
|
||||||
|
|
||||||
---
|
|
||||||
# RoleBinding
|
|
||||||
apiVersion: rbac.authorization.k8s.io/v1
|
|
||||||
kind: ClusterRoleBinding
|
|
||||||
metadata:
|
|
||||||
name: runner-cleanup
|
|
||||||
roleRef:
|
|
||||||
apiGroup: rbac.authorization.k8s.io
|
|
||||||
kind: ClusterRole
|
|
||||||
name: runner-cleanup
|
|
||||||
subjects:
|
|
||||||
- kind: ServiceAccount
|
|
||||||
name: runner-cleanup
|
|
||||||
namespace: ci
|
|
||||||
@@ -1,131 +0,0 @@
|
|||||||
apiVersion: tekton.dev/v1
|
|
||||||
kind: Task
|
|
||||||
metadata:
|
|
||||||
name: agent-memory-migration
|
|
||||||
namespace: tekton-pipelines
|
|
||||||
spec:
|
|
||||||
description: Apply agent memory schema migration (004) to production database
|
|
||||||
params:
|
|
||||||
- name: migration-version
|
|
||||||
description: Migration version number
|
|
||||||
default: "004"
|
|
||||||
- name: database-name
|
|
||||||
description: Database name
|
|
||||||
default: "memory"
|
|
||||||
workspaces:
|
|
||||||
- name: source
|
|
||||||
description: Git source with migrations
|
|
||||||
- name: db-credentials
|
|
||||||
description: Database credentials secret
|
|
||||||
steps:
|
|
||||||
- name: apply-migration
|
|
||||||
image: postgres:16-alpine
|
|
||||||
workingDir: $(workspaces.source.path)
|
|
||||||
env:
|
|
||||||
- name: PGPASSWORD
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-db-app
|
|
||||||
key: password
|
|
||||||
- name: PGHOST
|
|
||||||
value: memory-db-rw.poimen.svc.cluster.local
|
|
||||||
- name: PGUSER
|
|
||||||
value: app
|
|
||||||
- name: PGDATABASE
|
|
||||||
value: $(params.database-name)
|
|
||||||
script: |
|
|
||||||
#!/bin/sh
|
|
||||||
set -e
|
|
||||||
|
|
||||||
echo "Applying migration $(params.migration-version)_agent_memory_schema.sql"
|
|
||||||
|
|
||||||
# Wait for database to be ready
|
|
||||||
until pg_isready -h $PGHOST -U $PGUSER -d $PGDATABASE; do
|
|
||||||
echo "Waiting for database..."
|
|
||||||
sleep 2
|
|
||||||
done
|
|
||||||
|
|
||||||
# Apply migration
|
|
||||||
psql -h $PGHOST -U $PGUSER -d $PGDATABASE \
|
|
||||||
-f migrations/$(params.migration-version)_agent_memory_schema.sql
|
|
||||||
|
|
||||||
# Verify tables created
|
|
||||||
TABLES=$(psql -h $PGHOST -U $PGUSER -d $PGDATABASE -t -c \
|
|
||||||
"SELECT count(*) FROM information_schema.tables WHERE table_schema='public' AND table_name IN ('agent_prompt', 'agent_skill', 'agent_decision', 'role_prompt_mapping', 'agent_metrics')")
|
|
||||||
|
|
||||||
if [ "$TABLES" -eq 5 ]; then
|
|
||||||
echo "✓ All agent memory tables created successfully"
|
|
||||||
exit 0
|
|
||||||
else
|
|
||||||
echo "✗ Migration failed: expected 5 tables, found $TABLES"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: verify-indexes
|
|
||||||
image: postgres:16-alpine
|
|
||||||
env:
|
|
||||||
- name: PGPASSWORD
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-db-app
|
|
||||||
key: password
|
|
||||||
- name: PGHOST
|
|
||||||
value: memory-db-rw.poimen.svc.cluster.local
|
|
||||||
- name: PGUSER
|
|
||||||
value: app
|
|
||||||
- name: PGDATABASE
|
|
||||||
value: $(params.database-name)
|
|
||||||
script: |
|
|
||||||
#!/bin/sh
|
|
||||||
set -e
|
|
||||||
|
|
||||||
echo "Verifying indexes..."
|
|
||||||
|
|
||||||
INDEXES=$(psql -h $PGHOST -U $PGUSER -d $PGDATABASE -t -c \
|
|
||||||
"SELECT count(*) FROM pg_indexes WHERE schemaname='public' AND tablename LIKE 'agent_%'")
|
|
||||||
|
|
||||||
if [ "$INDEXES" -gt 0 ]; then
|
|
||||||
echo "✓ Found $INDEXES indexes on agent tables"
|
|
||||||
psql -h $PGHOST -U $PGUSER -d $PGDATABASE -c \
|
|
||||||
"SELECT indexname FROM pg_indexes WHERE schemaname='public' AND tablename LIKE 'agent_%' ORDER BY indexname;"
|
|
||||||
else
|
|
||||||
echo "✗ No indexes found on agent tables"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: verify-schemas
|
|
||||||
image: postgres:16-alpine
|
|
||||||
env:
|
|
||||||
- name: PGPASSWORD
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-db-app
|
|
||||||
key: password
|
|
||||||
- name: PGHOST
|
|
||||||
value: memory-db-rw.poimen.svc.cluster.local
|
|
||||||
- name: PGUSER
|
|
||||||
value: app
|
|
||||||
- name: PGDATABASE
|
|
||||||
value: $(params.database-name)
|
|
||||||
script: |
|
|
||||||
#!/bin/sh
|
|
||||||
set -e
|
|
||||||
|
|
||||||
echo "Verifying table schemas..."
|
|
||||||
|
|
||||||
# Verify agent_prompt table
|
|
||||||
psql -h $PGHOST -U $PGUSER -d $PGDATABASE -c "
|
|
||||||
SELECT column_name, data_type, is_nullable
|
|
||||||
FROM information_schema.columns
|
|
||||||
WHERE table_name='agent_prompt'
|
|
||||||
ORDER BY ordinal_position;"
|
|
||||||
|
|
||||||
echo "✓ Agent prompt schema verified"
|
|
||||||
|
|
||||||
# Verify role_prompt_mapping has foreign key
|
|
||||||
psql -h $PGHOST -U $PGUSER -d $PGDATABASE -c "
|
|
||||||
SELECT constraint_name, constraint_type
|
|
||||||
FROM information_schema.table_constraints
|
|
||||||
WHERE table_name='role_prompt_mapping';"
|
|
||||||
|
|
||||||
echo "✓ All table schemas verified"
|
|
||||||
@@ -1,76 +0,0 @@
|
|||||||
---
|
|
||||||
# PipelineRun: Agent Memory Feature Testing
|
|
||||||
# Tests role-to-prompt mapping with API Platform Engineer role requirements
|
|
||||||
# Runs migrations, integration tests, and validates all constraints
|
|
||||||
|
|
||||||
apiVersion: tekton.dev/v1
|
|
||||||
kind: PipelineRun
|
|
||||||
metadata:
|
|
||||||
name: agent-memory-test-run
|
|
||||||
namespace: poimen
|
|
||||||
generateName: agent-memory-test-
|
|
||||||
spec:
|
|
||||||
pipelineRef:
|
|
||||||
name: poimen-ci
|
|
||||||
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
value: "forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:latest"
|
|
||||||
- name: registry-user
|
|
||||||
value: "riotpiao-poimen"
|
|
||||||
- name: registry-token
|
|
||||||
value: "${FORGEJO_REGISTRY_TOKEN}" # Injected by ArgoCD/SOPS
|
|
||||||
|
|
||||||
workspaces:
|
|
||||||
- name: source
|
|
||||||
emptyDir: {} # Or use PVC for persistent builds
|
|
||||||
|
|
||||||
serviceAccountName: tekton-builder
|
|
||||||
|
|
||||||
timeouts:
|
|
||||||
pipeline: "1h"
|
|
||||||
tasks: "30m"
|
|
||||||
|
|
||||||
---
|
|
||||||
# ServiceAccount for Tekton Pipeline (builder with DB access)
|
|
||||||
apiVersion: v1
|
|
||||||
kind: ServiceAccount
|
|
||||||
metadata:
|
|
||||||
name: tekton-builder
|
|
||||||
namespace: poimen
|
|
||||||
|
|
||||||
---
|
|
||||||
# ClusterRoleBinding: Allow pipeline to query database via pod exec
|
|
||||||
apiVersion: rbac.authorization.k8s.io/v1
|
|
||||||
kind: ClusterRoleBinding
|
|
||||||
metadata:
|
|
||||||
name: tekton-builder-db-access
|
|
||||||
roleRef:
|
|
||||||
apiGroup: rbac.authorization.k8s.io
|
|
||||||
kind: ClusterRole
|
|
||||||
name: tekton-builder-db-access
|
|
||||||
subjects:
|
|
||||||
- kind: ServiceAccount
|
|
||||||
name: tekton-builder
|
|
||||||
namespace: poimen
|
|
||||||
|
|
||||||
---
|
|
||||||
# ClusterRole: Database access for migrations
|
|
||||||
apiVersion: rbac.authorization.k8s.io/v1
|
|
||||||
kind: ClusterRole
|
|
||||||
metadata:
|
|
||||||
name: tekton-builder-db-access
|
|
||||||
rules:
|
|
||||||
- apiGroups: [""]
|
|
||||||
resources: ["pods"]
|
|
||||||
verbs: ["get", "list"]
|
|
||||||
- apiGroups: [""]
|
|
||||||
resources: ["pods/exec"]
|
|
||||||
verbs: ["create"]
|
|
||||||
- apiGroups: [""]
|
|
||||||
resources: ["secrets"]
|
|
||||||
resourceNames: ["memory-db-app"]
|
|
||||||
verbs: ["get"]
|
|
||||||
- apiGroups: [""]
|
|
||||||
resources: ["services"]
|
|
||||||
verbs: ["get", "list"]
|
|
||||||
@@ -1,150 +0,0 @@
|
|||||||
---
|
|
||||||
# Tekton Task: Integration Tests for Poimen Memory Service
|
|
||||||
#
|
|
||||||
# Executes:
|
|
||||||
# 1. Database migrations
|
|
||||||
# 2. Integration test suites (cargo test)
|
|
||||||
# 3. Reports results
|
|
||||||
#
|
|
||||||
# Parameters:
|
|
||||||
# - image: Docker image with SHA to test
|
|
||||||
#
|
|
||||||
# Results:
|
|
||||||
# - summary: Test summary (pass/fail + count)
|
|
||||||
|
|
||||||
apiVersion: tekton.dev/v1
|
|
||||||
kind: Task
|
|
||||||
metadata:
|
|
||||||
name: poimen-integration-test
|
|
||||||
namespace: poimen
|
|
||||||
spec:
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
type: string
|
|
||||||
description: "Docker image SHA to test (e.g., forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:abc123)"
|
|
||||||
|
|
||||||
results:
|
|
||||||
- name: summary
|
|
||||||
description: "Test summary: PASS or FAIL + test count"
|
|
||||||
|
|
||||||
steps:
|
|
||||||
# Step 1: Apply database migrations
|
|
||||||
- name: migrate
|
|
||||||
image: $(params.image)
|
|
||||||
env:
|
|
||||||
- name: DB_HOST
|
|
||||||
value: "memory-db-rw.poimen.svc.cluster.local"
|
|
||||||
- name: DB_PORT
|
|
||||||
value: "5432"
|
|
||||||
- name: DB_NAME
|
|
||||||
value: "memory"
|
|
||||||
- name: DB_USER
|
|
||||||
value: "app"
|
|
||||||
- name: DB_PASSWORD
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-db-app
|
|
||||||
key: password
|
|
||||||
|
|
||||||
script: |
|
|
||||||
#!/bin/bash
|
|
||||||
set -e
|
|
||||||
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Step 1: Database Migrations"
|
|
||||||
echo "=========================================="
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Run migrations
|
|
||||||
/app/migrations/run_migrations.sh
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "✓ Migrations complete"
|
|
||||||
|
|
||||||
# Step 2: Run integration tests
|
|
||||||
- name: test
|
|
||||||
image: $(params.image)
|
|
||||||
env:
|
|
||||||
- name: DATABASE_URL
|
|
||||||
value: "postgresql://[email protected]:5432/memory"
|
|
||||||
- name: RUST_LOG
|
|
||||||
value: "info,mem_cli=debug,mem_ingest=debug,mem_store=debug"
|
|
||||||
- name: MEM_AUTH_MODE
|
|
||||||
value: "none"
|
|
||||||
- name: SQLX_OFFLINE
|
|
||||||
value: "true"
|
|
||||||
- name: PGPASSWORD
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef:
|
|
||||||
name: memory-db-app
|
|
||||||
key: password
|
|
||||||
|
|
||||||
script: |
|
|
||||||
#!/bin/bash
|
|
||||||
set -e
|
|
||||||
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Step 2: Integration Tests"
|
|
||||||
echo "=========================================="
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
TEST_SUITES=(
|
|
||||||
"it_phase3_phase4"
|
|
||||||
"it_unified_query_4_6"
|
|
||||||
"it_temporal_filtering_4_2_fixed"
|
|
||||||
)
|
|
||||||
|
|
||||||
PASSED=0
|
|
||||||
FAILED=0
|
|
||||||
|
|
||||||
for suite in "${TEST_SUITES[@]}"; do
|
|
||||||
echo "Running: $suite"
|
|
||||||
if cargo test --test "$suite" --lib 2>&1 | tail -50; then
|
|
||||||
((PASSED++))
|
|
||||||
echo "✓ $suite passed"
|
|
||||||
else
|
|
||||||
((FAILED++))
|
|
||||||
echo "✗ $suite failed"
|
|
||||||
fi
|
|
||||||
echo ""
|
|
||||||
done
|
|
||||||
|
|
||||||
# Run unit tests
|
|
||||||
echo "Running unit tests..."
|
|
||||||
if cargo test --lib mem_ingest 2>&1 | tail -100; then
|
|
||||||
echo "✓ mem_ingest passed"
|
|
||||||
else
|
|
||||||
((FAILED++))
|
|
||||||
echo "✗ mem_ingest failed"
|
|
||||||
fi
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
if cargo test --lib mem_cli::query 2>&1 | tail -100; then
|
|
||||||
echo "✓ mem_cli::query passed"
|
|
||||||
else
|
|
||||||
((FAILED++))
|
|
||||||
echo "✗ mem_cli::query failed"
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Test Summary: $PASSED passed, $FAILED failed"
|
|
||||||
echo "=========================================="
|
|
||||||
|
|
||||||
if [ $FAILED -eq 0 ]; then
|
|
||||||
echo "PASS: All integration tests passed"
|
|
||||||
echo "PASS: All integration tests passed" > /tekton/results/summary
|
|
||||||
exit 0
|
|
||||||
else
|
|
||||||
echo "FAIL: $FAILED test suite(s) failed"
|
|
||||||
echo "FAIL: $FAILED test suite(s) failed" > /tekton/results/summary
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
resources:
|
|
||||||
requests:
|
|
||||||
memory: "1Gi"
|
|
||||||
cpu: "500m"
|
|
||||||
limits:
|
|
||||||
memory: "2Gi"
|
|
||||||
cpu: "2000m"
|
|
||||||
@@ -1,140 +0,0 @@
|
|||||||
---
|
|
||||||
# Tekton Pipeline: Poimen Memory Service CI/CD
|
|
||||||
#
|
|
||||||
# Orchestrates:
|
|
||||||
# 1. integration-test-task: Run integration tests against image
|
|
||||||
# 2. (Future) build-task: Build Docker image
|
|
||||||
# 3. (Future) promote-task: Promote image to :latest
|
|
||||||
#
|
|
||||||
# Parameters:
|
|
||||||
# - image: Docker image with SHA to test
|
|
||||||
# - registry-user: Registry credentials
|
|
||||||
# - registry-token: Registry credentials
|
|
||||||
|
|
||||||
apiVersion: tekton.dev/v1
|
|
||||||
kind: Pipeline
|
|
||||||
metadata:
|
|
||||||
name: poimen-ci
|
|
||||||
namespace: poimen
|
|
||||||
spec:
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
type: string
|
|
||||||
description: "Docker image SHA to test (e.g., forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:abc123)"
|
|
||||||
|
|
||||||
- name: registry-user
|
|
||||||
type: string
|
|
||||||
description: "Registry username"
|
|
||||||
default: ""
|
|
||||||
|
|
||||||
- name: registry-token
|
|
||||||
type: string
|
|
||||||
description: "Registry token/password"
|
|
||||||
default: ""
|
|
||||||
|
|
||||||
workspaces:
|
|
||||||
- name: source
|
|
||||||
description: "Git source repository with migrations"
|
|
||||||
|
|
||||||
tasks:
|
|
||||||
# Task 0: Apply Agent Memory Migrations
|
|
||||||
- name: agent-memory-migration
|
|
||||||
taskRef:
|
|
||||||
name: agent-memory-migration
|
|
||||||
params:
|
|
||||||
- name: migration-version
|
|
||||||
value: "004"
|
|
||||||
- name: database-name
|
|
||||||
value: "memory"
|
|
||||||
workspaces:
|
|
||||||
- name: source
|
|
||||||
workspace: source
|
|
||||||
|
|
||||||
# Task 1: Integration Tests (runs after migration)
|
|
||||||
- name: integration-tests
|
|
||||||
runAfter:
|
|
||||||
- agent-memory-migration
|
|
||||||
taskRef:
|
|
||||||
name: poimen-integration-test
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
value: $(params.image)
|
|
||||||
|
|
||||||
# Task 2: Gate on test results
|
|
||||||
- name: gate-on-tests
|
|
||||||
runAfter:
|
|
||||||
- integration-tests
|
|
||||||
taskSpec:
|
|
||||||
steps:
|
|
||||||
- name: check-results
|
|
||||||
image: alpine:latest
|
|
||||||
script: |
|
|
||||||
#!/bin/sh
|
|
||||||
set -e
|
|
||||||
echo "✓ Integration tests passed, proceeding with promotion"
|
|
||||||
|
|
||||||
# Task 3: Promote image (placeholder - will be implemented)
|
|
||||||
- name: promote-image
|
|
||||||
runAfter:
|
|
||||||
- gate-on-tests
|
|
||||||
taskSpec:
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
type: string
|
|
||||||
- name: registry-user
|
|
||||||
type: string
|
|
||||||
- name: registry-token
|
|
||||||
type: string
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- name: promote
|
|
||||||
image: docker:latest
|
|
||||||
env:
|
|
||||||
- name: IMAGE
|
|
||||||
value: $(params.image)
|
|
||||||
- name: REGISTRY_USER
|
|
||||||
value: $(params.registry-user)
|
|
||||||
- name: REGISTRY_TOKEN
|
|
||||||
value: $(params.registry-token)
|
|
||||||
script: |
|
|
||||||
#!/bin/sh
|
|
||||||
set -e
|
|
||||||
|
|
||||||
echo "Promoting image to :latest..."
|
|
||||||
|
|
||||||
# Extract registry and repo from image
|
|
||||||
# e.g., forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:abc123
|
|
||||||
REGISTRY=$(echo $IMAGE | cut -d/ -f1)
|
|
||||||
REPO=$(echo $IMAGE | cut -d: -f1)
|
|
||||||
SHA=$(echo $IMAGE | cut -d: -f2)
|
|
||||||
|
|
||||||
echo "Registry: $REGISTRY"
|
|
||||||
echo "Repo: $REPO"
|
|
||||||
echo "SHA: $SHA"
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Login and promote
|
|
||||||
echo "$REGISTRY_TOKEN" | docker login -u "$REGISTRY_USER" --password-stdin "$REGISTRY"
|
|
||||||
docker pull "$IMAGE"
|
|
||||||
docker tag "$IMAGE" "${REPO}:latest"
|
|
||||||
docker push "${REPO}:latest"
|
|
||||||
|
|
||||||
echo "✓ Promoted to :latest"
|
|
||||||
|
|
||||||
params:
|
|
||||||
- name: image
|
|
||||||
value: $(params.image)
|
|
||||||
- name: registry-user
|
|
||||||
value: $(params.registry-user)
|
|
||||||
- name: registry-token
|
|
||||||
value: $(params.registry-token)
|
|
||||||
|
|
||||||
finally:
|
|
||||||
- name: cleanup
|
|
||||||
taskSpec:
|
|
||||||
steps:
|
|
||||||
- name: cleanup-tasks
|
|
||||||
image: alpine:latest
|
|
||||||
script: |
|
|
||||||
#!/bin/sh
|
|
||||||
echo "Pipeline execution complete"
|
|
||||||
@@ -1,142 +0,0 @@
|
|||||||
-- Agent Memory Schema (Phase 6)
|
|
||||||
-- Stores agent prompts, skills, and decisions with role-to-prompt mapping
|
|
||||||
-- Follows API Platform Engineer contract-first design (agency-agents role)
|
|
||||||
|
|
||||||
CREATE TABLE IF NOT EXISTS agent_prompt (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
name VARCHAR(512) NOT NULL,
|
|
||||||
template TEXT NOT NULL,
|
|
||||||
target_model VARCHAR(128),
|
|
||||||
task_category VARCHAR(128) NOT NULL,
|
|
||||||
usage_count BIGINT DEFAULT 0,
|
|
||||||
avg_quality FLOAT DEFAULT 0.0,
|
|
||||||
last_used TIMESTAMP WITH TIME ZONE,
|
|
||||||
active BOOLEAN DEFAULT true,
|
|
||||||
version INTEGER DEFAULT 1,
|
|
||||||
tags TEXT[] DEFAULT '{}',
|
|
||||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
updated_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
UNIQUE(project_id, name, version)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_agent_prompt_project_active ON agent_prompt(project_id, active);
|
|
||||||
CREATE INDEX idx_agent_prompt_task_category ON agent_prompt(task_category);
|
|
||||||
CREATE INDEX idx_agent_prompt_tags ON agent_prompt USING GIN(tags);
|
|
||||||
|
|
||||||
-- Agent Skill: linked capabilities with effectiveness tracking
|
|
||||||
CREATE TABLE IF NOT EXISTS agent_skill (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
agent_id VARCHAR(255) NOT NULL,
|
|
||||||
name VARCHAR(512) NOT NULL,
|
|
||||||
description TEXT NOT NULL,
|
|
||||||
trigger_patterns TEXT[] DEFAULT '{}',
|
|
||||||
success_rate FLOAT DEFAULT 0.0,
|
|
||||||
invocation_count BIGINT DEFAULT 0,
|
|
||||||
avg_latency_ms BIGINT DEFAULT 0,
|
|
||||||
linked_prompts UUID[] DEFAULT '{}',
|
|
||||||
enabled BOOLEAN DEFAULT true,
|
|
||||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
updated_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
UNIQUE(project_id, agent_id, name)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_agent_skill_agent ON agent_skill(project_id, agent_id);
|
|
||||||
CREATE INDEX idx_agent_skill_enabled ON agent_skill(enabled);
|
|
||||||
CREATE INDEX idx_agent_skill_linked_prompts ON agent_skill USING GIN(linked_prompts);
|
|
||||||
|
|
||||||
-- Agent Decision: reasoning and outcome tracking
|
|
||||||
CREATE TABLE IF NOT EXISTS agent_decision (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
agent_id VARCHAR(255) NOT NULL,
|
|
||||||
action VARCHAR(512) NOT NULL,
|
|
||||||
reasoning TEXT NOT NULL,
|
|
||||||
alternatives TEXT[] DEFAULT '{}',
|
|
||||||
confidence FLOAT DEFAULT 0.0,
|
|
||||||
context_entities UUID[] DEFAULT '{}',
|
|
||||||
tool VARCHAR(255),
|
|
||||||
task VARCHAR(255),
|
|
||||||
outcome_success BOOLEAN,
|
|
||||||
outcome_quality FLOAT,
|
|
||||||
outcome_feedback TEXT,
|
|
||||||
outcome_recorded_at TIMESTAMP WITH TIME ZONE,
|
|
||||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
updated_at TIMESTAMP WITH TIME ZONE DEFAULT NOW()
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_agent_decision_agent ON agent_decision(project_id, agent_id);
|
|
||||||
CREATE INDEX idx_agent_decision_action ON agent_decision(action);
|
|
||||||
CREATE INDEX idx_agent_decision_context ON agent_decision USING GIN(context_entities);
|
|
||||||
|
|
||||||
-- Agent Registration: lifecycle management
|
|
||||||
CREATE TABLE IF NOT EXISTS agent_registry (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
agent_id VARCHAR(255) NOT NULL,
|
|
||||||
capabilities TEXT[] NOT NULL,
|
|
||||||
webhook_url VARCHAR(2048),
|
|
||||||
rate_limit INTEGER DEFAULT 1000,
|
|
||||||
metadata JSONB DEFAULT '{}',
|
|
||||||
status VARCHAR(32) DEFAULT 'active',
|
|
||||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
updated_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
UNIQUE(project_id, agent_id)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_agent_registry_project ON agent_registry(project_id);
|
|
||||||
CREATE INDEX idx_agent_registry_status ON agent_registry(status);
|
|
||||||
|
|
||||||
-- Role-to-Prompt Mapping: maps agent roles to prompt templates
|
|
||||||
CREATE TABLE IF NOT EXISTS role_prompt_mapping (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
role_name VARCHAR(255) NOT NULL,
|
|
||||||
prompt_id UUID NOT NULL REFERENCES agent_prompt(id) ON DELETE CASCADE,
|
|
||||||
priority INTEGER DEFAULT 0,
|
|
||||||
active BOOLEAN DEFAULT true,
|
|
||||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
updated_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
UNIQUE(project_id, role_name, prompt_id),
|
|
||||||
FOREIGN KEY (project_id) REFERENCES projects(id) ON DELETE CASCADE
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_role_prompt_mapping_role ON role_prompt_mapping(project_id, role_name, active);
|
|
||||||
CREATE INDEX idx_role_prompt_mapping_prompt ON role_prompt_mapping(prompt_id);
|
|
||||||
|
|
||||||
-- Agent Metrics: performance tracking
|
|
||||||
CREATE TABLE IF NOT EXISTS agent_metrics (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
agent_id VARCHAR(255) NOT NULL,
|
|
||||||
requests_total BIGINT DEFAULT 0,
|
|
||||||
requests_success BIGINT DEFAULT 0,
|
|
||||||
requests_failed BIGINT DEFAULT 0,
|
|
||||||
average_latency_ms FLOAT DEFAULT 0.0,
|
|
||||||
p95_latency_ms FLOAT DEFAULT 0.0,
|
|
||||||
p99_latency_ms FLOAT DEFAULT 0.0,
|
|
||||||
error_rate FLOAT DEFAULT 0.0,
|
|
||||||
recorded_at TIMESTAMP WITH TIME ZONE DEFAULT NOW(),
|
|
||||||
UNIQUE(project_id, agent_id, DATE(recorded_at))
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_agent_metrics_agent ON agent_metrics(project_id, agent_id, recorded_at DESC);
|
|
||||||
|
|
||||||
-- Prompt Usage Log: detailed invocation tracking
|
|
||||||
CREATE TABLE IF NOT EXISTS prompt_usage_log (
|
|
||||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
||||||
project_id VARCHAR(255) NOT NULL,
|
|
||||||
prompt_id UUID NOT NULL REFERENCES agent_prompt(id) ON DELETE CASCADE,
|
|
||||||
agent_id VARCHAR(255),
|
|
||||||
model_used VARCHAR(128),
|
|
||||||
input_tokens INTEGER,
|
|
||||||
output_tokens INTEGER,
|
|
||||||
quality_score FLOAT,
|
|
||||||
duration_ms BIGINT,
|
|
||||||
error_message TEXT,
|
|
||||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT NOW()
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_prompt_usage_log_prompt ON prompt_usage_log(prompt_id, created_at DESC);
|
|
||||||
CREATE INDEX idx_prompt_usage_log_agent ON prompt_usage_log(agent_id, created_at DESC);
|
|
||||||
@@ -1,127 +0,0 @@
|
|||||||
#!/bin/bash
|
|
||||||
|
|
||||||
# Database Migration Runner
|
|
||||||
# Used by K8s Job to apply all migrations before integration tests
|
|
||||||
#
|
|
||||||
# Environment variables (from K8s):
|
|
||||||
# DB_HOST - PostgreSQL host
|
|
||||||
# DB_PORT - PostgreSQL port
|
|
||||||
# DB_NAME - Database name
|
|
||||||
# DB_USER - Database user
|
|
||||||
# DB_PASSWORD - Database password (from Secret)
|
|
||||||
|
|
||||||
set -e
|
|
||||||
|
|
||||||
DB_HOST="${DB_HOST:-memory-db-rw.poimen.svc.cluster.local}"
|
|
||||||
DB_PORT="${DB_PORT:-5432}"
|
|
||||||
DB_NAME="${DB_NAME:-memory}"
|
|
||||||
DB_USER="${DB_USER:-app}"
|
|
||||||
|
|
||||||
if [ -z "$DB_PASSWORD" ]; then
|
|
||||||
echo "ERROR: DB_PASSWORD not set"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Database Migration Runner"
|
|
||||||
echo "=========================================="
|
|
||||||
echo ""
|
|
||||||
echo "Configuration:"
|
|
||||||
echo " Host: $DB_HOST:$DB_PORT"
|
|
||||||
echo " Database: $DB_NAME"
|
|
||||||
echo " User: $DB_USER"
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Export for psql
|
|
||||||
export PGPASSWORD="$DB_PASSWORD"
|
|
||||||
|
|
||||||
# Get migration directory (where this script is)
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
||||||
MIGRATION_DIR="$SCRIPT_DIR"
|
|
||||||
|
|
||||||
echo "Migration directory: $MIGRATION_DIR"
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Collect all SQL files
|
|
||||||
MIGRATIONS=($(ls -1 "$MIGRATION_DIR"/*.sql 2>/dev/null | sort))
|
|
||||||
|
|
||||||
if [ ${#MIGRATIONS[@]} -eq 0 ]; then
|
|
||||||
echo "ERROR: No migration files found in $MIGRATION_DIR"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo "Found ${#MIGRATIONS[@]} migration(s):"
|
|
||||||
for m in "${MIGRATIONS[@]}"; do
|
|
||||||
echo " - $(basename $m)"
|
|
||||||
done
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Wait for DB to be ready
|
|
||||||
echo "Waiting for database to be ready..."
|
|
||||||
for i in {1..30}; do
|
|
||||||
if psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "SELECT 1;" >/dev/null 2>&1; then
|
|
||||||
echo "✓ Database is ready"
|
|
||||||
break
|
|
||||||
fi
|
|
||||||
if [ $i -eq 30 ]; then
|
|
||||||
echo "✗ Database not ready after 30 attempts"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
echo " Attempt $i/30..."
|
|
||||||
sleep 1
|
|
||||||
done
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Running Migrations"
|
|
||||||
echo "=========================================="
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
SUCCESS=0
|
|
||||||
FAILED=0
|
|
||||||
|
|
||||||
for migration in "${MIGRATIONS[@]}"; do
|
|
||||||
name=$(basename "$migration")
|
|
||||||
echo -n "▶ $name ... "
|
|
||||||
|
|
||||||
if psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -f "$migration" >/dev/null 2>&1; then
|
|
||||||
echo "✓"
|
|
||||||
((SUCCESS++))
|
|
||||||
else
|
|
||||||
echo "✗ FAILED"
|
|
||||||
echo ""
|
|
||||||
echo "Error output:"
|
|
||||||
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -f "$migration" 2>&1 | sed 's/^/ /'
|
|
||||||
((FAILED++))
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Migration Summary"
|
|
||||||
echo "=========================================="
|
|
||||||
echo " Success: $SUCCESS"
|
|
||||||
echo " Failed: $FAILED"
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
if [ $FAILED -eq 0 ]; then
|
|
||||||
echo "✓ All migrations applied successfully"
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "Verifying schema..."
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Verify key tables exist
|
|
||||||
for table in memory_entity memory_edge ingest_jobs; do
|
|
||||||
if psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "SELECT 1 FROM information_schema.tables WHERE table_name='$table';" 2>&1 | grep -q "1 row"; then
|
|
||||||
echo " ✓ Table $table exists"
|
|
||||||
else
|
|
||||||
echo " ⚠ Table $table not found"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
exit 0
|
|
||||||
else
|
|
||||||
echo "✗ Some migrations failed"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
@@ -1,608 +0,0 @@
|
|||||||
// Integration test: Agent Memory with API Platform Engineer role requirements
|
|
||||||
// Tests contract-first design per agency-agents/engineering/engineering-api-platform-engineer.md
|
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use serde_json::{json, Value};
|
|
||||||
|
|
||||||
// Test constants aligned with API Platform Engineer role
|
|
||||||
const API_VERSION: &str = "v1";
|
|
||||||
const PROJECT_ID: &str = "poimen";
|
|
||||||
const TEST_AGENT_ID: &str = "api-platform-engineer";
|
|
||||||
const API_PLATFORM_ENGINEER_ROLE: &str = "api-platform-engineer";
|
|
||||||
|
|
||||||
// API Platform Engineer role prompt templates
|
|
||||||
const CONTRACT_FIRST_PROMPT: &str = r#"
|
|
||||||
You are an API Platform Engineer designing a contract-first API.
|
|
||||||
|
|
||||||
Task: Review the following API specification for:
|
|
||||||
1. Naming consistency (pick snake_case or camelCase and never waver)
|
|
||||||
2. Backward compatibility (no breaking changes without versioning)
|
|
||||||
3. Error responses (consistent structure, stable codes, correct HTTP status semantics)
|
|
||||||
4. Rate limiting (communicated, not just enforced)
|
|
||||||
5. Documentation (SDKs and docs generated from spec, never drift)
|
|
||||||
|
|
||||||
Specification:
|
|
||||||
{{spec}}
|
|
||||||
|
|
||||||
Output JSON with:
|
|
||||||
{
|
|
||||||
"contract_valid": boolean,
|
|
||||||
"breaking_changes": [string],
|
|
||||||
"naming_inconsistencies": [string],
|
|
||||||
"error_issues": [string],
|
|
||||||
"rate_limit_issues": [string],
|
|
||||||
"recommendations": [string]
|
|
||||||
}
|
|
||||||
"#;
|
|
||||||
|
|
||||||
const BACKWARD_COMPATIBILITY_PROMPT: &str = r#"
|
|
||||||
You are an API versioning expert.
|
|
||||||
|
|
||||||
Analyze the proposed change:
|
|
||||||
{{change}}
|
|
||||||
|
|
||||||
Determine:
|
|
||||||
1. Is this a breaking change?
|
|
||||||
2. Does it require a new version?
|
|
||||||
3. What's the migration path?
|
|
||||||
4. What deprecation runway is needed?
|
|
||||||
|
|
||||||
Output JSON with:
|
|
||||||
{
|
|
||||||
"breaking": boolean,
|
|
||||||
"requires_new_version": boolean,
|
|
||||||
"migration_path": string,
|
|
||||||
"deprecation_runway_days": number,
|
|
||||||
"is_safe_additive": boolean
|
|
||||||
}
|
|
||||||
"#;
|
|
||||||
|
|
||||||
const SDK_GENERATION_PROMPT: &str = r#"
|
|
||||||
You are an SDK generation specialist.
|
|
||||||
|
|
||||||
Given this OpenAPI spec:
|
|
||||||
{{spec}}
|
|
||||||
|
|
||||||
Generate SDK requirements for:
|
|
||||||
1. Language: {{language}}
|
|
||||||
2. Idiomatic patterns for that language
|
|
||||||
3. Error handling
|
|
||||||
4. Retry logic and idempotency
|
|
||||||
5. Type safety
|
|
||||||
|
|
||||||
Output JSON with:
|
|
||||||
{
|
|
||||||
"sdk_structure": object,
|
|
||||||
"error_handling": string,
|
|
||||||
"idempotency_strategy": string,
|
|
||||||
"type_safety_level": string,
|
|
||||||
"generated_package_version": string
|
|
||||||
}
|
|
||||||
"#;
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_contract_first_api_specification() {
|
|
||||||
// Contract-first principle: OpenAPI spec is source of truth
|
|
||||||
let api_spec = json!({
|
|
||||||
"openapi": "3.0.0",
|
|
||||||
"info": {
|
|
||||||
"title": "Poimen Agent Memory API",
|
|
||||||
"version": API_VERSION,
|
|
||||||
"description": "Agent memory with role-to-prompt mapping"
|
|
||||||
},
|
|
||||||
"paths": {
|
|
||||||
"/memory/agents/{project_id}/prompts": {
|
|
||||||
"post": {
|
|
||||||
"operationId": "createPrompt",
|
|
||||||
"parameters": [
|
|
||||||
{
|
|
||||||
"name": "project_id",
|
|
||||||
"in": "path",
|
|
||||||
"required": true,
|
|
||||||
"schema": { "type": "string" }
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"requestBody": {
|
|
||||||
"required": true,
|
|
||||||
"content": {
|
|
||||||
"application/json": {
|
|
||||||
"schema": {
|
|
||||||
"type": "object",
|
|
||||||
"required": ["name", "template", "task_category"],
|
|
||||||
"properties": {
|
|
||||||
"name": { "type": "string", "minLength": 1 },
|
|
||||||
"template": { "type": "string", "description": "Prompt template with {{placeholders}}" },
|
|
||||||
"target_model": { "type": "string", "example": "ornith:35b" },
|
|
||||||
"task_category": { "type": "string", "enum": ["extraction", "reasoning", "summarization", "validation"] },
|
|
||||||
"tags": { "type": "array", "items": { "type": "string" } }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"responses": {
|
|
||||||
"201": {
|
|
||||||
"description": "Prompt created",
|
|
||||||
"content": {
|
|
||||||
"application/json": {
|
|
||||||
"schema": { "$ref": "#/components/schemas/Prompt" }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"400": { "$ref": "#/components/responses/BadRequest" },
|
|
||||||
"429": { "$ref": "#/components/responses/RateLimited" }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"/memory/agents/{project_id}/roles": {
|
|
||||||
"post": {
|
|
||||||
"operationId": "mapRoleToPrompt",
|
|
||||||
"requestBody": {
|
|
||||||
"required": true,
|
|
||||||
"content": {
|
|
||||||
"application/json": {
|
|
||||||
"schema": {
|
|
||||||
"type": "object",
|
|
||||||
"required": ["role_name", "prompt_id"],
|
|
||||||
"properties": {
|
|
||||||
"role_name": { "type": "string", "minLength": 1 },
|
|
||||||
"prompt_id": { "type": "string", "format": "uuid" },
|
|
||||||
"priority": { "type": "integer", "default": 0 }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"responses": {
|
|
||||||
"200": { "description": "Mapping created" },
|
|
||||||
"400": { "$ref": "#/components/responses/BadRequest" }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"/memory/agents/{project_id}/roles/{role_name}/prompts": {
|
|
||||||
"get": {
|
|
||||||
"operationId": "getRolePrompts",
|
|
||||||
"responses": {
|
|
||||||
"200": { "description": "List of prompts for role" },
|
|
||||||
"404": { "$ref": "#/components/responses/NotFound" }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"components": {
|
|
||||||
"schemas": {
|
|
||||||
"Prompt": {
|
|
||||||
"type": "object",
|
|
||||||
"required": ["id", "name", "template", "task_category"],
|
|
||||||
"properties": {
|
|
||||||
"id": { "type": "string", "format": "uuid" },
|
|
||||||
"name": { "type": "string" },
|
|
||||||
"template": { "type": "string" },
|
|
||||||
"target_model": { "type": "string", "nullable": true },
|
|
||||||
"task_category": { "type": "string" },
|
|
||||||
"usage_count": { "type": "integer" },
|
|
||||||
"avg_quality": { "type": "number", "format": "float" },
|
|
||||||
"version": { "type": "integer" },
|
|
||||||
"created_at": { "type": "string", "format": "date-time" }
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"Error": {
|
|
||||||
"type": "object",
|
|
||||||
"required": ["code", "message"],
|
|
||||||
"properties": {
|
|
||||||
"code": { "type": "string", "description": "Machine-readable error code" },
|
|
||||||
"message": { "type": "string", "description": "Human-readable error message" },
|
|
||||||
"details": { "type": "object", "description": "Field-level or contextual detail" },
|
|
||||||
"request_id": { "type": "string", "description": "Trace this to support" }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"responses": {
|
|
||||||
"BadRequest": {
|
|
||||||
"description": "Bad request",
|
|
||||||
"content": {
|
|
||||||
"application/json": { "schema": { "$ref": "#/components/schemas/Error" } }
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"NotFound": {
|
|
||||||
"description": "Resource not found",
|
|
||||||
"content": {
|
|
||||||
"application/json": { "schema": { "$ref": "#/components/schemas/Error" } }
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"RateLimited": {
|
|
||||||
"description": "Rate limited",
|
|
||||||
"headers": {
|
|
||||||
"Retry-After": { "schema": { "type": "integer" } },
|
|
||||||
"X-RateLimit-Limit": { "schema": { "type": "integer" } },
|
|
||||||
"X-RateLimit-Remaining": { "schema": { "type": "integer" } },
|
|
||||||
"X-RateLimit-Reset": { "schema": { "type": "integer" } }
|
|
||||||
},
|
|
||||||
"content": {
|
|
||||||
"application/json": { "schema": { "$ref": "#/components/schemas/Error" } }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
// Validate contract structure
|
|
||||||
assert_eq!(api_spec["openapi"], "3.0.0");
|
|
||||||
assert_eq!(api_spec["info"]["version"], API_VERSION);
|
|
||||||
|
|
||||||
// Validate error schema is consistent
|
|
||||||
let error_schema = &api_spec["components"]["schemas"]["Error"];
|
|
||||||
assert!(error_schema["required"]
|
|
||||||
.as_array()
|
|
||||||
.unwrap()
|
|
||||||
.contains(&Value::String("code".to_string())));
|
|
||||||
assert!(error_schema["required"]
|
|
||||||
.as_array()
|
|
||||||
.unwrap()
|
|
||||||
.contains(&Value::String("message".to_string())));
|
|
||||||
|
|
||||||
// Validate naming consistency (snake_case)
|
|
||||||
assert!(
|
|
||||||
api_spec["paths"]["/memory/agents/{project_id}/prompts"]["post"]["operationId"]
|
|
||||||
.as_str()
|
|
||||||
.unwrap()
|
|
||||||
.contains("createPrompt")
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
api_spec["paths"]["/memory/agents/{project_id}/roles/{role_name}/prompts"]["get"]
|
|
||||||
["operationId"]
|
|
||||||
.as_str()
|
|
||||||
.unwrap()
|
|
||||||
.contains("getRolePrompts")
|
|
||||||
);
|
|
||||||
|
|
||||||
// Validate backward compatibility: all fields are optional except required ones
|
|
||||||
let create_prompt_schema = &api_spec["paths"]["/memory/agents/{project_id}/prompts"]
|
|
||||||
["post"]["requestBody"]["content"]["application/json"]["schema"];
|
|
||||||
assert_eq!(
|
|
||||||
create_prompt_schema["required"].as_array().unwrap(),
|
|
||||||
&vec![
|
|
||||||
Value::String("name".to_string()),
|
|
||||||
Value::String("template".to_string()),
|
|
||||||
Value::String("task_category".to_string())
|
|
||||||
]
|
|
||||||
);
|
|
||||||
|
|
||||||
println!("✓ Contract-first API specification validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_backward_compatibility_rules() {
|
|
||||||
// Rule 1: Adding optional fields is safe
|
|
||||||
let safe_change = json!({
|
|
||||||
"type": "add_field",
|
|
||||||
"field": "metadata",
|
|
||||||
"required": false,
|
|
||||||
"breaking": false
|
|
||||||
});
|
|
||||||
assert!(!safe_change["breaking"].as_bool().unwrap());
|
|
||||||
|
|
||||||
// Rule 2: Removing fields is breaking
|
|
||||||
let breaking_change = json!({
|
|
||||||
"type": "remove_field",
|
|
||||||
"field": "template",
|
|
||||||
"breaking": true,
|
|
||||||
"requires_version_bump": true
|
|
||||||
});
|
|
||||||
assert!(breaking_change["breaking"].as_bool().unwrap());
|
|
||||||
assert!(breaking_change["requires_version_bump"].as_bool().unwrap());
|
|
||||||
|
|
||||||
// Rule 3: Adding new enum value is safe if clients tolerate unknowns
|
|
||||||
let safe_enum_addition = json!({
|
|
||||||
"type": "add_enum_value",
|
|
||||||
"enum": "task_category",
|
|
||||||
"new_value": "planning",
|
|
||||||
"breaking": false,
|
|
||||||
"requires_documentation": true
|
|
||||||
});
|
|
||||||
assert!(!safe_enum_addition["breaking"].as_bool().unwrap());
|
|
||||||
|
|
||||||
// Rule 4: Changing field type is breaking
|
|
||||||
let breaking_type_change = json!({
|
|
||||||
"type": "change_field_type",
|
|
||||||
"field": "usage_count",
|
|
||||||
"old_type": "integer",
|
|
||||||
"new_type": "string",
|
|
||||||
"breaking": true,
|
|
||||||
"requires_version_bump": true,
|
|
||||||
"migration_path": "Convert all consumers to parse as string"
|
|
||||||
});
|
|
||||||
assert!(breaking_type_change["breaking"].as_bool().unwrap());
|
|
||||||
|
|
||||||
println!("✓ Backward compatibility rules validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_rate_limiting_communication() {
|
|
||||||
// Rate limits must be communicated in response headers
|
|
||||||
let response_headers = json!({
|
|
||||||
"X-RateLimit-Limit": 1000,
|
|
||||||
"X-RateLimit-Remaining": 847,
|
|
||||||
"X-RateLimit-Reset": 1720483200,
|
|
||||||
"Retry-After": 30
|
|
||||||
});
|
|
||||||
|
|
||||||
// All required rate limit headers present
|
|
||||||
assert!(response_headers.get("X-RateLimit-Limit").is_some());
|
|
||||||
assert!(response_headers.get("X-RateLimit-Remaining").is_some());
|
|
||||||
assert!(response_headers.get("X-RateLimit-Reset").is_some());
|
|
||||||
|
|
||||||
// On 429, Retry-After present
|
|
||||||
let rate_limited_response = json!({
|
|
||||||
"status": 429,
|
|
||||||
"error": {
|
|
||||||
"code": "rate_limit_exceeded",
|
|
||||||
"message": "1000 req/hr exceeded; retry after 30s",
|
|
||||||
"request_id": "req_a1b2"
|
|
||||||
},
|
|
||||||
"headers": {
|
|
||||||
"Retry-After": 30
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
assert_eq!(rate_limited_response["status"], 429);
|
|
||||||
assert_eq!(
|
|
||||||
rate_limited_response["error"]["code"],
|
|
||||||
"rate_limit_exceeded"
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
rate_limited_response["headers"]["Retry-After"]
|
|
||||||
.as_i64()
|
|
||||||
.unwrap()
|
|
||||||
> 0
|
|
||||||
);
|
|
||||||
|
|
||||||
println!("✓ Rate limiting communication validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_error_response_consistency() {
|
|
||||||
// Error responses must have consistent structure everywhere
|
|
||||||
let errors = vec![
|
|
||||||
json!({
|
|
||||||
"code": "invalid_request",
|
|
||||||
"message": "name field required",
|
|
||||||
"details": { "field": "name" },
|
|
||||||
"request_id": "req-123"
|
|
||||||
}),
|
|
||||||
json!({
|
|
||||||
"code": "not_found",
|
|
||||||
"message": "Prompt not found",
|
|
||||||
"details": { "prompt_id": "uuid-456" },
|
|
||||||
"request_id": "req-789"
|
|
||||||
}),
|
|
||||||
json!({
|
|
||||||
"code": "permission_denied",
|
|
||||||
"message": "Insufficient capabilities",
|
|
||||||
"details": { "required": "memory:write" },
|
|
||||||
"request_id": "req-999"
|
|
||||||
}),
|
|
||||||
];
|
|
||||||
|
|
||||||
for error in errors {
|
|
||||||
// All errors have required structure
|
|
||||||
assert!(error["code"].is_string());
|
|
||||||
assert!(error["message"].is_string());
|
|
||||||
assert!(error["request_id"].is_string());
|
|
||||||
|
|
||||||
// No 200 with error (must use proper HTTP status)
|
|
||||||
assert_ne!(error["code"], ""); // code is stable, machine-readable
|
|
||||||
}
|
|
||||||
|
|
||||||
println!("✓ Error response consistency validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_deprecation_lifecycle() {
|
|
||||||
// Deprecation requires: Announce → Signal → Runway → Monitor → Sunset
|
|
||||||
let deprecation_plan = json!({
|
|
||||||
"endpoint": "/agents/{id}",
|
|
||||||
"lifecycle": {
|
|
||||||
"phase": "announced",
|
|
||||||
"deprecation_date": "2025-06-01",
|
|
||||||
"sunset_date": "2026-06-01",
|
|
||||||
"runway_days": 365
|
|
||||||
},
|
|
||||||
"signals": {
|
|
||||||
"deprecation_header": "Deprecation: true",
|
|
||||||
"sunset_header": "Sunset: Sun, 01 Jun 2026 00:00:00 GMT",
|
|
||||||
"warning_in_response": true
|
|
||||||
},
|
|
||||||
"migration_guide": "Use /agents/v2/{id} instead",
|
|
||||||
"monitoring": {
|
|
||||||
"track_usage_by_consumer": true,
|
|
||||||
"alert_on_remaining_usage": true
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
assert_eq!(deprecation_plan["lifecycle"]["runway_days"], 365);
|
|
||||||
assert!(deprecation_plan["signals"]["deprecation_header"]
|
|
||||||
.as_str()
|
|
||||||
.unwrap()
|
|
||||||
.contains("Deprecation"));
|
|
||||||
assert!(deprecation_plan["monitoring"]["track_usage_by_consumer"]
|
|
||||||
.as_bool()
|
|
||||||
.unwrap());
|
|
||||||
|
|
||||||
println!("✓ Deprecation lifecycle validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_idempotency_and_retry_safety() {
|
|
||||||
// Write operations must be idempotent via Idempotency-Key
|
|
||||||
let request_with_key = json!({
|
|
||||||
"method": "POST",
|
|
||||||
"path": "/memory/agents/project1/prompts",
|
|
||||||
"headers": {
|
|
||||||
"Idempotency-Key": "req-unique-uuid-123"
|
|
||||||
},
|
|
||||||
"body": {
|
|
||||||
"name": "extract-entities",
|
|
||||||
"template": "Extract entities from {{text}}"
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
assert!(request_with_key["headers"]["Idempotency-Key"].is_string());
|
|
||||||
|
|
||||||
// Retry with same key returns cached response
|
|
||||||
let response_1 = json!({
|
|
||||||
"status": 201,
|
|
||||||
"id": "prompt-uuid-456"
|
|
||||||
});
|
|
||||||
|
|
||||||
let response_2_retry = json!({
|
|
||||||
"status": 201,
|
|
||||||
"id": "prompt-uuid-456",
|
|
||||||
"cached": true
|
|
||||||
});
|
|
||||||
|
|
||||||
// Both return same result → safe to retry
|
|
||||||
assert_eq!(response_1["id"], response_2_retry["id"]);
|
|
||||||
|
|
||||||
println!("✓ Idempotency and retry safety validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_api_platform_engineer_role_requirements() {
|
|
||||||
// Comprehensive validation per api-platform-engineer.md role
|
|
||||||
let role_requirements = json!({
|
|
||||||
"role": API_PLATFORM_ENGINEER_ROLE,
|
|
||||||
"requirements": {
|
|
||||||
"contract_first": {
|
|
||||||
"openapi_spec": "required",
|
|
||||||
"source_of_truth_before_code": true,
|
|
||||||
"consistency_reviewed": true
|
|
||||||
},
|
|
||||||
"backward_compatibility": {
|
|
||||||
"no_silent_breaking_changes": true,
|
|
||||||
"additive_changes_allowed": true,
|
|
||||||
"versioning_policy": "major version in path (/v1, /v2)",
|
|
||||||
"deprecation_runway": "6-12+ months"
|
|
||||||
},
|
|
||||||
"error_handling": {
|
|
||||||
"consistent_structure": true,
|
|
||||||
"stable_machine_readable_code": true,
|
|
||||||
"correct_http_status": true,
|
|
||||||
"request_id_for_tracing": true
|
|
||||||
},
|
|
||||||
"rate_limiting": {
|
|
||||||
"communicated_headers": true,
|
|
||||||
"no_ambush_429": true,
|
|
||||||
"retry_after_provided": true
|
|
||||||
},
|
|
||||||
"sdk_and_docs": {
|
|
||||||
"generated_from_spec": true,
|
|
||||||
"never_drift": true,
|
|
||||||
"typed_idiomatic": true,
|
|
||||||
"multiple_languages": true
|
|
||||||
},
|
|
||||||
"idempotency": {
|
|
||||||
"write_operations_idempotent": true,
|
|
||||||
"idempotency_key_support": true,
|
|
||||||
"safe_retry": true
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
// Validate all requirements
|
|
||||||
assert!(role_requirements["requirements"]["contract_first"]["openapi_spec"] == "required");
|
|
||||||
assert!(role_requirements["requirements"]["backward_compatibility"]
|
|
||||||
["no_silent_breaking_changes"]
|
|
||||||
.as_bool()
|
|
||||||
.unwrap());
|
|
||||||
assert!(
|
|
||||||
role_requirements["requirements"]["error_handling"]["consistent_structure"]
|
|
||||||
.as_bool()
|
|
||||||
.unwrap()
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
role_requirements["requirements"]["rate_limiting"]["communicated_headers"]
|
|
||||||
.as_bool()
|
|
||||||
.unwrap()
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
role_requirements["requirements"]["sdk_and_docs"]["generated_from_spec"]
|
|
||||||
.as_bool()
|
|
||||||
.unwrap()
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
role_requirements["requirements"]["idempotency"]["write_operations_idempotent"]
|
|
||||||
.as_bool()
|
|
||||||
.unwrap()
|
|
||||||
);
|
|
||||||
|
|
||||||
println!("✓ API Platform Engineer role requirements validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_agent_prompts_for_api_platform_engineer() {
|
|
||||||
// Agent prompts aligned with API Platform Engineer role
|
|
||||||
let agent_prompts = vec![
|
|
||||||
("contract-review", CONTRACT_FIRST_PROMPT, "extraction"),
|
|
||||||
(
|
|
||||||
"compatibility-check",
|
|
||||||
BACKWARD_COMPATIBILITY_PROMPT,
|
|
||||||
"reasoning",
|
|
||||||
),
|
|
||||||
("sdk-generation", SDK_GENERATION_PROMPT, "generation"),
|
|
||||||
];
|
|
||||||
|
|
||||||
for (name, template, category) in agent_prompts {
|
|
||||||
let prompt = json!({
|
|
||||||
"name": name,
|
|
||||||
"template": template,
|
|
||||||
"task_category": category,
|
|
||||||
"target_model": "ornith:35b"
|
|
||||||
});
|
|
||||||
|
|
||||||
assert!(!prompt["template"].as_str().unwrap().is_empty());
|
|
||||||
assert!(
|
|
||||||
prompt["template"].as_str().unwrap().contains("{{")
|
|
||||||
|| prompt["template"].as_str().unwrap().contains("output")
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
println!("✓ Agent prompts for API Platform Engineer validated");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn test_role_to_prompt_mapping_consistency() {
|
|
||||||
// Role mappings ensure consistent prompt selection
|
|
||||||
let role_mappings = json!({
|
|
||||||
"api-platform-engineer": [
|
|
||||||
{
|
|
||||||
"prompt": "contract-review",
|
|
||||||
"priority": 1,
|
|
||||||
"for_task": "API specification review"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"prompt": "compatibility-check",
|
|
||||||
"priority": 2,
|
|
||||||
"for_task": "Breaking change validation"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"prompt": "sdk-generation",
|
|
||||||
"priority": 3,
|
|
||||||
"for_task": "SDK generation planning"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
});
|
|
||||||
|
|
||||||
let engineer_prompts = role_mappings["api-platform-engineer"].as_array().unwrap();
|
|
||||||
assert_eq!(engineer_prompts.len(), 3);
|
|
||||||
|
|
||||||
// Prompts ordered by priority
|
|
||||||
assert!(
|
|
||||||
engineer_prompts[0]["priority"].as_i64().unwrap()
|
|
||||||
< engineer_prompts[1]["priority"].as_i64().unwrap()
|
|
||||||
);
|
|
||||||
|
|
||||||
println!("✓ Role-to-prompt mapping consistency validated");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Reference in New Issue
Block a user