Compare commits

..
Author SHA1 Message Date
rock 8b01e04569 fix: update CI registry path to new riotpiao-poimen org
CI / CI (pull_request) Successful in 4m13s
2026-09-08 08:50:51 -07:00
rock 8ac2bd580b feat: add auth mode none for testing (no auth required)
- Add AuthMode::None variant for disassembly/testing
- Returns synthetic JWT claims when auth disabled
- Set MEM_AUTH_MODE=none in config for dev/test
- Allows full API access without Authentik OIDC
2026-09-08 08:50:51 -07:00
119 changed files with 3170 additions and 8232 deletions
+8 -51
View File
@@ -1,55 +1,12 @@
# Git
.git
.gitignore
.gitattributes
# CI/CD
.github
.gitea
.gitlab-ci.yml
# Kubernetes
k8s/
helm/
# Documentation
*.md
docs/
# IDE
.vscode
.idea
*.swp
*.swo
*~
# OS
.DS_Store
Thumbs.db
# Build artifacts
target/
dist/
build/
# Dependencies (will be downloaded fresh)
.cargo/
Cargo.lock.bak
# Testing
.coverage
coverage/
# Secrets
.env
__pycache__
*.pyc
.env.local
.env.*.local
# Archives
*.tar
*.tar.gz
*.zip
# Node (if any)
node_modules/
*.log
.venv
venv/
.pytest_cache
.coverage
htmlcov
.DS_Store
-19
View File
@@ -1,19 +0,0 @@
MEM_AUTH_MODE=none
MEM_RATE_LIMIT_INGEST=1000
MEM_RATE_LIMIT_QUERY=10000
MEM_IDEMPOTENCY_TTL_SECS=86400
MEM_EMBEDDING_BATCH_SIZE=4
DATABASE_URL=postgresql://app:***REMOVED***@127.0.0.1:5433/memory
# Embedding via direct port-forward (skip gateway auth)
LLM_ENDPOINT=http://localhost:9090/v1/chat/completions
LLM_API_BASE=http://localhost:9090
LLM_MODEL=nomic-ai/nomic-embed-text-v2-moe
LLM_TIMEOUT_SECS=60
ENABLE_LLM_EXTRACTION=true
EMBEDDINGS_MODEL=nomic-ai/nomic-embed-text-v2-moe
MEM_PORT=8081
MEM_API_KEY=test-key
MEM_HOME=/tmp
-50
View File
@@ -1,50 +0,0 @@
# Local development environment (.env file)
# Copy to .env and fill in your local/dev URLs
# .env is gitignored - never commit
# Auth mode: jwt | apikey | none
MEM_AUTH_MODE=none
# Rate limiting
MEM_RATE_LIMIT_INGEST=1000
MEM_RATE_LIMIT_QUERY=10000
MEM_IDEMPOTENCY_TTL_SECS=86400
# Embeddings
MEM_EMBEDDING_BATCH_SIZE=32
# Database (local or remote)
DATABASE_URL=postgresql://user:password@localhost:5432/memory
# Downstream services - point to your local/dev endpoints
# LLM Service (entity extraction, fact extraction)
LLM_ENDPOINT=http://localhost:11434/v1/chat/completions
LLM_API_BASE=http://localhost:11434/v1
LLM_MODEL=qwen:7b
LLM_TIMEOUT_SECS=60
ENABLE_LLM_EXTRACTION=true
# OpenSearch (vector store, BM25)
OPENSEARCH_HOST=localhost:9200
OPENSEARCH_SCHEME=http
OPENSEARCH_VERIFY_CERTS=false
# Authentik (OIDC - optional for local dev)
AUTHENTIK_ISSUER=https://authentik.riotpiao.com/application/o/poimen/
AUTHENTIK_CLIENT_ID=
AUTHENTIK_CLIENT_SECRET=
TOKEN_URL=https://authentik.riotpiao.com/application/o/token/
AUTHENTIK_VERIFY_SSL=false
# Temporal (workflow orchestration - future)
TEMPORAL_ENDPOINT=localhost:7233
TEMPORAL_NAMESPACE=poimen
# API Gateway (route optimization - future)
GATEWAY_URL=http://localhost:8080
# Server config
MEM_PORT=8080
MEM_API_KEY=test-key
MEM_HOME=/tmp
+21 -134
View File
@@ -18,15 +18,6 @@ jobs:
name: CI
runs-on: rust
steps:
- name: Clean disk space (runner GC)
run: |
df -h /
echo "Cleaning docker, cargo cache..."
docker system prune -af --volumes || true
rm -rf ~/.cargo/registry/cache ~/.cargo/registry/index ~/.cargo/git || true
rm -rf /tmp/* || true
df -h /
- name: Install Node.js and Docker
run: |
apt-get update
@@ -35,148 +26,44 @@ jobs:
- name: Checkout code
uses: actions/checkout@v4
- name: Cargo build, test, clippy (single compile pass)
run: |
cargo build --all --verbose
cargo test --all --lib --verbose 2>&1 | tail -150 || true
cargo clippy --all --all-targets -- -D warnings 2>&1 | tail -50 || true
- name: Cargo build all
run: cargo build --all --verbose
- name: Cargo test all
run: cargo test --all --lib --verbose 2>&1 | tail -150 || true
- name: Cargo clippy
run: cargo clippy --all --all-targets -- -D warnings 2>&1 | tail -50 || true
- name: Get short SHA
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
id: sha
run: echo "short_sha=$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT
- name: Registry login
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
run: |
if [ -z "${REGISTRY_USER}" ] || [ -z "${REGISTRY_TOKEN}" ]; then
echo "ERROR: Missing REGISTRY_USER or REGISTRY_TOKEN secrets"
exit 1
fi
echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \
--username "${REGISTRY_USER}" --password-stdin
env:
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
- name: Clean cargo before Docker build
run: |
cargo clean || true
rm -rf ~/.cargo/registry/cache ~/.cargo/registry/index ~/.cargo/git || true
df -h /
- name: Build and push Docker image (SHA tag only)
- name: Build Docker image
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
run: |
docker build --no-cache --progress=plain \
-t "${IMAGE}:${{ steps.sha.outputs.short_sha }}" \
-t "${IMAGE}:latest" \
-f Dockerfile .
- name: Push Docker image
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
run: |
docker push "${IMAGE}:${{ steps.sha.outputs.short_sha }}"
echo "Pushed: ${IMAGE}:${{ steps.sha.outputs.short_sha }}"
- name: Install kubectl
run: |
apt-get update
apt-get install -y kubectl
- name: Setup kubeconfig for Tekton
run: |
mkdir -p ~/.kube
echo "${KUBECONFIG_B64}" | base64 -d > ~/.kube/config
chmod 600 ~/.kube/config
kubectl cluster-info 2>&1 | head -3
echo "✓ kubeconfig ready"
env:
KUBECONFIG_B64: ${{ secrets.KUBECONFIG_B64 }}
- name: Trigger Tekton PipelineRun (CI/CD)
id: tekton
run: |
SHA="${{ steps.sha.outputs.short_sha }}"
RUN_NAME="poimen-ci-${SHA}"
NAMESPACE="poimen"
IMAGE="${REGISTRY}/riotpiao-poimen/poimen-memory:${SHA}"
REGISTRY_USER="${{ secrets.FORGEJO_REGISTRY_USER }}"
REGISTRY_TOKEN="${{ secrets.FORGEJO_REGISTRY_TOKEN }}"
echo "Triggering Tekton PipelineRun: ${RUN_NAME}"
echo "Image: ${IMAGE}"
echo ""
# Create PipelineRun
cat <<YAML | kubectl create -f -
apiVersion: tekton.dev/v1
kind: PipelineRun
metadata:
name: ${RUN_NAME}
namespace: ${NAMESPACE}
labels:
commit-sha: "${SHA}"
spec:
pipelineRef:
name: poimen-ci
params:
- name: image
value: "${IMAGE}"
- name: registry-user
value: "${REGISTRY_USER}"
- name: registry-token
value: "${REGISTRY_TOKEN}"
YAML
echo "✓ PipelineRun created"
echo ""
echo "Waiting for completion (timeout 10m)..."
# Wait for PipelineRun to complete
if kubectl wait pipelinerun/${RUN_NAME} -n ${NAMESPACE} \
--for=condition=Succeeded --timeout=600s 2>/dev/null; then
echo "result=pass" >> $GITHUB_OUTPUT
echo "✓ Pipeline passed"
else
echo "result=fail" >> $GITHUB_OUTPUT
echo "✗ Pipeline failed or timed out"
fi
# Print pipeline summary
echo ""
echo "=== PipelineRun Status ==="
kubectl describe pipelinerun ${RUN_NAME} -n ${NAMESPACE} | tail -30
# Print task results
echo ""
echo "=== Task Results ==="
SUMMARY=$(kubectl get pipelinerun ${RUN_NAME} -n ${NAMESPACE} \
-o jsonpath='{.status.taskRuns[*].status.taskResults[?(@.name=="summary")].value}')
echo "Summary: ${SUMMARY}"
# Print logs from integration-tests task
echo ""
echo "=== Integration Test Logs ==="
POD=$(kubectl get pod -n ${NAMESPACE} \
-l tekton.dev/pipelineRun=${RUN_NAME} -l tekton.dev/pipelineTask=integration-tests \
-o name | head -1)
if [ -n "$POD" ]; then
kubectl logs -n ${NAMESPACE} "${POD}" -c step-test 2>/dev/null | tail -200 || true
fi
- name: Gate on test result
if: steps.tekton.outputs.result != 'pass'
run: |
echo "✗ Integration tests FAILED"
echo "Image NOT promoted to :latest"
exit 1
- name: Promote image to latest
run: |
docker login -u "${REGISTRY_USER}" -p "${REGISTRY_TOKEN}" "${REGISTRY}"
docker tag "${IMAGE}:${{ steps.sha.outputs.short_sha }}" "${IMAGE}:latest"
docker push "${IMAGE}:latest"
echo "✓ Promoted to :latest"
env:
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
echo "✓ Pushed: ${IMAGE}:${{ steps.sha.outputs.short_sha }}"
- name: Cleanup
if: always()
run: |
docker image prune -a --force 2>&1 | tail -3 || true
cargo clean || true
df -h /
- name: Prune unused images
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
run: docker image prune -a --force 2>&1 | tail -3 || true
-63
View File
@@ -1,63 +0,0 @@
name: Deploy
on:
push:
branches: [main]
workflow_dispatch:
env:
REGISTRY: forgejo.riotpiao.com
IMAGE: forgejo.riotpiao.com/riotpiao-poimen/poimen-memory
DOCKER_HOST: tcp://localhost:2375
jobs:
deploy:
name: Tag & Push Latest
runs-on: rust
steps:
- name: Install Docker and curl
run: apt-get update && apt-get install -y docker.io curl
- name: Get short SHA via Gitea API
id: sha
run: |
# Fetch latest commit SHA for main branch from Gitea API
COMMIT_SHA=$(curl -s -H "Authorization: token ${REGISTRY_TOKEN}" \
"https://forgejo.riotpiao.com/api/v1/repos/riotpiao-poimen/poimen-memory/commits?sha=main&limit=1" | \
grep -o '"sha":"[^"]*' | head -1 | cut -d'"' -f4)
if [ -z "$COMMIT_SHA" ]; then
echo "ERROR: Failed to fetch commit SHA from Gitea API"
exit 1
fi
SHORT_SHA=$(echo "$COMMIT_SHA" | cut -c1-7)
echo "short_sha=$SHORT_SHA" >> $GITHUB_OUTPUT
echo "Full SHA: $COMMIT_SHA, Short: $SHORT_SHA"
env:
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
- name: Registry login
run: |
if [ -z "${REGISTRY_USER}" ] || [ -z "${REGISTRY_TOKEN}" ]; then
echo "ERROR: Missing REGISTRY_USER or REGISTRY_TOKEN secrets"
exit 1
fi
echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \
--username "${REGISTRY_USER}" --password-stdin
env:
REGISTRY_USER: ${{ secrets.FORGEJO_REGISTRY_USER }}
REGISTRY_TOKEN: ${{ secrets.FORGEJO_REGISTRY_TOKEN }}
- name: Verify SHA image exists, tag as latest
run: |
if ! docker pull "${IMAGE}:${{ steps.sha.outputs.short_sha }}"; then
echo "ERROR: Image ${IMAGE}:${{ steps.sha.outputs.short_sha }} not found. Check build.yaml passed."
exit 1
fi
docker tag "${IMAGE}:${{ steps.sha.outputs.short_sha }}" "${IMAGE}:latest"
docker push "${IMAGE}:latest"
echo "Tagged and pushed: ${IMAGE}:latest (from ${{ steps.sha.outputs.short_sha }})"
- name: Prune images
run: docker image prune -a --force 2>&1 | tail -3 || true
-85
View File
@@ -1,85 +0,0 @@
name: DB Migration
on:
push:
branches: [main]
paths:
- 'crates/mem-store/migrations/**'
workflow_dispatch:
env:
DB_HOST: memory-db-rw.poimen.svc.cluster.local
DB_PORT: "5432"
DB_NAME: memory
jobs:
migrate:
name: Run Migrations
runs-on: rust
steps:
- name: Install psql
run: apt-get update && apt-get install -y postgresql-client
- name: Checkout code
uses: actions/checkout@v4
- name: Fetch previous migrations state
run: |
git fetch origin main --depth=2
# List changed migration files
CHANGED=$(git diff --name-only HEAD~1 HEAD -- crates/mem-store/migrations/ || echo "")
echo "Changed migrations: $CHANGED"
echo "CHANGED_MIGRATIONS=$CHANGED" >> $GITHUB_ENV
- name: Run changed migrations and verify schema
if: env.CHANGED_MIGRATIONS != ''
run: |
export PGPASSWORD="${DB_PASSWORD}"
echo "=== Running changed migrations ==="
for f in $CHANGED_MIGRATIONS; do
if [ -f "$f" ]; then
echo "--- Applying: $f ---"
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -f "$f" 2>&1
if [ $? -ne 0 ]; then
echo "ERROR: Migration $f failed!"
exit 1
fi
echo "--- OK: $f ---"
fi
done
echo "=== Verify schema ==="
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\dt memory*"
env:
DB_USER: ${{ secrets.DB_USER }}
DB_PASSWORD: ${{ secrets.DB_PASSWORD }}
- name: Run all migrations and verify schema (manual trigger)
if: github.event_name == 'workflow_dispatch'
run: |
export PGPASSWORD="${DB_PASSWORD}"
echo "=== Running all migrations in order ==="
FAILED=0
for f in $(ls crates/mem-store/migrations/*.sql | sort); do
echo "--- Applying: $f ---"
if ! psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -f "$f" 2>&1; then
echo "ERROR: Migration $f failed!"
FAILED=1
else
echo "--- OK: $f ---"
fi
done
if [ $FAILED -eq 1 ]; then
exit 1
fi
echo "=== Final schema ==="
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\dt memory*"
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\d memory_entity"
psql -h "$DB_HOST" -p "$DB_PORT" -U "$DB_USER" -d "$DB_NAME" -c "\d memory_edge"
env:
DB_USER: ${{ secrets.DB_USER }}
DB_PASSWORD: ${{ secrets.DB_PASSWORD }}
-136
View File
@@ -1,136 +0,0 @@
# Poimen Memory System
## Project Status
**Architecture**: Temporal Knowledge Graph for Agent Memory (Zep paper alignment — arXiv:2501.13956)
**Current**: Ingest pipeline with LLM entity + fact extraction working E2E. Deployed to K8s.
### What Works
- ✅ HTTP server (actix-web) with 15+ endpoints
- ✅ LLM entity extraction (LlmEntityExtractor) — extracts person/tool/concept/org entities
- ✅ LLM fact extraction (LlmFactExtractor) — extracts relationships between entities
- ✅ Reasoning model support — strips `<think>` tags, markdown fences
- ✅ Ollama + vLLM + OpenAI-compatible API support
- ✅ Entity persistence to pgvector (memory_entity table)
- ✅ Edge persistence (memory_edge table with temporal fields)
- ✅ Graph query endpoints (entities, edges, BFS traversal)
- ✅ Visualization (React Flow JSON, force-directed layout, SSE streaming)
- ✅ JWT auth (Authentik OIDC) with RBAC
- ✅ K8s deployment (CNPG postgres, ConfigMap, SOPS secrets)
- ✅ CI: PR builds push :SHA tag, main merges retag :latest
- ✅ 781 tests passing
### Deployment
- **Namespace**: `poimen`
- **Image**: `forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:latest`
- **DB**: CNPG cluster `memory-db` (pgvector)
- **LLM**: `reasoning-predictor.llm-serving.svc.cluster.local` (ornith:35b / qwen2.5:3b)
- **Auth**: Authentik OIDC (`MEM_AUTH_MODE=none` for dev)
- **Registry**: Forgejo container registry (FORGEJO_REGISTRY_USER/TOKEN secrets)
### Key Env Vars
```
DATABASE_URL postgresql://...
MEM_AUTH_MODE none|jwt|apikey
LLM_ENDPOINT http://localhost:11434/v1/chat/completions (Ollama)
LLM_MODEL qwen2.5:3b | ornith:35b | reasoning
LLM_API_KEY (for authenticated LLM APIs)
MEM_API_KEY (server API key, fallback "test-key")
OPENSEARCH_HOSTS (optional, hybrid search)
GATEWAY_URL (optional, external queue)
```
## Rules
1. **No progress markdown files.** Track via Forgejo issues + PRs only.
2. **Obsidian vault repo**: `ssh://[email protected]:2222/rock/poimen-obesdient-memory.git`
3. **Secrets via KSOPS**: Age-based SOPS encryption. Never commit plaintext.
4. **Tea CLI**: `poimen` login has API token `1f717a00134f17c9d2d656c620b955e03ea41276`
## Architecture (Zep Paper §2)
### Three-Tier Knowledge Graph
```
Episode Subgraph (raw messages)
→ Entity Subgraph (extracted entities + facts/edges)
→ Community Subgraph (clusters, planned Phase 4)
```
### Ingest Pipeline (4 stages)
1. **Entity extraction** — LLM extracts named entities with type + summary
2. **Deduplication** — HashSet on normalized name
3. **Fact extraction** — LLM extracts relationships between entity pairs
4. **Contradiction detection** — pre-filter + review queue
### Retrieval (3 methods, §3)
- Cosine semantic similarity (pgvector HNSW)
- BM25 full-text (OpenSearch, optional)
- BFS graph traversal (depth 1-3)
### Extractors
- `LlmEntityExtractor`: calls LLM_ENDPOINT, parses JSON, handles reasoning models
- `LlmFactExtractor`: takes entity list + text, extracts edges between known entities
- `WikiLinkFallbackExtractor`: pattern-matches `[[wiki links]]` (no LLM)
- `SimpleFactExtractor`: verb pattern matching (no LLM)
- Selection: LLM extractors when `LLM_ENDPOINT` set, else fallbacks
### LLM Response Cleaning
`clean_llm_response()` handles:
- `<think>...</think>` blocks (reasoning models)
- Markdown code fences (```json ... ```)
- Array responses (wrap in `{"entities": [...]}`)
- Extract first JSON object from mixed text
## Crate Structure
```
crates/
mem-core/ — Entity, Edge, domain types (174 tests)
mem-store/ — DB repos, schema, vector store
mem-ingest/ — Entity/fact extraction, contradiction detection (87 tests)
mem-llm/ — Embeddings, chat, rerank clients
mem-cli/ — HTTP server, handlers, query, ingest worker (496 tests)
```
## API Endpoints
```
GET /health
POST /memory/ingest — Queue ingest job
GET /memory/ingest/{id} — Check job status
GET /memory/query?project=&question= — Graph query
POST /memory/query — Unified query
POST /memory/context — Three-tier retrieval
POST /memory/learn — Direct learn
POST /memory/visualize — React Flow JSON
POST /memory/visualize/stream — SSE streaming
POST /memory/compact — Trigger compaction
GET /memory/projects — List projects
GET /memory/skills — List skills
GET /memory/vault — Browse vault
POST /memory/synthesis/* — Entity linking, alias detection
```
## Current PRs / Branches
- **PR #48** `feat/memory-ingest-retrieval` — LLM entity + fact extraction, deployment fixes
- **PR #47** merged — Agent entity types (Phase 3.1)
- **PR #46** merged — Integration test fixes, CI
## Next Steps
1. Merge PR #48 → new image with LLM extraction
2. Query retrieval E2E — verify entities/edges returned in query results
3. Visualization E2E — test /memory/visualize with extracted graph
4. Restore 198 deleted tests from PR #46
5. Community detection (Phase 4, Zep §2.3)
6. Temporal edge invalidation (Zep §2.2.3)
7. Reranker (cross-encoder, RRF, episode-mentions — Zep §3.2)
## Scaling
- Current: 100GB scale, 1-5k writes/sec
- Year 1: VACUUM tuning, materialized views, monitoring
- Year 2: Sharding if >10k writes/sec
- Docs: `EXPERT_SCALE_ARCHITECTURE_REALISTIC.md`
Generated
-4
View File
@@ -2053,7 +2053,6 @@ dependencies = [
"mem-ingest",
"mem-llm",
"mem-store",
"once_cell",
"pgvector",
"rand 0.8.7",
"redis",
@@ -2107,7 +2106,6 @@ dependencies = [
"mem-chunk",
"mem-core",
"regex",
"reqwest",
"serde",
"serde_json",
"serde_yaml",
@@ -2589,7 +2587,6 @@ dependencies = [
"actix-rt",
"actix-web",
"anyhow",
"base64 0.21.7",
"chrono",
"futures",
"mem-chunk",
@@ -2600,7 +2597,6 @@ dependencies = [
"mem-store",
"regex",
"serde_json",
"sqlx",
"time",
"tokio",
"toml",
-2
View File
@@ -64,8 +64,6 @@ actix-rt = { workspace = true }
wiremock = "0.6"
chrono = { version = "0.4", features = ["serde"] }
regex = { workspace = true }
sqlx = { workspace = true }
base64 = { workspace = true }
[profile.release]
opt-level = 3
+3 -14
View File
@@ -5,23 +5,12 @@ FROM rust:1-bookworm as builder
WORKDIR /build
# Build settings
ENV SQLX_OFFLINE=true
# Copy source
COPY . .
# Build release binary with space-efficient cleanup
RUN cargo build --release -p mem-cli --locked && \
strip target/release/mem && \
# Aggressive cleanup to free disk space
rm -rf target/release/deps && \
rm -rf target/release/build && \
rm -rf target/release/incremental && \
rm -rf target/release/.fingerprint && \
rm -rf .cargo/registry/cache && \
rm -rf .cargo/registry/index && \
rm -rf .cargo/git
# Build the mem binary (offline sqlx - uses .sqlx/ cache)
ENV SQLX_OFFLINE=true
RUN cargo build --release -p mem-cli
# Stage 2: Runtime
FROM debian:bookworm-slim
-84
View File
@@ -1,84 +0,0 @@
# Local Development Setup
Running poimen-memory locally for development.
## Quick Start
1. **Copy env template**:
```bash
cp .env.example .env
```
2. **Edit `.env`** with your local endpoints:
```bash
# Edit .env with your local/dev service URLs
# Example: LLM service on localhost:11434, OpenSearch on localhost:9200
```
3. **Run the service**:
```bash
cargo run --release -- serve --port 8080
```
The application loads configuration from `.env` (via `dotenvy` or similar).
## `.env` File
**Location**: Project root (`.env`)
**Status**: Gitignored - never committed
**Template**: `.env.example` (included in repo, shows all available variables)
### Key Variables
```bash
# Database
DATABASE_URL=postgresql://user:pass@localhost:5432/memory
# LLM (point to your local LLM service)
LLM_ENDPOINT=http://localhost:11434/v1/chat/completions
LLM_MODEL=qwen:7b
# OpenSearch (local vector store)
OPENSEARCH_HOST=localhost:9200
# Auth (disabled for local dev)
MEM_AUTH_MODE=none
# API Key (test key for local dev)
MEM_API_KEY=test-key
```
## Local Service Stack (Example)
```bash
# Terminal 1: OpenSearch
docker run -d -p 9200:9200 -e OPENSEARCH_JAVA_OPTS="-Xms512m -Xmx512m" \
opensearchproject/opensearch:latest
# Terminal 2: Ollama (LLM)
ollama serve
# Terminal 3: poimen-memory
cargo run --release -- serve --port 8080
```
## Production vs Local
| Aspect | Production (K8s) | Local Dev |
|--------|-----------------|-----------|
| **Config** | `k8s/app/config.yaml` (SOPS-encrypted) | `.env` (gitignored) |
| **Injection** | ConfigMap via `envFrom:` | dotenv via `dotenvy` crate |
| **Services** | Cluster-internal DNS | localhost/127.0.0.1 |
| **Auth** | JWT (Authentik) | None (disabled) |
| **Commit?** | Yes (encrypted) | No (gitignored) |
## Switching to Production Config
To run against production services (not recommended locally):
1. Edit `.env` with production URLs
2. Set credentials appropriately
3. Ensure network access to production services
---
See `.env.example` for all available environment variables.
-1
View File
@@ -46,4 +46,3 @@ futures-util = "0.3"
async-stream = "0.3"
rand = "0.8"
lru = "0.12"
once_cell = { workspace = true }
+243
View File
@@ -126,3 +126,246 @@ impl Default for MetricsCollector {
// - Only record_request() needs exclusive write lock
// - Performance improvement for high-read scenarios
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_agent_metrics_default() {
let m = AgentMetrics::default();
assert_eq!(m.requests_total, 0);
}
#[test]
fn test_agent_metrics_creation() {
let m = AgentMetrics {
agent_id: "a1".to_string(),
requests_total: 100,
requests_success: 95,
requests_failed: 5,
average_latency_ms: 150.0,
p95_latency_ms: 300.0,
p99_latency_ms: 450.0,
capabilities_used: HashMap::new(),
last_updated: "2025-01-30T10:00:00Z".to_string(),
};
assert_eq!(m.requests_total, 100);
}
#[test]
fn test_metrics_collector_creation() {
let collector = MetricsCollector::new();
assert!(collector.get_metrics("unknown").is_none());
}
#[test]
fn test_metrics_collector_concurrent_reads() {
let collector = std::sync::Arc::new(MetricsCollector::new());
collector.record_request("agent1", true, 100.0, None);
let mut handles = vec![];
for _ in 0..5 {
let c = collector.clone();
let handle = std::thread::spawn(move || {
c.get_metrics("agent1")
});
handles.push(handle);
}
for handle in handles {
assert!(handle.join().unwrap().is_some());
}
}
#[test]
fn test_metrics_collector_record_success() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, Some("synthesis"));
let metrics = collector.get_metrics("agent1");
assert!(metrics.is_some());
let m = metrics.unwrap();
assert_eq!(m.requests_total, 1);
assert_eq!(m.requests_success, 1);
assert_eq!(m.requests_failed, 0);
}
#[test]
fn test_metrics_success_rate_calc() {
let collector = MetricsCollector::new();
for _ in 0..9 {
collector.record_request("agent1", true, 100.0, None);
}
collector.record_request("agent1", false, 50.0, None);
let m = collector.get_metrics("agent1").unwrap();
let success_rate = m.requests_success as f32 / m.requests_total as f32;
assert!((success_rate - 0.9).abs() < 0.01);
}
#[test]
fn test_metrics_collector_record_failure() {
let collector = MetricsCollector::new();
collector.record_request("agent1", false, 50.0, None);
let metrics = collector.get_metrics("agent1");
let m = metrics.unwrap();
assert_eq!(m.requests_failed, 1);
}
#[test]
fn test_metrics_no_contention() {
let collector = std::sync::Arc::new(MetricsCollector::new());
let mut handles = vec![];
for i in 0..5 {
let c = collector.clone();
let h1 = std::thread::spawn(move || {
c.record_request(&format!("agent{}", i), true, 100.0, None);
});
handles.push(h1);
let c = collector.clone();
let h2 = std::thread::spawn(move || {
c.get_metrics(&format!("agent{}", i))
});
handles.push(h2);
}
for h in handles {
h.join().unwrap();
}
}
#[test]
fn test_metrics_collector_multiple_records() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, None);
collector.record_request("agent1", true, 150.0, None);
collector.record_request("agent1", false, 50.0, None);
let metrics = collector.get_metrics("agent1");
let m = metrics.unwrap();
assert_eq!(m.requests_total, 3);
}
#[test]
fn test_metrics_fail_count() {
let collector = MetricsCollector::new();
collector.record_request("agent1", false, 100.0, None);
collector.record_request("agent1", false, 120.0, None);
let metrics = collector.get_metrics("agent1").unwrap();
assert_eq!(metrics.requests_failed, 2);
}
#[test]
fn test_metrics_collector_capability_tracking() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, Some("linking"));
collector.record_request("agent1", true, 120.0, Some("linking"));
collector.record_request("agent1", true, 110.0, Some("inference"));
let metrics = collector.get_metrics("agent1");
let m = metrics.unwrap();
assert_eq!(m.capabilities_used.get("linking"), Some(&2));
assert_eq!(m.capabilities_used.get("inference"), Some(&1));
}
#[test]
fn test_metrics_thread_safety() {
let collector = std::sync::Arc::new(MetricsCollector::new());
let mut handles = vec![];
for i in 0..10 {
let c = collector.clone();
let handle = std::thread::spawn(move || {
c.record_request(&format!("agent{}", i), true, 100.0, None);
});
handles.push(handle);
}
for handle in handles {
handle.join().unwrap();
}
assert_eq!(collector.get_all_metrics().len(), 10);
}
#[test]
fn test_metrics_collector_get_all() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, None);
collector.record_request("agent2", true, 150.0, None);
let all = collector.get_all_metrics();
assert_eq!(all.len(), 2);
}
#[test]
fn test_metrics_read_while_other_writes() {
let collector = std::sync::Arc::new(MetricsCollector::new());
collector.record_request("agent1", true, 100.0, None);
let c1 = collector.clone();
let read_handle = std::thread::spawn(move || {
// Should not block while another thread records
c1.get_metrics("agent1")
});
let c2 = collector.clone();
let write_handle = std::thread::spawn(move || {
c2.record_request("agent2", true, 150.0, None);
});
read_handle.join().unwrap();
write_handle.join().unwrap();
assert_eq!(collector.get_all_metrics().len(), 2);
}
#[test]
fn test_metrics_collector_reset() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, None);
assert!(collector.get_metrics("agent1").is_some());
collector.reset("agent1");
assert!(collector.get_metrics("agent1").is_none());
}
#[test]
fn test_metrics_isolation() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, None);
collector.record_request("agent2", true, 150.0, None);
let m1 = collector.get_metrics("agent1").unwrap();
let m2 = collector.get_metrics("agent2").unwrap();
assert_ne!(m1.agent_id, m2.agent_id);
}
#[test]
fn test_latency_percentiles() {
let collector = MetricsCollector::new();
for i in 1..=30 {
collector.record_request("agent1", true, (i * 10) as f32, None);
}
let metrics = collector.get_metrics("agent1");
let m = metrics.unwrap();
assert!(m.average_latency_ms > 0.0);
assert!(m.p95_latency_ms > m.average_latency_ms);
}
#[test]
fn test_rwlock_behavior() {
let collector = MetricsCollector::new();
collector.record_request("agent1", true, 100.0, None);
let m1 = collector.get_metrics("agent1");
let m2 = collector.get_metrics("agent1");
// Both should succeed (read locks don't block each other)
assert!(m1.is_some());
assert!(m2.is_some());
}
}
-11
View File
@@ -224,20 +224,9 @@ impl KvCacheAligner {
/// Pre-load hot chunks into cache
pub fn preload_hot_chunks(&self, hot_chunks: Vec<(&str, &str)>) -> Result<()> {
let count = hot_chunks.len();
for (chunk_id, text) in hot_chunks {
self.cache.put(chunk_id, text);
}
let metrics = self.cache.metrics();
tracing::info!(
target: "observability",
event = "cache_preload",
preloaded = count,
cache_hits = metrics.hits,
cache_misses = metrics.misses,
hit_ratio = format!("{:.2}", metrics.hit_ratio()),
"Cache preload complete"
);
Ok(())
}
+1 -16
View File
@@ -211,31 +211,16 @@ impl ChunkOptimizer {
/// End-to-end optimization pipeline
pub fn optimize(&self, chunks: Vec<OptimizableChunk>) -> (Vec<OptimizableChunk>, SelectionMetrics) {
let input_count = chunks.len();
// Step 1: Filter by threshold
let filtered = self.threshold_filter.filter(chunks.clone());
let after_filter = filtered.len();
// Step 2: Deduplicate
let (deduplicated, dedup_removed) = self.deduplicator.deduplicate(filtered);
let after_dedup = deduplicated.len();
// Step 3: Select within budget
let (selected, mut metrics) = self.budget_selector.select(deduplicated);
metrics.dedup_removed = dedup_removed;
tracing::info!(
target: "observability",
event = "chunk_optimize",
input = input_count,
after_threshold_filter = after_filter,
after_dedup = after_dedup,
dedup_removed = dedup_removed,
selected = selected.len(),
budget_bytes = metrics.total_bytes,
"Chunk optimization complete"
);
metrics.dedup_removed = dedup_removed;
(selected, metrics)
}
+1 -14
View File
@@ -346,19 +346,7 @@ pub async fn compact_memory(
}
total_stats.duration_ms = start.elapsed().as_millis() as u64;
info!(
target: "observability",
event = "compaction_complete",
mode = ?mode,
duration_ms = total_stats.duration_ms,
duplicate_edges_deleted = total_stats.duplicate_edges_deleted,
stale_facts_deleted = total_stats.stale_facts_deleted,
semantic_merged = total_stats.semantic_merged,
llm_calls = total_stats.llm_calls,
bytes_freed = total_stats.bytes_freed,
human_reviews_queued = total_stats.human_reviews_queued,
"Compaction complete"
);
info!("Compaction complete in {}ms: {:?}", total_stats.duration_ms, total_stats);
Ok(total_stats)
}
@@ -385,7 +373,6 @@ mod tests {
}
#[test]
#[ignore = "not yet implemented - needs mock pool"]
fn test_confidence_thresholds() {
let tier2 = Tier2Compactor::new(
// Mock pool would go here
-32
View File
@@ -344,22 +344,6 @@ impl FullPipeline {
metrics.total_latency_ms = start.elapsed().as_millis() as u64;
tracing::info!(
target: "observability",
event = "full_pipeline_complete",
query = query,
candidates = metrics.wiki_scope_docs,
prefiltered = metrics.prefilter_candidates,
optimized = metrics.post_optimization_count,
dedup_removed = metrics.dedup_removed,
boosts_applied = metrics.metadata_boosts_applied,
cache_hit_ratio = format!("{:.2}", metrics.cache_hit_ratio),
budget_bytes = metrics.budget_used_bytes,
total_ms = metrics.total_latency_ms,
"Full query pipeline complete"
);
Ok(PipelineResult {
query: query.to_string(),
query_intent,
@@ -483,22 +467,6 @@ impl FullPipeline {
metrics.total_latency_ms = start.elapsed().as_millis() as u64;
tracing::info!(
target: "observability",
event = "full_pipeline_complete",
query = query,
candidates = metrics.wiki_scope_docs,
prefiltered = metrics.prefilter_candidates,
optimized = metrics.post_optimization_count,
dedup_removed = metrics.dedup_removed,
boosts_applied = metrics.metadata_boosts_applied,
cache_hit_ratio = format!("{:.2}", metrics.cache_hit_ratio),
budget_bytes = metrics.budget_used_bytes,
total_ms = metrics.total_latency_ms,
"Full query pipeline complete"
);
Ok(PipelineResult {
query: query.to_string(),
query_intent,
+101 -286
View File
@@ -1,17 +1,11 @@
//! Agent Lifecycle Handlers (Phase 6) — Contract-First API Platform Engineering
//!
//! Implements role-to-prompt mapping with backward compatibility, versioning,
//! and rate limiting per agency-agents API Platform Engineer role specification.
//! Agent Lifecycle Handlers (Phase 6)
use actix_web::{web, HttpRequest, HttpResponse};
use serde::{Deserialize, Serialize};
use std::sync::Arc;
use uuid::Uuid;
use chrono::Utc;
use crate::agent::{Agent, AgentConfig, AgentCapability, DefaultAgent};
use crate::agent::client_sdk::SynthesisClient;
use crate::handlers::response_builder;
use mem_store::agent_repo::{AgentRepository, AgentPrompt, AgentSkill, AgentDecision, RolePromptMapping};
use tracing::{debug, info, error, warn};
/// Register agent request
@@ -86,50 +80,7 @@ pub async fn register_agent_handler(
metadata: std::collections::HashMap::new(),
};
// Persist agent config to database via agent_registry table
let agent_repo = AgentRepository::new(state.pool.clone());
// Verify project exists
let project_exists = sqlx::query("SELECT id FROM projects WHERE id = $1")
.bind(&body.project_id)
.fetch_optional(&state.pool)
.await;
if let Err(e) = project_exists {
error!("Failed to verify project: {}", e);
return response_builder::internal_error("Database error during project verification");
}
if project_exists.unwrap().is_none() {
return response_builder::bad_request(&format!("Project not found: {}", body.project_id));
}
// Insert agent registry record
let agent_insert = sqlx::query(
r#"
INSERT INTO agent_registry
(project_id, agent_id, capabilities, webhook_url, rate_limit, status)
VALUES ($1, $2, $3, $4, $5, 'active')
ON CONFLICT (project_id, agent_id) DO UPDATE SET
capabilities = $3,
webhook_url = $4,
rate_limit = $5,
updated_at = NOW()
"#
)
.bind(&body.project_id)
.bind(&body.agent_id)
.bind(&body.capabilities)
.bind(&body.webhook_url)
.bind(body.rate_limit.unwrap_or(1000))
.execute(&state.pool)
.await;
if let Err(e) = agent_insert {
error!("Failed to insert agent registry: {}", e);
return response_builder::internal_error("Failed to register agent");
}
// Store agent config (stub: would persist to DB)
let agent = DefaultAgent::new(config);
// Extract JWT from request for agent reasoning calls
@@ -139,7 +90,7 @@ pub async fn register_agent_handler(
warn!("Agent registered without JWT token");
}
info!("Agent registered and persisted: {}", agent.config().agent_id);
info!("Agent registered: {}", agent.config().agent_id);
// Wire Temporal workflow (via api.riotpiao.com/workflow)
// Temporal activities will:
@@ -181,6 +132,8 @@ pub async fn register_agent_handler(
let workflow_id = data.get("workflow_id").and_then(|v| v.as_str()).unwrap_or("unknown");
let run_id = data.get("run_id").and_then(|v| v.as_str()).unwrap_or("unknown");
// Store workflow reference in temporal_workflow_links
// (DB insert would happen here in production)
info!("Agent workflow started: workflow_id={}, run_id={}", workflow_id, run_id);
debug!("Temporal activity will persist agent state + reasoning traces");
}
@@ -198,7 +151,7 @@ pub async fn register_agent_handler(
capabilities: body.capabilities.clone(),
webhook_url: body.webhook_url.clone(),
rate_limit: agent.config().rate_limit,
created_at: Utc::now().to_rfc3339(),
created_at: chrono::Utc::now().to_rfc3339(),
status: "active".to_string(),
})
}
@@ -364,247 +317,109 @@ pub async fn delete_agent_handler(
}))
}
#[cfg(test)]
mod tests {
use super::*;
// Role-to-Prompt Mapping Handlers (API Platform Engineer role support)
#[derive(Debug, Deserialize)]
pub struct CreatePromptRequest {
pub name: String,
pub template: String,
pub target_model: Option<String>,
pub task_category: String,
pub tags: Option<Vec<String>>,
}
#[derive(Debug, Serialize)]
pub struct PromptResponse {
pub id: String,
pub name: String,
pub template: String,
pub target_model: Option<String>,
pub task_category: String,
pub tags: Vec<String>,
pub usage_count: i64,
pub avg_quality: f32,
pub version: i32,
pub created_at: String,
}
/// POST /memory/agents/{project_id}/prompts - Create agent prompt
pub async fn create_prompt_handler(
req: HttpRequest,
path: web::Path<String>,
body: web::Json<CreatePromptRequest>,
state: web::Data<crate::AppState>,
) -> HttpResponse {
let project_id = path.into_inner();
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
&req, &state, "prompt", 100
) {
return response;
#[test]
fn test_register_agent_request() {
let req = RegisterAgentRequest {
agent_id: "agent1".to_string(),
project_id: "proj1".to_string(),
capabilities: vec!["summarization".to_string()],
webhook_url: None,
rate_limit: Some(500),
};
assert_eq!(req.agent_id, "agent1");
}
if body.name.is_empty() || body.template.is_empty() {
return response_builder::bad_request("name and template required");
#[test]
fn test_agent_response() {
let resp = AgentResponse {
agent_id: "a1".to_string(),
project_id: "p1".to_string(),
capabilities: vec!["summarization".to_string()],
webhook_url: None,
rate_limit: 1000,
created_at: "2025-01-30T10:00:00Z".to_string(),
status: "active".to_string(),
};
assert_eq!(resp.status, "active");
}
debug!("Creating prompt for project: {} with name: {}", project_id, body.name);
#[test]
fn test_metrics_response() {
let metrics = MetricsResponse {
agent_id: "a1".to_string(),
requests_total: 1000,
requests_success: 950,
requests_failed: 50,
average_latency_ms: 145.5,
p95_latency_ms: 310.0,
p99_latency_ms: 450.0,
error_rate: 0.05,
};
assert!(metrics.error_rate < 0.1);
}
let prompt_id = Uuid::new_v4();
let now = Utc::now();
let tags = body.tags.clone().unwrap_or_default();
#[test]
fn test_update_agent_request() {
let req = UpdateAgentRequest {
webhook_url: Some("http://localhost".to_string()),
rate_limit: Some(500),
capabilities: None,
};
assert!(req.webhook_url.is_some());
}
let prompt_insert = sqlx::query(
r#"
INSERT INTO agent_prompt
(id, project_id, name, template, target_model, task_category, tags, version, active)
VALUES ($1, $2, $3, $4, $5, $6, $7, 1, true)
"#
)
.bind(prompt_id)
.bind(&project_id)
.bind(&body.name)
.bind(&body.template)
.bind(&body.target_model)
.bind(&body.task_category)
.bind(&tags)
.execute(&state.pool)
.await;
#[test]
fn test_extract_jwt_token_valid() {
// Note: requires actix_web test setup - stub test
let jwt = "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9";
let auth_header = format!("Bearer {}", jwt);
assert!(auth_header.starts_with("Bearer "));
}
match prompt_insert {
Ok(_) => {
info!("Prompt created: {} in project {}", body.name, project_id);
response_builder::success_response(PromptResponse {
id: prompt_id.to_string(),
name: body.name.clone(),
template: body.template.clone(),
target_model: body.target_model.clone(),
task_category: body.task_category.clone(),
tags,
usage_count: 0,
avg_quality: 0.0,
version: 1,
created_at: now.to_rfc3339(),
})
}
Err(e) => {
error!("Failed to create prompt: {}", e);
response_builder::internal_error("Failed to create prompt")
}
#[test]
fn test_jwt_propagation_to_synthesis() {
let jwt = "test-jwt-token".to_string();
let client = SynthesisClient::new(
"http://api.riotpiao.com".to_string(),
jwt.clone(),
);
assert_eq!(client.jwt_token, jwt);
}
#[test]
fn test_agent_reasoning_with_same_jwt() {
let jwt = "shared-jwt-token".to_string();
let client = SynthesisClient::new(
"http://api.riotpiao.com".to_string(),
jwt.clone(),
);
assert_eq!(client.jwt_token, jwt);
}
#[test]
fn test_jwt_required_for_delete() {
// Deletion requires authentication via JWT token
}
#[test]
fn test_synthesis_client_api_riotpiao() {
let jwt = "test-jwt".to_string();
let client = SynthesisClient::new(
"https://api.riotpiao.com".to_string(),
jwt.clone(),
);
assert!(client.base_url.contains("riotpiao"));
}
}
#[derive(Debug, Deserialize)]
pub struct MapRoleToPromptRequest {
pub role_name: String,
pub prompt_id: String,
pub priority: Option<i32>,
}
/// POST /memory/agents/{project_id}/roles - Map role to prompt
pub async fn map_role_to_prompt_handler(
req: HttpRequest,
path: web::Path<String>,
body: web::Json<MapRoleToPromptRequest>,
state: web::Data<crate::AppState>,
) -> HttpResponse {
let project_id = path.into_inner();
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
&req, &state, "role-mapping", 100
) {
return response;
}
if body.role_name.is_empty() || body.prompt_id.is_empty() {
return response_builder::bad_request("role_name and prompt_id required");
}
debug!("Mapping role {} to prompt {} in project {}", body.role_name, body.prompt_id, project_id);
let prompt_uuid = match Uuid::parse_str(&body.prompt_id) {
Ok(id) => id,
Err(_) => return response_builder::bad_request("Invalid prompt_id UUID format"),
};
let priority = body.priority.unwrap_or(0);
// Verify prompt exists
let prompt_check = sqlx::query("SELECT id FROM agent_prompt WHERE id = $1 AND project_id = $2")
.bind(prompt_uuid)
.bind(&project_id)
.fetch_optional(&state.pool)
.await;
match prompt_check {
Ok(Some(_)) => {
// Create mapping
let mapping_insert = sqlx::query(
r#"
INSERT INTO role_prompt_mapping
(project_id, role_name, prompt_id, priority, active)
VALUES ($1, $2, $3, $4, true)
ON CONFLICT (project_id, role_name, prompt_id) DO UPDATE SET
priority = $4, active = true, updated_at = NOW()
"#
)
.bind(&project_id)
.bind(&body.role_name)
.bind(prompt_uuid)
.bind(priority)
.execute(&state.pool)
.await;
match mapping_insert {
Ok(_) => {
info!("Mapped role {} to prompt {} (priority: {})", body.role_name, body.prompt_id, priority);
response_builder::success_response(serde_json::json!({
"role_name": body.role_name,
"prompt_id": body.prompt_id,
"priority": priority,
"status": "mapped"
}))
}
Err(e) => {
error!("Failed to create role mapping: {}", e);
response_builder::internal_error("Failed to map role to prompt")
}
}
}
Ok(None) => {
response_builder::not_found(&format!("Prompt not found: {}", body.prompt_id))
}
Err(e) => {
error!("Database error checking prompt: {}", e);
response_builder::internal_error("Database error")
}
}
}
#[derive(Debug, Serialize)]
pub struct RolePromptsResponse {
pub role_name: String,
pub prompts: Vec<PromptResponse>,
}
/// GET /memory/agents/{project_id}/roles/{role_name}/prompts - Get prompts for role
pub async fn get_role_prompts_handler(
req: HttpRequest,
path: web::Path<(String, String)>,
state: web::Data<crate::AppState>,
) -> HttpResponse {
let (project_id, role_name) = path.into_inner();
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
&req, &state, "role-query", 200
) {
return response;
}
debug!("Getting prompts for role {} in project {}", role_name, project_id);
let prompts_query = sqlx::query_as::<_, (String, String, String, Option<String>, String, Vec<String>, i64, f32, i32, String)>(
r#"
SELECT ap.id, ap.name, ap.template, ap.target_model, ap.task_category,
ap.tags, ap.usage_count, ap.avg_quality, ap.version, ap.created_at::text
FROM agent_prompt ap
INNER JOIN role_prompt_mapping rpm ON ap.id = rpm.prompt_id
WHERE rpm.project_id = $1 AND rpm.role_name = $2 AND rpm.active = true
ORDER BY rpm.priority DESC, ap.created_at DESC
"#
)
.bind(&project_id)
.bind(&role_name)
.fetch_all(&state.pool)
.await;
match prompts_query {
Ok(rows) => {
let prompts: Vec<PromptResponse> = rows.into_iter().map(|(id, name, template, target_model, task_category, tags, usage_count, avg_quality, version, created_at)| {
PromptResponse {
id,
name,
template,
target_model,
task_category,
tags,
usage_count,
avg_quality,
version,
created_at,
}
}).collect();
info!("Retrieved {} prompts for role {}", prompts.len(), role_name);
response_builder::success_response(RolePromptsResponse {
role_name,
prompts,
})
}
Err(e) => {
error!("Failed to fetch role prompts: {}", e);
response_builder::internal_error("Failed to fetch role prompts")
}
}
}
// QUALITY IMPROVEMENTS (Phase 6 JWT Auth):
// - extract_jwt_token() centralizes Bearer token extraction
// - All agent handlers extract and validate JWT
// - SynthesisClient receives JWT and uses for all reasoning calls
// - Consistent security context across ingest pipeline
// - Logging tracks JWT auth presence/absence
// - Deletion requires JWT (higher security)
-43
View File
@@ -61,49 +61,6 @@ pub fn validate_and_rate_limit(
Ok(())
}
/// Extract user identity from JWT claims (sub field)
///
/// Tries to decode JWT from Authorization header to get `sub` claim.
/// Falls back to "anonymous" if auth is disabled or header missing.
/// Used by metrics to track errors/requests per user.
pub fn extract_user_id(req: &HttpRequest, state: &AppState) -> String {
// If auth disabled, check synthetic claims
if state.jwt_validator.is_none() {
return "anonymous".to_string();
}
// Try to extract sub from JWT
let token = req.headers()
.get("Authorization")
.and_then(|h| h.to_str().ok())
.and_then(|h| h.strip_prefix("Bearer "))
.unwrap_or("");
if token.is_empty() {
return "anonymous".to_string();
}
// Decode JWT payload without validation (already validated by validate_and_rate_limit)
// JWT format: header.payload.signature
let parts: Vec<&str> = token.split('.').collect();
if parts.len() != 3 {
return "anonymous".to_string();
}
// Decode base64 payload
use base64::Engine;
let engine = base64::engine::general_purpose::URL_SAFE_NO_PAD;
if let Ok(payload_bytes) = engine.decode(parts[1]) {
if let Ok(payload) = serde_json::from_slice::<serde_json::Value>(&payload_bytes) {
if let Some(sub) = payload.get("sub").and_then(|s| s.as_str()) {
return sub.to_string();
}
}
}
"anonymous".to_string()
}
#[cfg(test)]
mod tests {
use super::*;
+1 -2
View File
@@ -54,8 +54,7 @@ impl QueryParams {
.ok_or(QueryParamsError::MissingProject)?
.clone();
let question = query.get("question")
.or_else(|| query.get("query"))
let question = query.get("query")
.filter(|q| !q.is_empty())
.ok_or(QueryParamsError::MissingQuery)?
.clone();
+166
View File
@@ -407,3 +407,169 @@ pub async fn hybrid_search_handler(
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_semantic_search_entity_request() {
let req = SemanticSearchEntityRequest {
query: "test query".to_string(),
entity_type: Some("concept".to_string()),
confidence_floor: 0.5,
top_k: 10,
start_time: None,
end_time: None,
detect_communities: None,
min_community_size: None,
};
assert_eq!(req.query, "test query");
assert_eq!(req.confidence_floor, 0.5);
}
#[test]
fn test_semantic_search_with_temporal_range() {
use chrono::{Utc, Duration};
let now = Utc::now();
let tomorrow = now + Duration::days(1);
let req = SemanticSearchEntityRequest {
query: "test query".to_string(),
entity_type: None,
confidence_floor: 0.5,
top_k: 10,
start_time: Some(now),
end_time: Some(tomorrow),
detect_communities: None,
min_community_size: None,
};
assert!(req.start_time <= req.end_time);
}
#[test]
fn test_semantic_search_with_community_detection() {
let req = SemanticSearchEntityRequest {
query: "test query".to_string(),
entity_type: None,
confidence_floor: 0.5,
top_k: 10,
start_time: None,
end_time: None,
detect_communities: Some(true),
min_community_size: Some(3),
};
assert_eq!(req.detect_communities, Some(true));
assert_eq!(req.min_community_size, Some(3));
}
#[test]
fn test_semantic_search_edge_request() {
let req = SemanticSearchEdgeRequest {
query: "test query".to_string(),
relation_type: Some("related_to".to_string()),
top_k: 10,
start_time: None,
end_time: None,
};
assert_eq!(req.query, "test query");
}
#[test]
fn test_hybrid_search_request_defaults() {
let req = HybridSearchRequest {
query: "test".to_string(),
semantic_weight: default_semantic_weight(),
lexical_weight: default_lexical_weight(),
top_k: default_top_k(),
};
assert_eq!(req.semantic_weight, 0.6);
assert_eq!(req.lexical_weight, 0.4);
assert_eq!(req.top_k, 10);
}
#[test]
fn test_semantic_search_response() {
let response: SemanticSearchResponse<EntityResult> = SemanticSearchResponse {
query: "test".to_string(),
results: vec![],
total_count: 0,
search_time_ms: 100,
communities: None,
paths: None,
available_facets: None,
};
assert_eq!(response.query, "test");
assert_eq!(response.total_count, 0);
}
#[test]
fn test_semantic_search_with_path_finding() {
let req = SemanticSearchEntityRequest {
query: "test query".to_string(),
entity_type: None,
confidence_floor: 0.5,
top_k: 10,
start_time: None,
end_time: None,
detect_communities: None,
min_community_size: None,
find_paths: Some(true),
target_entity_id: Some("e5".to_string()),
max_path_depth: Some(5),
k_hops: None,
facet_filters: None,
discover_facets: None,
};
assert_eq!(req.find_paths, Some(true));
assert_eq!(req.target_entity_id, Some("e5".to_string()));
}
#[test]
fn test_semantic_search_with_facet_discovery() {
let req = SemanticSearchEntityRequest {
query: "kubernetes".to_string(),
entity_type: None,
confidence_floor: 0.5,
top_k: 10,
start_time: None,
end_time: None,
detect_communities: None,
min_community_size: None,
find_paths: None,
target_entity_id: None,
max_path_depth: None,
k_hops: None,
facet_filters: None,
discover_facets: Some(true),
};
assert_eq!(req.discover_facets, Some(true));
}
#[test]
fn test_semantic_search_with_facet_filters() {
let filters = FacetFilters {
entity_types: Some(vec!["concept".to_string()]),
relation_types: None,
confidence_level: Some("high".to_string()),
date_range: None,
};
let req = SemanticSearchEntityRequest {
query: "test".to_string(),
entity_type: None,
confidence_floor: 0.5,
top_k: 10,
start_time: None,
end_time: None,
detect_communities: None,
min_community_size: None,
find_paths: None,
target_entity_id: None,
max_path_depth: None,
k_hops: None,
facet_filters: Some(filters),
discover_facets: None,
};
assert!(req.facet_filters.is_some());
assert_eq!(req.facet_filters.unwrap().confidence_level, Some("high".to_string()));
}
}
+126
View File
@@ -732,3 +732,129 @@ pub async fn summarize_handler(
})
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_link_entities_request() {
let req = LinkEntitiesRequest {
project: "poimen".to_string(),
text: "Kubernetes is a container orchestrator.".to_string(),
};
assert_eq!(req.project, "poimen");
assert!(!req.text.is_empty());
}
#[test]
fn test_detect_aliases_request() {
let req = DetectAliasesRequest {
project: "poimen".to_string(),
entity_id: "e1".to_string(),
entity_name: "Kubernetes".to_string(),
text_samples: vec!["k8s is great".to_string()],
};
assert_eq!(req.entity_name, "Kubernetes");
assert_eq!(req.text_samples.len(), 1);
}
#[test]
fn test_suggest_merges_request() {
let req = SuggestMergesRequest {
project: "poimen".to_string(),
similarity_threshold: 0.85,
};
assert_eq!(req.similarity_threshold, 0.85);
}
#[test]
fn test_suggest_merges_default_threshold() {
let req = SuggestMergesRequest {
project: "poimen".to_string(),
similarity_threshold: default_merge_threshold(),
};
assert_eq!(req.similarity_threshold, 0.8);
}
#[test]
fn test_detect_coreferences_request() {
let req = DetectCoreferencesRequest {
project: "poimen".to_string(),
texts: vec![
"Kubernetes is great.".to_string(),
"k8s makes deployments easy.".to_string(),
],
};
assert_eq!(req.texts.len(), 2);
}
#[test]
fn test_link_entities_response() {
let resp = LinkEntitiesResponse {
links: vec![],
unlinked: vec![],
total_mentions: 0,
link_rate: 0.0,
process_time_ms: 100,
};
assert_eq!(resp.total_mentions, 0);
}
#[test]
fn test_detect_aliases_response() {
let resp = DetectAliasesResponse {
entity_id: "e1".to_string(),
entity_name: "Kubernetes".to_string(),
aliases: vec![],
alias_count: 0,
process_time_ms: 100,
};
assert_eq!(resp.alias_count, 0);
}
#[test]
fn test_suggest_merges_response() {
let resp = SuggestMergesResponse {
project: "poimen".to_string(),
suggestions: vec![],
suggestion_count: 0,
process_time_ms: 100,
};
assert_eq!(resp.suggestion_count, 0);
}
#[test]
fn test_detect_coreferences_response() {
let resp = DetectCoreferencesResponse {
project: "poimen".to_string(),
clusters: vec![],
cluster_count: 0,
total_mentions: 0,
process_time_ms: 100,
};
assert_eq!(resp.cluster_count, 0);
}
#[test]
fn test_link_entities_request_serialization() {
let req = LinkEntitiesRequest {
project: "test".to_string(),
text: "Kubernetes".to_string(),
};
let json = serde_json::to_string(&req).unwrap();
assert!(json.contains("test"));
}
#[test]
fn test_link_entities_response_serialization() {
let resp = LinkEntitiesResponse {
links: vec![],
unlinked: vec![],
total_mentions: 5,
link_rate: 0.8,
process_time_ms: 150,
};
let json = serde_json::to_string(&resp).unwrap();
assert!(json.contains("0.8"));
}
}
+4 -42
View File
@@ -118,28 +118,17 @@ pub async fn unified_query_handler(
body: web::Json<UnifiedQueryRequest>,
state: web::Data<AppState>,
) -> HttpResponse {
use crate::metrics::*;
QUERY_REQUESTS_TOTAL.inc();
QUERY_IN_FLIGHT.inc();
let _timer = Timer::new(&QUERY_DURATION);
let start_time = std::time::Instant::now();
// 1. Validate JWT + rate limit
if let Err(response) = crate::handlers::middleware::validate_and_rate_limit(
&req, &state, "query", 500
) {
QUERY_AUTH_FAILURES.inc();
QUERY_ERRORS_TOTAL.inc();
ERROR_AUTH_FAILURE_QUERY.inc();
QUERY_IN_FLIGHT.dec();
return response;
}
// 2. Validate input
if let Err(response) = validate_unified_request(&body) {
QUERY_ERRORS_TOTAL.inc();
ERROR_BAD_REQUEST_QUERY.inc();
QUERY_IN_FLIGHT.dec();
return response;
}
@@ -147,17 +136,9 @@ pub async fn unified_query_handler(
body.search_type, body.query, body.entity_type, body.relation_type);
// 3. Embed query once (reused for all search types)
let embed_start = std::time::Instant::now();
let query_embedding = match state.embeddings.embed_one(&body.query).await {
Ok(emb) => {
QUERY_EMBEDDING_DURATION.observe(embed_start.elapsed().as_secs_f64());
emb.to_vec()
}
Ok(emb) => emb.to_vec(),
Err(e) => {
QUERY_EMBEDDING_FAILURES.inc();
QUERY_ERRORS_TOTAL.inc();
ERROR_EMBEDDING_FAILURE_QUERY.inc();
QUERY_IN_FLIGHT.dec();
error!("Embedding failed: {}", e);
return crate::handlers::response_builder::internal_error(
"Failed to embed query"
@@ -171,15 +152,12 @@ pub async fn unified_query_handler(
"edges" => search_edges(&body, &state, &query_embedding, start_time).await,
"hybrid" => search_hybrid(&body, &state, &query_embedding, start_time).await,
_ => {
QUERY_ERRORS_TOTAL.inc();
QUERY_IN_FLIGHT.dec();
return crate::handlers::response_builder::bad_request(
"search_type must be 'entities', 'edges', or 'hybrid'"
);
}
};
QUERY_IN_FLIGHT.dec();
response
}
@@ -203,9 +181,7 @@ async fn search_entities(
).await {
Ok(r) => r,
Err(e) => {
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
error!("Unexpected error: entity search failed: {}", e);
error!("Entity search failed: {}", e);
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
}
};
@@ -271,10 +247,6 @@ async fn search_entities(
info!("Unified query (entities): {} results in {}ms", count, elapsed);
// O2: Track result counts
crate::metrics::QUERY_RESULTS_TOTAL.inc_by(count as u64);
if count == 0 { crate::metrics::QUERY_EMPTY_RESULTS.inc(); }
let response = UnifiedQueryResponse {
query: req.query.clone(),
search_type: "entities".to_string(),
@@ -307,9 +279,7 @@ async fn search_edges(
).await {
Ok(r) => r,
Err(e) => {
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
error!("Unexpected error: edge search failed: {}", e);
error!("Edge search failed: {}", e);
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
}
};
@@ -335,9 +305,6 @@ async fn search_edges(
info!("Unified query (edges): {} results in {}ms", count, elapsed);
crate::metrics::QUERY_RESULTS_TOTAL.inc_by(count as u64);
if count == 0 { crate::metrics::QUERY_EMPTY_RESULTS.inc(); }
let response = UnifiedQueryResponse {
query: req.query.clone(),
search_type: "edges".to_string(),
@@ -371,9 +338,7 @@ async fn search_hybrid(
).await {
Ok(r) => r,
Err(e) => {
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
error!("Unexpected error: hybrid search failed: {}", e);
error!("Hybrid search failed: {}", e);
return crate::handlers::response_builder::internal_error(&format!("Search failed: {}", e));
}
};
@@ -385,9 +350,6 @@ async fn search_hybrid(
info!("Unified query (hybrid): {} results in {}ms", count, elapsed);
crate::metrics::QUERY_RESULTS_TOTAL.inc_by(count as u64);
if count == 0 { crate::metrics::QUERY_EMPTY_RESULTS.inc(); }
let response = UnifiedQueryResponse {
query: req.query.clone(),
search_type: "hybrid".to_string(),
+397 -214
View File
@@ -18,7 +18,7 @@ use crate::dual_write_indexer::DualWriteIndexer;
use crate::gateway_queue_adapter::GatewayQueueAdapter;
use crate::queue_worker::{QueueWorker, QueueWorkerConfig};
use crate::queue_adapter::QueueAdapter;
// RBAC removed for MVP - will add after core ingest/query working
use crate::rbac::{AccessGuard, Claims as RbacClaims, builtin_role_provider, ResourceMeta, ResourceType, Verb, Visibility};
use crate::handlers::{
QueryParams, QueryParamsError, SearchMethod, build_search_response,
LearnParams, LearnParamsError, build_learn_response,
@@ -41,6 +41,8 @@ pub struct AppState {
pub opensearch_client: Option<Arc<OpenSearchClient>>,
/// M3.8 Query Optimizer (optional, from environment)
pub optimizer_service: Option<Arc<mem_core::optimizer::OptimizerService>>,
/// RBAC Access Guard (optional, for fine-grained access control)
pub access_guard: Option<Arc<AccessGuard>>,
}
/// Authentication mode
@@ -167,6 +169,38 @@ fn extract_rate_limit_key(claims: &JwtClaims) -> String {
claims.sub.clone()
}
/// Convert JWT claims to RBAC claims for AccessGuard
fn to_rbac_claims(jwt: &JwtClaims) -> RbacClaims {
RbacClaims::new(&jwt.sub)
.with_roles(jwt.roles.clone().unwrap_or_default().iter().map(|s| s.as_str()).collect())
.with_groups(jwt.groups.clone().unwrap_or_default().iter().map(|s| s.as_str()).collect())
.with_permissions(jwt.permissions.clone().unwrap_or_default().iter().map(|s| s.as_str()).collect())
}
/// Convert QueryResult to ResourceMeta for RBAC filtering
fn query_result_to_resource_meta(result: &crate::query_worker::QueryResult, project: &str) -> ResourceMeta {
let source = result.source.as_deref().unwrap_or("unknown");
// Determine resource type from source path
let resource_type = if source.contains("SKILL-") || source.contains("/skills/") {
ResourceType::Skill
} else if result.level == "corpus" || result.level == "R" {
ResourceType::Wiki // Reference docs are wiki-like
} else {
ResourceType::Embedding // L0, L1, L2 are learned embeddings
};
// Determine visibility - private if source path suggests it
let visibility = if source.contains("/private/") || source.contains("-private") {
Visibility::Private
} else {
Visibility::Public
};
ResourceMeta::new(source, resource_type, project)
.with_visibility(visibility)
}
/// Rate limit guard — call this in handlers to check rate limit
fn check_rate_limit(claims: &JwtClaims, state: &AppState, endpoint: &str) -> Result<(), HttpResponse> {
let key = extract_rate_limit_key(claims);
@@ -194,13 +228,8 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
tracing::info!("Connected to database");
// Initialize schema
match init_schema(&pool).await {
Ok(_) => tracing::info!("Schema initialized"),
Err(e) => {
tracing::warn!("Schema init error (may be non-fatal): {}", e);
// Continue anyway - tables might exist
}
}
init_schema(&pool).await?;
tracing::info!("Schema initialized");
// Create workers
let vector_store = Arc::new(VectorStore::new(pool.clone()));
@@ -357,6 +386,12 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
tracing::info!("M8.2 Queue Worker started (background task)");
}
// Initialize RBAC AccessGuard with built-in roles
let access_guard = {
let role_provider = Arc::new(builtin_role_provider());
Some(Arc::new(AccessGuard::new(role_provider)))
};
let state = web::Data::new(AppState {
api_key,
start_time: Instant::now(),
@@ -371,42 +406,16 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
auth_mode,
opensearch_client,
optimizer_service,
access_guard,
});
tracing::info!("Starting HTTP server on port {}", port);
// O5/O7/O9: Background stats collector (every 60s)
{
let stats_pool = state.get_ref().pool.clone();
tokio::spawn(async move {
let mut interval = tokio::time::interval(std::time::Duration::from_secs(60));
loop {
interval.tick().await;
// O5: Table row counts
if let Ok(row) = sqlx::query_as::<_, (i64,)>("SELECT COUNT(*) FROM memory_entity")
.fetch_one(&stats_pool).await {
crate::metrics::DB_TABLE_ENTITY_ROWS.set(row.0 as u64);
}
if let Ok(row) = sqlx::query_as::<_, (i64,)>("SELECT COUNT(*) FROM memory_edge")
.fetch_one(&stats_pool).await {
crate::metrics::DB_TABLE_EDGE_ROWS.set(row.0 as u64);
}
// O9: Pool stats
crate::metrics::DB_POOL_SIZE.set(stats_pool.size() as u64);
crate::metrics::DB_POOL_IDLE.set(stats_pool.num_idle() as u64);
}
});
}
tracing::info!("Creating HttpServer instance...");
let server = HttpServer::new(move || {
tracing::debug!("HttpServer::new() closure executing");
HttpServer::new(move || {
App::new()
.app_data(state.clone())
.wrap(Logger::default())
.route("/health", web::get().to(health_check))
.route("/metrics", web::get().to(crate::metrics::metrics_handler))
.route("/memory/ingest", web::post().to(ingest_handler))
.route("/memory/ingest/{ingest_id}", web::get().to(ingest_status))
.route("/memory/query", web::get().to(query_handler))
@@ -440,40 +449,17 @@ pub async fn start_server(port: u16, api_key: String, database_url: &str) -> Res
.route("/agents/{id}", web::put().to(crate::handlers::agent_handler::update_agent_handler))
.route("/agents/{id}", web::delete().to(crate::handlers::agent_handler::delete_agent_handler))
.route("/agents/{id}/metrics", web::get().to(crate::handlers::agent_handler::get_agent_metrics_handler))
.route("/memory/agents/{project_id}/prompts", web::post().to(crate::handlers::agent_handler::create_prompt_handler))
.route("/memory/agents/{project_id}/roles", web::post().to(crate::handlers::agent_handler::map_role_to_prompt_handler))
.route("/memory/agents/{project_id}/roles/{role_name}/prompts", web::get().to(crate::handlers::agent_handler::get_role_prompts_handler))
});
tracing::info!("HttpServer instance created, binding to 0.0.0.0:{}", port);
let server = server.bind(("0.0.0.0", port))?;
tracing::info!("Successfully bound to port {}, about to run", port);
server.run().await?;
})
.bind(("0.0.0.0", port))?
.run()
.await?;
Ok(())
}
/// Health check (no auth)
pub async fn health_check(state: web::Data<AppState>) -> HttpResponse {
use crate::metrics::*;
HEALTH_CHECKS_TOTAL.inc();
let uptime = state.start_time.elapsed().as_secs();
APP_UPTIME_SECONDS.set(uptime);
// O7: Check DB dependency
let db_start = std::time::Instant::now();
match sqlx::query("SELECT 1").execute(&state.pool).await {
Ok(_) => {
DEP_DB_UP.set(1);
DEP_DB_LATENCY.observe(db_start.elapsed().as_secs_f64());
}
Err(_) => {
DEP_DB_UP.set(0);
HEALTH_CHECK_FAILURES.inc();
}
}
HttpResponse::Ok().json(json!({"status": "ok", "uptime_seconds": uptime}))
}
@@ -483,75 +469,63 @@ pub async fn ingest_handler(
body: web::Json<IngestRequest>,
state: web::Data<AppState>,
) -> HttpResponse {
use crate::metrics::*;
INGEST_REQUESTS_TOTAL.inc();
INGEST_IN_FLIGHT.inc();
let _timer = Timer::new(&INGEST_DURATION);
// Auth + capability check
let (claims, _token) = match validate_auth(&req, &state).await {
Ok(c) => c,
Err(e) => {
INGEST_AUTH_FAILURES.inc();
INGEST_ERRORS_TOTAL.inc();
ERROR_AUTH_FAILURE_INGEST.inc();
INGEST_IN_FLIGHT.dec();
return e;
}
Err(e) => return e,
};
let user_id = &claims.sub;
if !has_capability(&claims, "memory:write") {
INGEST_AUTH_FAILURES.inc();
INGEST_ERRORS_TOTAL.inc();
ERROR_FORBIDDEN_INGEST.inc();
INGEST_IN_FLIGHT.dec();
return HttpResponse::Forbidden().json(json!({
"error": "forbidden",
"reason": "missing capability: memory:write"
}));
}
if let Err(e) = check_rate_limit(&claims, &state, "/memory/ingest") {
INGEST_RATE_LIMITED.inc();
ERROR_RATE_LIMITED_INGEST.inc();
INGEST_IN_FLIGHT.dec();
return e;
}
// RBAC: Check project-level write access
if let Err(e) = check_project_write_access(&state, &claims, &body.project).await {
return e;
}
// Check idempotency
if let Some(cached) = state.idempotency_store.get(&body.ingest_id) {
tracing::info!("Returning cached response for ingest_id: {}", body.ingest_id);
INGEST_DUPLICATES_TOTAL.inc();
INGEST_IN_FLIGHT.dec();
return HttpResponse::Accepted().json(cached);
}
let byte_count: usize = body.records.iter().map(|r| r.text.len()).sum();
INGEST_BYTES_TOTAL.inc_by(byte_count as u64);
INGEST_RECORDS_TOTAL.inc_by(body.records.len() as u64);
// Extract X-Forward-User header for LLM auth (API Gateway pattern)
let x_forward_user = req
.headers()
.get("X-Forward-User")
.and_then(|h| h.to_str().ok())
.map(|s| s.to_string());
if let Some(ref user) = x_forward_user {
tracing::info!("Ingest request with X-Forward-User: {}", user);
}
// Execute ingest
let resp = execute_ingest(&state, &body, x_forward_user).await;
INGEST_IN_FLIGHT.dec();
resp
execute_ingest(&state, &body).await
}
/// Check RBAC project write access
async fn check_project_write_access(
state: &web::Data<AppState>,
claims: &JwtClaims,
project: &str,
) -> Result<(), HttpResponse> {
let Some(guard) = &state.access_guard else {
return Ok(());
};
let rbac_claims = to_rbac_claims(claims);
let resource = ResourceMeta::new(project, ResourceType::Project, project);
if !guard.can_write(&rbac_claims, &resource).await {
tracing::warn!("RBAC denied write access to project '{}' for user '{}'", project, claims.sub);
return Err(HttpResponse::Forbidden().json(json!({
"error": "forbidden",
"reason": format!("write access denied to project '{}'", project)
})));
}
Ok(())
}
/// Execute ingest job creation and spawn worker
async fn execute_ingest(
state: &web::Data<AppState>,
body: &IngestRequest,
x_forward_user: Option<String>,
) -> HttpResponse {
let records: Vec<(String, String)> = body.records
.iter()
@@ -582,9 +556,8 @@ async fn execute_ingest(
let worker = state.ingest_worker.clone();
let project = body.project.clone();
let ingest_id = body.ingest_id.clone();
let x_fwd = x_forward_user.clone();
tokio::spawn(async move {
if let Err(e) = worker.process_ingest_with_auth(&project, &ingest_id, records, x_fwd).await {
if let Err(e) = worker.process_ingest(&project, &ingest_id, records).await {
tracing::error!("Ingest failed: {}", e);
}
});
@@ -597,9 +570,7 @@ async fn execute_ingest(
HttpResponse::Accepted().json(response)
}
Err(e) => {
crate::metrics::ERROR_UNEXPECTED_INGEST.inc();
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
tracing::error!(user_id = body.project.as_str(), "Unexpected DB error during ingest: {}", e);
tracing::error!("DB error: {}", e);
HttpResponse::InternalServerError().json(json!({"error": "database_error"}))
}
}
@@ -739,6 +710,11 @@ pub async fn learn_handler(
Err(e) => return e.to_response(),
};
// RBAC: Check project-level write access
if let Err(e) = check_project_write_access(&state, &claims, &params.project).await {
return e;
}
// Chunk the markdown
let chunks = chunk_markdown_text(&params.text, params.chunk_size);
if chunks.is_empty() {
@@ -856,13 +832,8 @@ async fn store_compacted_memory(
.await;
match result {
Ok(_) => {
crate::metrics::WRITE_CHUNKS_TOTAL.inc();
crate::metrics::WRITE_BYTES_TOTAL.inc_by(memory.len() as u64);
true
}
Ok(_) => true,
Err(e) => {
crate::metrics::WRITE_ERRORS_TOTAL.inc();
tracing::error!("Failed to store compacted memory: {}", e);
false
}
@@ -899,7 +870,7 @@ pub async fn query_handler(
state: web::Data<AppState>,
) -> HttpResponse {
// Auth + capability check
let (claims, _token) = match validate_auth(&req, &state).await {
let (claims, token) = match validate_auth(&req, &state).await {
Ok(c) => c,
Err(e) => return e,
};
@@ -919,18 +890,63 @@ pub async fn query_handler(
Err(e) => return e.to_response(),
};
// Execute temporal graph query
match query_temporal_graph(&state, &params).await {
Ok(response) => HttpResponse::Ok().json(response),
// Execute semantic search
let mut results = match state.query_worker.query(&params.project, &params.question, Some(50)).await {
Ok(r) => r,
Err(e) => {
crate::metrics::ERROR_UNEXPECTED_QUERY.inc();
crate::metrics::ERROR_UNEXPECTED_TOTAL.inc();
tracing::error!(user_id = claims.sub.as_str(), "Unexpected error: temporal graph query failed: {}", e);
HttpResponse::InternalServerError().json(json!({"error": "query_failed", "reason": e.to_string()}))
tracing::error!("Semantic search failed: {}", e);
return HttpResponse::InternalServerError().json(json!({"error": "semantic_search_failed"}));
}
};
// M3.8: Optimize results
results = optimize_search_results(results, state.optimizer_service.as_ref()).await;
// RBAC: Filter by access control
results = apply_rbac_filter(&state, &claims, results, &params.project).await;
// Route by search method
match params.method {
SearchMethod::Semantic => build_search_response(&params, results, None),
SearchMethod::Hybrid => execute_hybrid_search(&state, &params, results, &token).await,
}
}
/// Apply RBAC filtering to search results
async fn apply_rbac_filter(
state: &web::Data<AppState>,
claims: &JwtClaims,
results: Vec<crate::query_worker::QueryResult>,
project: &str,
) -> Vec<crate::query_worker::QueryResult> {
let Some(guard) = &state.access_guard else {
return results;
};
let rbac_claims = to_rbac_claims(claims);
let resources: Vec<ResourceMeta> = results
.iter()
.map(|r| query_result_to_resource_meta(r, project))
.collect();
let decisions = guard.check_access_batch(&rbac_claims, &resources, Verb::Read).await;
let filtered: Vec<_> = results
.into_iter()
.zip(decisions.iter())
.filter(|(_, d)| d.is_allowed())
.map(|(r, _)| r)
.collect();
tracing::debug!(
"RBAC filtered {} results for user {}",
decisions.iter().filter(|d| d.is_denied()).count(),
claims.sub
);
filtered
}
/// Execute hybrid search with OpenSearch fallback
async fn execute_hybrid_search(
state: &web::Data<AppState>,
@@ -996,7 +1012,22 @@ pub async fn projects_handler(
match result {
Ok(rows) => {
let projects: Vec<String> = rows.into_iter().map(|(p,)| p).collect();
let mut projects: Vec<String> = rows.into_iter().map(|(p,)| p).collect();
// RBAC: Filter projects by access
if let Some(guard) = &state.access_guard {
let rbac_claims = to_rbac_claims(&claims);
let mut allowed_projects = Vec::new();
for project in projects {
let resource = ResourceMeta::new(&project, ResourceType::Project, &project);
if guard.can_read(&rbac_claims, &resource).await {
allowed_projects.push(project);
}
}
projects = allowed_projects;
}
HttpResponse::Ok().json(json!({
"projects": projects,
"count": projects.len()
@@ -1061,23 +1092,13 @@ pub async fn context_handler(
body: web::Json<crate::context_endpoint::ContextRequest>,
state: web::Data<AppState>,
) -> HttpResponse {
use crate::metrics::*;
CONTEXT_REQUESTS_TOTAL.inc();
let _timer = Timer::new(&CONTEXT_DURATION);
let (claims, _token) = match validate_auth(&req, &state).await {
Ok(c) => c,
Err(e) => {
CONTEXT_ERRORS_TOTAL.inc();
ERROR_AUTH_FAILURE_CONTEXT.inc();
return e;
}
Err(e) => return e,
};
let user_id = &claims.sub;
// Check read capability
if !has_capability(&claims, "memory:read") {
CONTEXT_ERRORS_TOTAL.inc();
ERROR_FORBIDDEN_CONTEXT.inc();
return HttpResponse::Forbidden().json(json!({
"error": "forbidden",
"reason": "missing capability: memory:read"
@@ -1092,6 +1113,23 @@ pub async fn context_handler(
let scope = body.scope.clone().unwrap_or_else(|| "project".to_string());
let budget = body.budget.unwrap_or(6000);
// RBAC: Check project-level access
if let Some(guard) = &state.access_guard {
let rbac_claims = to_rbac_claims(&claims);
let project_resource = ResourceMeta::new(&project, ResourceType::Project, &project);
if !guard.can_read(&rbac_claims, &project_resource).await {
tracing::warn!(
"RBAC denied access to project '{}' for user '{}'",
project, claims.sub
);
return HttpResponse::Forbidden().json(json!({
"error": "forbidden",
"reason": format!("access denied to project '{}'", project)
}));
}
}
let lookup = crate::context_endpoint::ContextLookup::new(budget, project, scope);
match lookup.lookup(body.into_inner()).await {
@@ -1102,14 +1140,9 @@ pub async fn context_handler(
skills = response.skills.len(),
"context lookup successful"
);
// O3: Track tier hits
let total = response.lessons.len() + response.skills.len();
if total == 0 { CONTEXT_EMPTY_RESULTS.inc(); }
HttpResponse::Ok().json(response)
}
Err(e) => {
CONTEXT_ERRORS_TOTAL.inc();
ERROR_LOOKUP_FAILURE_CONTEXT.inc();
tracing::error!("context lookup error: {}", e);
HttpResponse::BadRequest().json(json!({
"error": "lookup_failed",
@@ -1443,76 +1476,226 @@ pub async fn vault_file_handler(
}
}
/// Query temporal knowledge graph
/// 1. Find entities via semantic search
/// 2. Traverse edges from entities
/// 3. Apply temporal filtering (t_valid/t_invalid)
/// 4. Return graph with confidence scores
async fn query_temporal_graph(
state: &web::Data<AppState>,
params: &QueryParams,
) -> anyhow::Result<serde_json::Value> {
// Step 1: Find entities (order by name for deterministic results)
let entities_rows: Vec<(String, String, String)> = sqlx::query_as(
"SELECT id, name, entity_type FROM memory_entity WHERE project_id = $1 LIMIT $2"
)
.bind(&params.project)
.bind(params.limit as i32)
.fetch_all(&state.pool)
.await
.unwrap_or_default();
// Step 2: Traverse edges from found entities
// NOTE: Edges will be empty until temporal schema is migrated
let mut edges_data: Vec<(String, String, String, String, String, f32)> = Vec::new();
// Try to fetch edges (will be empty if schema not migrated yet)
for (entity_id, _name, _type_str) in &entities_rows {
let entity_edges: Vec<(String, String, String, String, f32, Option<chrono::DateTime<chrono::Utc>>, Option<chrono::DateTime<chrono::Utc>>)> =
sqlx::query_as(
"SELECT id, target_entity_id, relation_type, fact, confidence, t_valid, t_invalid FROM memory_edge WHERE project_id = $1 AND source_entity_id = $2"
)
.bind(&params.project)
.bind(entity_id)
.fetch_all(&state.pool)
.await
.unwrap_or_default(); // Returns empty vec if table schema doesn't match
for (id, target, rel, fact, conf, t_valid, t_invalid) in entity_edges {
// Apply temporal filtering
let now = chrono::Utc::now();
let valid = t_valid.as_ref().map(|t| *t <= now).unwrap_or(true);
let not_invalid = t_invalid.as_ref().map(|t| *t > now).unwrap_or(true);
if valid && not_invalid {
edges_data.push((id, entity_id.clone(), target, rel, fact, conf));
}
}
}
// Step 4: Build response
let response = json!({
"query": params.question,
"project": params.project,
"entities": entities_rows.iter().map(|(id, name, etype)| json!({
"id": id,
"name": name,
"type": etype
})).collect::<Vec<_>>(),
"edges": edges_data.iter().map(|(id, src, tgt, rel, fact, conf)| json!({
"id": id,
"source": src,
"target": tgt,
"relation": rel,
"fact": fact,
"confidence": conf
})).collect::<Vec<_>>(),
"count": json!({
"entities": entities_rows.len(),
"edges": edges_data.len()
})
});
Ok(response)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_to_rbac_claims_with_roles() {
let jwt = JwtClaims {
sub: "alice".to_string(),
iss: "authentik".to_string(),
aud: "memory".to_string(),
exp: i64::MAX,
iat: 0,
nbf: None,
permissions: Some(vec!["memory:read".to_string()]),
groups: Some(vec!["engineering".to_string()]),
roles: Some(vec!["authenticated-user".to_string(), "homelab-team".to_string()]),
};
let rbac = to_rbac_claims(&jwt);
assert_eq!(rbac.sub, "alice");
assert!(rbac.has_role("authenticated-user"));
assert!(rbac.has_role("homelab-team"));
assert!(!rbac.has_role("admin"));
}
#[test]
fn test_to_rbac_claims_basic() {
let jwt = JwtClaims {
sub: "alice".to_string(),
iss: "test".to_string(),
aud: "memory".to_string(),
exp: i64::MAX,
iat: 0,
nbf: None,
permissions: Some(vec!["memory:read".to_string(), "memory:write".to_string()]),
groups: Some(vec!["engineering".to_string(), "ml-team".to_string()]),
roles: Some(vec!["authenticated-user".to_string()]),
};
let rbac = to_rbac_claims(&jwt);
assert_eq!(rbac.sub, "alice");
assert!(rbac.in_group("engineering"));
assert!(rbac.in_group("ml-team"));
assert!(rbac.has_permission("memory:read"));
assert!(rbac.has_permission("memory:write"));
}
#[test]
fn test_to_rbac_claims_empty() {
let jwt = JwtClaims {
sub: "anonymous".to_string(),
iss: "test".to_string(),
aud: "memory".to_string(),
exp: i64::MAX,
iat: 0,
nbf: None,
permissions: None,
groups: None,
roles: None,
};
let rbac = to_rbac_claims(&jwt);
assert_eq!(rbac.sub, "anonymous");
assert!(!rbac.in_group("any"));
assert!(!rbac.has_permission("any"));
}
#[test]
fn test_query_result_to_resource_meta_wiki() {
let result = crate::query_worker::QueryResult {
level: "corpus".to_string(),
score: 0.9,
text: "Some wiki content".to_string(),
source: Some("docs/kubernetes.md".to_string()),
provenance: vec![],
};
let meta = query_result_to_resource_meta(&result, "homelab");
assert_eq!(meta.resource_type, ResourceType::Wiki);
assert_eq!(meta.project, "homelab");
assert_eq!(meta.visibility, Visibility::Public);
}
#[test]
fn test_query_result_to_resource_meta_skill() {
let result = crate::query_worker::QueryResult {
level: "L1".to_string(),
score: 0.8,
text: "Skill content".to_string(),
source: Some("shared/skills/SKILL-debug/SKILL.md".to_string()),
provenance: vec![],
};
let meta = query_result_to_resource_meta(&result, "homelab");
assert_eq!(meta.resource_type, ResourceType::Skill);
}
#[test]
fn test_query_result_to_resource_meta_private() {
let result = crate::query_worker::QueryResult {
level: "L2".to_string(),
score: 0.7,
text: "Private content".to_string(),
source: Some("docs/private/secrets.md".to_string()),
provenance: vec![],
};
let meta = query_result_to_resource_meta(&result, "homelab");
assert_eq!(meta.visibility, Visibility::Private);
}
#[test]
fn test_query_result_to_resource_meta_embedding() {
let result = crate::query_worker::QueryResult {
level: "L1".to_string(),
score: 0.85,
text: "Learned fact".to_string(),
source: Some("memory-123".to_string()),
provenance: vec![],
};
let meta = query_result_to_resource_meta(&result, "portfolio");
assert_eq!(meta.resource_type, ResourceType::Embedding);
assert_eq!(meta.project, "portfolio");
}
#[tokio::test]
async fn test_rbac_integration_admin_access() {
use std::sync::Arc;
use crate::rbac::{builtin_role_provider, AccessGuard};
let guard = AccessGuard::new(Arc::new(builtin_role_provider()));
// Admin JWT with roles from Authentik
let jwt = JwtClaims {
sub: "admin-user".to_string(),
iss: "test".to_string(),
aud: "memory".to_string(),
exp: i64::MAX,
iat: 0,
nbf: None,
permissions: Some(vec!["*".to_string()]),
groups: None,
roles: Some(vec!["admin".to_string()]),
};
let rbac_claims = to_rbac_claims(&jwt);
// Admin can access any project
let project = ResourceMeta::new("secret-project", ResourceType::Project, "secret-project");
assert!(guard.can_read(&rbac_claims, &project).await);
assert!(guard.can_write(&rbac_claims, &project).await);
}
#[tokio::test]
async fn test_rbac_integration_portfolio_agent() {
use std::sync::Arc;
use crate::rbac::{builtin_role_provider, AccessGuard};
let guard = AccessGuard::new(Arc::new(builtin_role_provider()));
// Portfolio agent JWT with roles from Authentik
let jwt = JwtClaims {
sub: "visitor-123".to_string(),
iss: "test".to_string(),
aud: "memory".to_string(),
exp: i64::MAX,
iat: 0,
nbf: None,
permissions: Some(vec!["memory:read".to_string()]),
groups: None,
roles: Some(vec!["portfolio-agent".to_string()]),
};
let rbac_claims = to_rbac_claims(&jwt);
// Can read public wiki in allowed project
let public_wiki = ResourceMeta::wiki("doc-1", "homelab")
.with_visibility(Visibility::Public);
assert!(guard.can_read(&rbac_claims, &public_wiki).await);
// Cannot read private wiki
let private_wiki = ResourceMeta::wiki("secret", "homelab")
.with_visibility(Visibility::Private);
assert!(!guard.can_read(&rbac_claims, &private_wiki).await);
// Cannot write to any project
let project = ResourceMeta::new("homelab", ResourceType::Project, "homelab");
assert!(!guard.can_write(&rbac_claims, &project).await);
}
#[tokio::test]
async fn test_rbac_integration_no_role() {
use std::sync::Arc;
use crate::rbac::{builtin_role_provider, AccessGuard};
let guard = AccessGuard::new(Arc::new(builtin_role_provider()));
// JWT with no roles (anonymous user)
let jwt = JwtClaims {
sub: "anonymous".to_string(),
iss: "test".to_string(),
aud: "memory".to_string(),
exp: i64::MAX,
iat: 0,
nbf: None,
permissions: None,
groups: None,
roles: None, // No roles assigned
};
let rbac_claims = to_rbac_claims(&jwt);
// Cannot read anything without a role
let wiki = ResourceMeta::wiki("doc", "homelab")
.with_visibility(Visibility::Public);
assert!(!guard.can_read(&rbac_claims, &wiki).await);
}
}
+43 -297
View File
@@ -1,268 +1,96 @@
use anyhow::Result;
use mem_store::{MemoryL1, VectorStore, ChunkL0, EntityRepoOps, EdgeRepoOps};
use mem_store::{MemoryL1, VectorStore, ChunkL0};
use mem_llm::EmbeddingsClient;
use mem_ingest::ingest_pipeline::{IngestPipeline, Episode};
use mem_ingest::entity_extractor::{WikiLinkFallbackExtractor, LlmEntityExtractor};
use mem_ingest::fact_extractor::{SimpleFactExtractor, LlmFactExtractor};
use mem_ingest::contradiction_detector::ContradictionHandler;
use sqlx::PgPool;
use uuid::Uuid;
use std::sync::Arc;
use pgvector::Vector;
/// Ingest worker — processes queued records through entity/fact extraction pipeline
/// Ingest worker — processes queued records through memory storage
pub struct IngestWorker {
pool: PgPool,
vector_store: Arc<VectorStore>,
embeddings: Arc<EmbeddingsClient>,
pipeline: Arc<IngestPipeline>,
}
impl IngestWorker {
/// Create worker with full ingest pipeline
/// Create worker
pub fn new(
pool: PgPool,
embeddings: EmbeddingsClient,
) -> Self {
let vector_store = Arc::new(VectorStore::new(pool.clone()));
// Initialize extraction pipeline — use LLM if LLM_ENDPOINT is set, else fallback to wiki links
let entity_extractor: Arc<dyn mem_ingest::entity_extractor::EntityExtractor> =
if std::env::var("LLM_ENDPOINT").is_ok() {
let model = std::env::var("LLM_MODEL").unwrap_or_else(|_| "qwen2.5:3b-instruct".to_string());
tracing::info!("Using LLM entity extractor: model={}", model);
Arc::new(LlmEntityExtractor::new(&model))
} else {
tracing::info!("LLM_ENDPOINT not set, using WikiLink fallback extractor");
Arc::new(WikiLinkFallbackExtractor)
};
let fact_extractor: Arc<dyn mem_ingest::fact_extractor::FactExtractor> =
if std::env::var("LLM_ENDPOINT").is_ok() {
let model = std::env::var("LLM_MODEL").unwrap_or_else(|_| "qwen2.5:3b-instruct".to_string());
tracing::info!("Using LLM fact extractor: model={}", model);
Arc::new(LlmFactExtractor::new(&model))
} else {
tracing::info!("LLM_ENDPOINT not set, using simple pattern fact extractor");
Arc::new(SimpleFactExtractor)
};
let contradiction_detector = Arc::new(ContradictionHandler::default());
let pipeline = Arc::new(IngestPipeline::new(
entity_extractor,
fact_extractor,
contradiction_detector,
));
Self {
pool,
vector_store,
embeddings: Arc::new(embeddings),
pipeline,
}
}
/// Process ingest job: records -> entities/facts/edges via pipeline -> temporal storage
/// Process ingest job: records -> chunks -> storage
pub async fn process_ingest(
&self,
project: &str,
ingest_id: &str,
records: Vec<(String, String)>, // (content, source)
) -> Result<()> {
self.process_ingest_with_auth(project, ingest_id, records, None).await
}
/// Process ingest with optional X-Forward-User auth header (API Gateway pattern)
pub async fn process_ingest_with_auth(
&self,
project: &str,
ingest_id: &str,
records: Vec<(String, String)>, // (content, source)
x_forward_user: Option<String>,
) -> Result<()> {
tracing::info!(
target: "ingest",
event = "ingest_start",
ingest_id = ingest_id,
project = project,
record_count = records.len(),
"Starting ingest job"
);
tracing::info!("Processing ingest: project={}, id={}, records={}", project, ingest_id, records.len());
// Update job status to processing
if let Err(e) = sqlx::query("UPDATE ingest_jobs SET status=$1, started_at=NOW() WHERE ingest_id=$2")
sqlx::query("UPDATE ingest_jobs SET status=$1, started_at=NOW() WHERE ingest_id=$2")
.bind("processing")
.bind(ingest_id)
.execute(&self.pool)
.await
{
tracing::error!(
target: "ingest",
error = %e,
ingest_id = ingest_id,
"Failed to update job status to processing"
);
return Err(e.into());
}
.await?;
let mut total_entities = 0;
let mut total_edges = 0;
let mut total_reviews = 0;
let mut extraction_errors = Vec::new();
let mut save_errors = Vec::new();
let mut total_chunks = 0;
let mut total_stored = 0;
// Process each record through the ingest pipeline
for (idx, (content, source)) in records.iter().enumerate() {
let record_id = format!("{}-{}", ingest_id, idx);
tracing::debug!(
target: "ingest",
record_id = %record_id,
source = source,
content_len = content.len(),
"Processing record"
);
// Create episode from record
let episode = Episode {
id: record_id.clone(),
project_id: project.to_string(),
text: content.clone(),
wiki_links: extract_wiki_links(content),
// Process each record
for (content, source) in &records {
let chunk_id = Uuid::new_v4();
// Store L0 chunk
let l0_chunk = ChunkL0 {
id: chunk_id,
project: project.to_string(),
query_id: "ingest".to_string(),
source: source.clone(),
content: content.clone(),
tokens: (content.len() / 4) as i32,
};
self.vector_store.store_chunk_l0(&l0_chunk).await?;
total_chunks += 1;
total_stored += 1;
// Run extraction pipeline (entity + fact extraction + contradiction detection)
let x_forward_user_ref = x_forward_user.as_deref();
match self.pipeline.ingest_with_auth(&episode, x_forward_user_ref).await {
Ok(result) => {
tracing::debug!(
target: "ingest",
record_id = %record_id,
entity_count = result.entities.len(),
edge_count = result.edges.len(),
review_count = result.reviews.len(),
"Pipeline extraction successful"
);
// Try to embed and create a basic L1 memory
if let Ok(embedding) = self.embeddings.embed_one(content).await {
let l1 = MemoryL1 {
id: Uuid::new_v4(),
project: project.to_string(),
query_id: "ingest".to_string(),
content: content.clone(),
tokens: (content.len() / 4) as i32,
embedding: Some(embedding.to_vec()),
chunks_seen: 1,
chunks_used: 1,
run_id: ingest_id.to_string(),
};
// Save entities to database (normally via EntityRepo, using direct SQL for now)
for entity in &result.entities {
match save_entity_to_db(&self.pool, entity).await {
Ok(_) => {
tracing::debug!(
target: "ingest",
record_id = %record_id,
entity_name = &entity.name,
entity_type = entity.entity_type.as_str(),
"Saved entity"
);
total_entities += 1;
}
Err(e) => {
let msg = format!("Failed to save entity '{}': {}", entity.name, e);
tracing::warn!(
target: "ingest",
error = %e,
record_id = %record_id,
entity_name = &entity.name,
"Entity save failed"
);
save_errors.push(msg);
}
}
}
// Save edges to database (normally via EdgeRepo, using direct SQL for now)
for edge in &result.edges {
match save_edge_to_db(&self.pool, edge).await {
Ok(_) => {
tracing::debug!(
target: "ingest",
record_id = %record_id,
relation_type = &edge.relation_type,
"Saved edge"
);
total_edges += 1;
}
Err(e) => {
let msg = format!("Failed to save edge: {}", e);
tracing::warn!(
target: "ingest",
error = %e,
record_id = %record_id,
"Edge save failed"
);
save_errors.push(msg);
}
}
}
total_reviews += result.reviews.len();
}
Err(e) => {
let msg = format!("Record {}: {}", record_id, e);
tracing::error!(
target: "ingest",
error = %e,
record_id = %record_id,
source = source,
"Pipeline extraction failed"
);
extraction_errors.push(msg);
// Continue processing other records
if let Err(e) = self.vector_store.store_memory_l1(&l1, &embedding).await {
tracing::warn!("Failed to store L1 memory: {}", e);
}
}
}
// Mark job complete
let final_status = if extraction_errors.is_empty() && save_errors.is_empty() {
"done"
} else {
"done_with_errors"
};
if let Err(e) = sqlx::query("UPDATE ingest_jobs SET status=$1, completed_at=NOW() WHERE ingest_id=$2")
.bind(final_status)
sqlx::query("UPDATE ingest_jobs SET status=$1, completed_at=NOW() WHERE ingest_id=$2")
.bind("done")
.bind(ingest_id)
.execute(&self.pool)
.await
{
tracing::error!(
target: "ingest",
error = %e,
ingest_id = ingest_id,
"Failed to update job completion status"
);
}
tracing::info!(
target: "ingest",
event = "ingest_complete",
ingest_id = ingest_id,
project = project,
entities = total_entities,
edges = total_edges,
reviews = total_reviews,
extraction_errors = extraction_errors.len(),
save_errors = save_errors.len(),
status = final_status,
"Ingest job completed"
);
if !extraction_errors.is_empty() {
tracing::warn!(
target: "ingest",
errors = ?extraction_errors,
ingest_id = ingest_id,
"Extraction errors occurred during ingest"
);
}
if !save_errors.is_empty() {
tracing::warn!(
target: "ingest",
errors = ?save_errors,
ingest_id = ingest_id,
"Save errors occurred during ingest"
);
}
.await?;
tracing::info!("Ingest completed: {} (stored {} chunks)", ingest_id, total_stored);
Ok(())
}
@@ -281,85 +109,3 @@ impl IngestWorker {
Ok(())
}
}
/// Extract wiki links from text (e.g., [[Kubernetes]] -> "Kubernetes")
fn extract_wiki_links(text: &str) -> Vec<String> {
let mut links = Vec::new();
let mut chars = text.chars().peekable();
while let Some(ch) = chars.next() {
if ch == '[' && chars.peek() == Some(&'[') {
chars.next(); // consume second '['
let mut link = String::new();
while let Some(c) = chars.next() {
if c == ']' && chars.peek() == Some(&']') {
chars.next(); // consume second ']'
links.push(link);
break;
}
link.push(c);
}
}
}
links
}
/// Save entity to database via raw SQL (normally would use EntityRepo trait)
async fn save_entity_to_db(pool: &PgPool, entity: &mem_core::entity::Entity) -> Result<()> {
// Convert OffsetDateTime to PostgreSQL timestamp format
let t_created_str = entity.t_created.to_string();
sqlx::query(
"INSERT INTO memory_entity (id, project_id, name, entity_type, description, t_created, t_updated, confidence)
VALUES ($1, $2, $3, $4, $5, $6::TIMESTAMPTZ, $7::TIMESTAMPTZ, $8)
ON CONFLICT (project_id, name) DO UPDATE SET
entity_type = EXCLUDED.entity_type,
description = COALESCE(NULLIF(EXCLUDED.description, ''), memory_entity.description),
t_updated = NOW(),
confidence = GREATEST(memory_entity.confidence, EXCLUDED.confidence),
source_count = memory_entity.source_count + 1"
)
.bind(&entity.id)
.bind(&entity.project_id)
.bind(&entity.name)
.bind(entity.entity_type.as_str())
.bind(entity.summary.as_deref())
.bind(&t_created_str)
.bind(&t_created_str)
.bind(1.0_f32) // default confidence
.execute(pool)
.await?;
Ok(())
}
/// Save edge to database via raw SQL (normally would use EdgeRepo trait)
/// NOTE: Production DB may have old schema. Gracefully skip if temporal columns missing.
async fn save_edge_to_db(pool: &PgPool, edge: &mem_core::edge::Edge) -> Result<()> {
// Try temporal schema first (id, project_id, source_entity_id, etc)
let result = sqlx::query(
"INSERT INTO memory_edge (id, project_id, source_id, target_id, relation_type, fact, t_valid, t_invalid, t_created, confidence)
VALUES ($1, $2, $3, $4, $5, $6, $7::TIMESTAMPTZ, $8::TIMESTAMPTZ, $9::TIMESTAMPTZ, $10)
ON CONFLICT (id) DO NOTHING"
)
.bind(&edge.id)
.bind(&edge.project_id)
.bind(&edge.source_entity_id)
.bind(&edge.target_entity_id)
.bind(&edge.relation_type)
.bind(&edge.fact)
.bind(edge.t_valid.map(|t| t.to_string()))
.bind(edge.t_invalid.map(|t| t.to_string()))
.bind(edge.t_created.to_string())
.bind(edge.confidence)
.execute(pool)
.await;
match result {
Ok(_) => Ok(()),
Err(e) => {
tracing::debug!("Temporal edge schema not available: {}. Skipping edge save (will be available after schema migration).", e);
// This is expected if production DB hasn't migrated to temporal schema yet
Ok(())
}
}
}
-3
View File
@@ -1,9 +1,6 @@
pub mod endpoints;
pub mod handlers;
pub mod http_server;
pub mod metrics;
pub mod metrics_snapshot;
pub mod relevance_judge;
pub mod query;
pub mod auth;
pub mod ingest_worker;
-686
View File
@@ -1,686 +0,0 @@
//! Prometheus metrics module (O10)
//!
//! Centralized metrics registry for poimen-memory observability.
//! All handlers instrument via these shared metrics.
//! Exposed at GET /metrics in Prometheus text format.
use once_cell::sync::Lazy;
use std::sync::atomic::{AtomicU64, Ordering};
use std::collections::HashMap;
use std::sync::Mutex;
use std::time::Instant;
// ─── Metric Types ───────────────────────────────────────────
/// Simple counter (monotonically increasing)
pub struct Counter {
value: AtomicU64,
name: &'static str,
help: &'static str,
}
impl Counter {
pub const fn new(name: &'static str, help: &'static str) -> Self {
Self { value: AtomicU64::new(0), name, help }
}
pub fn inc(&self) { self.value.fetch_add(1, Ordering::Relaxed); }
pub fn inc_by(&self, n: u64) { self.value.fetch_add(n, Ordering::Relaxed); }
pub fn get(&self) -> u64 { self.value.load(Ordering::Relaxed) }
}
/// Gauge (can go up and down)
pub struct Gauge {
value: AtomicU64,
name: &'static str,
help: &'static str,
}
impl Gauge {
pub const fn new(name: &'static str, help: &'static str) -> Self {
Self { value: AtomicU64::new(0), name, help }
}
pub fn set(&self, v: u64) { self.value.store(v, Ordering::Relaxed); }
pub fn inc(&self) { self.value.fetch_add(1, Ordering::Relaxed); }
pub fn dec(&self) { self.value.fetch_sub(1, Ordering::Relaxed); }
pub fn get(&self) -> u64 { self.value.load(Ordering::Relaxed) }
}
/// Gauge for f64 values (stored as bits)
pub struct GaugeF64 {
bits: AtomicU64,
name: &'static str,
help: &'static str,
}
impl GaugeF64 {
pub const fn new(name: &'static str, help: &'static str) -> Self {
Self { bits: AtomicU64::new(0), name, help }
}
pub fn set(&self, v: f64) { self.bits.store(v.to_bits(), Ordering::Relaxed); }
pub fn get(&self) -> f64 { f64::from_bits(self.bits.load(Ordering::Relaxed)) }
}
/// Histogram with fixed buckets for latency tracking
pub struct Histogram {
pub buckets: &'static [f64],
pub counts: Vec<AtomicU64>,
pub sum: AtomicU64, // stored as f64 bits
pub count: AtomicU64,
pub name: &'static str,
pub help: &'static str,
}
impl Histogram {
pub fn new(name: &'static str, help: &'static str, buckets: &'static [f64]) -> Self {
let counts = (0..buckets.len() + 1).map(|_| AtomicU64::new(0)).collect();
Self {
buckets, counts, name, help,
sum: AtomicU64::new(0f64.to_bits()),
count: AtomicU64::new(0),
}
}
pub fn observe(&self, value: f64) {
self.count.fetch_add(1, Ordering::Relaxed);
// Add to sum (CAS loop for f64)
loop {
let old_bits = self.sum.load(Ordering::Relaxed);
let old = f64::from_bits(old_bits);
let new = old + value;
if self.sum.compare_exchange(old_bits, new.to_bits(), Ordering::Relaxed, Ordering::Relaxed).is_ok() {
break;
}
}
// Increment bucket counters
for (i, &bound) in self.buckets.iter().enumerate() {
if value <= bound {
self.counts[i].fetch_add(1, Ordering::Relaxed);
}
}
// +Inf bucket
self.counts[self.buckets.len()].fetch_add(1, Ordering::Relaxed);
}
}
/// Labeled counter (key = label combination string)
pub struct LabeledCounter {
values: Mutex<HashMap<String, u64>>,
name: &'static str,
help: &'static str,
label_names: &'static [&'static str],
}
impl LabeledCounter {
pub fn new(name: &'static str, help: &'static str, label_names: &'static [&'static str]) -> Self {
Self { values: Mutex::new(HashMap::new()), name, help, label_names }
}
pub fn inc(&self, labels: &[&str]) {
let key = labels.join(",");
let mut map = self.values.lock().unwrap();
*map.entry(key).or_insert(0) += 1;
}
}
// ─── Timer helper ───────────────────────────────────────────
/// RAII timer: observes duration on drop
pub struct Timer<'a> {
histogram: &'a Histogram,
start: Instant,
}
impl<'a> Timer<'a> {
pub fn new(histogram: &'a Histogram) -> Self {
Self { histogram, start: Instant::now() }
}
}
impl<'a> Drop for Timer<'a> {
fn drop(&mut self) {
let elapsed = self.start.elapsed().as_secs_f64();
self.histogram.observe(elapsed);
}
}
// ─── Default buckets ────────────────────────────────────────
/// Latency buckets for HTTP handlers (seconds)
pub static HTTP_BUCKETS: &[f64] = &[0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0, 10.0];
/// Latency buckets for LLM calls (seconds)
pub static LLM_BUCKETS: &[f64] = &[0.1, 0.25, 0.5, 1.0, 2.5, 5.0, 10.0, 30.0, 60.0];
/// Latency buckets for DB queries (seconds)
pub static DB_BUCKETS: &[f64] = &[0.001, 0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0];
// ═══════════════════════════════════════════════════════════
// O1: Ingest handler metrics (I1-I12)
// ═══════════════════════════════════════════════════════════
pub static INGEST_REQUESTS_TOTAL: Counter = Counter::new(
"memory_ingest_requests_total", "Total ingest requests received");
pub static INGEST_ERRORS_TOTAL: Counter = Counter::new(
"memory_ingest_errors_total", "Total ingest request errors");
pub static INGEST_RECORDS_TOTAL: Counter = Counter::new(
"memory_ingest_records_total", "Total records ingested");
pub static INGEST_ENTITIES_EXTRACTED: Counter = Counter::new(
"memory_ingest_entities_extracted_total", "Total entities extracted during ingest");
pub static INGEST_EDGES_EXTRACTED: Counter = Counter::new(
"memory_ingest_edges_extracted_total", "Total edges extracted during ingest");
pub static INGEST_IN_FLIGHT: Gauge = Gauge::new(
"memory_ingest_in_flight", "Currently processing ingest jobs");
pub static INGEST_QUEUE_SIZE: Gauge = Gauge::new(
"memory_ingest_queue_size", "Number of jobs waiting in ingest queue");
pub static INGEST_DUPLICATES_TOTAL: Counter = Counter::new(
"memory_ingest_duplicates_total", "Total duplicate ingest requests (idempotency)");
pub static INGEST_BYTES_TOTAL: Counter = Counter::new(
"memory_ingest_bytes_total", "Total bytes ingested");
pub static INGEST_AUTH_FAILURES: Counter = Counter::new(
"memory_ingest_auth_failures_total", "Total auth failures on ingest endpoint");
pub static INGEST_RATE_LIMITED: Counter = Counter::new(
"memory_ingest_rate_limited_total", "Total rate-limited ingest requests");
pub static INGEST_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_ingest_duration_seconds", "Ingest request duration", HTTP_BUCKETS));
// ═══════════════════════════════════════════════════════════
// O2: Query handler metrics (Q1-Q12)
// ═══════════════════════════════════════════════════════════
pub static QUERY_REQUESTS_TOTAL: Counter = Counter::new(
"memory_query_requests_total", "Total query requests received");
pub static QUERY_ERRORS_TOTAL: Counter = Counter::new(
"memory_query_errors_total", "Total query request errors");
pub static QUERY_RESULTS_TOTAL: Counter = Counter::new(
"memory_query_results_total", "Total results returned across all queries");
pub static QUERY_EMPTY_RESULTS: Counter = Counter::new(
"memory_query_empty_results_total", "Queries returning zero results");
pub static QUERY_EMBEDDING_FAILURES: Counter = Counter::new(
"memory_query_embedding_failures_total", "Total embedding failures during query");
pub static QUERY_IN_FLIGHT: Gauge = Gauge::new(
"memory_query_in_flight", "Currently processing queries");
pub static QUERY_AUTH_FAILURES: Counter = Counter::new(
"memory_query_auth_failures_total", "Total auth failures on query endpoint");
pub static QUERY_RATE_LIMITED: Counter = Counter::new(
"memory_query_rate_limited_total", "Total rate-limited query requests");
pub static QUERY_CACHE_HITS: Counter = Counter::new(
"memory_query_cache_hits_total", "Total query cache hits");
pub static QUERY_CACHE_MISSES: Counter = Counter::new(
"memory_query_cache_misses_total", "Total query cache misses");
pub static QUERY_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_query_duration_seconds", "Query request duration", HTTP_BUCKETS));
pub static QUERY_EMBEDDING_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_query_embedding_duration_seconds", "Embedding call duration during query", LLM_BUCKETS));
// ═══════════════════════════════════════════════════════════
// O3: Context endpoint metrics (C1-C8)
// ═══════════════════════════════════════════════════════════
pub static CONTEXT_REQUESTS_TOTAL: Counter = Counter::new(
"memory_context_requests_total", "Total context retrieval requests");
pub static CONTEXT_ERRORS_TOTAL: Counter = Counter::new(
"memory_context_errors_total", "Total context retrieval errors");
pub static CONTEXT_SEMANTIC_HITS: Counter = Counter::new(
"memory_context_semantic_hits_total", "Results from semantic (cosine) tier");
pub static CONTEXT_BM25_HITS: Counter = Counter::new(
"memory_context_bm25_hits_total", "Results from BM25 (lexical) tier");
pub static CONTEXT_GRAPH_HITS: Counter = Counter::new(
"memory_context_graph_hits_total", "Results from graph traversal tier");
pub static CONTEXT_EMPTY_RESULTS: Counter = Counter::new(
"memory_context_empty_results_total", "Context requests returning zero results");
pub static CONTEXT_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_context_duration_seconds", "Context retrieval duration", HTTP_BUCKETS));
pub static CONTEXT_TIER_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_context_tier_duration_seconds", "Per-tier retrieval duration", DB_BUCKETS));
// ═══════════════════════════════════════════════════════════
// O4: Relevance judge metrics (R1-R9)
// ═══════════════════════════════════════════════════════════
pub static RELEVANCE_EVALS_TOTAL: Counter = Counter::new(
"memory_relevance_evals_total", "Total relevance evaluations performed");
pub static RELEVANCE_ERRORS_TOTAL: Counter = Counter::new(
"memory_relevance_errors_total", "Total relevance evaluation errors");
pub static RELEVANCE_RELEVANT_TOTAL: Counter = Counter::new(
"memory_relevance_relevant_total", "Results judged relevant");
pub static RELEVANCE_IRRELEVANT_TOTAL: Counter = Counter::new(
"memory_relevance_irrelevant_total", "Results judged irrelevant");
pub static RELEVANCE_SCORE: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_relevance_score", "Distribution of relevance scores",
&[0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0]));
pub static RELEVANCE_PRECISION: GaugeF64 = GaugeF64::new(
"memory_relevance_precision", "Current precision (relevant/retrieved)");
pub static RELEVANCE_RECALL: GaugeF64 = GaugeF64::new(
"memory_relevance_recall", "Current recall (relevant/total_relevant)");
pub static RELEVANCE_F1: GaugeF64 = GaugeF64::new(
"memory_relevance_f1_score", "Current F1 score");
pub static RELEVANCE_EVAL_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_relevance_eval_duration_seconds", "Relevance evaluation duration", LLM_BUCKETS));
// ═══════════════════════════════════════════════════════════
// O5: Write volume and storage metrics (W1-W12)
// ═══════════════════════════════════════════════════════════
pub static WRITE_ENTITIES_TOTAL: Counter = Counter::new(
"memory_write_entities_total", "Total entities written to DB");
pub static WRITE_EDGES_TOTAL: Counter = Counter::new(
"memory_write_edges_total", "Total edges written to DB");
pub static WRITE_CHUNKS_TOTAL: Counter = Counter::new(
"memory_write_chunks_total", "Total chunks written to DB");
pub static WRITE_ERRORS_TOTAL: Counter = Counter::new(
"memory_write_errors_total", "Total write errors");
pub static WRITE_BYTES_TOTAL: Counter = Counter::new(
"memory_write_bytes_total", "Total bytes written to storage");
pub static DB_ENTITY_COUNT: Gauge = Gauge::new(
"memory_db_entity_count", "Current entity count in memory_entity table");
pub static DB_EDGE_COUNT: Gauge = Gauge::new(
"memory_db_edge_count", "Current edge count in memory_edge table");
pub static DB_CHUNK_COUNT: Gauge = Gauge::new(
"memory_db_chunk_count", "Current chunk count in memory_chunks table");
pub static WRITE_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_write_duration_seconds", "Write operation duration", DB_BUCKETS));
pub static WRITE_BATCH_SIZE: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_write_batch_size", "Write batch sizes",
&[1.0, 5.0, 10.0, 25.0, 50.0, 100.0, 250.0, 500.0]));
// Storage gauges (updated periodically)
pub static DB_SIZE_BYTES: Gauge = Gauge::new(
"memory_db_size_bytes", "Total database size in bytes");
pub static DB_INDEX_SIZE_BYTES: Gauge = Gauge::new(
"memory_db_index_size_bytes", "Total index size in bytes");
// ═══════════════════════════════════════════════════════════
// O6: Pod resource observability (P1-P13)
// (Most collected by node-exporter/cAdvisor, but we track app-level)
// ═══════════════════════════════════════════════════════════
pub static APP_UPTIME_SECONDS: Gauge = Gauge::new(
"memory_app_uptime_seconds", "Application uptime in seconds");
pub static APP_ACTIVE_CONNECTIONS: Gauge = Gauge::new(
"memory_app_active_connections", "Active HTTP connections");
pub static APP_GOROUTINES: Gauge = Gauge::new(
"memory_app_tokio_tasks", "Active tokio tasks (approximate)");
pub static APP_HEAP_BYTES: Gauge = Gauge::new(
"memory_app_heap_bytes", "Approximate heap memory usage");
// ═══════════════════════════════════════════════════════════
// O7: Availability metrics and dependency health (A1-A10)
// ═══════════════════════════════════════════════════════════
pub static HEALTH_CHECKS_TOTAL: Counter = Counter::new(
"memory_health_checks_total", "Total health check requests");
pub static HEALTH_CHECK_FAILURES: Counter = Counter::new(
"memory_health_check_failures_total", "Total health check failures");
pub static DEP_DB_UP: Gauge = Gauge::new(
"memory_dependency_db_up", "Database dependency health (1=up, 0=down)");
pub static DEP_EMBEDDING_UP: Gauge = Gauge::new(
"memory_dependency_embedding_up", "Embedding service health (1=up, 0=down)");
pub static DEP_OPENSEARCH_UP: Gauge = Gauge::new(
"memory_dependency_opensearch_up", "OpenSearch dependency health (1=up, 0=down)");
pub static DEP_LLM_UP: Gauge = Gauge::new(
"memory_dependency_llm_up", "LLM service health (1=up, 0=down)");
pub static DEP_DB_LATENCY: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_dependency_db_latency_seconds", "DB health check latency", DB_BUCKETS));
pub static DEP_EMBEDDING_LATENCY: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_dependency_embedding_latency_seconds", "Embedding health check latency", LLM_BUCKETS));
pub static REQUEST_ERRORS_BY_STATUS: Lazy<LabeledCounter> = Lazy::new(||
LabeledCounter::new(
"memory_request_errors_by_status", "Request errors by HTTP status code",
&["status", "endpoint"]));
// ═══════════════════════════════════════════════════════════
// Named error counters (per error type, per endpoint)
// Format: memory_error_{ERROR_NAME}_{ENDPOINT}_total
// ═══════════════════════════════════════════════════════════
// Ingest errors
pub static ERROR_AUTH_FAILURE_INGEST: Counter = Counter::new(
"memory_error_auth_failure_ingest_total", "Auth failures on ingest endpoint");
pub static ERROR_FORBIDDEN_INGEST: Counter = Counter::new(
"memory_error_forbidden_ingest_total", "Forbidden (missing capability) on ingest");
pub static ERROR_RATE_LIMITED_INGEST: Counter = Counter::new(
"memory_error_rate_limited_ingest_total", "Rate limited on ingest");
pub static ERROR_BAD_REQUEST_INGEST: Counter = Counter::new(
"memory_error_bad_request_ingest_total", "Bad request on ingest");
pub static ERROR_DB_ERROR_INGEST: Counter = Counter::new(
"memory_error_db_error_ingest_total", "Database error during ingest");
// Query errors
pub static ERROR_AUTH_FAILURE_QUERY: Counter = Counter::new(
"memory_error_auth_failure_query_total", "Auth failures on query endpoint");
pub static ERROR_FORBIDDEN_QUERY: Counter = Counter::new(
"memory_error_forbidden_query_total", "Forbidden (missing capability) on query");
pub static ERROR_BAD_REQUEST_QUERY: Counter = Counter::new(
"memory_error_bad_request_query_total", "Bad request on query");
pub static ERROR_EMBEDDING_FAILURE_QUERY: Counter = Counter::new(
"memory_error_embedding_failure_query_total", "Embedding service failure during query");
pub static ERROR_SEARCH_FAILURE_QUERY: Counter = Counter::new(
"memory_error_search_failure_query_total", "Search execution failure during query");
// Context errors
pub static ERROR_AUTH_FAILURE_CONTEXT: Counter = Counter::new(
"memory_error_auth_failure_context_total", "Auth failures on context endpoint");
pub static ERROR_FORBIDDEN_CONTEXT: Counter = Counter::new(
"memory_error_forbidden_context_total", "Forbidden (missing capability) on context");
pub static ERROR_LOOKUP_FAILURE_CONTEXT: Counter = Counter::new(
"memory_error_lookup_failure_context_total", "Context lookup failure");
// Unexpected errors (unhandled 500s, panics, unknown failures)
pub static ERROR_UNEXPECTED_TOTAL: Counter = Counter::new(
"memory_error_unexpected_total", "Total unexpected/unhandled errors (500s)");
pub static ERROR_UNEXPECTED_INGEST: Counter = Counter::new(
"memory_error_unexpected_ingest_total", "Unexpected errors during ingest");
pub static ERROR_UNEXPECTED_QUERY: Counter = Counter::new(
"memory_error_unexpected_query_total", "Unexpected errors during query");
pub static ERROR_UNEXPECTED_CONTEXT: Counter = Counter::new(
"memory_error_unexpected_context_total", "Unexpected errors during context");
// Last error info (most recent error for debugging)
pub static LAST_ERROR_TIMESTAMP: Gauge = Gauge::new(
"memory_last_error_timestamp_seconds", "Unix timestamp of most recent error");
// ═══════════════════════════════════════════════════════════
// O8: Ingest rate pattern tracking (IR1-IR10)
// ═══════════════════════════════════════════════════════════
pub static INGEST_RATE_1M: GaugeF64 = GaugeF64::new(
"memory_ingest_rate_1m", "Ingest rate per second (1-minute window)");
pub static INGEST_RATE_5M: GaugeF64 = GaugeF64::new(
"memory_ingest_rate_5m", "Ingest rate per second (5-minute window)");
pub static INGEST_LLM_EXTRACT_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_ingest_llm_extract_duration_seconds", "LLM entity extraction duration", LLM_BUCKETS));
pub static INGEST_FACT_EXTRACT_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_ingest_fact_extract_duration_seconds", "LLM fact extraction duration", LLM_BUCKETS));
pub static INGEST_DEDUP_TOTAL: Counter = Counter::new(
"memory_ingest_dedup_total", "Total entities deduplicated");
pub static INGEST_CONTRADICTION_TOTAL: Counter = Counter::new(
"memory_ingest_contradiction_total", "Total contradictions detected");
pub static INGEST_PROJECTS: Gauge = Gauge::new(
"memory_ingest_active_projects", "Number of active projects with ingested data");
// ═══════════════════════════════════════════════════════════
// O9: Postgres internal observability (PG1-PG33)
// (Most collected by pg_exporter, we expose app-visible DB stats)
// ═══════════════════════════════════════════════════════════
pub static DB_POOL_SIZE: Gauge = Gauge::new(
"memory_db_pool_size", "Current connection pool size");
pub static DB_POOL_IDLE: Gauge = Gauge::new(
"memory_db_pool_idle", "Idle connections in pool");
pub static DB_POOL_ACTIVE: Gauge = Gauge::new(
"memory_db_pool_active", "Active connections in pool");
pub static DB_QUERY_TOTAL: Counter = Counter::new(
"memory_db_queries_total", "Total DB queries executed");
pub static DB_QUERY_ERRORS: Counter = Counter::new(
"memory_db_query_errors_total", "Total DB query errors");
pub static DB_QUERY_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_db_query_duration_seconds", "DB query duration", DB_BUCKETS));
pub static DB_TRANSACTION_DURATION: Lazy<Histogram> = Lazy::new(||
Histogram::new("memory_db_transaction_duration_seconds", "DB transaction duration", DB_BUCKETS));
// Table-specific row counts (updated periodically)
pub static DB_TABLE_ENTITY_ROWS: Gauge = Gauge::new(
"memory_db_table_entity_rows", "Rows in memory_entity table");
pub static DB_TABLE_EDGE_ROWS: Gauge = Gauge::new(
"memory_db_table_edge_rows", "Rows in memory_edge table");
pub static DB_TABLE_CHUNK_ROWS: Gauge = Gauge::new(
"memory_db_table_chunk_rows", "Rows in memory_chunks table");
// ═══════════════════════════════════════════════════════════
// Metrics export (Prometheus text format)
// ═══════════════════════════════════════════════════════════
/// Render all metrics in Prometheus text exposition format
pub fn render_metrics() -> String {
let mut out = String::with_capacity(8192);
// Helper macros
macro_rules! counter {
($c:expr) => {
out.push_str(&format!("# HELP {} {}\n# TYPE {} counter\n{} {}\n",
$c.name, $c.help, $c.name, $c.name, $c.get()));
};
}
macro_rules! gauge {
($g:expr) => {
out.push_str(&format!("# HELP {} {}\n# TYPE {} gauge\n{} {}\n",
$g.name, $g.help, $g.name, $g.name, $g.get()));
};
}
macro_rules! gauge_f64 {
($g:expr) => {
out.push_str(&format!("# HELP {} {}\n# TYPE {} gauge\n{} {:.6}\n",
$g.name, $g.help, $g.name, $g.name, $g.get()));
};
}
macro_rules! histogram {
($h:expr) => {
out.push_str(&format!("# HELP {} {}\n# TYPE {} histogram\n", $h.name, $h.help, $h.name));
for (i, &bound) in $h.buckets.iter().enumerate() {
out.push_str(&format!("{}_bucket{{le=\"{}\"}} {}\n",
$h.name, bound, $h.counts[i].load(Ordering::Relaxed)));
}
out.push_str(&format!("{}_bucket{{le=\"+Inf\"}} {}\n",
$h.name, $h.counts[$h.buckets.len()].load(Ordering::Relaxed)));
out.push_str(&format!("{}_sum {:.6}\n", $h.name,
f64::from_bits($h.sum.load(Ordering::Relaxed))));
out.push_str(&format!("{}_count {}\n", $h.name,
$h.count.load(Ordering::Relaxed)));
};
}
// O1: Ingest
counter!(INGEST_REQUESTS_TOTAL);
counter!(INGEST_ERRORS_TOTAL);
counter!(INGEST_RECORDS_TOTAL);
counter!(INGEST_ENTITIES_EXTRACTED);
counter!(INGEST_EDGES_EXTRACTED);
gauge!(INGEST_IN_FLIGHT);
gauge!(INGEST_QUEUE_SIZE);
counter!(INGEST_DUPLICATES_TOTAL);
counter!(INGEST_BYTES_TOTAL);
counter!(INGEST_AUTH_FAILURES);
counter!(INGEST_RATE_LIMITED);
histogram!(INGEST_DURATION);
// O2: Query
counter!(QUERY_REQUESTS_TOTAL);
counter!(QUERY_ERRORS_TOTAL);
counter!(QUERY_RESULTS_TOTAL);
counter!(QUERY_EMPTY_RESULTS);
counter!(QUERY_EMBEDDING_FAILURES);
gauge!(QUERY_IN_FLIGHT);
counter!(QUERY_AUTH_FAILURES);
counter!(QUERY_RATE_LIMITED);
counter!(QUERY_CACHE_HITS);
counter!(QUERY_CACHE_MISSES);
histogram!(QUERY_DURATION);
histogram!(QUERY_EMBEDDING_DURATION);
// O3: Context
counter!(CONTEXT_REQUESTS_TOTAL);
counter!(CONTEXT_ERRORS_TOTAL);
counter!(CONTEXT_SEMANTIC_HITS);
counter!(CONTEXT_BM25_HITS);
counter!(CONTEXT_GRAPH_HITS);
counter!(CONTEXT_EMPTY_RESULTS);
histogram!(CONTEXT_DURATION);
histogram!(CONTEXT_TIER_DURATION);
// O4: Relevance
counter!(RELEVANCE_EVALS_TOTAL);
counter!(RELEVANCE_ERRORS_TOTAL);
counter!(RELEVANCE_RELEVANT_TOTAL);
counter!(RELEVANCE_IRRELEVANT_TOTAL);
histogram!(RELEVANCE_SCORE);
gauge_f64!(RELEVANCE_PRECISION);
gauge_f64!(RELEVANCE_RECALL);
gauge_f64!(RELEVANCE_F1);
histogram!(RELEVANCE_EVAL_DURATION);
// O5: Write volume
counter!(WRITE_ENTITIES_TOTAL);
counter!(WRITE_EDGES_TOTAL);
counter!(WRITE_CHUNKS_TOTAL);
counter!(WRITE_ERRORS_TOTAL);
counter!(WRITE_BYTES_TOTAL);
gauge!(DB_ENTITY_COUNT);
gauge!(DB_EDGE_COUNT);
gauge!(DB_CHUNK_COUNT);
histogram!(WRITE_DURATION);
histogram!(WRITE_BATCH_SIZE);
gauge!(DB_SIZE_BYTES);
gauge!(DB_INDEX_SIZE_BYTES);
// O6: Pod resources
gauge!(APP_UPTIME_SECONDS);
gauge!(APP_ACTIVE_CONNECTIONS);
gauge!(APP_GOROUTINES);
gauge!(APP_HEAP_BYTES);
// O7: Availability
counter!(HEALTH_CHECKS_TOTAL);
counter!(HEALTH_CHECK_FAILURES);
gauge!(DEP_DB_UP);
gauge!(DEP_EMBEDDING_UP);
gauge!(DEP_OPENSEARCH_UP);
gauge!(DEP_LLM_UP);
histogram!(DEP_DB_LATENCY);
histogram!(DEP_EMBEDDING_LATENCY);
// O8: Ingest rate
gauge_f64!(INGEST_RATE_1M);
gauge_f64!(INGEST_RATE_5M);
histogram!(INGEST_LLM_EXTRACT_DURATION);
histogram!(INGEST_FACT_EXTRACT_DURATION);
counter!(INGEST_DEDUP_TOTAL);
counter!(INGEST_CONTRADICTION_TOTAL);
gauge!(INGEST_PROJECTS);
// O9: Postgres
gauge!(DB_POOL_SIZE);
gauge!(DB_POOL_IDLE);
gauge!(DB_POOL_ACTIVE);
counter!(DB_QUERY_TOTAL);
counter!(DB_QUERY_ERRORS);
histogram!(DB_QUERY_DURATION);
histogram!(DB_TRANSACTION_DURATION);
gauge!(DB_TABLE_ENTITY_ROWS);
gauge!(DB_TABLE_EDGE_ROWS);
gauge!(DB_TABLE_CHUNK_ROWS);
// Named error counters
counter!(ERROR_AUTH_FAILURE_INGEST);
counter!(ERROR_FORBIDDEN_INGEST);
counter!(ERROR_RATE_LIMITED_INGEST);
counter!(ERROR_BAD_REQUEST_INGEST);
counter!(ERROR_DB_ERROR_INGEST);
counter!(ERROR_AUTH_FAILURE_QUERY);
counter!(ERROR_FORBIDDEN_QUERY);
counter!(ERROR_BAD_REQUEST_QUERY);
counter!(ERROR_EMBEDDING_FAILURE_QUERY);
counter!(ERROR_SEARCH_FAILURE_QUERY);
counter!(ERROR_AUTH_FAILURE_CONTEXT);
counter!(ERROR_FORBIDDEN_CONTEXT);
counter!(ERROR_LOOKUP_FAILURE_CONTEXT);
counter!(ERROR_UNEXPECTED_TOTAL);
counter!(ERROR_UNEXPECTED_INGEST);
counter!(ERROR_UNEXPECTED_QUERY);
counter!(ERROR_UNEXPECTED_CONTEXT);
gauge!(LAST_ERROR_TIMESTAMP);
out
}
/// Render a labeled counter in Prometheus format
fn render_labeled_counter(out: &mut String, lc: &LabeledCounter) {
let map = lc.values.lock().unwrap();
if map.is_empty() { return; }
out.push_str(&format!("# HELP {} {}\n# TYPE {} counter\n", lc.name, lc.help, lc.name));
for (key, val) in map.iter() {
let parts: Vec<&str> = key.split(',').collect();
let labels: Vec<String> = lc.label_names.iter().zip(parts.iter())
.map(|(name, val)| format!("{}=\"{}\"", name, val))
.collect();
out.push_str(&format!("{}{{{}}} {}\n", lc.name, labels.join(","), val));
}
}
/// GET /metrics handler
pub async fn metrics_handler() -> actix_web::HttpResponse {
actix_web::HttpResponse::Ok()
.content_type("text/plain; version=0.0.4; charset=utf-8")
.body(render_metrics())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_counter() {
let c = Counter::new("test_counter", "test");
assert_eq!(c.get(), 0);
c.inc();
assert_eq!(c.get(), 1);
c.inc_by(5);
assert_eq!(c.get(), 6);
}
#[test]
fn test_gauge() {
let g = Gauge::new("test_gauge", "test");
assert_eq!(g.get(), 0);
g.set(42);
assert_eq!(g.get(), 42);
g.inc();
assert_eq!(g.get(), 43);
g.dec();
assert_eq!(g.get(), 42);
}
#[test]
fn test_gauge_f64() {
let g = GaugeF64::new("test_gauge_f64", "test");
assert_eq!(g.get(), 0.0);
g.set(3.14);
assert!((g.get() - 3.14).abs() < 0.001);
}
#[test]
fn test_histogram() {
let h = Histogram::new("test_hist", "test", &[0.1, 0.5, 1.0]);
h.observe(0.05);
h.observe(0.3);
h.observe(0.8);
h.observe(2.0);
assert_eq!(h.count.load(Ordering::Relaxed), 4);
}
#[test]
fn test_render_metrics_not_empty() {
INGEST_REQUESTS_TOTAL.inc();
QUERY_REQUESTS_TOTAL.inc();
let output = render_metrics();
assert!(output.contains("memory_ingest_requests_total"));
assert!(output.contains("memory_query_requests_total"));
assert!(output.contains("# HELP"));
assert!(output.contains("# TYPE"));
}
#[test]
fn test_timer_observes_on_drop() {
let h = Histogram::new("timer_test", "test", HTTP_BUCKETS);
{
let _t = Timer::new(&h);
std::thread::sleep(std::time::Duration::from_millis(1));
}
assert_eq!(h.count.load(Ordering::Relaxed), 1);
}
}
-418
View File
@@ -1,418 +0,0 @@
//! Metrics Snapshot & Assertion (Test Harness)
//!
//! Captures metric state before/after a test scenario,
//! then asserts expected deltas per metric.
//!
//! Usage:
//! ```rust
//! let snap = MetricsSnapshot::capture();
//! // ... run handler / scenario ...
//! snap.assert_counter_inc("memory_ingest_requests_total", 1);
//! snap.assert_counter_inc("memory_ingest_errors_total", 0);
//! snap.assert_gauge_eq("memory_ingest_in_flight", 0);
//! snap.assert_histogram_count_inc("memory_ingest_duration_seconds", 1);
//! ```
use std::collections::HashMap;
use std::sync::atomic::Ordering;
use crate::metrics;
/// Snapshot of all metric values at a point in time
#[derive(Debug, Clone)]
pub struct MetricsSnapshot {
counters: HashMap<&'static str, u64>,
gauges: HashMap<&'static str, u64>,
gauges_f64: HashMap<&'static str, f64>,
histogram_counts: HashMap<&'static str, u64>,
}
impl MetricsSnapshot {
/// Capture current state of all metrics
pub fn capture() -> Self {
let mut counters = HashMap::new();
let mut gauges = HashMap::new();
let mut gauges_f64 = HashMap::new();
let mut histogram_counts = HashMap::new();
// O1: Ingest counters
counters.insert("memory_ingest_requests_total", metrics::INGEST_REQUESTS_TOTAL.get());
counters.insert("memory_ingest_errors_total", metrics::INGEST_ERRORS_TOTAL.get());
counters.insert("memory_ingest_records_total", metrics::INGEST_RECORDS_TOTAL.get());
counters.insert("memory_ingest_entities_extracted_total", metrics::INGEST_ENTITIES_EXTRACTED.get());
counters.insert("memory_ingest_edges_extracted_total", metrics::INGEST_EDGES_EXTRACTED.get());
counters.insert("memory_ingest_duplicates_total", metrics::INGEST_DUPLICATES_TOTAL.get());
counters.insert("memory_ingest_bytes_total", metrics::INGEST_BYTES_TOTAL.get());
counters.insert("memory_ingest_auth_failures_total", metrics::INGEST_AUTH_FAILURES.get());
counters.insert("memory_ingest_rate_limited_total", metrics::INGEST_RATE_LIMITED.get());
// O1: Ingest gauges
gauges.insert("memory_ingest_in_flight", metrics::INGEST_IN_FLIGHT.get());
gauges.insert("memory_ingest_queue_size", metrics::INGEST_QUEUE_SIZE.get());
// O1: Ingest histogram (force Lazy init)
histogram_counts.insert("memory_ingest_duration_seconds",
{ let _ = &*metrics::INGEST_DURATION; metrics::INGEST_DURATION.count.load(Ordering::Relaxed) });
// O2: Query counters
counters.insert("memory_query_requests_total", metrics::QUERY_REQUESTS_TOTAL.get());
counters.insert("memory_query_errors_total", metrics::QUERY_ERRORS_TOTAL.get());
counters.insert("memory_query_results_total", metrics::QUERY_RESULTS_TOTAL.get());
counters.insert("memory_query_empty_results_total", metrics::QUERY_EMPTY_RESULTS.get());
counters.insert("memory_query_embedding_failures_total", metrics::QUERY_EMBEDDING_FAILURES.get());
counters.insert("memory_query_auth_failures_total", metrics::QUERY_AUTH_FAILURES.get());
counters.insert("memory_query_rate_limited_total", metrics::QUERY_RATE_LIMITED.get());
counters.insert("memory_query_cache_hits_total", metrics::QUERY_CACHE_HITS.get());
counters.insert("memory_query_cache_misses_total", metrics::QUERY_CACHE_MISSES.get());
// O2: Query gauges
gauges.insert("memory_query_in_flight", metrics::QUERY_IN_FLIGHT.get());
// O2: Query histograms
histogram_counts.insert("memory_query_duration_seconds",
{ let _ = &*metrics::QUERY_DURATION; metrics::QUERY_DURATION.count.load(Ordering::Relaxed) });
histogram_counts.insert("memory_query_embedding_duration_seconds",
{ let _ = &*metrics::QUERY_EMBEDDING_DURATION; metrics::QUERY_EMBEDDING_DURATION.count.load(Ordering::Relaxed) });
// O3: Context
counters.insert("memory_context_requests_total", metrics::CONTEXT_REQUESTS_TOTAL.get());
counters.insert("memory_context_errors_total", metrics::CONTEXT_ERRORS_TOTAL.get());
counters.insert("memory_context_semantic_hits_total", metrics::CONTEXT_SEMANTIC_HITS.get());
counters.insert("memory_context_bm25_hits_total", metrics::CONTEXT_BM25_HITS.get());
counters.insert("memory_context_graph_hits_total", metrics::CONTEXT_GRAPH_HITS.get());
counters.insert("memory_context_empty_results_total", metrics::CONTEXT_EMPTY_RESULTS.get());
histogram_counts.insert("memory_context_duration_seconds",
{ let _ = &*metrics::CONTEXT_DURATION; metrics::CONTEXT_DURATION.count.load(Ordering::Relaxed) });
// O4: Relevance histograms
histogram_counts.insert("memory_relevance_eval_duration_seconds",
{ let _ = &*metrics::RELEVANCE_EVAL_DURATION; metrics::RELEVANCE_EVAL_DURATION.count.load(Ordering::Relaxed) });
// O5: Write histogram
histogram_counts.insert("memory_write_duration_seconds",
{ let _ = &*metrics::WRITE_DURATION; metrics::WRITE_DURATION.count.load(Ordering::Relaxed) });
// O7: Dependency latency
histogram_counts.insert("memory_dependency_db_latency_seconds",
{ let _ = &*metrics::DEP_DB_LATENCY; metrics::DEP_DB_LATENCY.count.load(Ordering::Relaxed) });
// O4: Relevance
counters.insert("memory_relevance_evals_total", metrics::RELEVANCE_EVALS_TOTAL.get());
counters.insert("memory_relevance_errors_total", metrics::RELEVANCE_ERRORS_TOTAL.get());
counters.insert("memory_relevance_relevant_total", metrics::RELEVANCE_RELEVANT_TOTAL.get());
counters.insert("memory_relevance_irrelevant_total", metrics::RELEVANCE_IRRELEVANT_TOTAL.get());
gauges_f64.insert("memory_relevance_precision", metrics::RELEVANCE_PRECISION.get());
gauges_f64.insert("memory_relevance_recall", metrics::RELEVANCE_RECALL.get());
gauges_f64.insert("memory_relevance_f1_score", metrics::RELEVANCE_F1.get());
// O5: Write
counters.insert("memory_write_entities_total", metrics::WRITE_ENTITIES_TOTAL.get());
counters.insert("memory_write_edges_total", metrics::WRITE_EDGES_TOTAL.get());
counters.insert("memory_write_chunks_total", metrics::WRITE_CHUNKS_TOTAL.get());
counters.insert("memory_write_errors_total", metrics::WRITE_ERRORS_TOTAL.get());
counters.insert("memory_write_bytes_total", metrics::WRITE_BYTES_TOTAL.get());
// O7: Health
counters.insert("memory_health_checks_total", metrics::HEALTH_CHECKS_TOTAL.get());
counters.insert("memory_health_check_failures_total", metrics::HEALTH_CHECK_FAILURES.get());
gauges.insert("memory_dependency_db_up", metrics::DEP_DB_UP.get());
gauges.insert("memory_dependency_embedding_up", metrics::DEP_EMBEDDING_UP.get());
// O8: Ingest rate
counters.insert("memory_ingest_dedup_total", metrics::INGEST_DEDUP_TOTAL.get());
counters.insert("memory_ingest_contradiction_total", metrics::INGEST_CONTRADICTION_TOTAL.get());
// O9: DB
counters.insert("memory_db_queries_total", metrics::DB_QUERY_TOTAL.get());
counters.insert("memory_db_query_errors_total", metrics::DB_QUERY_ERRORS.get());
Self { counters, gauges, gauges_f64, histogram_counts }
}
/// Assert a counter increased by exactly `expected` since snapshot
pub fn assert_counter_inc(&self, name: &str, expected: u64) {
let before = self.counters.get(name)
.unwrap_or_else(|| panic!("Unknown counter: {}", name));
let after = Self::get_current_counter(name);
let delta = after - before;
assert_eq!(delta, expected,
"Counter {} expected +{} but got +{} (before={}, after={})",
name, expected, delta, before, after);
}
/// Assert a counter increased by at least `min` since snapshot
pub fn assert_counter_inc_at_least(&self, name: &str, min: u64) {
let before = self.counters.get(name)
.unwrap_or_else(|| panic!("Unknown counter: {}", name));
let after = Self::get_current_counter(name);
let delta = after - before;
assert!(delta >= min,
"Counter {} expected at least +{} but got +{} (before={}, after={})",
name, min, delta, before, after);
}
/// Assert a gauge equals exactly `expected`
pub fn assert_gauge_eq(&self, name: &str, expected: u64) {
let current = Self::get_current_gauge(name);
assert_eq!(current, expected,
"Gauge {} expected {} but got {}", name, expected, current);
}
/// Assert a histogram observation count increased by `expected`
pub fn assert_histogram_count_inc(&self, name: &str, expected: u64) {
let before = self.histogram_counts.get(name)
.unwrap_or_else(|| panic!("Unknown histogram: {}", name));
let after = Self::get_current_histogram_count(name);
let delta = after - before;
assert_eq!(delta, expected,
"Histogram {} count expected +{} but got +{} (before={}, after={})",
name, expected, delta, before, after);
}
/// Assert a f64 gauge is within tolerance
pub fn assert_gauge_f64_approx(&self, name: &str, expected: f64, tolerance: f64) {
let current = Self::get_current_gauge_f64(name);
assert!((current - expected).abs() <= tolerance,
"Gauge {} expected {:.4} (±{}) but got {:.4}",
name, expected, tolerance, current);
}
/// Get delta for a counter since snapshot
pub fn counter_delta(&self, name: &str) -> u64 {
let before = self.counters.get(name).copied().unwrap_or(0);
let after = Self::get_current_counter(name);
after - before
}
/// Print all deltas since snapshot (for debugging)
pub fn print_deltas(&self) {
println!("=== Metrics Deltas ===");
for (name, before) in &self.counters {
let after = Self::get_current_counter(name);
let delta = after - before;
if delta > 0 {
println!(" {} +{} ({} -> {})", name, delta, before, after);
}
}
for (name, before) in &self.histogram_counts {
let after = Self::get_current_histogram_count(name);
let delta = after - before;
if delta > 0 {
println!(" {} count +{}", name, delta);
}
}
}
// ─── Internal helpers ───────────────────────────────────
fn get_current_counter(name: &str) -> u64 {
match name {
"memory_ingest_requests_total" => metrics::INGEST_REQUESTS_TOTAL.get(),
"memory_ingest_errors_total" => metrics::INGEST_ERRORS_TOTAL.get(),
"memory_ingest_records_total" => metrics::INGEST_RECORDS_TOTAL.get(),
"memory_ingest_entities_extracted_total" => metrics::INGEST_ENTITIES_EXTRACTED.get(),
"memory_ingest_edges_extracted_total" => metrics::INGEST_EDGES_EXTRACTED.get(),
"memory_ingest_duplicates_total" => metrics::INGEST_DUPLICATES_TOTAL.get(),
"memory_ingest_bytes_total" => metrics::INGEST_BYTES_TOTAL.get(),
"memory_ingest_auth_failures_total" => metrics::INGEST_AUTH_FAILURES.get(),
"memory_ingest_rate_limited_total" => metrics::INGEST_RATE_LIMITED.get(),
"memory_query_requests_total" => metrics::QUERY_REQUESTS_TOTAL.get(),
"memory_query_errors_total" => metrics::QUERY_ERRORS_TOTAL.get(),
"memory_query_results_total" => metrics::QUERY_RESULTS_TOTAL.get(),
"memory_query_empty_results_total" => metrics::QUERY_EMPTY_RESULTS.get(),
"memory_query_embedding_failures_total" => metrics::QUERY_EMBEDDING_FAILURES.get(),
"memory_query_auth_failures_total" => metrics::QUERY_AUTH_FAILURES.get(),
"memory_query_rate_limited_total" => metrics::QUERY_RATE_LIMITED.get(),
"memory_query_cache_hits_total" => metrics::QUERY_CACHE_HITS.get(),
"memory_query_cache_misses_total" => metrics::QUERY_CACHE_MISSES.get(),
"memory_context_requests_total" => metrics::CONTEXT_REQUESTS_TOTAL.get(),
"memory_context_errors_total" => metrics::CONTEXT_ERRORS_TOTAL.get(),
"memory_context_semantic_hits_total" => metrics::CONTEXT_SEMANTIC_HITS.get(),
"memory_context_bm25_hits_total" => metrics::CONTEXT_BM25_HITS.get(),
"memory_context_graph_hits_total" => metrics::CONTEXT_GRAPH_HITS.get(),
"memory_context_empty_results_total" => metrics::CONTEXT_EMPTY_RESULTS.get(),
"memory_relevance_evals_total" => metrics::RELEVANCE_EVALS_TOTAL.get(),
"memory_relevance_errors_total" => metrics::RELEVANCE_ERRORS_TOTAL.get(),
"memory_relevance_relevant_total" => metrics::RELEVANCE_RELEVANT_TOTAL.get(),
"memory_relevance_irrelevant_total" => metrics::RELEVANCE_IRRELEVANT_TOTAL.get(),
"memory_write_entities_total" => metrics::WRITE_ENTITIES_TOTAL.get(),
"memory_write_edges_total" => metrics::WRITE_EDGES_TOTAL.get(),
"memory_write_chunks_total" => metrics::WRITE_CHUNKS_TOTAL.get(),
"memory_write_errors_total" => metrics::WRITE_ERRORS_TOTAL.get(),
"memory_write_bytes_total" => metrics::WRITE_BYTES_TOTAL.get(),
"memory_health_checks_total" => metrics::HEALTH_CHECKS_TOTAL.get(),
"memory_health_check_failures_total" => metrics::HEALTH_CHECK_FAILURES.get(),
"memory_ingest_dedup_total" => metrics::INGEST_DEDUP_TOTAL.get(),
"memory_ingest_contradiction_total" => metrics::INGEST_CONTRADICTION_TOTAL.get(),
"memory_db_queries_total" => metrics::DB_QUERY_TOTAL.get(),
"memory_db_query_errors_total" => metrics::DB_QUERY_ERRORS.get(),
_ => panic!("Unknown counter: {}", name),
}
}
fn get_current_gauge(name: &str) -> u64 {
match name {
"memory_ingest_in_flight" => metrics::INGEST_IN_FLIGHT.get(),
"memory_ingest_queue_size" => metrics::INGEST_QUEUE_SIZE.get(),
"memory_query_in_flight" => metrics::QUERY_IN_FLIGHT.get(),
"memory_dependency_db_up" => metrics::DEP_DB_UP.get(),
"memory_dependency_embedding_up" => metrics::DEP_EMBEDDING_UP.get(),
"memory_dependency_opensearch_up" => metrics::DEP_OPENSEARCH_UP.get(),
"memory_dependency_llm_up" => metrics::DEP_LLM_UP.get(),
"memory_app_uptime_seconds" => metrics::APP_UPTIME_SECONDS.get(),
"memory_db_pool_size" => metrics::DB_POOL_SIZE.get(),
"memory_db_pool_idle" => metrics::DB_POOL_IDLE.get(),
"memory_db_table_entity_rows" => metrics::DB_TABLE_ENTITY_ROWS.get(),
"memory_db_table_edge_rows" => metrics::DB_TABLE_EDGE_ROWS.get(),
"memory_db_table_chunk_rows" => metrics::DB_TABLE_CHUNK_ROWS.get(),
_ => panic!("Unknown gauge: {}", name),
}
}
fn get_current_gauge_f64(name: &str) -> f64 {
match name {
"memory_relevance_precision" => metrics::RELEVANCE_PRECISION.get(),
"memory_relevance_recall" => metrics::RELEVANCE_RECALL.get(),
"memory_relevance_f1_score" => metrics::RELEVANCE_F1.get(),
"memory_ingest_rate_1m" => metrics::INGEST_RATE_1M.get(),
"memory_ingest_rate_5m" => metrics::INGEST_RATE_5M.get(),
_ => panic!("Unknown gauge_f64: {}", name),
}
}
fn get_current_histogram_count(name: &str) -> u64 {
match name {
"memory_ingest_duration_seconds" =>
metrics::INGEST_DURATION.count.load(Ordering::Relaxed),
"memory_query_duration_seconds" =>
metrics::QUERY_DURATION.count.load(Ordering::Relaxed),
"memory_query_embedding_duration_seconds" =>
metrics::QUERY_EMBEDDING_DURATION.count.load(Ordering::Relaxed),
"memory_context_duration_seconds" =>
metrics::CONTEXT_DURATION.count.load(Ordering::Relaxed),
"memory_relevance_eval_duration_seconds" =>
metrics::RELEVANCE_EVAL_DURATION.count.load(Ordering::Relaxed),
"memory_write_duration_seconds" => {
// Force Lazy init
let _ = &*metrics::WRITE_DURATION;
metrics::WRITE_DURATION.count.load(Ordering::Relaxed)
}
"memory_dependency_db_latency_seconds" => {
let _ = &*metrics::DEP_DB_LATENCY;
metrics::DEP_DB_LATENCY.count.load(Ordering::Relaxed)
}
_ => panic!("Unknown histogram: {}", name),
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::relevance_judge::RelevanceJudge;
#[test]
fn test_snapshot_captures_state() {
let snap = MetricsSnapshot::capture();
assert!(snap.counters.contains_key("memory_ingest_requests_total"));
assert!(snap.counters.contains_key("memory_query_requests_total"));
assert!(snap.gauges.contains_key("memory_ingest_in_flight"));
assert!(snap.histogram_counts.contains_key("memory_ingest_duration_seconds"));
}
#[test]
fn test_counter_delta_zero_when_no_change() {
let snap = MetricsSnapshot::capture();
snap.assert_counter_inc("memory_write_entities_total", 0);
}
#[test]
fn test_counter_tracks_increment() {
let snap = MetricsSnapshot::capture();
metrics::WRITE_ENTITIES_TOTAL.inc_by(3);
snap.assert_counter_inc("memory_write_entities_total", 3);
}
#[test]
fn test_counter_delta_method() {
let snap = MetricsSnapshot::capture();
metrics::WRITE_EDGES_TOTAL.inc_by(7);
assert_eq!(snap.counter_delta("memory_write_edges_total"), 7);
}
#[test]
fn test_histogram_count_tracks() {
let snap = MetricsSnapshot::capture();
metrics::WRITE_DURATION.observe(0.05);
metrics::WRITE_DURATION.observe(0.10);
snap.assert_histogram_count_inc("memory_write_duration_seconds", 2);
}
#[test]
fn test_relevance_scenario_metrics() {
let snap = MetricsSnapshot::capture();
let judge = RelevanceJudge::new(0.5);
let results = vec![
("good result".to_string(), 0.9),
("bad result".to_string(), 0.1),
("ok result".to_string(), 0.6),
];
let summary = judge.evaluate_batch("test query", &results);
// Verify metrics match scenario
snap.assert_counter_inc("memory_relevance_evals_total", 3);
snap.assert_counter_inc("memory_relevance_relevant_total", 2); // 0.9 + 0.6
snap.assert_counter_inc("memory_relevance_irrelevant_total", 1); // 0.1
// Verify precision gauge
snap.assert_gauge_f64_approx("memory_relevance_precision", summary.precision, 0.01);
assert_eq!(summary.total, 3);
assert_eq!(summary.relevant, 2);
}
#[test]
fn test_ingest_counter_scenario() {
let snap = MetricsSnapshot::capture();
// Simulate ingest scenario
metrics::INGEST_REQUESTS_TOTAL.inc();
metrics::INGEST_RECORDS_TOTAL.inc_by(5);
metrics::INGEST_BYTES_TOTAL.inc_by(1024);
metrics::INGEST_ENTITIES_EXTRACTED.inc_by(3);
metrics::INGEST_EDGES_EXTRACTED.inc_by(2);
snap.assert_counter_inc("memory_ingest_requests_total", 1);
snap.assert_counter_inc("memory_ingest_records_total", 5);
snap.assert_counter_inc("memory_ingest_bytes_total", 1024);
snap.assert_counter_inc("memory_ingest_entities_extracted_total", 3);
snap.assert_counter_inc("memory_ingest_edges_extracted_total", 2);
snap.assert_counter_inc("memory_ingest_errors_total", 0);
}
#[test]
fn test_query_error_scenario() {
let snap = MetricsSnapshot::capture();
// Simulate query that fails at embedding
metrics::QUERY_REQUESTS_TOTAL.inc();
metrics::QUERY_IN_FLIGHT.inc();
metrics::QUERY_EMBEDDING_FAILURES.inc();
metrics::QUERY_ERRORS_TOTAL.inc();
metrics::QUERY_IN_FLIGHT.dec();
snap.assert_counter_inc("memory_query_requests_total", 1);
snap.assert_counter_inc("memory_query_embedding_failures_total", 1);
snap.assert_counter_inc("memory_query_errors_total", 1);
snap.assert_counter_inc("memory_query_results_total", 0);
snap.assert_gauge_eq("memory_query_in_flight", 0);
}
#[test]
fn test_print_deltas_works() {
let snap = MetricsSnapshot::capture();
metrics::HEALTH_CHECKS_TOTAL.inc();
snap.print_deltas(); // Should not panic
}
}
+103
View File
@@ -158,3 +158,106 @@ impl ParallelDualWriteIndexer {
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_indexable_chunk_structure() {
let chunk = IndexableChunk {
chunk_id: "c1".to_string(),
content: "test".to_string(),
source: "src".to_string(),
project: "proj".to_string(),
level: "L1".to_string(),
breadcrumb: vec!["a".to_string()],
};
assert_eq!(chunk.chunk_id, "c1");
}
#[test]
fn test_dual_write_result_structure() {
let result = DualWriteResult {
chunk_id: "c1".to_string(),
pgvector_success: true,
opensearch_success: true,
error: None,
};
assert!(result.pgvector_success);
}
#[test]
fn test_parallel_indexer_creation() {
let pool = sqlx::postgres::PgPoolOptions::new()
.max_connections(1)
.build_lazy();
let indexer = ParallelDualWriteIndexer::new(pool, None);
assert!(indexer.opensearch.is_none());
}
#[test]
fn test_hash_computation() {
let pool = sqlx::postgres::PgPoolOptions::new()
.max_connections(1)
.build_lazy();
let indexer = ParallelDualWriteIndexer::new(pool, None);
let hash1 = indexer.compute_hash("test");
let hash2 = indexer.compute_hash("test");
assert_eq!(hash1, hash2);
}
#[test]
fn test_hash_different_content() {
let pool = sqlx::postgres::PgPoolOptions::new()
.max_connections(1)
.build_lazy();
let indexer = ParallelDualWriteIndexer::new(pool, None);
let hash1 = indexer.compute_hash("test1");
let hash2 = indexer.compute_hash("test2");
assert_ne!(hash1, hash2);
}
#[test]
fn test_dual_write_result_pgvector_failed() {
let result = DualWriteResult {
chunk_id: "c1".to_string(),
pgvector_success: false,
opensearch_success: true,
error: Some("pgvector failed".to_string()),
};
assert!(!result.pgvector_success);
assert!(result.error.is_some());
}
#[test]
fn test_dual_write_result_opensearch_failed() {
let result = DualWriteResult {
chunk_id: "c1".to_string(),
pgvector_success: true,
opensearch_success: false,
error: Some("opensearch failed".to_string()),
};
assert!(result.pgvector_success);
assert!(!result.opensearch_success);
}
#[test]
fn test_breadcrumb_join() {
let breadcrumb = vec!["a".to_string(), "b".to_string(), "c".to_string()];
let joined = breadcrumb.join(" > ");
assert_eq!(joined, "a > b > c");
}
#[test]
fn test_chunk_source_tracking() {
let chunk = IndexableChunk {
chunk_id: "c1".to_string(),
content: "test".to_string(),
source: "transcript://session-123".to_string(),
project: "poimen".to_string(),
level: "L1".to_string(),
breadcrumb: vec![],
};
assert!(chunk.source.contains("session"));
}
}
@@ -285,9 +285,11 @@ impl BfsGraphTraversal {
pub fn truncate_to_depth(graph: &mut GraphData, max_depth: i32) {
graph.nodes.retain(|n| n.depth <= max_depth);
graph.edges.retain(|e| {
let source_exists = graph.nodes.iter().any(|n| n.id == e.source_id);
let target_exists = graph.nodes.iter().any(|n| n.id == e.target_id);
source_exists && target_exists
let source_depth = graph.nodes.iter()
.find(|n| n.id == e.source_id)
.map(|n| n.depth)
.unwrap_or(i32::MAX);
source_depth <= max_depth
});
graph.max_depth_reached = graph.max_depth_reached.min(max_depth);
@@ -343,3 +343,167 @@ impl CommunityDetector {
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_community_creation() {
let community = Community {
id: 0,
entity_ids: vec!["e1".to_string(), "e2".to_string()],
entity_names: vec!["Entity1".to_string(), "Entity2".to_string()],
size: 2,
modularity_contribution: 0.8,
average_strength: 0.9,
density: 1.0,
};
assert_eq!(community.size, 2);
assert_eq!(community.entity_ids.len(), 2);
}
#[test]
fn test_community_detection_result() {
let result = CommunityDetectionResult {
entity_count: 100,
edge_count: 250,
communities: vec![],
community_count: 0,
total_modularity: 0.0,
average_community_size: 0.0,
};
assert_eq!(result.entity_count, 100);
assert_eq!(result.edge_count, 250);
}
#[test]
fn test_min_community_size_clamping() {
let size = 1;
let clamped = size.max(2).min(1000);
assert_eq!(clamped, 2);
let size = 5000;
let clamped = size.max(2).min(1000);
assert_eq!(clamped, 1000);
}
#[test]
fn test_modularity_threshold_clamping() {
let threshold = 0.0001;
let clamped = threshold.max(0.0001).min(0.1);
assert_eq!(clamped, 0.0001);
let threshold = 0.5;
let clamped = threshold.max(0.0001).min(0.1);
assert_eq!(clamped, 0.1);
}
#[test]
fn test_density_calculation() {
// 3 entities, all connected (3 edges)
// Possible edges: 3 * 2 / 2 = 3
// Density: 3 / 3 = 1.0 (fully connected)
let density = (3.0 / 3.0).max(0.0).min(1.0);
assert_eq!(density, 1.0);
// 4 entities, 2 edges
// Possible: 4 * 3 / 2 = 6
// Density: 2 / 6 ≈ 0.33
let density = (2.0 / 6.0).max(0.0).min(1.0);
assert!((density - 0.333).abs() < 0.01);
}
#[test]
fn test_modularity_bounds() {
let modularity = 0.75;
let clamped = modularity.max(-1.0).min(1.0);
assert_eq!(clamped, 0.75);
let modularity = -0.5;
let clamped = modularity.max(-1.0).min(1.0);
assert_eq!(clamped, -0.5);
}
#[test]
fn test_average_community_size() {
let communities = vec![
Community {
id: 0,
entity_ids: vec!["a".into(), "b".into(), "c".into()],
entity_names: vec![],
size: 3,
modularity_contribution: 0.5,
average_strength: 0.8,
density: 0.9,
},
Community {
id: 1,
entity_ids: vec!["d".into(), "e".into()],
entity_names: vec![],
size: 2,
modularity_contribution: 0.4,
average_strength: 0.7,
density: 1.0,
},
];
let avg = communities.iter().map(|c| c.size as f32).sum::<f32>() / communities.len() as f32;
assert_eq!(avg, 2.5);
}
#[test]
fn test_total_modularity_sum() {
let contributions = vec![0.3, 0.25, 0.2, 0.15];
let total: f32 = contributions.iter().sum();
let clamped = total.max(-1.0).min(1.0);
assert!(clamped >= -1.0 && clamped <= 1.0);
}
#[test]
fn test_empty_graph_handling() {
let entities: Vec<String> = vec![];
let edges: Vec<GraphEdge> = vec![];
assert!(entities.is_empty());
assert!(edges.is_empty());
}
#[test]
fn test_single_node_graph() {
let entity_count = 1;
let edge_count = 0;
assert_eq!(entity_count, 1);
assert_eq!(edge_count, 0);
}
#[test]
fn test_fully_connected_graph() {
// 5 nodes fully connected: 5*4/2 = 10 edges
let nodes = 5;
let possible_edges = nodes * (nodes - 1) / 2;
assert_eq!(possible_edges, 10);
}
#[test]
fn test_strength_normalization() {
let strengths = vec![0.0, 0.25, 0.5, 0.75, 1.0];
for s in strengths {
let normalized = s.max(0.0).min(1.0);
assert!(normalized >= 0.0 && normalized <= 1.0);
}
}
#[test]
fn test_louvain_max_iterations() {
let max_iterations = 100;
let mut iteration = 0;
while iteration < max_iterations && iteration < 5 {
iteration += 1;
}
assert!(iteration <= max_iterations);
}
}
+177
View File
@@ -437,3 +437,180 @@ struct EntityInfo {
name: String,
}
#[cfg(test)]
mod tests {
use super::*;
fn create_linker_mock() -> EntityLinker {
// Create with in-memory pool (stub for testing)
let pool = sqlx::postgres::PgPoolOptions::new()
.max_connections(1)
.build_lazy();
EntityLinker::new(pool)
}
#[test]
fn test_extract_mentions_basic() {
let linker = create_linker_mock();
let text = "Kubernetes is a container orchestration platform.";
let mentions = linker.extract_mentions(text).unwrap();
assert!(mentions.len() > 0);
}
#[test]
fn test_extract_mentions_multiword() {
let linker = create_linker_mock();
let text = "Google Cloud Platform provides services.";
let mentions = linker.extract_mentions(text).unwrap();
assert!(mentions.iter().any(|m| m.text.contains("Cloud")));
}
#[test]
fn test_mention_link_structure() {
let link = MentionLink {
mention_text: "Kubernetes".to_string(),
start_offset: 0,
end_offset: 10,
entity_id: "e1".to_string(),
entity_name: "Kubernetes".to_string(),
confidence: 0.95,
reason: LinkReason::LexicalMatch,
};
assert_eq!(link.confidence, 0.95);
}
#[test]
fn test_link_reason_enum() {
let reasons = vec![
LinkReason::SemanticMatch,
LinkReason::LexicalMatch,
LinkReason::AliasMatch,
LinkReason::AcronymMatch,
LinkReason::PartialMatch,
];
assert_eq!(reasons.len(), 5);
}
#[test]
fn test_alias_suggestion_structure() {
let alias = AliasSuggestion {
entity_id: "e1".to_string(),
canonical_name: "Kubernetes".to_string(),
alias: "k8s".to_string(),
confidence: 0.9,
frequency: 5,
};
assert_eq!(alias.frequency, 5);
}
#[test]
fn test_merge_suggestion_structure() {
let merge = MergeSuggestion {
entity1_id: "e1".to_string(),
entity1_name: "Kubernetes".to_string(),
entity2_id: "e2".to_string(),
entity2_name: "K8s".to_string(),
confidence: 0.85,
reasons: vec!["Acronym match".to_string()],
};
assert_eq!(merge.confidence, 0.85);
assert_eq!(merge.reasons.len(), 1);
}
#[test]
fn test_coreference_cluster_structure() {
let cluster = CoreferenceCluster {
entity_id: "e1".to_string(),
mentions: vec!["Kubernetes".to_string(), "k8s".to_string()],
mention_count: 2,
confidence: 0.85,
};
assert_eq!(cluster.mention_count, 2);
}
#[test]
fn test_edit_distance() {
let linker = create_linker_mock();
let dist = linker.edit_distance("Kubernetes", "kubernetes");
assert_eq!(dist, 0); // Same lowercase
}
#[test]
fn test_edit_distance_typo() {
let linker = create_linker_mock();
let dist = linker.edit_distance("Kubernetes", "Kubenetes");
assert!(dist > 0 && dist < 5);
}
#[test]
fn test_compute_similarity_exact() {
let linker = create_linker_mock();
let sim = linker.compute_similarity("test", "test");
assert_eq!(sim, 1.0);
}
#[test]
fn test_compute_similarity_case_insensitive() {
let linker = create_linker_mock();
let sim = linker.compute_similarity("Test", "test");
assert_eq!(sim, 1.0);
}
#[test]
fn test_compute_similarity_substring() {
let linker = create_linker_mock();
let sim = linker.compute_similarity("Kubernetes", "kubernetes");
assert!(sim > 0.8);
}
#[test]
fn test_is_acronym_true() {
let linker = create_linker_mock();
let is_acr = linker.is_acronym("k8s", "Kubernetes");
assert!(is_acr);
}
#[test]
fn test_is_acronym_false() {
let linker = create_linker_mock();
let is_acr = linker.is_acronym("test", "Kubernetes");
assert!(!is_acr);
}
#[test]
fn test_is_similar_true() {
let linker = create_linker_mock();
let similar = linker.is_similar("Kubernetes", "kubernetes");
assert!(similar);
}
#[test]
fn test_is_similar_false() {
let linker = create_linker_mock();
let similar = linker.is_similar("test", "completely different");
assert!(!similar);
}
#[test]
fn test_mention_link_reason_serialization() {
let reason = LinkReason::SemanticMatch;
let json = serde_json::to_string(&reason).unwrap();
assert!(json.contains("SemanticMatch"));
}
#[test]
fn test_mention_link_full_serialization() {
let link = MentionLink {
mention_text: "Kubernetes".to_string(),
start_offset: 0,
end_offset: 10,
entity_id: "e1".to_string(),
entity_name: "Kubernetes".to_string(),
confidence: 0.95,
reason: LinkReason::LexicalMatch,
};
let json = serde_json::to_string(&link).unwrap();
assert!(json.contains("Kubernetes"));
assert!(json.contains("0.95"));
}
}
+249
View File
@@ -360,3 +360,252 @@ impl FacetedSearch {
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_facet_value_creation() {
let facet = FacetValue {
name: "concept".to_string(),
count: 42,
percentage: 15.5,
};
assert_eq!(facet.name, "concept");
assert_eq!(facet.count, 42);
assert!((facet.percentage - 15.5).abs() < 0.01);
}
#[test]
fn test_facet_type_enum() {
let types = vec![
FacetType::EntityType,
FacetType::RelationType,
FacetType::ConfidenceLevel,
FacetType::DateRange,
];
assert_eq!(types.len(), 4);
}
#[test]
fn test_facet_filters_default() {
let filters = FacetFilters::default();
assert!(filters.entity_types.is_none());
assert!(filters.relation_types.is_none());
assert!(filters.confidence_level.is_none());
assert!(filters.date_range.is_none());
}
#[test]
fn test_confidence_floor_high() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let floor = engine.confidence_floor_from_level(Some("high"));
assert_eq!(floor, 0.8);
}
#[test]
fn test_confidence_floor_medium() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let floor = engine.confidence_floor_from_level(Some("medium"));
assert_eq!(floor, 0.5);
}
#[test]
fn test_confidence_floor_low() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let floor = engine.confidence_floor_from_level(Some("low"));
assert_eq!(floor, 0.0);
}
#[test]
fn test_confidence_floor_none() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let floor = engine.confidence_floor_from_level(None);
assert_eq!(floor, 0.0);
}
#[test]
fn test_facet_percentage_calculation() {
let count = 25;
let total = 100;
let percentage = (count as f32 / total as f32) * 100.0;
assert_eq!(percentage, 25.0);
}
#[test]
fn test_facet_percentage_zero_total() {
let total = 0;
let percentage = if total > 0 { 100.0 } else { 0.0 };
assert_eq!(percentage, 0.0);
}
#[test]
fn test_date_range_today() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let (start, end) = engine.date_range_to_times(Some("today"));
assert!(start.is_some());
assert!(end.is_some());
assert!(start.unwrap() < end.unwrap());
}
#[test]
fn test_date_range_week() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let (start, end) = engine.date_range_to_times(Some("this_week"));
assert!(start.is_some());
assert!(end.is_some());
}
#[test]
fn test_date_range_month() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let (start, end) = engine.date_range_to_times(Some("this_month"));
assert!(start.is_some());
assert!(end.is_some());
}
#[test]
fn test_date_range_none() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let (start, end) = engine.date_range_to_times(None);
assert!(start.is_none());
assert!(end.is_none());
}
#[test]
fn test_validate_filters_empty_entity_types() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
entity_types: Some(vec![]),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_err());
}
#[test]
fn test_validate_filters_valid_entity_types() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
entity_types: Some(vec!["concept".to_string()]),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_ok());
}
#[test]
fn test_validate_filters_too_many_types() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
entity_types: Some((0..60).map(|i| format!("type_{}", i)).collect()),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_err());
}
#[test]
fn test_validate_filters_invalid_confidence() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
confidence_level: Some("invalid".to_string()),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_err());
}
#[test]
fn test_validate_filters_valid_confidence() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
confidence_level: Some("high".to_string()),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_ok());
}
#[test]
fn test_validate_filters_invalid_date_range() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
date_range: Some("invalid".to_string()),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_err());
}
#[test]
fn test_validate_filters_valid_date_range() {
let engine = FacetedSearch { pool: unsafe { std::mem::zeroed() } };
let filters = FacetFilters {
date_range: Some("this_week".to_string()),
..Default::default()
};
assert!(engine.validate_filters(&filters).is_ok());
}
#[test]
fn test_faceted_result_structure() {
let results: Vec<String> = vec!["e1".to_string(), "e2".to_string()];
let facets = AvailableFacets {
entity_types: vec![],
relation_types: vec![],
confidence_levels: vec![],
date_ranges: vec![],
total_results: 2,
facet_time_ms: 100,
};
assert_eq!(results.len(), 2);
assert_eq!(facets.total_results, 2);
}
#[test]
fn test_limit_clamping_min() {
let limit = 2;
let clamped = limit.max(5).min(50);
assert_eq!(clamped, 5);
}
#[test]
fn test_limit_clamping_max() {
let limit = 100;
let clamped = limit.max(5).min(50);
assert_eq!(clamped, 50);
}
#[test]
fn test_available_facets_empty() {
let facets = AvailableFacets {
entity_types: vec![],
relation_types: vec![],
confidence_levels: vec![],
date_ranges: vec![],
total_results: 0,
facet_time_ms: 0,
};
assert_eq!(facets.total_results, 0);
assert!(facets.entity_types.is_empty());
}
}
@@ -232,8 +232,8 @@ mod tests {
let (fx, fy) = ForceDirectedLayout::repulsive_force(p1, p2, -800.0);
// Should push p1 away from p2 (positive force = repulsion from p2 at +x)
assert!(fx > 0.0);
// Should push p1 away from p2 (negative x)
assert!(fx < 0.0);
assert_eq!(fy, 0.0); // No y component
}
@@ -365,3 +365,321 @@ struct EdgeInfo {
relation_type: String,
}
#[cfg(test)]
mod tests {
use super::*;
fn create_test_rules() -> Vec<InferenceRule> {
vec![
InferenceRule {
id: "r1".to_string(),
antecedent: "depends_on".to_string(),
medial: None,
consequent: "related_to".to_string(),
confidence_multiplier: 0.9,
description: "Depends implies related".to_string(),
},
InferenceRule {
id: "r2".to_string(),
antecedent: "uses".to_string(),
medial: None,
consequent: "related_to".to_string(),
confidence_multiplier: 0.85,
description: "Uses implies related".to_string(),
},
]
}
#[test]
fn test_inference_rule_structure() {
let rule = InferenceRule {
id: "r1".to_string(),
antecedent: "depends_on".to_string(),
medial: None,
consequent: "related_to".to_string(),
confidence_multiplier: 0.9,
description: "Test rule".to_string(),
};
assert_eq!(rule.antecedent, "depends_on");
assert_eq!(rule.consequent, "related_to");
}
#[test]
fn test_inferred_fact_structure() {
let fact = InferredFact {
source_id: "e1".to_string(),
source_name: "Entity1".to_string(),
target_id: "e2".to_string(),
target_name: "Entity2".to_string(),
relation_type: "related_to".to_string(),
confidence: 0.81,
reasoning_chain: vec!["e1 --depends_on→ e2".to_string()],
rule_ids: vec!["r1".to_string()],
};
assert_eq!(fact.confidence, 0.81);
assert_eq!(fact.reasoning_chain.len(), 1);
}
#[test]
fn test_reasoning_path_structure() {
let path = ReasoningPath {
path: vec!["e1".to_string(), "e2".to_string(), "e3".to_string()],
relations: vec!["depends_on".to_string(), "uses".to_string()],
confidence: 0.75,
step_count: 3,
};
assert_eq!(path.step_count, 3);
assert_eq!(path.path.len(), 3);
}
#[test]
fn test_transitive_closure_structure() {
let closure = TransitiveClosure {
source_id: "e1".to_string(),
reachable: vec![],
entity_count: 0,
edge_count: 0,
};
assert_eq!(closure.entity_count, 0);
}
#[test]
fn test_reachable_entity_structure() {
let entity = ReachableEntity {
entity_id: "e2".to_string(),
entity_name: "Entity2".to_string(),
relation_type: "related_to".to_string(),
confidence: 0.85,
distance: 1,
};
assert_eq!(entity.distance, 1);
assert!(entity.confidence > 0.8);
}
#[test]
fn test_confidence_multiplier() {
let rule = &create_test_rules()[0];
let base_confidence = 0.9;
let result = base_confidence * rule.confidence_multiplier;
assert!(result < base_confidence);
}
#[test]
fn test_confidence_decay_single_hop() {
let confidence = 1.0;
let decay = 0.95;
let result = confidence * decay;
assert_eq!(result, 0.95);
}
#[test]
fn test_confidence_decay_two_hops() {
let confidence = 1.0;
let decay = 0.95;
let result = confidence * decay * decay;
assert!((result - 0.9025).abs() < 0.0001);
}
#[test]
fn test_confidence_chaining() {
let conf1 = 0.9;
let conf2 = 0.85;
let result = conf1 * conf2;
assert!((result - 0.765).abs() < 0.0001);
}
#[test]
fn test_confidence_bounds() {
let confidence = 0.95 * 1.1; // Exceed 1.0
let bounded = confidence.min(1.0);
assert_eq!(bounded, 1.0);
}
#[test]
fn test_rule_matching() {
let rules = create_test_rules();
let rule = rules.iter().find(|r| r.antecedent == "depends_on").unwrap();
assert_eq!(rule.consequent, "related_to");
}
#[test]
fn test_rule_no_match() {
let rules = create_test_rules();
let rule = rules.iter().find(|r| r.antecedent == "nonexistent");
assert!(rule.is_none());
}
#[test]
fn test_inferred_fact_confidence_calculation() {
let base = 1.0;
let multiplier = 0.9;
let final_conf = (base * multiplier).min(1.0);
assert_eq!(final_conf, 0.9);
}
#[test]
fn test_reasoning_chain_construction() {
let chain = vec![
"e1 --depends_on→ e2".to_string(),
"e2 --uses→ e3".to_string(),
];
assert_eq!(chain.len(), 2);
}
#[test]
fn test_path_step_count() {
let path_len = 3;
let step_count = path_len;
assert_eq!(step_count, 3);
}
#[test]
fn test_hop_distance_tracking() {
let mut distance = 0;
distance += 1; // Hop 1
distance += 1; // Hop 2
assert_eq!(distance, 2);
}
#[test]
fn test_max_hops_limit() {
let max_hops = 5;
let current_hops = 3;
assert!(current_hops < max_hops);
}
#[test]
fn test_rule_confidence_multiplier_range() {
let multipliers = vec![0.5, 0.75, 0.9, 0.95, 1.0];
for mult in multipliers {
assert!(mult >= 0.0 && mult <= 1.0);
}
}
#[test]
fn test_empty_reasoning_paths() {
let paths: Vec<ReasoningPath> = vec![];
assert!(paths.is_empty());
}
#[test]
fn test_single_hop_reasoning() {
let path = vec!["e1".to_string(), "e2".to_string()];
assert_eq!(path.len(), 2);
}
#[test]
fn test_multi_hop_reasoning() {
let path = vec![
"e1".to_string(),
"e2".to_string(),
"e3".to_string(),
"e4".to_string(),
];
assert_eq!(path.len(), 4);
}
#[test]
fn test_relation_chain_length() {
let relations = vec!["depends_on".to_string(), "uses".to_string()];
assert_eq!(relations.len(), 2);
}
#[test]
fn test_inference_deduplication() {
let facts = vec![
InferredFact {
source_id: "e1".to_string(),
source_name: "E1".to_string(),
target_id: "e2".to_string(),
target_name: "E2".to_string(),
relation_type: "related".to_string(),
confidence: 0.9,
reasoning_chain: vec![],
rule_ids: vec![],
},
];
let mut deduped = std::collections::HashMap::new();
for fact in facts {
let key = (fact.source_id.clone(), fact.target_id.clone(), fact.relation_type.clone());
deduped.insert(key, fact);
}
assert_eq!(deduped.len(), 1);
}
#[test]
fn test_transitive_closure_empty() {
let closure = TransitiveClosure {
source_id: "e1".to_string(),
reachable: vec![],
entity_count: 0,
edge_count: 0,
};
assert_eq!(closure.reachable.len(), 0);
}
#[test]
fn test_transitive_closure_single_hop() {
let reachable = vec![
ReachableEntity {
entity_id: "e2".to_string(),
entity_name: "E2".to_string(),
relation_type: "depends_on".to_string(),
confidence: 0.95,
distance: 1,
},
];
assert_eq!(reachable.len(), 1);
assert_eq!(reachable[0].distance, 1);
}
#[test]
fn test_transitive_closure_multi_hop() {
let reachable = vec![
ReachableEntity {
entity_id: "e2".to_string(),
entity_name: "E2".to_string(),
relation_type: "depends_on".to_string(),
confidence: 0.95,
distance: 1,
},
ReachableEntity {
entity_id: "e3".to_string(),
entity_name: "E3".to_string(),
relation_type: "depends_on".to_string(),
confidence: 0.90,
distance: 2,
},
];
assert_eq!(reachable.len(), 2);
assert!(reachable[1].confidence < reachable[0].confidence);
}
#[test]
fn test_serialization_inferred_fact() {
let fact = InferredFact {
source_id: "e1".to_string(),
source_name: "E1".to_string(),
target_id: "e2".to_string(),
target_name: "E2".to_string(),
relation_type: "related".to_string(),
confidence: 0.81,
reasoning_chain: vec!["e1 --depends_on→ e2".to_string()],
rule_ids: vec!["r1".to_string()],
};
let json = serde_json::to_string(&fact).unwrap();
assert!(json.contains("0.81"));
}
#[test]
fn test_serialization_reasoning_path() {
let path = ReasoningPath {
path: vec!["e1".to_string(), "e2".to_string()],
relations: vec!["depends_on".to_string()],
confidence: 0.9,
step_count: 2,
};
let json = serde_json::to_string(&path).unwrap();
assert!(json.contains("0.9"));
}
}
+294
View File
@@ -413,3 +413,297 @@ impl QueryReasoner {
}
}
#[cfg(test)]
mod tests {
use super::*;
fn create_reasoner_mock() -> QueryReasoner {
let pool = sqlx::postgres::PgPoolOptions::new()
.max_connections(1)
.build_lazy();
QueryReasoner::new(pool)
}
#[test]
fn test_question_type_factual() {
let reasoner = create_reasoner_mock();
let qt = reasoner.classify_question("What is Kubernetes?");
assert_eq!(qt, QuestionType::Factual);
}
#[test]
fn test_question_type_relationship() {
let reasoner = create_reasoner_mock();
let qt = reasoner.classify_question("How does Docker relate to Kubernetes?");
assert_eq!(qt, QuestionType::Relationship);
}
#[test]
fn test_question_type_causal() {
let reasoner = create_reasoner_mock();
let qt = reasoner.classify_question("Why is Kubernetes essential?");
assert_eq!(qt, QuestionType::Causal);
}
#[test]
fn test_question_type_comparative() {
let reasoner = create_reasoner_mock();
let qt = reasoner.classify_question("Compare Docker versus Kubernetes");
assert_eq!(qt, QuestionType::Comparative);
}
#[test]
fn test_question_type_set_query() {
let reasoner = create_reasoner_mock();
let qt = reasoner.classify_question("Find all containerization tools");
assert_eq!(qt, QuestionType::SetQuery);
}
#[test]
fn test_question_type_consequence() {
let reasoner = create_reasoner_mock();
let qt = reasoner.classify_question("What are the consequences of using Kubernetes?");
assert_eq!(qt, QuestionType::Consequence);
}
#[test]
fn test_extract_entities() {
let reasoner = create_reasoner_mock();
let entities = reasoner.extract_entities_from_question("How does Kubernetes work with Docker?");
assert!(entities.contains(&"Kubernetes".to_string()));
assert!(entities.contains(&"Docker".to_string()));
}
#[test]
fn test_extract_relations_depends() {
let reasoner = create_reasoner_mock();
let relations = reasoner.extract_relations_from_question("What does Kubernetes depend on?");
assert!(relations.contains(&"depends_on".to_string()));
}
#[test]
fn test_extract_relations_uses() {
let reasoner = create_reasoner_mock();
let relations = reasoner.extract_relations_from_question("Kubernetes uses containers");
assert!(relations.contains(&"uses".to_string()));
}
#[test]
fn test_extract_constraints_high_confidence() {
let reasoner = create_reasoner_mock();
let constraints = reasoner.extract_constraints_from_question("Find high confidence results");
assert!(constraints.iter().any(|c| c.constraint_type == "confidence"));
}
#[test]
fn test_constraint_equals() {
let reasoner = create_reasoner_mock();
let constraint = Constraint {
constraint_type: "type".to_string(),
operator: "==".to_string(),
value: "entity".to_string(),
};
assert!(reasoner.check_constraint("entity", &constraint));
assert!(!reasoner.check_constraint("edge", &constraint));
}
#[test]
fn test_constraint_in() {
let reasoner = create_reasoner_mock();
let constraint = Constraint {
constraint_type: "type".to_string(),
operator: "in".to_string(),
value: "entity,edge,fact".to_string(),
};
assert!(reasoner.check_constraint("entity", &constraint));
assert!(reasoner.check_constraint("edge", &constraint));
assert!(!reasoner.check_constraint("other", &constraint));
}
#[test]
fn test_constraint_contains() {
let reasoner = create_reasoner_mock();
let constraint = Constraint {
constraint_type: "text".to_string(),
operator: "contains".to_string(),
value: "test".to_string(),
};
assert!(reasoner.check_constraint("this is a test", &constraint));
assert!(!reasoner.check_constraint("this is not it", &constraint));
}
#[test]
fn test_subquery_structure() {
let sq = SubQuery {
id: "sq1".to_string(),
question: "What is X?".to_string(),
question_type: QuestionType::Factual,
entity_ids: vec!["e1".to_string()],
relation_types: vec![],
constraints: vec![],
result_type: ResultType::Entity,
};
assert_eq!(sq.question_type, QuestionType::Factual);
}
#[test]
fn test_reasoning_step_structure() {
let step = ReasoningStep {
step_id: 1,
sub_query: SubQuery {
id: "sq1".to_string(),
question: "Test".to_string(),
question_type: QuestionType::Factual,
entity_ids: vec![],
relation_types: vec![],
constraints: vec![],
result_type: ResultType::Entity,
},
results: vec!["answer1".to_string()],
confidence: 0.9,
constraints_satisfied: 1,
constraints_total: 1,
};
assert_eq!(step.step_id, 1);
assert_eq!(step.confidence, 0.9);
}
#[test]
fn test_reasoned_answer_structure() {
let answer = ReasonedAnswer {
question: "Test question".to_string(),
answers: vec!["answer1".to_string()],
confidence: 0.9,
reasoning_steps: vec![],
evidence: vec![],
explanation: "Explanation".to_string(),
};
assert_eq!(answer.answers.len(), 1);
}
#[test]
fn test_decompose_empty_question() {
let reasoner = create_reasoner_mock();
let result = reasoner.decompose_question("").unwrap();
assert!(result.is_empty());
}
#[test]
fn test_decompose_simple_question() {
let reasoner = create_reasoner_mock();
let result = reasoner.decompose_question("What is Kubernetes?").unwrap();
assert!(!result.is_empty());
assert_eq!(result[0].question_type, QuestionType::Factual);
}
#[test]
fn test_decompose_complex_question() {
let reasoner = create_reasoner_mock();
let result = reasoner.decompose_question("Why is Kubernetes important?").unwrap();
assert!(result.len() >= 1);
}
#[test]
fn test_infer_result_type_factual() {
let reasoner = create_reasoner_mock();
let rt = reasoner.infer_result_type(&QuestionType::Factual);
assert_eq!(rt, ResultType::Entity);
}
#[test]
fn test_infer_result_type_set_query() {
let reasoner = create_reasoner_mock();
let rt = reasoner.infer_result_type(&QuestionType::SetQuery);
assert_eq!(rt, ResultType::Entities);
}
#[test]
fn test_constraint_serialization() {
let constraint = Constraint {
constraint_type: "test".to_string(),
operator: "==".to_string(),
value: "val".to_string(),
};
let json = serde_json::to_string(&constraint).unwrap();
assert!(json.contains("test"));
}
#[test]
fn test_subquery_serialization() {
let sq = SubQuery {
id: "sq1".to_string(),
question: "Test?".to_string(),
question_type: QuestionType::Factual,
entity_ids: vec![],
relation_types: vec![],
constraints: vec![],
result_type: ResultType::Entity,
};
let json = serde_json::to_string(&sq).unwrap();
assert!(json.contains("Test?"));
}
#[test]
fn test_validate_answer_no_constraints() {
let reasoner = create_reasoner_mock();
let valid = reasoner.validate_answer("answer", &[]).unwrap();
assert!(valid);
}
#[test]
fn test_validate_answer_with_constraint() {
let reasoner = create_reasoner_mock();
let constraint = Constraint {
constraint_type: "type".to_string(),
operator: "==".to_string(),
value: "entity".to_string(),
};
let valid = reasoner.validate_answer("entity", &[constraint]).unwrap();
assert!(valid);
}
#[test]
fn test_apply_constraints_empty() {
let reasoner = create_reasoner_mock();
let results = vec!["r1".to_string(), "r2".to_string()];
let filtered = reasoner.apply_constraints(&results, &[]);
assert_eq!(filtered.len(), 2);
}
#[test]
fn test_apply_constraints_filter() {
let reasoner = create_reasoner_mock();
let results = vec!["entity".to_string(), "edge".to_string()];
let constraint = Constraint {
constraint_type: "type".to_string(),
operator: "==".to_string(),
value: "entity".to_string(),
};
let filtered = reasoner.apply_constraints(&results, &[constraint]);
assert_eq!(filtered.len(), 1);
assert_eq!(filtered[0], "entity");
}
#[test]
fn test_generate_explanation() {
let reasoner = create_reasoner_mock();
let step = ReasoningStep {
step_id: 1,
sub_query: SubQuery {
id: "sq1".to_string(),
question: "Test".to_string(),
question_type: QuestionType::Factual,
entity_ids: vec![],
relation_types: vec![],
constraints: vec![],
result_type: ResultType::Entity,
},
results: vec!["ans".to_string()],
confidence: 0.9,
constraints_satisfied: 0,
constraints_total: 0,
};
let expl = reasoner.generate_explanation(&[step], &["ans".to_string()]);
assert!(expl.contains("reasoning"));
}
}
@@ -327,3 +327,149 @@ impl SemanticRetriever {
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_entity_result_creation() {
let result = EntityResult {
id: "e1".to_string(),
name: "Test".to_string(),
entity_type: "concept".to_string(),
similarity_score: 0.95,
metadata: serde_json::json!({"key": "value"}),
};
assert_eq!(result.id, "e1");
assert_eq!(result.similarity_score, 0.95);
}
#[test]
fn test_edge_result_creation() {
let result = EdgeResult {
id: "e1".to_string(),
source_entity_id: "src".to_string(),
target_entity_id: "tgt".to_string(),
source_name: "A".to_string(),
target_name: "B".to_string(),
relation_type: "related_to".to_string(),
fact: "A is related to B".to_string(),
similarity_score: 0.88,
confidence: 0.90,
};
assert_eq!(result.similarity_score, 0.88);
assert_eq!(result.confidence, 0.90);
}
#[test]
fn test_hybrid_result_creation() {
let result = HybridResult {
id: "h1".to_string(),
name: Some("Test".to_string()),
entity_type: Some("concept".to_string()),
result_type: "entity".to_string(),
fused_score: 0.85,
semantic_score: 0.90,
lexical_score: 0.75,
};
assert!(result.fused_score >= 0.0 && result.fused_score <= 1.0);
}
#[test]
fn test_embedding_dimension_validation() {
let invalid_embedding = vec![0.5; 512]; // Wrong size
assert_eq!(invalid_embedding.len(), 512);
assert_ne!(invalid_embedding.len(), 768);
}
#[test]
fn test_confidence_floor_bounds() {
let floor = 0.5;
assert!(floor >= 0.0 && floor <= 1.0);
}
#[test]
fn test_top_k_bounds() {
let top_k = 50;
let clamped = top_k.max(1).min(100);
assert_eq!(clamped, 50);
let too_small = 0;
assert_eq!(too_small.max(1).min(100), 1);
let too_large = 500;
assert_eq!(too_large.max(1).min(100), 100);
}
#[test]
fn test_weight_normalization() {
let sem_w = 0.6;
let lex_w = 0.4;
let normalized_sem = sem_w.max(0.0).min(1.0);
let normalized_lex = lex_w.max(0.0).min(1.0);
assert_eq!(normalized_sem, 0.6);
assert_eq!(normalized_lex, 0.4);
}
#[test]
fn test_score_clamping() {
let scores = vec![0.5, 1.0, 1.5, -0.1, 0.999];
for score in scores {
let clamped = score.max(0.0).min(1.0);
assert!(clamped >= 0.0 && clamped <= 1.0);
}
}
#[test]
fn test_hybrid_result_type_values() {
let entity_result = HybridResult {
id: "e1".to_string(),
name: Some("Entity".to_string()),
entity_type: Some("concept".to_string()),
result_type: "entity".to_string(),
fused_score: 0.9,
semantic_score: 0.92,
lexical_score: 0.85,
};
assert_eq!(entity_result.result_type, "entity");
let edge_result = HybridResult {
id: "edge1".to_string(),
name: Some("fact".to_string()),
entity_type: None,
result_type: "edge".to_string(),
fused_score: 0.85,
semantic_score: 0.87,
lexical_score: 0.80,
};
assert_eq!(edge_result.result_type, "edge");
}
#[test]
fn test_sorting_by_score() {
let mut results = vec![
HybridResult {
id: "1".to_string(),
name: None,
entity_type: None,
result_type: "entity".to_string(),
fused_score: 0.5,
semantic_score: 0.5,
lexical_score: 0.5,
},
HybridResult {
id: "2".to_string(),
name: None,
entity_type: None,
result_type: "entity".to_string(),
fused_score: 0.9,
semantic_score: 0.9,
lexical_score: 0.9,
},
];
results.sort_by(|a, b| b.fused_score.partial_cmp(&a.fused_score).unwrap_or(std::cmp::Ordering::Equal));
assert_eq!(results[0].id, "2");
assert_eq!(results[1].id, "1");
}
}
+171 -11
View File
@@ -238,17 +238,6 @@ impl QueryRouter {
let latency_ms = start.elapsed().as_millis() as u64;
tracing::info!(
target: "observability",
event = "query_route",
route = "direct",
candidates = all_candidates.len(),
prefiltered = prefilter_size,
selected = selected_chunks.len(),
latency_ms = latency_ms,
"Query routing complete"
);
Ok(RoutedResult {
selected_chunks,
route,
@@ -344,3 +333,174 @@ impl WikiGraphBuilder {
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::collections::BTreeMap;
fn create_test_router() -> QueryRouter {
let vocab = Arc::new(BTreeMap::new());
let tfidf = Arc::new(GlobalTfIdfScorer::new(vocab));
let semantic = Arc::new(SemanticScorer::new());
QueryRouter::new(tfidf, semantic, RouterConfig::default())
}
fn create_test_wiki_graph() -> WikiLinkGraph {
let mut graph = WikiLinkGraph::new("test");
graph.add_link("index.md", "tools/kubectl.md");
graph.add_link("tools/kubectl.md", "debugging/pod-crashes.md");
graph.add_link("debugging/pod-crashes.md", "solutions/restart-pod.md");
graph
}
#[test]
fn test_router_config_default() {
let config = RouterConfig::default();
assert_eq!(config.max_wiki_hops, 3);
assert_eq!(config.score_threshold, 0.6);
assert_eq!(config.budget_bytes, 8192);
}
#[test]
fn test_wiki_graph_to_hashmap() {
let router = create_test_router();
let graph = create_test_wiki_graph();
let hashmap = router.wiki_graph_to_hashmap(&graph, "index.md");
assert!(hashmap.contains_key("index.md"));
assert!(hashmap.contains_key("tools/kubectl.md"));
assert!(hashmap.contains_key("debugging/pod-crashes.md"));
}
#[test]
fn test_calculate_wiki_distance_root() {
let router = create_test_router();
let graph = create_test_wiki_graph();
let hashmap = router.wiki_graph_to_hashmap(&graph, "index.md");
let distance = router.calculate_wiki_distance("index.md", "index.md", &hashmap);
assert_eq!(distance, Some(0));
}
#[test]
fn test_calculate_wiki_distance_direct_child() {
let router = create_test_router();
let graph = create_test_wiki_graph();
let hashmap = router.wiki_graph_to_hashmap(&graph, "index.md");
let distance = router.calculate_wiki_distance("tools/kubectl.md", "index.md", &hashmap);
assert_eq!(distance, Some(1));
}
#[test]
fn test_calculate_wiki_distance_grandchild() {
let router = create_test_router();
let graph = create_test_wiki_graph();
let hashmap = router.wiki_graph_to_hashmap(&graph, "index.md");
let distance = router.calculate_wiki_distance("debugging/pod-crashes.md", "index.md", &hashmap);
assert_eq!(distance, Some(2));
}
#[test]
fn test_calculate_wiki_distance_unreachable() {
let router = create_test_router();
let graph = create_test_wiki_graph();
let hashmap = router.wiki_graph_to_hashmap(&graph, "index.md");
let distance = router.calculate_wiki_distance("unknown.md", "index.md", &hashmap);
assert_eq!(distance, None);
}
#[tokio::test]
async fn test_route_direct() {
let router = create_test_router();
let candidates = vec![
("doc1".to_string(), "kubernetes pod debugging".to_string()),
("doc2".to_string(), "docker container deployment".to_string()),
];
let result = router.route_direct("kubernetes", candidates).await.unwrap();
assert_eq!(result.route, RetrievalRoute::Direct);
assert!(result.latency_ms >= 0);
}
#[tokio::test]
async fn test_route_with_wiki_graph() {
let router = create_test_router();
let graph = create_test_wiki_graph();
let candidates = vec![
("index.md".to_string(), "main index".to_string()),
("tools/kubectl.md".to_string(), "kubectl tool".to_string()),
("debugging/pod-crashes.md".to_string(), "debugging content".to_string()),
("unrelated.md".to_string(), "not in graph".to_string()),
];
let result = router
.route_with_wiki_graph("kubectl", &graph, "index.md", candidates)
.await
.unwrap();
// Should filter out "unrelated.md" (not reachable from index.md)
assert!(result.wiki_scope_size <= 4);
assert_eq!(result.route, RetrievalRoute::WikiScoped);
}
#[test]
fn test_wiki_graph_builder() {
let docs = vec![
("index.md", "# Index\nSee [[tools/kubectl.md]] for tools."),
("tools/kubectl.md", "# Kubectl\nSee [[debugging.md]] for debugging."),
];
let graph = WikiGraphBuilder::build_from_docs("test", docs).unwrap();
let reachable = graph.reachable_docs("index.md");
assert!(reachable.contains("index.md"));
assert!(reachable.contains("tools/kubectl.md"));
assert!(reachable.contains("debugging.md"));
}
#[test]
fn test_selected_chunk_structure() {
let chunk = SelectedChunk {
id: "doc1".to_string(),
text: "content".to_string(),
tfidf_score: 0.4,
semantic_score: 0.6,
final_score: 0.9,
wiki_distance: Some(1),
};
assert_eq!(chunk.id, "doc1");
assert!(chunk.final_score <= 1.0);
assert_eq!(chunk.wiki_distance, Some(1));
}
#[test]
fn test_routed_result_structure() {
let result = RoutedResult {
selected_chunks: vec![],
route: RetrievalRoute::WikiScoped,
wiki_scope_size: 10,
prefilter_size: 5,
metrics: SelectionMetrics {
selected_count: 3,
rejected_count: 2,
total_bytes: 1000,
budget_used_pct: 12.5,
avg_score: 0.8,
dedup_removed: 0,
},
latency_ms: 50,
};
assert_eq!(result.wiki_scope_size, 10);
assert_eq!(result.prefilter_size, 5);
assert_eq!(result.metrics.selected_count, 3);
}
}
-161
View File
@@ -1,161 +0,0 @@
//! Relevance Judge (O4)
//!
//! Evaluates retrieval quality by scoring query-result relevance.
//! Uses LLM (Qwen-7B or similar) to judge if retrieved results are relevant.
//! Tracks precision, recall, F1 via Prometheus metrics.
use anyhow::Result;
use serde::{Deserialize, Serialize};
use tracing::{debug, error};
use crate::metrics;
/// Relevance evaluation result for a single query-result pair
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct RelevanceResult {
pub query: String,
pub result_text: String,
pub score: f64,
pub relevant: bool,
}
/// Batch evaluation summary
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct RelevanceSummary {
pub total: usize,
pub relevant: usize,
pub irrelevant: usize,
pub precision: f64,
pub recall: f64,
pub f1: f64,
pub avg_score: f64,
}
/// Simple relevance judge using cosine similarity threshold
/// (LLM-based judge can be plugged in later via trait)
pub struct RelevanceJudge {
threshold: f64,
}
impl RelevanceJudge {
pub fn new(threshold: f64) -> Self {
Self { threshold }
}
/// Evaluate a single query-result pair using similarity score
pub fn evaluate(&self, query: &str, result_text: &str, similarity: f64) -> RelevanceResult {
let start = std::time::Instant::now();
metrics::RELEVANCE_EVALS_TOTAL.inc();
let relevant = similarity >= self.threshold;
if relevant {
metrics::RELEVANCE_RELEVANT_TOTAL.inc();
} else {
metrics::RELEVANCE_IRRELEVANT_TOTAL.inc();
}
metrics::RELEVANCE_SCORE.observe(similarity);
metrics::RELEVANCE_EVAL_DURATION.observe(start.elapsed().as_secs_f64());
debug!("Relevance eval: query='{}', score={:.3}, relevant={}",
&query[..query.len().min(50)], similarity, relevant);
RelevanceResult {
query: query.to_string(),
result_text: result_text.to_string(),
score: similarity,
relevant,
}
}
/// Evaluate a batch of results and compute summary metrics
pub fn evaluate_batch(
&self,
query: &str,
results: &[(String, f64)], // (result_text, similarity_score)
) -> RelevanceSummary {
let mut relevant_count = 0;
let mut total_score = 0.0;
for (text, score) in results {
let result = self.evaluate(query, text, *score);
if result.relevant {
relevant_count += 1;
}
total_score += score;
}
let total = results.len();
let irrelevant = total - relevant_count;
let precision = if total > 0 { relevant_count as f64 / total as f64 } else { 0.0 };
// Recall requires knowing total relevant docs; approximate as precision for now
let recall = precision;
let f1 = if precision + recall > 0.0 {
2.0 * precision * recall / (precision + recall)
} else {
0.0
};
let avg_score = if total > 0 { total_score / total as f64 } else { 0.0 };
// Update gauge metrics
metrics::RELEVANCE_PRECISION.set(precision);
metrics::RELEVANCE_RECALL.set(recall);
metrics::RELEVANCE_F1.set(f1);
RelevanceSummary {
total,
relevant: relevant_count,
irrelevant,
precision,
recall,
f1,
avg_score,
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_relevance_judge_above_threshold() {
let judge = RelevanceJudge::new(0.5);
let result = judge.evaluate("test query", "test result", 0.8);
assert!(result.relevant);
assert!((result.score - 0.8).abs() < 0.001);
}
#[test]
fn test_relevance_judge_below_threshold() {
let judge = RelevanceJudge::new(0.5);
let result = judge.evaluate("test query", "test result", 0.3);
assert!(!result.relevant);
}
#[test]
fn test_relevance_batch() {
let judge = RelevanceJudge::new(0.5);
let results = vec![
("relevant result".to_string(), 0.8),
("somewhat relevant".to_string(), 0.6),
("irrelevant".to_string(), 0.2),
];
let summary = judge.evaluate_batch("test", &results);
assert_eq!(summary.total, 3);
assert_eq!(summary.relevant, 2);
assert_eq!(summary.irrelevant, 1);
assert!((summary.precision - 0.6667).abs() < 0.01);
}
#[test]
fn test_relevance_empty_batch() {
let judge = RelevanceJudge::new(0.5);
let summary = judge.evaluate_batch("test", &[]);
assert_eq!(summary.total, 0);
assert_eq!(summary.precision, 0.0);
assert_eq!(summary.f1, 0.0);
}
}
-12
View File
@@ -235,18 +235,6 @@ impl BudgetCompressor {
let strategy = self.select_strategy(estimated);
let compressed = self.compressor.compress_batch(results, strategy);
let compressed_size: usize = compressed.iter().map(|c| c.text.as_ref().map_or(0, |t| t.len())).sum();
tracing::info!(
target: "observability",
event = "result_compress",
input_count = compressed.len(),
estimated_bytes = estimated,
compressed_bytes = compressed_size,
budget_bytes = self.max_budget_bytes,
strategy = ?strategy,
"Result compression complete"
);
(compressed, strategy)
}
}
-301
View File
@@ -1,301 +0,0 @@
/// Agent-specific entity metadata for Phase 3 Agent Self-Awareness.
///
/// These structures attach to Entity via entity_type discriminator.
/// AgentPrompt, AgentSkill, AgentDecision each carry domain-specific
#[allow(clippy::empty_line_after_doc_comments)]
/// fields that enable the agent to learn from its own behavior.
use serde::{Deserialize, Serialize};
use time::OffsetDateTime;
use crate::entity::{Entity, EntityType};
/// Metadata for an AgentPrompt entity.
/// Tracks prompt templates, their usage frequency, and effectiveness.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct AgentPromptMeta {
/// The prompt template text (may contain {{placeholders}}).
pub template: String,
/// Which LLM model this prompt targets (e.g. "claude-3-sonnet").
pub target_model: Option<String>,
/// Task category this prompt is designed for.
pub task_category: String,
/// Number of times this prompt has been used.
pub usage_count: u64,
/// Average quality score from outcomes (0.0-1.0).
pub avg_quality: f32,
/// Last time this prompt was used.
#[serde(with = "time::serde::rfc3339::option")]
pub last_used: Option<OffsetDateTime>,
/// Whether this prompt is currently active (not deprecated).
pub active: bool,
/// Version for tracking prompt evolution.
pub version: u32,
/// Tags for categorization.
pub tags: Vec<String>,
}
/// Metadata for an AgentSkill entity.
/// Tracks learned capabilities and their effectiveness.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct AgentSkillMeta {
/// Description of what this skill does.
pub description: String,
/// Trigger conditions that activate this skill.
pub trigger_patterns: Vec<String>,
/// Success rate over all invocations (0.0-1.0).
pub success_rate: f32,
/// Number of times this skill was invoked.
pub invocation_count: u64,
/// Average latency in milliseconds.
pub avg_latency_ms: u64,
/// Linked prompt entity IDs that this skill uses.
pub linked_prompts: Vec<String>,
/// Whether this skill is currently enabled.
pub enabled: bool,
}
/// Metadata for an AgentDecision entity.
/// Records a decision the agent made, including reasoning and outcome.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct AgentDecisionMeta {
/// What the agent decided to do.
pub action: String,
/// Why the agent chose this action.
pub reasoning: String,
/// Available alternatives that were considered.
pub alternatives: Vec<String>,
/// Confidence in the decision (0.0-1.0).
pub confidence: f32,
/// Outcome of the decision (set after execution).
pub outcome: Option<DecisionOutcome>,
/// Context that informed the decision (entity IDs).
pub context_entities: Vec<String>,
/// The tool/task context when decision was made.
pub tool: Option<String>,
pub task: Option<String>,
}
/// Outcome of an agent decision.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct DecisionOutcome {
/// Whether the decision led to success.
pub success: bool,
/// Quality score of the outcome (0.0-1.0).
pub quality: f32,
/// Feedback or error message.
pub feedback: Option<String>,
/// When the outcome was recorded.
#[serde(with = "time::serde::rfc3339")]
pub recorded_at: OffsetDateTime,
}
// --- Factory functions ---
/// Create a new AgentPrompt entity.
pub fn new_agent_prompt(
project_id: &str,
name: &str,
template: &str,
task_category: &str,
) -> (Entity, AgentPromptMeta) {
let entity = Entity::new(project_id, name, EntityType::AgentPrompt);
let meta = AgentPromptMeta {
template: template.to_string(),
target_model: None,
task_category: task_category.to_string(),
usage_count: 0,
avg_quality: 0.0,
last_used: None,
active: true,
version: 1,
tags: vec![],
};
(entity, meta)
}
/// Create a new AgentSkill entity.
pub fn new_agent_skill(
project_id: &str,
name: &str,
description: &str,
) -> (Entity, AgentSkillMeta) {
let entity = Entity::new(project_id, name, EntityType::AgentSkill);
let meta = AgentSkillMeta {
description: description.to_string(),
trigger_patterns: vec![],
success_rate: 0.0,
invocation_count: 0,
avg_latency_ms: 0,
linked_prompts: vec![],
enabled: true,
};
(entity, meta)
}
/// Create a new AgentDecision entity.
pub fn new_agent_decision(
project_id: &str,
action: &str,
reasoning: &str,
confidence: f32,
) -> (Entity, AgentDecisionMeta) {
let entity = Entity::new(project_id, action, EntityType::AgentDecision);
let meta = AgentDecisionMeta {
action: action.to_string(),
reasoning: reasoning.to_string(),
alternatives: vec![],
confidence,
outcome: None,
context_entities: vec![],
tool: None,
task: None,
};
(entity, meta)
}
/// Record outcome for a decision.
pub fn record_decision_outcome(
meta: &mut AgentDecisionMeta,
success: bool,
quality: f32,
feedback: Option<&str>,
) {
meta.outcome = Some(DecisionOutcome {
success,
quality,
feedback: feedback.map(|s| s.to_string()),
recorded_at: OffsetDateTime::now_utc(),
});
}
/// Update prompt usage statistics.
pub fn record_prompt_usage(meta: &mut AgentPromptMeta, quality: f32) {
let total = meta.avg_quality * meta.usage_count as f32 + quality;
meta.usage_count += 1;
meta.avg_quality = total / meta.usage_count as f32;
meta.last_used = Some(OffsetDateTime::now_utc());
}
/// Update skill invocation statistics.
pub fn record_skill_invocation(meta: &mut AgentSkillMeta, success: bool, latency_ms: u64) {
let total_success = meta.success_rate * meta.invocation_count as f32
+ if success { 1.0 } else { 0.0 };
let total_latency = meta.avg_latency_ms * meta.invocation_count + latency_ms;
meta.invocation_count += 1;
meta.success_rate = total_success / meta.invocation_count as f32;
meta.avg_latency_ms = total_latency / meta.invocation_count;
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_new_agent_prompt() {
let (entity, meta) = new_agent_prompt(
"poimen",
"extract-entities",
"Extract entities from: {{text}}",
"extraction",
);
assert_eq!(entity.entity_type, EntityType::AgentPrompt);
assert_eq!(entity.name, "extract-entities");
assert_eq!(meta.template, "Extract entities from: {{text}}");
assert_eq!(meta.task_category, "extraction");
assert_eq!(meta.usage_count, 0);
assert!(meta.active);
}
#[test]
fn test_new_agent_skill() {
let (entity, meta) = new_agent_skill(
"poimen",
"diagnose-pod-failure",
"Diagnose Kubernetes pod CrashLoopBackOff",
);
assert_eq!(entity.entity_type, EntityType::AgentSkill);
assert_eq!(meta.description, "Diagnose Kubernetes pod CrashLoopBackOff");
assert!(meta.enabled);
assert_eq!(meta.invocation_count, 0);
}
#[test]
fn test_new_agent_decision() {
let (entity, meta) = new_agent_decision(
"poimen",
"restart-pod",
"Pod stuck in CrashLoopBackOff for 10 minutes",
0.85,
);
assert_eq!(entity.entity_type, EntityType::AgentDecision);
assert_eq!(meta.action, "restart-pod");
assert_eq!(meta.confidence, 0.85);
assert!(meta.outcome.is_none());
}
#[test]
fn test_record_decision_outcome() {
let (_, mut meta) = new_agent_decision("p", "act", "reason", 0.9);
assert!(meta.outcome.is_none());
record_decision_outcome(&mut meta, true, 0.95, Some("Pod recovered"));
assert!(meta.outcome.is_some());
let outcome = meta.outcome.unwrap();
assert!(outcome.success);
assert_eq!(outcome.quality, 0.95);
assert_eq!(outcome.feedback, Some("Pod recovered".to_string()));
}
#[test]
fn test_record_prompt_usage() {
let (_, mut meta) = new_agent_prompt("p", "test", "tmpl", "cat");
assert_eq!(meta.usage_count, 0);
assert_eq!(meta.avg_quality, 0.0);
record_prompt_usage(&mut meta, 0.8);
assert_eq!(meta.usage_count, 1);
assert_eq!(meta.avg_quality, 0.8);
record_prompt_usage(&mut meta, 1.0);
assert_eq!(meta.usage_count, 2);
assert!((meta.avg_quality - 0.9).abs() < 0.001);
}
#[test]
fn test_record_skill_invocation() {
let (_, mut meta) = new_agent_skill("p", "skill", "desc");
assert_eq!(meta.invocation_count, 0);
record_skill_invocation(&mut meta, true, 100);
assert_eq!(meta.invocation_count, 1);
assert_eq!(meta.success_rate, 1.0);
assert_eq!(meta.avg_latency_ms, 100);
record_skill_invocation(&mut meta, false, 200);
assert_eq!(meta.invocation_count, 2);
assert_eq!(meta.success_rate, 0.5);
assert_eq!(meta.avg_latency_ms, 150);
}
#[test]
fn test_entity_type_round_trip_agent_types() {
for ty in &[
EntityType::AgentPrompt,
EntityType::AgentSkill,
EntityType::AgentDecision,
] {
let s = ty.as_str();
assert_eq!(EntityType::from_str(s), *ty);
}
}
#[test]
fn test_agent_prompt_serialization() {
let (_, meta) = new_agent_prompt("p", "test", "tmpl {{x}}", "cat");
let json = serde_json::to_string(&meta).unwrap();
let deserialized: AgentPromptMeta = serde_json::from_str(&json).unwrap();
assert_eq!(deserialized.template, "tmpl {{x}}");
assert_eq!(deserialized.task_category, "cat");
}
}
-1
View File
@@ -1,6 +1,5 @@
/// Community domain model for temporal graph-RAG.
/// Single Responsibility: Community (cluster) storage and metadata.
#[allow(clippy::empty_line_after_doc_comments)]
/// Open/Closed: Algorithm field extensible for new clustering methods.
use serde::{Deserialize, Serialize};
-2
View File
@@ -1,6 +1,5 @@
/// Edge domain model for temporal graph-RAG.
/// Single Responsibility: Fact/relationship storage with bi-temporal validity.
#[allow(clippy::empty_line_after_doc_comments)]
/// Open/Closed: ContradictionStatus enum extensible.
use serde::{Deserialize, Serialize};
@@ -30,7 +29,6 @@ impl ContradictionStatus {
}
}
#[allow(clippy::should_implement_trait)]
pub fn from_str(s: &str) -> Self {
match s.to_lowercase().as_str() {
"active" => Self::Active,
+1 -29
View File
@@ -1,7 +1,6 @@
/// Entity domain model for temporal graph-RAG.
/// Single Responsibility: Entity identity and metadata.
/// Open/Closed: EntityType enum extensible.
#[allow(clippy::empty_line_after_doc_comments)]
/// Dependencies: Uses time::OffsetDateTime (consistent with mem-core).
use serde::{Deserialize, Serialize};
@@ -9,7 +8,7 @@ use time::OffsetDateTime;
use std::fmt;
/// Entity type classification (extensible enum).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Hash)]
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Hash)]
#[serde(rename_all = "snake_case")]
pub enum EntityType {
Person,
@@ -18,13 +17,6 @@ pub enum EntityType {
Location,
Event,
Organization,
/// Agent prompt template tracked as a first-class entity.
/// Enables the agent to learn which prompts produce good results.
AgentPrompt,
/// Agent skill — a reusable capability the agent has learned.
AgentSkill,
/// Agent decision — a recorded choice with reasoning and outcome.
AgentDecision,
Unknown,
}
@@ -37,14 +29,10 @@ impl EntityType {
Self::Location => "location",
Self::Event => "event",
Self::Organization => "organization",
Self::AgentPrompt => "agent_prompt",
Self::AgentSkill => "agent_skill",
Self::AgentDecision => "agent_decision",
Self::Unknown => "unknown",
}
}
#[allow(clippy::should_implement_trait)]
pub fn from_str(s: &str) -> Self {
match s.to_lowercase().as_str() {
"person" => Self::Person,
@@ -53,24 +41,11 @@ impl EntityType {
"location" => Self::Location,
"event" => Self::Event,
"organization" => Self::Organization,
"agent_prompt" => Self::AgentPrompt,
"agent_skill" => Self::AgentSkill,
"agent_decision" => Self::AgentDecision,
_ => Self::Unknown,
}
}
}
impl<'de> serde::Deserialize<'de> for EntityType {
fn deserialize<D>(deserializer: D) -> Result<Self, D::Error>
where
D: serde::Deserializer<'de>,
{
let s = String::deserialize(deserializer)?;
Ok(Self::from_str(&s))
}
}
impl fmt::Display for EntityType {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "{}", self.as_str())
@@ -200,9 +175,6 @@ mod tests {
EntityType::Person,
EntityType::Tool,
EntityType::Concept,
EntityType::AgentPrompt,
EntityType::AgentSkill,
EntityType::AgentDecision,
] {
let s = ty.as_str();
assert_eq!(EntityType::from_str(s), *ty);
+2 -1
View File
@@ -135,10 +135,11 @@ pub fn run_loop(
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_loop_basic() {
// Placeholder test to verify it compiles
assert!(true);
}
}
+7 -7
View File
@@ -403,7 +403,7 @@ pub fn lookup(sig: &Signature, lessons: &[Lesson], floor: f32) -> Option<Hit> {
let mut best: Option<(f32, &Lesson)> = None;
for l in lessons.iter().filter(|l| l.tool == sig.tool) {
let s = similarity(&sig.normalised, &l.normalised);
if s >= floor && best.is_none_or(|(bs, _)| s > bs) {
if s >= floor && best.map_or(true, |(bs, _)| s > bs) {
best = Some((s, l));
}
}
@@ -503,7 +503,7 @@ pub fn tool_of_cmd(cmd: &str) -> String {
"kubectl" | "k" => "kubectl".into(),
"docker" | "podman" => "docker".into(),
"terraform" | "tofu" => "terraform".into(),
"" => "unknown".into(),
other if other.is_empty() => "unknown".into(),
other => other.to_string(),
}
}
@@ -549,7 +549,7 @@ pub fn render_skill(tool: &str, lessons: &[Lesson]) -> String {
s.push_str("`confirmed`, which outranks inferred lessons at equal similarity.\n\n");
let mut sorted: Vec<&Lesson> = lessons.iter().collect();
sorted.sort_by_key(|a| std::cmp::Reverse(a.seen));
sorted.sort_by(|a, b| b.seen.cmp(&a.seen));
for l in sorted {
s.push_str(&format!("## {}\n\n", l.raw.trim()));
@@ -557,7 +557,7 @@ pub fn render_skill(tool: &str, lessons: &[Lesson]) -> String {
"- seen: {} | last: {} | confidence: {:?}\n",
l.seen, l.last_seen, l.confidence
));
s.push_str(&format!("- signature: `{}`\n", &l.sig_sha[..12]));
s.push_str(&format!("- signature: `{}`\n", l.sig_sha[..12].to_string()));
s.push_str("- resolved by:\n");
for r in &l.resolution {
s.push_str(&format!(" ```\n {r}\n ```\n"));
@@ -712,7 +712,7 @@ mod tests {
ev("t2", "npm pkg set overrides.react=19", 0, ""),
ev("t3", "npm ci", 0, "ok"),
];
let ls = derive_lessons(&events, tool_of_cmd);
let ls = derive_lessons(&events, |c| tool_of_cmd(c));
assert_eq!(ls.len(), 1);
assert_eq!(ls[0].resolution, vec!["npm pkg set overrides.react=19"]);
assert_eq!(ls[0].confidence, Confidence::Inferred);
@@ -775,7 +775,7 @@ mod tests {
output: "error: flaky".into(),
};
let events = vec![ev("npm ci", 1), ev("npm ci", 0)];
assert!(derive_lessons(&events, tool_of_cmd).is_empty());
assert!(derive_lessons(&events, |c| tool_of_cmd(c)).is_empty());
}
#[test]
@@ -798,7 +798,7 @@ mod tests {
sig_sha: "abc".into(),
rule: "r".into(),
};
assert_eq!(lookup(&exact, std::slice::from_ref(&l), 0.5).unwrap().tier, Tier::Exact);
assert_eq!(lookup(&exact, &[l.clone()], 0.5).unwrap().tier, Tier::Exact);
let unrelated = Signature {
tool: "npm".into(),
-2
View File
@@ -12,7 +12,6 @@ pub mod scoring;
pub mod entity;
pub mod edge;
pub mod community;
pub mod agent_entity;
pub use gate_parser::{GateResponse, ParseError, parse_gate_response};
@@ -31,4 +30,3 @@ pub use scoring::{DocumentScorer, ScoringPipeline, GlobalTfIdfScorer, ProjectTfI
pub use entity::{Entity, EntityType};
pub use edge::{Edge, ContradictionStatus};
pub use community::Community;
pub use agent_entity::{AgentPromptMeta, AgentSkillMeta, AgentDecisionMeta, DecisionOutcome};
+2 -2
View File
@@ -152,11 +152,11 @@ impl FormatHandler for CsvFormatter {
async fn format(&self, result: &OptimizationResult) -> Result<Vec<u8>, String> {
let output = format!(
"{},{},{},{:.2}\n",
"{},{},{},{}\n",
escape_csv(&result.plugin),
result.original.len(),
result.optimized.len(),
result.ratio
format!("{:.2}", result.ratio)
);
Ok(output.into_bytes())
}
+2 -2
View File
@@ -40,7 +40,7 @@ impl CcrStore {
// Remove oldest entry if at capacity
if cache.len() >= self.max_entries {
if let Some(oldest_key) = cache.keys().next().cloned() {
cache.swap_remove(&oldest_key);
cache.remove(&oldest_key);
}
}
@@ -57,7 +57,7 @@ impl CcrStore {
// Check if expired
let duration = OffsetDateTime::now_utc() - *timestamp;
if duration.whole_seconds() > self.ttl_secs as i64 {
cache.swap_remove(hash);
cache.remove(hash);
return Ok(None);
}
+5 -5
View File
@@ -7,7 +7,7 @@
//! - Drop: redundant homogeneous elements, long string values
use anyhow::Result;
use serde_json::Value;
use serde_json::{json, Value};
use std::collections::HashMap;
pub struct JsonCrusher;
@@ -45,8 +45,8 @@ impl JsonCrusher {
let mut result = Vec::new();
// Add start items
for item in items.iter().take(start_count.min(len)) {
result.push(item.clone());
for i in 0..start_count.min(len) {
result.push(items[i].clone());
}
// Select mid-array items by variance/importance
@@ -58,8 +58,8 @@ impl JsonCrusher {
// Add end items
if end_count > 0 {
for item in items.iter().skip(len.saturating_sub(end_count)) {
result.push(item.clone());
for i in (len - end_count)..len {
result.push(items[i].clone());
}
}
@@ -5,7 +5,7 @@
use super::plugin::OptimizerService;
use crate::prompt::CacheMetrics;
use crate::domain::Chunk;
use crate::domain::{Chunk, Record};
use anyhow::Result;
/// Query optimizer: compresses chunks before LLM processing
@@ -83,7 +83,7 @@ impl QueryOptimizer {
match service.optimize(&chunk_text, &content_type, Some("raw")).await {
Ok(bytes) => {
let text = String::from_utf8(bytes)
.unwrap_or(chunk_text);
.unwrap_or_else(|_| chunk_text);
Ok(text)
}
Err(_) => {
+1 -1
View File
@@ -42,7 +42,7 @@ impl ContentRouter {
/// Check if content is valid JSON
fn is_json(content: &str) -> bool {
let trimmed = content.trim();
if !(trimmed.starts_with('{') || trimmed.starts_with('[')) {
if !((trimmed.starts_with('{') || trimmed.starts_with('['))) {
return false;
}
serde_json::from_str::<serde_json::Value>(trimmed).is_ok()
+1 -1
View File
@@ -128,7 +128,7 @@ impl TextCompressor {
}
// Capitalization (usually proper nouns or emphatic)
if token.chars().next().is_some_and(|c| c.is_uppercase()) && token.len() > 1 {
if token.chars().next().map_or(false, |c| c.is_uppercase()) && token.len() > 1 {
score += 1.0;
}
+2 -4
View File
@@ -12,9 +12,7 @@ const CACHE_TURN: &str = include_str!("../../../templates/gru-mem-turn.txt");
const BUDGET_TOTAL: usize = 32768;
const BUDGET_RESPONSE: usize = 2048;
#[allow(dead_code)]
const BUDGET_SYSTEM: usize = 400;
#[allow(dead_code)]
const BUDGET_QUESTION: usize = 150;
const BUDGET_MEMORY_MAX: usize = 1024;
const BUDGET_CHUNK_MAX: usize = 5000;
@@ -370,7 +368,7 @@ fn estimate_tokens(text: &str) -> usize {
#[cfg(test)]
mod tests {
use super::*;
use crate::domain::{Chunk, Record, Role, Provenance};
use crate::domain::{Chunk, Record, Role, Provenance, Level};
use time::OffsetDateTime;
fn make_test_chunk(text: &str) -> Chunk {
@@ -647,7 +645,7 @@ mod tests {
let metrics = result.unwrap();
let ratio = metrics.compression_ratio();
assert!((0.0..=100.0).contains(&ratio));
assert!(ratio >= 0.0 && ratio <= 100.0);
}
#[test]
+2
View File
@@ -1,5 +1,7 @@
use crate::domain::{ProjectId, QueryId};
use anyhow::{anyhow, Result};
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::path::Path;
/// A single standing query.
+1 -7
View File
@@ -1,4 +1,4 @@
use crate::Level;
use crate::{Level, Query};
use anyhow::Result;
use serde::{Deserialize, Serialize};
@@ -17,12 +17,6 @@ pub struct QueryExecutor {
// For now: proof-of-concept with mock data
}
impl Default for QueryExecutor {
fn default() -> Self {
Self::new()
}
}
impl QueryExecutor {
/// Create executor.
pub fn new() -> Self {
+3 -2
View File
@@ -71,10 +71,11 @@ impl QueryLevels {
}
// Check level filter
if !self.level_filter.is_empty()
&& !self.level_filter.contains(&level.to_string()) {
if !self.level_filter.is_empty() {
if !self.level_filter.contains(&level.to_string()) {
return false;
}
}
// Check evidence/reference flags
if level == "R" {
-15
View File
@@ -6,7 +6,6 @@
/// - Single Responsibility: each scorer does one thing
/// - Open/Closed: add new scorers without modifying existing
/// - Liskov Substitution: all scorers implement DocumentScorer
#[allow(clippy::empty_line_after_doc_comments)]
/// - Dependency Inversion: depend on trait, not concrete types
use anyhow::Result;
@@ -54,7 +53,6 @@ impl DocumentScorer for GlobalTfIdfScorer {
}
/// Project-scoped TF-IDF Scorer: scoring within project boundaries
#[allow(dead_code)]
pub struct ProjectTfIdfScorer {
project: String,
vocabulary: Arc<std::collections::BTreeMap<String, f32>>,
@@ -95,18 +93,11 @@ impl DocumentScorer for ProjectTfIdfScorer {
}
/// Semantic Scorer: vector similarity (placeholder)
#[allow(dead_code)]
pub struct SemanticScorer {
_embeddings_client: Arc<()>, // Placeholder
_pgvector: Arc<()>, // Placeholder
}
impl Default for SemanticScorer {
fn default() -> Self {
Self::new()
}
}
impl SemanticScorer {
pub fn new() -> Self {
Self {
@@ -165,12 +156,6 @@ pub struct ScoringPipeline {
scorers: Vec<(String, f32, Arc<dyn DocumentScorer>)>, // name, weight, scorer
}
impl Default for ScoringPipeline {
fn default() -> Self {
Self::new()
}
}
impl ScoringPipeline {
pub fn new() -> Self {
Self {
+1 -2
View File
@@ -81,7 +81,6 @@ impl SymptomVector {
/// Internal structure for tokens during extraction
#[derive(Debug, Clone)]
#[allow(dead_code)]
struct SymptomTokens {
keywords: Vec<String>,
error_codes: Vec<String>,
@@ -393,7 +392,7 @@ mod tests {
let words: Vec<&str> = symptom.normalised.split_whitespace().collect();
for word in &words {
// Check if this word is a stop word
assert!(!STOP_WORDS.contains(word), "Stop word '{}' should be removed", word);
assert!(!STOP_WORDS.contains(&word), "Stop word '{}' should be removed", word);
}
// Should contain key terms
assert!(symptom.normalised.contains("resolve"));
+4 -2
View File
@@ -267,9 +267,11 @@ fn test_compression_handles_large_content() {
fn test_multi_chunk_search_consistency() {
let optimizer = ContextOptimizer::new().expect("optimizer init");
let chunks = ["ERROR: connection failed\nDEBUG: thread id=100",
let chunks = vec![
"ERROR: connection failed\nDEBUG: thread id=100",
"ERROR: timeout after 5000ms\nTRACE: stack unwinding",
"ERROR: retry attempt 2\nDEBUG: backoff delay=200ms"];
"ERROR: retry attempt 2\nDEBUG: backoff delay=200ms",
];
let optimized_chunks: Vec<_> = chunks
.iter()
+6 -4
View File
@@ -103,7 +103,7 @@ fn gate_metadata_preservation() {
// Verify we get a valid OptimizedChunk with proper fields
assert!(optimized.original_tokens > 0, "should track original tokens");
assert!(optimized.compressed_tokens <= optimized.original_tokens, "compressed should not exceed original");
assert!(optimized.compressed_tokens >= 0, "should track compressed tokens");
}
#[test]
@@ -122,7 +122,7 @@ fn gate_error_handling_graceful() {
match optimizer.optimize(case.as_str()) {
Ok(result) => {
// Valid compression
assert!(result.original_tokens > 0);
assert!(result.original_tokens >= 0);
}
Err(_) => {
// Acceptable to fail on edge cases, but should fail gracefully
@@ -196,6 +196,7 @@ fn gate_memory_bounded() {
// Should not panic from memory exhaustion
// If we get here, we passed the gate
assert!(true, "memory usage bounded");
}
#[test]
@@ -208,7 +209,7 @@ fn gate_no_regressions_existing_functionality() {
assert!(!result.compressed.is_empty(), "basic optimization should work");
assert!(result.original_tokens > 0, "should track tokens");
assert!(result.compressed_tokens <= result.original_tokens, "compressed should not exceed original");
assert!(result.compressed_tokens >= 0, "should have compressed tokens");
}
// ============================================================================
@@ -230,7 +231,7 @@ fn gate_compression_targets_met() {
];
for (content, name, min_compression) in fixtures.iter() {
let optimized = optimizer.optimize(content).unwrap_or_else(|_| panic!("optimize {}", name));
let optimized = optimizer.optimize(content).expect(&format!("optimize {}", name));
let ratio = optimized.compressed.len() as f32 / content.len() as f32;
// At least some compression should happen
@@ -331,4 +332,5 @@ fn gate_summary_report() {
println!("\n🚀 STATUS: M3.8 READY FOR PRODUCTION");
assert!(true); // Just for testing framework
}
-1
View File
@@ -20,7 +20,6 @@ walkdir = "2.5"
sha2 = { workspace = true }
regex = { workspace = true }
async-trait = { workspace = true }
reqwest = { workspace = true }
[dev-dependencies]
time = { workspace = true }
-184
View File
@@ -1,184 +0,0 @@
//! Authentik JWT Token Exchange
//!
//! Uses OAuth2 client credentials flow to obtain JWT tokens from Authentik
//! These tokens are used to authenticate with LLM gateway and S3
use anyhow::{Result, anyhow};
use serde::{Deserialize, Serialize};
use std::sync::Arc;
use std::sync::Mutex;
use std::time::{SystemTime, Duration};
/// JWT token response from Authentik
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TokenResponse {
pub access_token: String,
pub token_type: String,
pub expires_in: u64,
#[serde(skip)]
pub obtained_at: Option<SystemTime>,
}
impl TokenResponse {
/// Check if token is still valid
pub fn is_expired(&self) -> bool {
match self.obtained_at {
Some(time) => {
let elapsed = time.elapsed().unwrap_or(Duration::from_secs(u64::MAX));
elapsed.as_secs() >= self.expires_in - 60 // Refresh 60s before expiry
}
None => true, // No timestamp = expired
}
}
}
/// Authentik JWT issuer client
pub struct AuthentikJwtIssuer {
issuer_url: String,
client_id: String,
client_secret: String,
cached_token: Arc<Mutex<Option<TokenResponse>>>,
}
impl AuthentikJwtIssuer {
pub fn new(issuer_url: &str, client_id: &str, client_secret: &str) -> Self {
Self {
issuer_url: issuer_url.to_string(),
client_id: client_id.to_string(),
client_secret: client_secret.to_string(),
cached_token: Arc::new(Mutex::new(None)),
}
}
/// From environment: AUTHENTIK_ISSUER, AUTHENTIK_CLIENT_ID, AUTHENTIK_CLIENT_SECRET
pub fn from_env() -> Result<Self> {
// Support both naming conventions: AUTHENTIK_* and memory-agent-oidc secret keys
let issuer = std::env::var("AUTHENTIK_ISSUER")
.or_else(|_| std::env::var("ISSUER"))
.map_err(|_| anyhow!("AUTHENTIK_ISSUER or ISSUER not set"))?;
let client_id = std::env::var("AUTHENTIK_CLIENT_ID")
.or_else(|_| std::env::var("CLIENT_ID"))
.map_err(|_| anyhow!("AUTHENTIK_CLIENT_ID or CLIENT_ID not set"))?;
let client_secret = std::env::var("AUTHENTIK_CLIENT_SECRET")
.or_else(|_| std::env::var("CLIENT_SECRET"))
.map_err(|_| anyhow!("AUTHENTIK_CLIENT_SECRET or CLIENT_SECRET not set"))?;
tracing::info!(
target: "observability",
event = "authentik_jwt_init",
issuer = %issuer,
client_id = %client_id,
"Authentik JWT issuer initialized"
);
Ok(Self::new(&issuer, &client_id, &client_secret))
}
/// Get valid access token, using cache if available
pub async fn get_access_token(&self) -> Result<String> {
// Check cache
if let Ok(lock) = self.cached_token.lock() {
if let Some(token) = lock.as_ref() {
if !token.is_expired() {
tracing::debug!("Using cached Authentik token");
return Ok(token.access_token.clone());
}
}
}
// Fetch new token
let mut token = self.fetch_token().await?;
token.obtained_at = Some(SystemTime::now());
let access_token = token.access_token.clone();
// Cache it
if let Ok(mut lock) = self.cached_token.lock() {
*lock = Some(token);
}
Ok(access_token)
}
/// Exchange client credentials for JWT token
async fn fetch_token(&self) -> Result<TokenResponse> {
let client = reqwest::Client::new();
// Authentik OAuth2 token endpoint
// Use TOKEN_URL env var if set, otherwise derive from issuer
let token_url = std::env::var("TOKEN_URL")
.or_else(|_| std::env::var("AUTHENTIK_TOKEN_URL"))
.unwrap_or_else(|_| {
// Derive: strip app-specific path, use global token endpoint
// e.g., https://authentik.riotpiao.com/application/o/memory-agent/
// -> https://authentik.riotpiao.com/application/o/token/
if let Some(base) = self.issuer_url.rfind("/o/") {
format!("{}/o/token/", &self.issuer_url[..base])
} else {
format!("{}/token/", self.issuer_url.trim_end_matches('/'))
}
});
let params = [
("grant_type", "client_credentials"),
("client_id", &self.client_id),
("client_secret", &self.client_secret),
("scope", "openid roles"),
];
let response = client
.post(&token_url)
.form(&params)
.timeout(Duration::from_secs(10))
.send()
.await?;
if !response.status().is_success() {
return Err(anyhow!(
"Authentik token request failed: {} - {}",
response.status(),
response.text().await.unwrap_or_default()
));
}
let token_resp: TokenResponse = response.json().await?;
tracing::info!(
"Obtained Authentik JWT token (expires in {} seconds)",
token_resp.expires_in
);
Ok(token_resp)
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_token_expiry_check() {
let mut token = TokenResponse {
access_token: "test".to_string(),
token_type: "Bearer".to_string(),
expires_in: 3600,
obtained_at: Some(SystemTime::now()),
};
assert!(!token.is_expired());
// Simulate aged token
token.obtained_at = Some(SystemTime::now() - Duration::from_secs(3600));
assert!(token.is_expired());
}
#[test]
fn test_issuer_creation() {
let issuer = AuthentikJwtIssuer::new(
"https://example.com",
"client_id",
"client_secret",
);
assert_eq!(issuer.issuer_url, "https://example.com");
assert_eq!(issuer.client_id, "client_id");
}
}
@@ -83,7 +83,6 @@ impl ContradictionPreFilter {
/// LLM-based contradiction detector (stage 2)
/// Only called if pre-filter returns true (cost optimization)
#[allow(dead_code)]
pub struct LlmContradictionDetector {
model_name: String,
auto_confirm_threshold: f32,
+12 -265
View File
@@ -14,23 +14,16 @@ use async_trait::async_trait;
use mem_core::entity::{Entity, EntityType};
use serde::{Deserialize, Serialize};
use crate::speaker_extractor::SpeakerExtractor;
use crate::authentik_jwt::AuthentikJwtIssuer;
use std::sync::Arc;
use tokio::sync::Mutex;
/// Extracted entity from LLM (intermediate representation)
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ExtractedEntity {
pub name: String,
#[serde(alias = "type")]
pub entity_type: EntityType,
pub summary: String,
#[serde(default = "default_confidence")]
pub confidence: f32,
}
fn default_confidence() -> f32 { 0.8 }
impl ExtractedEntity {
/// Convert to domain model (Phase 1 type)
pub fn to_domain(&self, project_id: &str) -> Entity {
@@ -44,62 +37,24 @@ impl ExtractedEntity {
#[async_trait]
pub trait EntityExtractor: Send + Sync {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>>;
async fn extract_with_auth(&self, text: &str, x_forward_user: Option<&str>) -> Result<Vec<ExtractedEntity>> {
// Default: ignore auth header, use regular extract
self.extract(text).await
}
}
/// LLM-based extractor with reflection verification (stage 1 + 2)
/// Uses Authentik JWT tokens for authentication to LLM gateway
#[allow(dead_code)]
pub struct LlmEntityExtractor {
model_name: String,
enable_reflection: bool,
jwt_issuer: Option<Arc<Mutex<AuthentikJwtIssuer>>>,
}
impl LlmEntityExtractor {
pub fn new(model_name: &str) -> Self {
let jwt_issuer = AuthentikJwtIssuer::from_env().ok();
Self {
model_name: model_name.to_string(),
enable_reflection: true,
jwt_issuer: jwt_issuer.map(|iss| Arc::new(Mutex::new(iss))),
}
}
/// Parse extraction response JSON
/// Format: { "entities": [{ "name": "...", "type": "...", "summary": "..." }, ...] }
/// Clean LLM response: strip thinking tags, markdown fences, extract JSON
fn clean_llm_response(text: &str) -> String {
let mut result = text.to_string();
// Remove <think>...</think> blocks
while let Some(start) = result.find("<think>") {
if let Some(end) = result.find("</think>") {
result = format!("{}{}", &result[..start], &result[end + 8..]);
} else {
break;
}
}
// Remove markdown code fences
result = result.replace("```json", "").replace("```", "");
// Find JSON object
let trimmed = result.trim();
if let Some(start) = trimmed.find('{') {
if let Some(end) = trimmed.rfind('}') {
return trimmed[start..=end].to_string();
}
}
// Maybe it's a JSON array — wrap in object
if let Some(start) = trimmed.find('[') {
if let Some(end) = trimmed.rfind(']') {
return format!("{{\"entities\": {}}}", &trimmed[start..=end]);
}
}
trimmed.to_string()
}
fn parse_extraction(response: &str) -> Result<Vec<ExtractedEntity>> {
#[derive(Deserialize)]
struct Response {
@@ -125,115 +80,11 @@ impl LlmEntityExtractor {
Ok(parsed.verified.into_iter().map(|v| (v.name, v.present)).collect())
}
/// Call LLM via api.riotpiao.com using X-Forward-User auth/exchange
/// Supports: Authentik JWT, X-Forward-User header, or API key fallback
async fn call_llm_endpoint(&self, prompt: &str, x_forward_user: Option<&str>) -> Result<String> {
let endpoint = std::env::var("LLM_ENDPOINT")
.unwrap_or_else(|_| "http://api-internal.riotpiao.com:8000/v1/chat/completions".to_string());
let model = std::env::var("LLM_MODEL")
.unwrap_or_else(|_| "qwen:7b".to_string());
// Get auth header: prefer X-Forward-User, fallback to Authentik JWT, then API key
let auth_header = if let Some(user) = x_forward_user {
// Use X-Forward-User directly (API Gateway pattern)
tracing::info!("Using X-Forward-User for LLM auth: {}", user);
format!("X-Forward-User: {}", user)
} else if let Some(jwt_issuer) = &self.jwt_issuer {
let issuer = jwt_issuer.lock().await;
match issuer.get_access_token().await {
Ok(token) => {
tracing::info!("Using Authentik JWT for LLM auth");
format!("Bearer {}", token)
},
Err(e) => {
tracing::warn!("Failed to get Authentik JWT: {}", e);
// Fallback to env var
let api_key = std::env::var("LLM_API_KEY")
.or_else(|_| std::env::var("MEM_API_KEY"))
.unwrap_or_else(|_| "test-key".to_string());
tracing::info!("Falling back to LLM_API_KEY");
format!("Bearer {}", api_key)
}
}
} else {
// Fallback to env var if Authentik not configured
let api_key = std::env::var("LLM_API_KEY")
.or_else(|_| std::env::var("MEM_API_KEY"))
.unwrap_or_else(|_| "test-key".to_string());
tracing::info!("Using LLM_API_KEY for LLM auth");
format!("Bearer {}", api_key)
};
let client = reqwest::Client::new();
// OpenAI-compatible API call
let payload = serde_json::json!({
"model": model,
"messages": [
{"role": "system", "content": "You are an entity extraction specialist. Extract named entities from text in JSON format."},
{"role": "user", "content": prompt}
],
"temperature": 0.3,
"max_tokens": 12000
});
let mut request = client
.post(&endpoint)
.header("Content-Type", "application/json");
// Set auth header (varies by auth method)
if auth_header.starts_with("X-Forward-User") {
request = request.header("X-Forward-User", auth_header.split(": ").nth(1).unwrap_or("unknown"));
} else {
request = request.header("Authorization", auth_header);
}
let response = request
.json(&payload)
.timeout(std::time::Duration::from_secs(90))
.send()
.await?;
let status = response.status();
if !status.is_success() {
let error_text = response.text().await.unwrap_or_default();
tracing::error!(
"LLM API error: {} - {}",
status,
error_text
);
// Return error instead of silently returning empty array
return Err(anyhow::anyhow!("LLM API failed with status {}: {}", status, error_text));
}
let data: serde_json::Value = response.json().await?;
// Extract content — some models put JSON in "content", others in "reasoning"
let msg = &data["choices"][0]["message"];
let raw_content = msg["content"].as_str().unwrap_or("").to_string();
let raw_reasoning = msg["reasoning"].as_str().unwrap_or("").to_string();
// Use content if non-empty, otherwise try reasoning field
let raw = if !raw_content.trim().is_empty() { &raw_content } else { &raw_reasoning };
let content = Self::clean_llm_response(raw);
let tokens = &data["usage"];
tracing::info!(
target: "observability",
event = "llm_entity_call",
model = %model,
endpoint = %endpoint,
raw_len = raw.len(),
cleaned_len = content.len(),
prompt_tokens = %tokens["prompt_tokens"],
completion_tokens = %tokens["completion_tokens"],
has_reasoning = !raw_reasoning.is_empty(),
"LLM entity extraction call complete"
);
Ok(content)
}
/// Fallback mock LLM call (for testing without API)
fn simulate_llm(&self, _prompt: &str) -> Result<String> {
/// Mock LLM call - replace with real API in production
/// TODO (Phase 2.6): Integrate with api.riotpiao.com/v1/chat/completions
/// TODO (Phase 2.6): Add JWT authentication from Authentik OIDC
async fn simulate_llm(&self, _prompt: &str) -> Result<String> {
// Production: call api.riotpiao.com with Bearer JWT token
// Mock response for testing
Ok(r#"{
"entities": [
@@ -283,15 +134,7 @@ Respond in JSON:
text
);
// Try real LLM first, fallback to mock if not configured
let extraction_response = if std::env::var("LLM_ENDPOINT").is_ok() {
self.call_llm_endpoint(&prompt, None).await.unwrap_or_else(|e| {
tracing::error!("LLM entity extraction failed: {}, using mock", e);
self.simulate_llm(&prompt).unwrap_or_default()
})
} else {
self.simulate_llm(&prompt)?
};
let extraction_response = self.simulate_llm(&prompt).await?;
let extracted = Self::parse_extraction(&extraction_response)?;
entities.extend(extracted); // Add LLM-extracted entities after speaker
@@ -312,28 +155,11 @@ Respond in JSON:
text, entities
);
let reflection = if std::env::var("LLM_ENDPOINT").is_ok() {
self.call_llm_endpoint(&reflection_prompt, None).await.unwrap_or_else(|e| {
tracing::warn!("Reflection LLM call failed: {}, skipping verification", e);
String::new()
})
} else {
self.simulate_llm(&reflection_prompt)?
};
let reflection = self.simulate_llm(&reflection_prompt).await?;
let verified = Self::parse_reflection(&reflection)?;
// If reflection succeeded, filter entities; otherwise keep all
if !reflection.is_empty() {
match Self::parse_reflection(&reflection) {
Ok(verified) => {
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
}
Err(e) => {
tracing::warn!("Reflection parse failed: {}, keeping all entities", e);
}
}
} else {
tracing::info!("Reflection skipped, keeping {} unverified entities", entities.len());
}
// Filter: keep only entities marked present
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
// Adjust confidence for reflected entities (slight penalty for needing verification)
for entity in &mut entities {
@@ -343,85 +169,6 @@ Respond in JSON:
Ok(entities)
}
/// Extract with X-Forward-User auth header (API Gateway pattern)
async fn extract_with_auth(&self, text: &str, x_forward_user: Option<&str>) -> Result<Vec<ExtractedEntity>> {
let mut entities = vec![];
// Extract speaker if available
use crate::speaker_extractor::{HeuristicSpeakerExtractor, SpeakerConfig};
if let Ok(speaker_extractor) = HeuristicSpeakerExtractor::new(SpeakerConfig::default()) {
if let Ok(Some(speaker)) = speaker_extractor.extract_speaker(text).await {
entities.push(ExtractedEntity {
name: speaker.name,
entity_type: mem_core::entity::EntityType::Person,
summary: "Speaker in this episode".to_string(),
confidence: speaker.confidence,
});
}
}
// Extract entities with auth header
let prompt = format!(
r#"Extract named entities from this text.
For each entity provide:
- name: Canonical name (proper capitalization)
- type: One of [person, tool, concept, location, event, organization]
- summary: One sentence
CRITICAL: Only extract entities EXPLICITLY mentioned. No inference.
Text:
"{}"
Respond in JSON:
{{"entities": [{{"name": "...", "type": "...", "summary": "..."}}, ...]}}
"#,
text
);
// Use provided X-Forward-User for auth
let extraction_response = if std::env::var("LLM_ENDPOINT").is_ok() {
self.call_llm_endpoint(&prompt, x_forward_user).await.unwrap_or_else(|e| {
tracing::error!("LLM entity extraction with auth failed: {}", e);
self.simulate_llm(&prompt).unwrap_or_default()
})
} else {
self.simulate_llm(&prompt)?
};
let extracted = Self::parse_extraction(&extraction_response)?;
entities.extend(extracted);
// Optional: reflection verification with auth
if self.enable_reflection && std::env::var("LLM_ENDPOINT").is_ok() {
let reflection_prompt = format!(
r#"Verify these entities are explicitly in the text:
Text:
"{}"
Entities:
{:?}
Respond in JSON:
{{"verified": [{{"name": "...", "present": true/false}}, ...]}}
"#,
text, entities
);
if let Ok(reflection) = self.call_llm_endpoint(&reflection_prompt, x_forward_user).await {
if !reflection.is_empty() {
if let Ok(verified) = Self::parse_reflection(&reflection) {
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
}
}
}
}
Ok(entities)
}
}
/// Fallback extractor: Use wiki_links if LLM fails (stage 3)
@@ -440,7 +187,7 @@ impl EntityExtractor for WikiLinkFallbackExtractor {
entities.push(ExtractedEntity {
name: name_str.to_string(),
entity_type: EntityType::Unknown,
summary: "Mentioned in episode".to_string(),
summary: format!("Mentioned in episode"),
confidence: 0.7, // Lower confidence for fallback
});
}
@@ -528,6 +275,6 @@ mod tests {
let text = "[[Entity1]] and [[Entity2]]";
let entities = composite.extract(text).await.unwrap();
assert!(!entities.is_empty());
assert!(entities.len() > 0);
}
}
+30 -283
View File
@@ -1,12 +1,12 @@
//! Fact extraction: Identify relationships between entities
//!
//! Three implementations:
//! Two implementations:
//! 1. SimpleFactExtractor: Pattern-based (verbs + wiki links)
//! 2. LlmFactExtractor: LLM-based extraction with entity context
//! 3. Fallback chain: LLM → Simple pattern matching
//! 2. LlmFactExtractor: LLM-based (placeholder for production)
//!
//! Aligned with Zep paper §2.2.2: Facts as edges between entity pairs,
//! with temporal extraction and dedup against existing edges.
//! CRAP: 12 (Simple pattern matching + LLM placeholder)
//! SOLID: Trait-based (Open/Closed)
//! DRY: Reuses EntityExtractor pattern
use anyhow::Result;
use async_trait::async_trait;
@@ -27,18 +27,20 @@ pub struct ExtractedFact {
pub trait FactExtractor: Send + Sync {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>>;
/// Extract facts with entity context (Zep §2.2.2: facts between known entities)
/// Extract facts with GRM context (optional, defaults to extract())
async fn extract_with_context(
&self,
text: &str,
_entity_contexts: &[crate::grm_retriever::EntityContext],
) -> Result<Vec<ExtractedFact>> {
// Default: ignore context, use plain extraction
self.extract(text).await
}
}
/// Simple fact extractor based on verb patterns
/// Pattern: [[Entity1]] verb [[Entity2]]
/// Common verbs: uses, manages, runs, deployed_to, works_with
pub struct SimpleFactExtractor;
#[async_trait]
@@ -46,15 +48,17 @@ impl FactExtractor for SimpleFactExtractor {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>> {
let mut facts = vec![];
// Extract [[Entity]] patterns
let entity_pattern = Regex::new(r"\[\[([^\]]+)\]\]")?;
let _entities: Vec<String> = entity_pattern
let entities: Vec<String> = entity_pattern
.captures_iter(text)
.filter_map(|cap| cap.get(1).map(|m| m.as_str().to_string()))
.collect();
let verbs = ["uses", "manages", "runs", "deployed_to", "works_with",
"depends_on", "contains", "extends", "implements", "connects_to"];
// Common relationship verbs
let verbs = ["uses", "manages", "runs", "deployed_to", "works_with"];
// Simple heuristic: if two entities appear close together with a verb between them
for verb in &verbs {
let pattern = format!(
r"\[\[([^\]]+)\]\].*?{}.*?\[\[([^\]]+)\]\]",
@@ -67,7 +71,12 @@ impl FactExtractor for SimpleFactExtractor {
source_entity_id: src.as_str().to_string(),
target_entity_id: tgt.as_str().to_string(),
relation_type: verb.to_uppercase(),
fact: format!("{} {} {}", src.as_str(), verb, tgt.as_str()),
fact: format!(
"{} {} {}",
src.as_str(),
verb,
tgt.as_str()
),
});
}
}
@@ -78,251 +87,18 @@ impl FactExtractor for SimpleFactExtractor {
}
}
/// LLM-based fact extractor (Zep §2.2.2 alignment)
/// Extracts relationships between entity pairs using LLM
pub struct LlmFactExtractor {
model_name: String,
jwt_issuer: Option<std::sync::Arc<tokio::sync::Mutex<crate::authentik_jwt::AuthentikJwtIssuer>>>,
}
impl LlmFactExtractor {
pub fn new(model_name: &str) -> Self {
let jwt_issuer = crate::authentik_jwt::AuthentikJwtIssuer::from_env().ok();
Self {
model_name: model_name.to_string(),
jwt_issuer: jwt_issuer.map(|iss| std::sync::Arc::new(tokio::sync::Mutex::new(iss))),
}
}
/// Clean LLM response: strip thinking tags, markdown fences, extract JSON
fn clean_llm_response(text: &str) -> String {
let mut result = text.to_string();
while let Some(start) = result.find("<think>") {
if let Some(end) = result.find("</think>") {
result = format!("{}{}", &result[..start], &result[end + 8..]);
} else { break; }
}
result = result.replace("```json", "").replace("```", "");
let trimmed = result.trim();
if let Some(start) = trimmed.find('{') {
if let Some(end) = trimmed.rfind('}') {
return trimmed[start..=end].to_string();
}
}
if let Some(start) = trimmed.find('[') {
if let Some(end) = trimmed.rfind(']') {
return format!("{{\"facts\": {}}}", &trimmed[start..=end]);
}
}
trimmed.to_string()
}
async fn call_llm(&self, prompt: &str) -> Result<String> {
let endpoint = std::env::var("LLM_ENDPOINT")
.unwrap_or_else(|_| "http://localhost:11434/v1/chat/completions".to_string());
// Get auth header: Authentik JWT if configured, else API key
let auth_header = if let Some(jwt_issuer) = &self.jwt_issuer {
let issuer = jwt_issuer.lock().await;
match issuer.get_access_token().await {
Ok(token) => format!("Bearer {}", token),
Err(e) => {
tracing::warn!(target: "observability", event = "fact_jwt_fallback", error = %e, "JWT failed, using API key");
let key = std::env::var("LLM_API_KEY").unwrap_or_else(|_| "default-key".to_string());
format!("Bearer {}", key)
}
}
} else {
let key = std::env::var("LLM_API_KEY")
.or_else(|_| std::env::var("MEM_API_KEY"))
.unwrap_or_else(|_| "default-key".to_string());
format!("Bearer {}", key)
};
let start = std::time::Instant::now();
let client = reqwest::Client::new();
let payload = serde_json::json!({
"model": self.model_name,
"messages": [
{"role": "system", "content": "You are a fact extraction specialist. Extract relationships between entities from text. Output ONLY valid JSON."},
{"role": "user", "content": prompt}
],
"max_tokens": 12000,
"temperature": 0.1
});
let response = client
.post(&endpoint)
.header("Authorization", &auth_header)
.header("Content-Type", "application/json")
.json(&payload)
.timeout(std::time::Duration::from_secs(120))
.send()
.await?;
let status = response.status();
if !status.is_success() {
let body = response.text().await.unwrap_or_default();
tracing::warn!(target: "observability", event = "fact_llm_error", status = %status, body = %body, "Fact LLM call failed");
return Err(anyhow::anyhow!("LLM API error: {}", status));
}
let elapsed = start.elapsed();
let data: serde_json::Value = response.json().await?;
// Handle both content and reasoning fields (ornith uses reasoning)
let msg = &data["choices"][0]["message"];
let raw_content = msg["content"].as_str().unwrap_or("").to_string();
let raw_reasoning = msg["reasoning"].as_str().unwrap_or("").to_string();
let raw = if !raw_content.trim().is_empty() { &raw_content } else { &raw_reasoning };
let cleaned = Self::clean_llm_response(raw);
let tokens = &data["usage"];
tracing::info!(
target: "observability",
event = "llm_fact_call",
model = %self.model_name,
endpoint = %endpoint,
raw_len = raw.len(),
cleaned_len = cleaned.len(),
prompt_tokens = %tokens["prompt_tokens"],
completion_tokens = %tokens["completion_tokens"],
duration_ms = elapsed.as_millis() as u64,
has_reasoning = !raw_reasoning.is_empty(),
"LLM fact extraction call complete"
);
Ok(cleaned)
}
}
/// LLM-based fact extractor (placeholder for production)
/// TODO (Phase 2.6): Implement with real LLM API
/// TODO (Phase 2.6): Support complex relationships (3-way, temporal, conditional)
pub struct LlmFactExtractor;
#[async_trait]
impl FactExtractor for LlmFactExtractor {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedFact>> {
self.extract_with_context(text, &[]).await
}
async fn extract_with_context(
&self,
text: &str,
entity_contexts: &[crate::grm_retriever::EntityContext],
) -> Result<Vec<ExtractedFact>> {
// Build entity list for prompt
let entity_names: Vec<&str> = entity_contexts
.iter()
.map(|e| e.entity_name.as_str())
.collect();
if entity_names.is_empty() {
tracing::debug!("No entities provided, skipping fact extraction");
return Ok(vec![]);
}
let prompt = format!(
r#"Extract relationships (facts) between these entities from the text.
Entities: {:?}
Text:
"{}"
For each relationship provide:
- source: Entity name (must be from the list above)
- target: Entity name (must be from the list above)
- relation: Verb/predicate describing the relationship (e.g., "uses", "manages", "is_part_of", "deployed_on")
- fact: One-sentence natural language description
CRITICAL: Only extract relationships EXPLICITLY stated or strongly implied. Source and target must both be from the entity list.
Respond in JSON:
{{"facts": [{{"source": "...", "target": "...", "relation": "...", "fact": "..."}}, ...]}}
"#,
entity_names, text
);
let llm_ok = std::env::var("LLM_ENDPOINT").is_ok();
let response = if llm_ok {
match self.call_llm(&prompt).await {
Ok(r) => r,
Err(e) => {
tracing::warn!("Fact extraction LLM failed: {}, returning empty", e);
return Ok(vec![]);
}
}
} else {
tracing::debug!("LLM_ENDPOINT not set, skipping LLM fact extraction");
return Ok(vec![]);
};
// Parse response
#[derive(Deserialize)]
struct FactResponse {
facts: Vec<RawFact>,
}
#[derive(Deserialize)]
struct RawFact {
source: String,
target: String,
relation: String,
fact: String,
}
// Try parsing, if trailing chars error try trimming to valid JSON
let parsed = match serde_json::from_str::<FactResponse>(&response) {
Ok(r) => Ok(r),
Err(e) if e.to_string().contains("trailing") => {
// Find the closing of the top-level object and retry
let mut depth = 0i32;
let mut end = 0;
for (i, c) in response.char_indices() {
match c {
'{' | '[' => depth += 1,
'}' | ']' => { depth -= 1; if depth == 0 { end = i + 1; break; } },
_ => {}
}
}
if end > 0 {
serde_json::from_str::<FactResponse>(&response[..end])
} else {
Err(e)
}
}
Err(e) => Err(e),
};
match parsed {
Ok(parsed) => {
let facts: Vec<ExtractedFact> = parsed.facts
.into_iter()
.filter(|f| {
// Validate source and target are known entities
let src_ok = entity_names.iter().any(|e| e.eq_ignore_ascii_case(&f.source));
let tgt_ok = entity_names.iter().any(|e| e.eq_ignore_ascii_case(&f.target));
if !src_ok || !tgt_ok {
tracing::debug!(
"Dropping fact with unknown entity: {} -> {}",
f.source, f.target
);
}
src_ok && tgt_ok && f.source != f.target
})
.map(|f| ExtractedFact {
source_entity_id: f.source,
target_entity_id: f.target,
relation_type: f.relation.to_uppercase(),
fact: f.fact,
})
.collect();
tracing::info!(
"LLM fact extraction: {} facts from {} entities",
facts.len(), entity_names.len()
);
Ok(facts)
}
Err(e) => {
tracing::warn!("Fact extraction JSON parse failed: {}", e);
Ok(vec![])
}
}
async fn extract(&self, _text: &str) -> Result<Vec<ExtractedFact>> {
// TODO (Phase 2.6): Implement LLM-based extraction
// Pattern: Send text to api.riotpiao.com with prompt
// Parse response for [source, relation, target] tuples
Ok(vec![])
}
}
@@ -334,38 +110,9 @@ mod tests {
async fn test_simple_fact_extraction() {
let extractor = SimpleFactExtractor;
let text = "[[Rock]] uses [[Kubernetes]] and [[ArgoCD]]";
let facts = extractor.extract(text).await.unwrap();
assert!(!facts.is_empty());
assert!(facts.len() > 0);
assert!(facts.iter().any(|f| f.relation_type == "USES"));
}
#[tokio::test]
async fn test_simple_no_wiki_links() {
let extractor = SimpleFactExtractor;
let text = "Kubernetes uses etcd for storage";
let facts = extractor.extract(text).await.unwrap();
assert!(facts.is_empty()); // No [[wiki links]]
}
#[test]
fn test_clean_llm_response() {
let input = r#"<think>reasoning here</think>{"facts": [{"source": "A", "target": "B", "relation": "uses", "fact": "A uses B"}]}"#;
let cleaned = LlmFactExtractor::clean_llm_response(input);
assert!(cleaned.starts_with("{"));
assert!(cleaned.contains("facts"));
}
#[test]
fn test_strip_thinking_no_tags() {
let input = r#"{"facts": []}"#;
let cleaned = LlmFactExtractor::clean_llm_response(input);
assert_eq!(cleaned, input);
}
#[tokio::test]
async fn test_llm_fact_no_entities_returns_empty() {
let extractor = LlmFactExtractor::new("test");
let facts = extractor.extract_with_context("some text", &[]).await.unwrap();
assert!(facts.is_empty());
}
}
+4 -1
View File
@@ -10,7 +10,10 @@
use anyhow::Result;
use async_trait::async_trait;
use serde::{Deserialize, Serialize};
use tracing::debug;
use std::collections::HashMap;
use tracing::{debug, info};
use mem_core::entity::Entity;
use mem_core::edge::Edge;
/// Memorability decision for entity or fact
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq)]
+2 -8
View File
@@ -59,15 +59,10 @@ impl IngestPipeline {
/// Execute extraction pipeline for episode
/// CRAP: 14 (Low: orchestration only, delegates to stages)
pub async fn ingest(&self, episode: &Episode) -> Result<ExtractionResult> {
self.ingest_with_auth(episode, None).await
}
/// Ingest with optional X-Forward-User auth header
pub async fn ingest_with_auth(&self, episode: &Episode, x_forward_user: Option<&str>) -> Result<ExtractionResult> {
debug!("Starting ingest for episode: {}", episode.id);
// Stage 1: Extract entities (with optional auth header)
let extracted_entities = self.entity_extractor.extract_with_auth(&episode.text, x_forward_user).await?;
// Stage 1: Extract entities
let extracted_entities = self.entity_extractor.extract(&episode.text).await?;
debug!("Extracted {} entities", extracted_entities.len());
// Convert to domain entities
@@ -149,7 +144,6 @@ impl IngestPipeline {
/// Async queue worker: Process episodes from queue
/// CRAP: 12 (Async loop, straightforward)
#[allow(dead_code)]
pub struct QueueWorker {
pipeline: Arc<IngestPipeline>,
batch_size: usize,
-1
View File
@@ -1,6 +1,5 @@
pub mod pi_session;
pub mod claude_transcript;
pub mod authentik_jwt;
pub mod doc_corpus;
pub mod derived_filter;
pub mod obsidian_ref_source;
+2 -2
View File
@@ -14,7 +14,7 @@ use tracing::{debug, info};
use crate::grm_retriever::{
EntityContext, FactContext, GraphContextRetriever, MemorabilityDecision, GrmConfig, MockGrmRetriever,
};
use mem_core::entity::Entity;
use mem_core::entity::{Entity, EntityType};
use mem_core::edge::Edge;
/// Entity filtering result
@@ -88,7 +88,7 @@ impl MemorabilityGate {
let (filtered, reason) = match context.decision {
MemorabilityDecision::Keep => {
if context.matched_entity_id.is_some() {
(true, "Existing entity (merge required)".to_string())
(true, format!("Existing entity (merge required)"))
} else {
(false, format!("New entity (score: {:.2})", context.memorability_score))
}
+1 -5
View File
@@ -20,7 +20,6 @@ pub struct RefMetadata {
}
/// Obsidian REST API client
#[allow(dead_code)]
pub struct ObsidianClient {
base_url: String,
}
@@ -48,7 +47,6 @@ impl ObsidianClient {
}
/// ObsidianRefSource: Fetches & chunks reference documents from Obsidian vault
#[allow(dead_code)]
pub struct ObsidianRefSource {
client: ObsidianClient,
project: String,
@@ -70,13 +68,11 @@ impl ObsidianRefSource {
}
/// Check if a file path is allowed (matches configured prefixes)
#[allow(dead_code)]
fn is_allowed_path(&self, path: &str) -> bool {
self.allowed_paths.iter().any(|prefix| path.starts_with(prefix))
}
/// Chunk reference document via heading-boundary logic
#[allow(dead_code)]
fn chunk_document(&self, path: &str, content: &str) -> Vec<Record> {
// M3.6.1 heading-boundary chunking
// - Split by headings
@@ -207,7 +203,7 @@ mod tests {
let chunks = source.chunk_document("docs/test.md", content);
// Should split by headings
assert!(!chunks.is_empty());
assert!(chunks.len() > 0);
}
#[test]
+2 -1
View File
@@ -60,7 +60,8 @@ impl MetricsCollector {
self.by_project
.lock()
.unwrap()
.get(project).cloned()
.get(project)
.map(|m| m.clone())
}
/// Get all project metrics.
+1 -1
View File
@@ -306,7 +306,7 @@ impl QueryMetricsRepository {
let mut repo = self.metrics.lock().unwrap();
repo.get_mut(query_id)
.ok_or_else(|| format!("Query {} not found", query_id))
.map(f)
.map(|metrics| f(metrics))
}
/// Get progress for a query
+3 -5
View File
@@ -4,10 +4,9 @@
///
/// Used to scope queries to project namespaces and enable graph traversal.
/// For example: poimen/tools/kubectl.md [[debugging.md]] creates an edge
#[allow(clippy::empty_line_after_doc_comments)]
/// from tools/kubectl to debugging (within same project).
use anyhow::Result;
use anyhow::{anyhow, Result};
use regex::Regex;
use std::collections::{HashMap, HashSet};
use std::path::{Path, PathBuf};
@@ -80,7 +79,6 @@ impl WikiLinkParser {
}
/// Graph Index: Stores and queries wiki-link relationships
#[allow(dead_code)]
pub struct WikiLinkGraph {
/// Forward links: source -> [targets]
forward_links: HashMap<String, Vec<String>>,
@@ -102,11 +100,11 @@ impl WikiLinkGraph {
/// Add a wiki-link edge
pub fn add_link(&mut self, source: &str, target: &str) {
self.forward_links.entry(source.to_string())
.or_default()
.or_insert_with(Vec::new)
.push(target.to_string());
self.backward_links.entry(target.to_string())
.or_default()
.or_insert_with(Vec::new)
.push(source.to_string());
}
+4 -4
View File
@@ -34,7 +34,7 @@ pub enum AuthMode {
impl AuthMode {
/// Detect from base URL or explicit env var.
pub fn detect(_base_url: &str, api_key: &str) -> Self {
pub fn detect(base_url: &str, api_key: &str) -> Self {
if api_key.is_empty() {
return Self::None;
}
@@ -87,7 +87,6 @@ struct Choice {
}
#[derive(Debug, Deserialize)]
#[allow(dead_code)]
struct MessageResponse {
role: String,
content: String,
@@ -209,11 +208,12 @@ impl ChatClient {
Ok(r) => r,
Err(e) => {
last_error = Some(anyhow!("Request failed: {}", e));
if (e.is_timeout() || e.is_status())
&& attempt < self.max_retries - 1 {
if e.is_timeout() || e.is_status() {
if attempt < self.max_retries - 1 {
tokio::time::sleep(Duration::from_millis(100 * 2_u64.pow(attempt))).await;
continue;
}
}
return Err(last_error.unwrap());
}
};
+4 -85
View File
@@ -27,7 +27,6 @@ struct EmbeddingRequest {
}
#[derive(Debug, Deserialize)]
#[allow(dead_code)]
#[serde(untagged)]
enum EmbeddingResponse {
Success {
@@ -43,7 +42,6 @@ enum EmbeddingResponse {
}
#[derive(Debug, Deserialize)]
#[allow(dead_code)]
struct EmbeddingData {
embedding: Vec<f32>,
#[serde(default)]
@@ -122,10 +120,10 @@ impl EmbeddingsClient {
/// Embed a single text string, returning a 768-dim vector
pub async fn embed_one(&self, text: &str) -> Result<Vector> {
let embeddings = self.embed(&[text.to_string()]).await?;
embeddings
Ok(embeddings
.into_iter()
.next()
.ok_or_else(|| anyhow!("empty embedding response"))
.ok_or_else(|| anyhow!("empty embedding response"))?)
}
/// Embed multiple texts, batched at ≤32 per request, preserving input order
@@ -168,18 +166,8 @@ impl EmbeddingsClient {
}
let resp = builder.json(&req).send().await?;
let status = resp.status();
let raw_body = resp.text().await?;
if !status.is_success() {
tracing::error!("Embedding API returned {}: {}", status, &raw_body[..raw_body.len().min(500)]);
return Err(anyhow!("Embedding API returned {}: {}", status, &raw_body[..raw_body.len().min(200)]));
}
let body: EmbeddingResponse = serde_json::from_str(&raw_body).map_err(|e| {
tracing::error!("Failed to parse embedding response: {}. Raw body: {}", e, &raw_body[..raw_body.len().min(500)]);
anyhow!("Failed to parse embedding response: {}. Raw: {}", e, &raw_body[..raw_body.len().min(200)])
})?;
let _status = resp.status();
let body: EmbeddingResponse = resp.json().await?;
match body {
EmbeddingResponse::Error { error } => {
@@ -214,73 +202,4 @@ mod tests {
assert_eq!(BATCH_SIZE, 32);
assert_eq!(EMBEDDINGS_DIM, 768);
}
#[test]
fn test_parse_real_embedding_response() {
// Exact format returned by embeddings-predictor service
let raw = r#"{"object":"list","data":[{"object":"embedding","embedding":[0.1,0.2,0.3],"index":0}],"model":"nomic-ai/nomic-embed-text-v2-moe","usage":{"prompt_tokens":3,"total_tokens":3}}"#;
let parsed: EmbeddingResponse = serde_json::from_str(raw).expect("should parse");
match parsed {
EmbeddingResponse::Success { data, .. } => {
assert_eq!(data.len(), 1);
assert_eq!(data[0].embedding.len(), 3);
assert_eq!(data[0].index, 0);
}
EmbeddingResponse::Error { error } => panic!("parsed as error: {:?}", error),
}
}
#[test]
fn test_parse_embedding_error_response() {
let raw = r#"{"error":"model not found"}"#;
let parsed: EmbeddingResponse = serde_json::from_str(raw).expect("should parse");
match parsed {
EmbeddingResponse::Error { error } => {
assert_eq!(error.as_str().unwrap(), "model not found");
}
EmbeddingResponse::Success { .. } => panic!("should be error"),
}
}
#[test]
fn test_parse_768_dim_response() {
// 768 floats
let embedding: Vec<f32> = (0..768).map(|i| i as f32 * 0.001).collect();
let raw = format!(
r#"{{"object":"list","data":[{{"object":"embedding","embedding":{},"index":0}}],"model":"test","usage":{{}}}}"#,
serde_json::to_string(&embedding).unwrap()
);
let parsed: EmbeddingResponse = serde_json::from_str(&raw).expect("should parse 768-dim");
match parsed {
EmbeddingResponse::Success { data, .. } => {
assert_eq!(data[0].embedding.len(), 768);
}
_ => panic!("should be success"),
}
}
#[test]
fn test_parse_html_fails_gracefully() {
// Simulates gateway returning HTML error page
let raw = "<html><body>502 Bad Gateway</body></html>";
let result: Result<EmbeddingResponse, _> = serde_json::from_str(raw);
assert!(result.is_err(), "HTML should fail to parse as JSON");
let err_msg = result.unwrap_err().to_string();
assert!(err_msg.contains("expected"), "Error should mention parsing: {}", err_msg);
}
#[test]
fn test_parse_multi_input_response() {
// Array input returns multiple embeddings
let raw = r#"{"object":"list","data":[{"object":"embedding","embedding":[0.1,0.2,0.3],"index":0},{"object":"embedding","embedding":[0.4,0.5,0.6],"index":1}],"model":"test","usage":{}}"#;
let parsed: EmbeddingResponse = serde_json::from_str(raw).expect("should parse");
match parsed {
EmbeddingResponse::Success { data, .. } => {
assert_eq!(data.len(), 2);
assert_eq!(data[0].index, 0);
assert_eq!(data[1].index, 1);
}
_ => panic!("should be success"),
}
}
}
@@ -1,67 +0,0 @@
-- Migration 009: Temporal edge schema (Zep paper §2.2.2)
-- Replaces old memory_edge (child_sha/parent_sha node graph)
-- with temporal edge schema supporting relation types, facts, and validity periods.
-- Idempotent: safe to run multiple times.
-- Rename old table if it still exists (skip if already migrated)
DO $$
BEGIN
IF EXISTS (SELECT 1 FROM information_schema.tables WHERE table_name = 'memory_edge'
AND EXISTS (SELECT 1 FROM information_schema.columns
WHERE table_name = 'memory_edge' AND column_name = 'child_sha'))
THEN
ALTER TABLE memory_edge RENAME TO memory_edge_legacy;
END IF;
END $$;
-- Create temporal edge table
CREATE TABLE IF NOT EXISTS memory_edge (
id TEXT PRIMARY KEY,
project_id TEXT NOT NULL DEFAULT 'default',
source_id TEXT NOT NULL,
target_id TEXT NOT NULL,
relation_type TEXT NOT NULL DEFAULT '',
fact TEXT NOT NULL DEFAULT '',
weight REAL NOT NULL DEFAULT 1.0,
strength REAL DEFAULT 1.0,
confidence REAL DEFAULT 0.8,
t_valid TIMESTAMPTZ,
t_invalid TIMESTAMPTZ,
t_created TIMESTAMPTZ NOT NULL DEFAULT NOW(),
t_expired TIMESTAMPTZ,
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
episode_id TEXT,
deleted_at TIMESTAMPTZ
);
-- Ensure app user owns the table
DO $$ BEGIN
IF EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'app') THEN
ALTER TABLE memory_edge OWNER TO app;
END IF;
END $$;
CREATE INDEX IF NOT EXISTS idx_memory_edge_source ON memory_edge(source_id);
CREATE INDEX IF NOT EXISTS idx_memory_edge_target ON memory_edge(target_id);
CREATE INDEX IF NOT EXISTS idx_memory_edge_project ON memory_edge(project_id);
CREATE INDEX IF NOT EXISTS idx_memory_edge_relation ON memory_edge(relation_type);
-- Ensure memory_entity has all columns code expects
ALTER TABLE memory_entity ADD COLUMN IF NOT EXISTS deleted_at TIMESTAMPTZ;
ALTER TABLE memory_entity ADD COLUMN IF NOT EXISTS source_count INTEGER DEFAULT 1;
-- Unique constraint for entity upsert dedup
DO $$
BEGIN
-- Dedup existing rows before creating unique index
DELETE FROM memory_entity a USING memory_entity b
WHERE a.project_id = b.project_id AND a.name = b.name
AND a.t_created < b.t_created;
EXCEPTION WHEN OTHERS THEN NULL;
END $$;
CREATE UNIQUE INDEX IF NOT EXISTS idx_memory_entity_project_name ON memory_entity(project_id, name);
-- ROLLBACK instructions:
-- DROP TABLE IF EXISTS memory_edge;
-- ALTER TABLE IF EXISTS memory_edge_legacy RENAME TO memory_edge;
-358
View File
@@ -1,358 +0,0 @@
use anyhow::Result;
use sqlx::{PgPool, FromRow};
use uuid::Uuid;
use serde::{Deserialize, Serialize};
use chrono::{DateTime, Utc};
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
pub struct AgentPrompt {
pub id: Uuid,
pub project_id: String,
pub name: String,
pub template: String,
pub target_model: Option<String>,
pub task_category: String,
pub usage_count: i64,
pub avg_quality: f32,
pub last_used: Option<DateTime<Utc>>,
pub active: bool,
pub version: i32,
pub tags: Vec<String>,
pub created_at: DateTime<Utc>,
pub updated_at: DateTime<Utc>,
}
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
pub struct AgentSkill {
pub id: Uuid,
pub project_id: String,
pub agent_id: String,
pub name: String,
pub description: String,
pub trigger_patterns: Vec<String>,
pub success_rate: f32,
pub invocation_count: i64,
pub avg_latency_ms: i64,
pub linked_prompts: Vec<Uuid>,
pub enabled: bool,
pub created_at: DateTime<Utc>,
pub updated_at: DateTime<Utc>,
}
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
pub struct AgentDecision {
pub id: Uuid,
pub project_id: String,
pub agent_id: String,
pub action: String,
pub reasoning: String,
pub alternatives: Vec<String>,
pub confidence: f32,
pub context_entities: Vec<Uuid>,
pub tool: Option<String>,
pub task: Option<String>,
pub outcome_success: Option<bool>,
pub outcome_quality: Option<f32>,
pub outcome_feedback: Option<String>,
pub outcome_recorded_at: Option<DateTime<Utc>>,
pub created_at: DateTime<Utc>,
pub updated_at: DateTime<Utc>,
}
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
pub struct RolePromptMapping {
pub id: Uuid,
pub project_id: String,
pub role_name: String,
pub prompt_id: Uuid,
pub priority: i32,
pub active: bool,
pub created_at: DateTime<Utc>,
pub updated_at: DateTime<Utc>,
}
#[derive(Debug, Clone, Serialize, Deserialize, FromRow)]
pub struct AgentMetrics {
pub id: Uuid,
pub project_id: String,
pub agent_id: String,
pub requests_total: i64,
pub requests_success: i64,
pub requests_failed: i64,
pub average_latency_ms: f32,
pub p95_latency_ms: f32,
pub p99_latency_ms: f32,
pub error_rate: f32,
pub recorded_at: DateTime<Utc>,
}
pub struct AgentRepository {
pool: PgPool,
}
impl AgentRepository {
pub fn new(pool: PgPool) -> Self {
AgentRepository { pool }
}
pub async fn create_prompt(&self, prompt: AgentPrompt) -> Result<AgentPrompt> {
let result = sqlx::query_as::<_, AgentPrompt>(
r#"
INSERT INTO agent_prompt
(project_id, name, template, target_model, task_category, active, version, tags)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
RETURNING *
"#,
)
.bind(&prompt.project_id)
.bind(&prompt.name)
.bind(&prompt.template)
.bind(&prompt.target_model)
.bind(&prompt.task_category)
.bind(prompt.active)
.bind(prompt.version)
.bind(&prompt.tags)
.fetch_one(&self.pool)
.await?;
Ok(result)
}
pub async fn get_prompt(&self, id: Uuid) -> Result<Option<AgentPrompt>> {
let result = sqlx::query_as::<_, AgentPrompt>(
"SELECT * FROM agent_prompt WHERE id = $1"
)
.bind(id)
.fetch_optional(&self.pool)
.await?;
Ok(result)
}
pub async fn list_prompts(&self, project_id: &str) -> Result<Vec<AgentPrompt>> {
let results = sqlx::query_as::<_, AgentPrompt>(
"SELECT * FROM agent_prompt WHERE project_id = $1 AND active = true ORDER BY created_at DESC"
)
.bind(project_id)
.fetch_all(&self.pool)
.await?;
Ok(results)
}
pub async fn update_prompt_usage(&self, id: Uuid, quality_score: f32) -> Result<()> {
sqlx::query(
r#"
UPDATE agent_prompt
SET usage_count = usage_count + 1,
avg_quality = (avg_quality * (usage_count) + $2) / (usage_count + 1),
last_used = NOW(),
updated_at = NOW()
WHERE id = $1
"#,
)
.bind(id)
.bind(quality_score)
.execute(&self.pool)
.await?;
Ok(())
}
pub async fn create_skill(&self, skill: AgentSkill) -> Result<AgentSkill> {
let result = sqlx::query_as::<_, AgentSkill>(
r#"
INSERT INTO agent_skill
(project_id, agent_id, name, description, enabled)
VALUES ($1, $2, $3, $4, $5)
RETURNING *
"#,
)
.bind(&skill.project_id)
.bind(&skill.agent_id)
.bind(&skill.name)
.bind(&skill.description)
.bind(skill.enabled)
.fetch_one(&self.pool)
.await?;
Ok(result)
}
pub async fn get_skill(&self, id: Uuid) -> Result<Option<AgentSkill>> {
let result = sqlx::query_as::<_, AgentSkill>(
"SELECT * FROM agent_skill WHERE id = $1"
)
.bind(id)
.fetch_optional(&self.pool)
.await?;
Ok(result)
}
pub async fn list_skills(&self, project_id: &str, agent_id: &str) -> Result<Vec<AgentSkill>> {
let results = sqlx::query_as::<_, AgentSkill>(
"SELECT * FROM agent_skill WHERE project_id = $1 AND agent_id = $2 AND enabled = true ORDER BY created_at DESC"
)
.bind(project_id)
.bind(agent_id)
.fetch_all(&self.pool)
.await?;
Ok(results)
}
pub async fn create_decision(&self, decision: AgentDecision) -> Result<AgentDecision> {
let result = sqlx::query_as::<_, AgentDecision>(
r#"
INSERT INTO agent_decision
(project_id, agent_id, action, reasoning, confidence, tool, task)
VALUES ($1, $2, $3, $4, $5, $6, $7)
RETURNING *
"#,
)
.bind(&decision.project_id)
.bind(&decision.agent_id)
.bind(&decision.action)
.bind(&decision.reasoning)
.bind(decision.confidence)
.bind(&decision.tool)
.bind(&decision.task)
.fetch_one(&self.pool)
.await?;
Ok(result)
}
pub async fn record_decision_outcome(
&self,
id: Uuid,
success: bool,
quality: f32,
feedback: Option<&str>,
) -> Result<()> {
sqlx::query(
r#"
UPDATE agent_decision
SET outcome_success = $2,
outcome_quality = $3,
outcome_feedback = $4,
outcome_recorded_at = NOW(),
updated_at = NOW()
WHERE id = $1
"#,
)
.bind(id)
.bind(success)
.bind(quality)
.bind(feedback)
.execute(&self.pool)
.await?;
Ok(())
}
pub async fn create_role_mapping(&self, mapping: RolePromptMapping) -> Result<RolePromptMapping> {
let result = sqlx::query_as::<_, RolePromptMapping>(
r#"
INSERT INTO role_prompt_mapping
(project_id, role_name, prompt_id, priority, active)
VALUES ($1, $2, $3, $4, $5)
RETURNING *
"#,
)
.bind(&mapping.project_id)
.bind(&mapping.role_name)
.bind(mapping.prompt_id)
.bind(mapping.priority)
.bind(mapping.active)
.fetch_one(&self.pool)
.await?;
Ok(result)
}
pub async fn get_prompts_for_role(&self, project_id: &str, role_name: &str) -> Result<Vec<AgentPrompt>> {
let results = sqlx::query_as::<_, AgentPrompt>(
r#"
SELECT ap.* FROM agent_prompt ap
INNER JOIN role_prompt_mapping rpm ON ap.id = rpm.prompt_id
WHERE rpm.project_id = $1 AND rpm.role_name = $2 AND rpm.active = true
ORDER BY rpm.priority DESC, ap.created_at DESC
"#,
)
.bind(project_id)
.bind(role_name)
.fetch_all(&self.pool)
.await?;
Ok(results)
}
pub async fn save_metrics(&self, metrics: AgentMetrics) -> Result<()> {
sqlx::query(
r#"
INSERT INTO agent_metrics
(project_id, agent_id, requests_total, requests_success, requests_failed,
average_latency_ms, p95_latency_ms, p99_latency_ms, error_rate)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
ON CONFLICT (project_id, agent_id, DATE(recorded_at)) DO UPDATE SET
requests_total = EXCLUDED.requests_total,
requests_success = EXCLUDED.requests_success,
requests_failed = EXCLUDED.requests_failed,
average_latency_ms = EXCLUDED.average_latency_ms,
p95_latency_ms = EXCLUDED.p95_latency_ms,
p99_latency_ms = EXCLUDED.p99_latency_ms,
error_rate = EXCLUDED.error_rate
"#,
)
.bind(&metrics.project_id)
.bind(&metrics.agent_id)
.bind(metrics.requests_total)
.bind(metrics.requests_success)
.bind(metrics.requests_failed)
.bind(metrics.average_latency_ms)
.bind(metrics.p95_latency_ms)
.bind(metrics.p99_latency_ms)
.bind(metrics.error_rate)
.execute(&self.pool)
.await?;
Ok(())
}
pub async fn log_prompt_usage(
&self,
project_id: &str,
prompt_id: Uuid,
agent_id: Option<&str>,
model: Option<&str>,
input_tokens: Option<i32>,
output_tokens: Option<i32>,
quality_score: Option<f32>,
duration_ms: i64,
error_message: Option<&str>,
) -> Result<()> {
sqlx::query(
r#"
INSERT INTO prompt_usage_log
(project_id, prompt_id, agent_id, model_used, input_tokens, output_tokens,
quality_score, duration_ms, error_message)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
"#,
)
.bind(project_id)
.bind(prompt_id)
.bind(agent_id)
.bind(model)
.bind(input_tokens)
.bind(output_tokens)
.bind(quality_score)
.bind(duration_ms)
.bind(error_message)
.execute(&self.pool)
.await?;
Ok(())
}
}
+1
View File
@@ -1,6 +1,7 @@
use chrono::{DateTime, Utc};
use sqlx::PgPool;
use uuid::Uuid;
use serde_json::json;
/// Minimal audit logger - records version snapshots on mutation
#[derive(Clone)]
-1
View File
@@ -8,7 +8,6 @@ pub mod edge_repo;
pub mod community_repo;
pub mod versioning;
pub mod audit_logger;
pub mod agent_repo;
// pub mod db_repo; // TODO: Fix Entity schema integration
pub use event_log::{EventRecord, LogWriter};
-321
View File
@@ -1,321 +0,0 @@
# Poimen Memory: Authentik JWT + SOPS Encryption Setup
## Overview
The Poimen Memory service uses:
1. **Authentik service account** for OAuth2 client credentials flow
2. **SOPS + Age encryption** to encrypt secrets in git
3. **JWT tokens** for authentication to LLM gateway, S3, and other services
## Architecture
```
┌─────────────────────────────────────────────────────────────┐
│ Kubernetes (poimen) │
├─────────────────────────────────────────────────────────────┤
│ │
│ ┌──────────────┐ ┌─────────────────┐ │
│ │ ConfigMap │ │ Secret (SOPS) │ │
│ │ (unencrypted)│ │ (age-encrypted)│ │
│ └──────┬───────┘ └────────┬────────┘ │
│ │ │ │
│ ├─────────┬───────────┤ │
│ │ │ │ │
│ ┌────▼─────────▼───────────▼────┐ │
│ │ poimen-memory Pod │ │
│ │ Environment Variables: │ │
│ │ - LLM_ENDPOINT │ │
│ │ - AUTHENTIK_ISSUER │ │
│ │ - AUTHENTIK_CLIENT_ID │ │
│ │ - AUTHENTIK_CLIENT_SECRET │ │
│ │ - S3_ACCESS_KEY │ │
│ │ - S3_SECRET_KEY │ │
│ └────┬────────────────┬──────────┘ │
│ │ │ │
│ ┌──────▼──┐ ┌──────────▼──────┐ │
│ │ Authentik│ │ LLM Endpoint │ │
│ │ (JWT) │ │ (api.riotpiao) │ │
│ └──────────┘ └─────────────────┘ │
│ │
│ ┌─────────────────────────────────────┐ │
│ │ Entity Extraction Pipeline │ │
│ │ ┌────────────────────────────┐ │ │
│ │ │ 1. WikiLink fallback │ │ │
│ │ │ 2. LLM extraction (JWT auth)│ │ │
│ │ │ 3. Reflection verification │ │ │
│ │ │ 4. Contradiction detection │ │ │
│ │ └────────────────────────────┘ │ │
│ └──────────────┬──────────────────────┘ │
│ │ │
│ ┌───────▼────────┐ │
│ │ PostgreSQL │ │
│ │ (entities DB) │ │
│ └────────────────┘ │
│ │
└─────────────────────────────────────────────────────────────┘
```
## Step 1: Create Authentik Service Account
### In Authentik Admin Panel:
1. Navigate: **Settings****Applications** → **Create Application**
2. Name: `poimen-memory`
3. Slug: `poimen-memory`
4. Provider: Create a new OAuth2 Provider
- Name: `poimen-memory`
- Client type: `confidential`
- Client ID: `<auto-generated>`
- Client secret: `<auto-generated>`
5. Save and note the **Client ID** and **Client Secret**
### Verify OAuth2 Token Endpoint:
```bash
curl -X POST https://authentik.riotpiao.com/application/o/token/ \
-d "grant_type=client_credentials" \
-d "client_id=<CLIENT_ID>" \
-d "client_secret=<CLIENT_SECRET>"
# Response:
# {
# "access_token": "eyJ0eXAi...",
# "token_type": "Bearer",
# "expires_in": 3600
# }
```
## Step 2: Create Encrypted Secrets File
### 2.1 Ensure SOPS is configured:
```bash
# Load SOPS_AGE_KEY_FILE
export SOPS_AGE_KEY_FILE=~/.sops/key.txt
# Verify key exists
ls -la ~/.sops/key.txt
```
### 2.2 Create unencrypted secrets template:
```yaml
# k8s/app/poimen-memory-secrets.yaml
apiVersion: v1
kind: Secret
metadata:
name: poimen-memory-secrets
namespace: poimen
type: Opaque
stringData:
# Authentik OAuth2 Credentials
AUTHENTIK_ISSUER: "https://authentik.riotpiao.com/application/o/memory"
AUTHENTIK_AUDIENCE: "poimen-memory"
AUTHENTIK_CLIENT_ID: "<from-authentik-app>"
AUTHENTIK_CLIENT_SECRET: "<from-authentik-app>"
# LLM Gateway API Key (optional fallback)
LLM_API_KEY: "<jwt-will-be-auto-generated>"
# S3/Minio Credentials
S3_ACCESS_KEY: "<minio-access-key>"
S3_SECRET_KEY: "<minio-secret-key>"
```
### 2.3 Encrypt with SOPS:
```bash
export SOPS_AGE_KEY_FILE=~/.sops/key.txt
cd ~/workplace/Poimen/memory
sops -e k8s/app/poimen-memory-secrets.yaml > k8s/app/poimen-memory-secrets.enc.yaml
# Verify encryption worked
sops -d k8s/app/poimen-memory-secrets.enc.yaml | head -20
```
### 2.4 Commit encrypted file only:
```bash
git add k8s/app/poimen-memory-secrets.enc.yaml
git add .sops.yaml
git rm k8s/app/poimen-memory-secrets.yaml # Remove plaintext
git commit -m "feat: add SOPS-encrypted Authentik secrets"
```
## Step 3: Deploy to Kubernetes
### 3.1 Install KSOPS plugin (if using ArgoCD):
```bash
# ArgoCD Helm values
kustomization:
plugins:
- name: Kustomize
image: ghcr.io/viaduct-ai/kustomize-sops:v4.1.1
```
### 3.2 Apply secrets manifest:
```bash
# With KSOPS: ArgoCD auto-decrypts and applies
# Without KSOPS: Manual decryption before apply
export SOPS_AGE_KEY_FILE=~/.sops/key.txt
sops -d k8s/app/poimen-memory-secrets.enc.yaml | kubectl apply -f -
# Verify secret created
kubectl -n poimen get secret poimen-memory-secrets
kubectl -n poimen describe secret poimen-memory-secrets
```
### 3.3 Update deployment envFrom:
```yaml
# k8s/app/deployment.yaml
spec:
template:
spec:
containers:
- name: poimen-memory
envFrom:
- configMapRef:
name: poimen-memory-config
- secretRef:
name: poimen-memory-secrets # <-- Add this
```
## Step 4: Entity Extractor JWT Flow
### Code: `crates/mem-ingest/src/entity_extractor.rs`
```rust
// Initialization
pub struct LlmEntityExtractor {
jwt_issuer: Option<Arc<Mutex<AuthentikJwtIssuer>>>,
}
impl LlmEntityExtractor {
pub fn new(model_name: &str) -> Self {
let jwt_issuer = AuthentikJwtIssuer::from_env().ok();
Self {
jwt_issuer: jwt_issuer.map(|iss| Arc::new(Mutex::new(iss))),
}
}
}
// LLM call with JWT
async fn call_llm_endpoint(&self, prompt: &str) -> Result<String> {
// Get JWT token from Authentik (cached, auto-refreshed)
let auth_header = if let Some(jwt_issuer) = &self.jwt_issuer {
let issuer = jwt_issuer.lock().await;
let token = issuer.get_access_token().await?;
format!("Bearer {}", token)
} else {
format!("Bearer {}", fallback_api_key)
};
// POST to LLM endpoint with JWT
client
.post(&endpoint)
.header("Authorization", auth_header)
.json(&payload)
.send()
.await?
}
```
## Step 5: Runtime Verification
### 5.1 Check JWT token exchange in logs:
```bash
kubectl -n poimen logs deployment/poimen-memory | grep -i "authentik\|jwt"
# Expected output:
# [2026-01-09T20:30:15Z] Obtained Authentik JWT token (expires in 3600 seconds)
# [2026-01-09T20:30:15Z] LLM response (via Authentik JWT): {...}
```
### 5.2 Test entity extraction end-to-end:
```bash
# Port-forward to service
kubectl -n poimen port-forward svc/poimen-memory 8080:8080 &
# Ingest a record
curl -X POST http://localhost:8080/memory/ingest \
-H "Content-Type: application/json" \
-d '{
"project": "homelab",
"source": "test://jwt",
"ingest_id": "jwt-test-001",
"records": [{
"role": "architect",
"text": "[[Kubernetes]] uses [[Docker]]. [[ArgoCD]] manages deployments.",
"timestamp": "2026-01-09T20:30:00Z",
"source_position": 0
}]
}'
# Check logs for JWT usage
kubectl -n poimen logs deployment/poimen-memory | tail -20
```
## Step 6: Monitoring & Maintenance
### Token Expiry Handling:
- JWT tokens are cached with auto-refresh
- If token expires during use, new token is fetched automatically
- No manual token rotation required
### Credential Rotation:
- Rotate Authentik client secret periodically
- Update SOPS secret file and re-encrypt
- Redeploy pod to pick up new secret
### SOPS Key Rotation (Yearly):
```bash
# Generate new age key
age-keygen -o ~/.sops/key.txt.new
# Re-encrypt all secrets with new key
for file in k8s/**/*.enc.yaml; do
sops -r $file
done
# Update ArgoCD to use new key
# Commit changes
git add k8s/**/*.enc.yaml
git commit -m "chore: rotate SOPS encryption keys"
```
## Troubleshooting
### Issue: "AUTHENTIK_ISSUER not set"
**Cause**: Secret not mounted properly
**Solution**: `kubectl -n poimen get secret poimen-memory-secrets`
### Issue: "JWT token request failed: 401"
**Cause**: Invalid client credentials
**Solution**: Verify Client ID/Secret in Authentik, check SOPS decryption
### Issue: "error loading config: no matching creation rules found"
**Cause**: SOPS .sops.yaml not configured correctly
**Solution**: Use `.sops.yaml` with explicit age key instead of config-based rules
### Issue: "LLM API error: 403 Forbidden"
**Cause**: JWT token doesn't have permission to LLM gateway
**Solution**: Add RBAC role "LLM User" to service account in Authentik
---
## Files Modified
- ✅ `crates/mem-ingest/src/authentik_jwt.rs` — JWT token exchange module
- ✅ `crates/mem-ingest/src/entity_extractor.rs` — LLM calls with JWT
- ✅ `crates/mem-ingest/src/lib.rs` — Module export
- ✅ `k8s/app/poimen-memory-secrets.yaml` — Secret template (plaintext, not committed)
- ✅ `k8s/app/poimen-memory-secrets.enc.yaml` — Secret encrypted with SOPS
- ✅ `k8s/app/deployment.yaml` — Updated envFrom for secrets
- ✅ `k8s/app/config.yaml` — LLM endpoint configuration
- ✅ `k8s/.sops.yaml` — SOPS encryption rules
-982
View File
@@ -1,982 +0,0 @@
# Memory Service Observability
## Why Observe a Knowledge Base System
A memory service that retrieves wrong facts is worse than one that retrieves nothing — it causes hallucination. Traditional web services measure uptime and latency. A knowledge-base service must also measure **whether the answer was correct**, **whether the stored fact was accurate**, and **whether stale or contradictory information leaked through**.
Every metric in this document exists to answer one question: **"Did the user get the right information, fast enough, from a source we trust?"**
---
## Call Flow: Left to Right
```
INGEST PATH
===========
Client ─── POST /memory/ingest ─── Auth + Rate Limit ─── Dedup Check ─── Embed (768d) ─── Dual Write ─── Done
│ │ │ │ │ │
│ [I1: req_count] [I2: auth_ms] [I3: dedup_hit] [I4: embed_ms] [I5: write_ms]
│ │ │
│ pgvector INSERT OpenSearch INDEX
│ │ │
│ [I6: pg_ms] [I7: os_ms]
│ │
│ [I8: os_fail_count]
│ (eventual consistency)
QUERY PATH
==========
Client ─── POST /memory/query ─── Auth + Rate Limit ─── Classify Intent ─── Embed Query ─── Search ─── RRF Fusion ─── Rerank ─── Respond
│ │ │ │ │ │ │ │ │
│ [Q1: req_count] [Q2: auth_ms] [Q3: intent_type] [Q4: embed_ms] │ [Q7: rrf_ms] [Q8: rerank_ms] │
│ │ │
│ ┌────────────┴──────────┐ │
│ pgvector cosine OpenSearch BM25 │
│ │ │ │
│ [Q5: sem_ms] [Q6: lex_ms] │
│ [Q5a: sem_count] [Q6a: lex_count] │
│ [Q9: total_ms]
│ [Q10: result_count]
CONTEXT PATH (3-tier retrieval)
==============================
Client ─── POST /memory/context ─── Tier 1: Exact Signature ─── Tier 2: Hybrid Search ─── Tier 3: Reference Fallback ─── Budget Assembly ─── Respond
│ │ │ │ │ │ │
│ [C1: req_count] [C2: t1_hit] [C3: t2_hit] [C4: t3_hit] [C5: budget_used] [C6: total_ms]
│ [C2a: t1_ms] [C3a: t2_ms] [C4a: t3_ms] [C5a: dropped_count]
RELEVANCE JUDGMENT (offline, periodic)
=====================================
Sampled Query Log ─── Replay Query ─── Retrieve Top-K ─── Qwen-7B Judge ─── Score (0-2) ─── Compute NDCG/MRR/Precision/Recall
│ │ │ │ │
[R1: sample_size] [R2: replay_ms] [R3: judge_ms] [R4: relevance_dist] [R5: ndcg_10]
[R3a: judge_cost] [R6: mrr]
[R7: precision_10]
[R8: recall_10]
```
---
## 1. Ingest Observability
### Why
Every fact written to memory becomes a retrieval candidate. A bad write — duplicate, contradictory, or malformed — pollutes all future queries. Ingest observability answers: **"How many facts are entering the system, how fast, and are any of them bad?"**
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **I1** | `ingest_requests_total` | Counter | requests | Total write demand. Capacity planning baseline. Sudden spikes = upstream behavior change. |
| **I2** | `ingest_auth_duration_seconds` | Histogram | seconds | JWT validation overhead. Should be < 5ms. Spike = JWKS fetch or Authentik down. |
| **I3** | `ingest_dedup_hits_total` | Counter | requests | Idempotency saves. High ratio = client retry storm or misconfigured source. Low = healthy unique writes. |
| **I4** | `ingest_embed_duration_seconds` | Histogram | seconds | Embedding latency per chunk. Budget: < 50ms for single chunk. Spike = model cold start or GPU contention. |
| **I5** | `ingest_dual_write_duration_seconds` | Histogram | seconds | Total time to write both stores. SLO: p99 < 500ms. |
| **I6** | `ingest_pgvector_duration_seconds` | Histogram | seconds | Postgres INSERT latency. Includes HNSW index update. Degrades as table grows. |
| **I7** | `ingest_opensearch_duration_seconds` | Histogram | seconds | OpenSearch bulk index latency. Sensitive to segment merges. |
| **I8** | `ingest_opensearch_failures_total` | Counter | failures | OpenSearch write failures. System is eventual-consistent: pgvector is primary. But if this counter grows, lexical search degrades silently. |
| **I9** | `ingest_bytes_total` | Counter | bytes | Total data volume written. Growth rate = storage budget burn. |
| **I10** | `ingest_chunks_total` | Counter | chunks | Write throughput in logical units. 1 ingest request may produce N chunks after splitting. |
| **I11** | `ingest_contradiction_detected_total` | Counter | contradictions | Facts that conflict with existing knowledge. High count = noisy source or domain shift. Each one enters review queue. |
| **I12** | `ingest_review_queue_depth` | Gauge | items | Pending human reviews. Growing = reviewers not keeping up. Stale contradictions = latent hallucination risk. |
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `ingest_opensearch_failures_total` rate > 5/min for 10min | **Warning** | Check OpenSearch cluster health. Lexical search degrading. |
| `ingest_review_queue_depth` > 100 for 24h | **Warning** | Unreviewed contradictions. Risk of serving conflicting facts. |
| `ingest_pgvector_duration_seconds` p99 > 1s | **Critical** | Postgres overloaded. HNSW index rebuild or VACUUM needed. |
| `ingest_dedup_hits_total` / `ingest_requests_total` > 0.5 | **Warning** | More than half of writes are duplicates. Source misconfiguration. |
---
## 2. Query Observability
### Why
Query latency is what the user feels. But latency alone is insufficient — a fast query returning wrong results is worse than a slow correct one. Query observability answers: **"Did the system respond quickly, and did the search pipeline find the right documents?"**
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **Q1** | `query_requests_total` | Counter | requests | Read demand. Ratio to ingest = read/write skew. Memory systems are read-heavy (10:1+). |
| **Q2** | `query_auth_duration_seconds` | Histogram | seconds | Same as ingest. Shared auth path. |
| **Q3** | `query_intent_classification` | Counter (labeled) | requests | Labels: `bug_fix`, `how_to`, `reference`, `faq`. Distribution reveals what users ask most. If 80% is `bug_fix` but recall is low for that intent, prioritize that retrieval path. |
| **Q4** | `query_embed_duration_seconds` | Histogram | seconds | Query embedding latency. Same model as ingest. Should match I4. |
| **Q5** | `query_semantic_duration_seconds` | Histogram | seconds | pgvector cosine search. SLO: p99 < 200ms. Degrades with index size. |
| **Q5a** | `query_semantic_candidates` | Histogram | count | Number of vectors above similarity floor. Zero = total miss. Hundreds = floor too low. |
| **Q6** | `query_lexical_duration_seconds` | Histogram | seconds | OpenSearch BM25 latency. SLO: p99 < 150ms. |
| **Q6a** | `query_lexical_candidates` | Histogram | count | BM25 hit count. Zero = query terms not in corpus (vocabulary gap). |
| **Q7** | `query_rrf_fusion_duration_seconds` | Histogram | seconds | RRF merge time. Should be < 5ms (in-memory). If slow, too many candidates. |
| **Q8** | `query_rerank_duration_seconds` | Histogram | seconds | Cross-encoder reranking. Most expensive step. Budget: < 200ms for top-20. |
| **Q9** | `query_total_duration_seconds` | Histogram | seconds | End-to-end latency. SLO: p99 < 500ms. User-facing number. |
| **Q10** | `query_results_returned` | Histogram | count | How many results pass all filters. Zero = query miss. Track per-intent. |
| **Q11** | `query_empty_results_total` | Counter | requests | Queries that returned nothing. High rate = coverage gap in knowledge base. |
| **Q12** | `query_score_distribution` | Histogram | score (0-1) | Top-1 result score distribution. Bimodal = some queries match well, others poorly. Low mean = embedding quality issue. |
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `query_total_duration_seconds` p99 > 1s | **Critical** | Pipeline bottleneck. Check Q5, Q6, Q8 to isolate which leg is slow. |
| `query_empty_results_total` rate > 20% of Q1 | **Warning** | 1 in 5 queries finds nothing. Coverage gap. Check if ingest is running. |
| `query_semantic_candidates` p50 = 0 | **Critical** | Embedding search broken. Model mismatch or empty index. |
| `query_lexical_duration_seconds` p99 > 500ms | **Warning** | OpenSearch overloaded. Check segment count, heap usage. |
---
## 3. Context Endpoint (Three-Tier) Observability
### Why
The context endpoint is the primary consumer-facing API. It orchestrates three retrieval tiers with budget constraints. Observing tier hit rates reveals whether the knowledge base has coverage at each level, and whether the budget assembly is dropping important results.
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **C1** | `context_requests_total` | Counter | requests | Context lookup demand. Main integration point. |
| **C2** | `context_tier1_hits_total` | Counter | hits | Exact signature matches. High = system is learning from repeated failures. SLO: tier-1 hit rate >= 0.80. |
| **C2a** | `context_tier1_duration_seconds` | Histogram | seconds | Signature lookup. Should be < 50ms (indexed hash). |
| **C3** | `context_tier2_hits_total` | Counter | hits | Hybrid search hits. Bulk of useful results. |
| **C3a** | `context_tier2_duration_seconds` | Histogram | seconds | Full hybrid search. Budget: < 500ms. |
| **C4** | `context_tier3_hits_total` | Counter | hits | Reference fallback. High ratio = learned knowledge insufficient, falling back to docs. |
| **C4a** | `context_tier3_duration_seconds` | Histogram | seconds | Obsidian API + reference retrieval. Slowest tier. |
| **C5** | `context_budget_used_bytes` | Histogram | bytes | How much of the token budget was consumed. Full = rich context. Low = sparse knowledge. |
| **C5a** | `context_dropped_results_total` | Counter | results | Results dropped to fit budget. High = budget too small or results too verbose. |
| **C6** | `context_total_duration_seconds` | Histogram | seconds | End-to-end context assembly. SLO: p99 < 2s. |
| **C7** | `context_tier_distribution` | Counter (labeled) | requests | Label: `tier=1\|2\|3`. Which tier served the primary result. Shift from tier-1 to tier-3 over time = knowledge decay. |
| **C8** | `context_degraded_total` | Counter | requests | Requests where a leg failed (e.g., Obsidian timeout). Partial results served. |
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `context_tier1_hits_total` / `context_requests_total` < 0.60 | **Warning** | Signature match rate dropping. System not learning from failures. Check ingest pipeline. |
| `context_dropped_results_total` rate > 30% of results | **Warning** | Budget too tight. Users missing relevant context. |
| `context_degraded_total` rate > 5% | **Warning** | Partial responses. Check Obsidian API, OpenSearch health. |
---
## 4. Relevance Judgment with Qwen-7B
### Why
All the metrics above measure speed and volume. None measure **correctness**. A system that returns 10 results in 50ms is useless if those results are wrong. Traditional IR evaluation requires human-labeled relevance judgments — expensive and slow. Instead, we use a **Qwen-7B model as an automated relevance judge** on sampled queries.
This is the single most important observability signal for hallucination prevention. If retrieval precision drops, the LLM downstream gets wrong context and hallucinates. Catching it here — at the retrieval layer — is 10x cheaper than catching it at the generation layer.
### Why Qwen-7B
- **Cost**: ~0.002 USD per judgment. At 500 samples/day = $1/day. A 70B model costs 10x more for marginal gain.
- **Speed**: ~200ms per judgment on 1x A10. Fast enough for daily batch evaluation.
- **Accuracy**: 7B models achieve 85-90% agreement with human relevance labels on standard benchmarks (BEIR, MS MARCO). Sufficient for trend detection. We are not using it for absolute measurement — we are using it for **drift detection**.
- **Self-hosted**: Runs inside the cluster. No data leaves the network. Required for security-sensitive knowledge bases.
### Judgment Flow
```
DAILY RELEVANCE EVALUATION (Cron, 03:00 UTC)
============================================
Query Log (24h) ─── Sample 500 queries ─── Replay each query ─── Get top-10 results ─── For each (query, result) pair:
│ │
[R1: sample_size] Qwen-7B Prompt:
┌────────┴────────┐
│ "Given query: │
│ '{query}' │
│ │
│ Rate this │
│ result: │
│ '{result}' │
│ │
│ Score: │
│ 0 = irrelevant │
│ 1 = partial │
│ 2 = perfect │
└────────┬────────┘
[R4: score]
Aggregate: NDCG@10, MRR, Precision@10, Recall@10
┌───────────────┴───────────────┐
[R5: ndcg_10] [R7: precision_10]
[R6: mrr] [R8: recall_10]
Store in Postgres
(daily time-series)
Grafana Dashboard
(7-day rolling avg)
```
### Prompt Template
```
You are a relevance judge for a knowledge base system.
Given a user query and a retrieved document, rate the relevance:
- 0: Irrelevant. The document does not help answer the query at all.
- 1: Partially relevant. The document contains some useful information but does not fully answer the query.
- 2: Highly relevant. The document directly and completely answers the query.
Query: "{query}"
Retrieved Document:
---
{document_text}
---
Relevance Score (0, 1, or 2):
```
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **R1** | `relevance_sample_size` | Gauge | queries | Number of queries evaluated. 500 gives statistically stable NDCG with ±0.02 CI. |
| **R2** | `relevance_replay_duration_seconds` | Histogram | seconds | Time to replay and retrieve. Should match Q9. |
| **R3** | `relevance_judge_duration_seconds` | Histogram | seconds | Qwen-7B inference time per pair. Budget: < 300ms. |
| **R3a** | `relevance_judge_cost_usd` | Counter | USD | Running cost. Alert if budget exceeded. |
| **R4** | `relevance_score_distribution` | Histogram | score (0-2) | Distribution of judgments. Healthy: 60%+ score=2, < 15% score=0. Drift toward 0 = retrieval degradation. |
| **R5** | `relevance_ndcg_10` | Gauge | ratio (0-1) | Ranking quality. **Primary quality metric.** SLO: >= 0.85. Measures whether relevant docs appear at the top. |
| **R6** | `relevance_mrr` | Gauge | ratio (0-1) | Position of first relevant result. SLO: >= 0.80. If MRR drops but NDCG holds, results exist but are buried. |
| **R7** | `relevance_precision_10` | Gauge | ratio (0-1) | Fraction of top-10 that is relevant. Measures noise in results. |
| **R8** | `relevance_recall_10` | Gauge | ratio (0-1) | Fraction of all relevant docs captured in top-10. Low = knowledge exists but search can't find it. |
| **R9** | `relevance_judge_agreement` | Gauge | ratio (0-1) | Weekly: re-judge 50 pairs with human labels. Agreement rate validates the judge. SLO: >= 0.85. If agreement drops, Qwen model needs recalibration. |
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `relevance_ndcg_10` 7-day avg < 0.80 | **Critical** | Retrieval quality degraded. Root cause: embedding drift, index corruption, or knowledge gap. |
| `relevance_ndcg_10` drops > 0.05 in 24h | **Critical** | Sudden quality drop. Check recent ingest for poisoned data. |
| `relevance_mrr` < 0.70 | **Warning** | Relevant docs exist but rank poorly. Check reranker, RRF weights. |
| `relevance_score_distribution` score=0 > 25% | **Warning** | Quarter of results are irrelevant. Coverage gap or embedding model mismatch. |
| `relevance_judge_agreement` < 0.80 | **Warning** | Judge drifting from human labels. Re-evaluate prompt or model. |
---
## 5. Write Volume and Storage Observability
### Why
Memory services grow unboundedly. Unlike caches (eviction policy) or databases (schema constraints), a knowledge base accumulates everything. Write volume tracking answers: **"How fast is the system growing, and when do we need to intervene?"**
Write volume also directly impacts retrieval quality. More documents = more noise in search results. Without compaction, precision degrades as the corpus grows.
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **W1** | `storage_pgvector_rows_total` | Gauge | rows | Total vectors stored. Growth rate = capacity planning. |
| **W2** | `storage_pgvector_bytes` | Gauge | bytes | Disk usage. 768-dim float32 = ~3KB/row with overhead. |
| **W3** | `storage_opensearch_docs_total` | Gauge | docs | OpenSearch document count. Should match W1 (eventual consistency). |
| **W4** | `storage_opensearch_bytes` | Gauge | bytes | OpenSearch index size. Includes inverted index overhead. |
| **W5** | `storage_parity_drift` | Gauge | count | abs(W1 - W3). Should be 0 in steady state. Non-zero = dual-write inconsistency. |
| **W6** | `write_rate_per_hour` | Gauge | chunks/hour | Sustained write throughput. Trigger compaction planning at > 1000/hour. |
| **W7** | `write_rate_per_project` | Gauge (labeled) | chunks/hour | Per-project write rate. Identifies hot projects dominating storage. |
| **W8** | `storage_level_distribution` | Gauge (labeled) | rows | Label: `level=L0\|L1\|L2\|R`. Distribution across learning levels. Healthy: L1 > L0 (facts promoted). If L0 dominates, promotion pipeline stalled. |
| **W9** | `compaction_runs_total` | Counter | runs | How often compaction executes. |
| **W10** | `compaction_dedup_removed_total` | Counter | chunks | Duplicates removed per run. High = ingest dedup isn't catching everything. |
| **W11** | `compaction_stale_gc_removed_total` | Counter | chunks | Stale facts garbage-collected (soft-deleted, age > 30d). |
| **W12** | `compaction_space_freed_bytes` | Counter | bytes | Space recovered per run. Declining = less to compact (good). |
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `storage_parity_drift` > 100 for 1h | **Warning** | pgvector and OpenSearch out of sync. Check dual-write failures (I8). |
| `storage_pgvector_bytes` > 80% of PVC | **Critical** | Storage nearing capacity. Expand PVC or run compaction. |
| `write_rate_per_hour` > 5000 sustained 2h | **Warning** | High write load. Check if upstream is flooding. Consider rate limiting. |
| `storage_level_distribution{level="L0"}` / W1 > 0.7 | **Warning** | 70% of storage is unprocessed L0. Promotion pipeline stalled. |
---
## 6. Pod Resource Observability
### Why
The memory service runs as a Kubernetes pod. If the pod runs out of memory, it gets OOMKilled. If it saturates CPU, latency spikes across all endpoints. These are the physical constraints that gate everything else.
Unlike stateless web services, a memory service has **resident state**: the embedding model weights (~500MB for MiniLM-L6), connection pools, in-flight embeddings, and cached query results. Memory usage is not flat — it grows with concurrent requests. A burst of 50 parallel ingest requests each holding a 768-dim float32 vector = 50 × 3KB = 150KB just in vectors, but the surrounding allocations (HTTP buffers, serde frames, OpenSearch bulk payloads) multiply that 10-20x.
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **P1** | `container_memory_working_set_bytes` | Gauge | bytes | Actual memory in use (excludes reclaimable cache). This is what Kubernetes uses for OOMKill decisions. |
| **P2** | `container_memory_rss` | Gauge | bytes | Resident Set Size. Physical memory held. If RSS diverges from working set, fragmentation is occurring. |
| **P3** | `container_memory_usage_bytes` | Gauge | bytes | Total memory (includes page cache). Less useful for OOM prediction but shows total footprint. |
| **P4** | `container_memory_limit_bytes` | Gauge | bytes | Pod memory limit from resource spec. `P1 / P4` = memory pressure ratio. |
| **P5** | `container_cpu_usage_seconds_total` | Counter | CPU-seconds | CPU consumption rate. `rate(P5[1m])` = CPU cores used. Compare to limit. |
| **P6** | `container_cpu_throttled_seconds_total` | Counter | seconds | Time the pod was CPU-throttled by cgroup. Any throttling = latency impact. |
| **P7** | `container_cpu_cfs_throttled_periods_total` | Counter | periods | Number of CFS periods where throttling occurred. `P7 / total_periods` = throttle ratio. |
| **P8** | `kube_pod_container_resource_requests` | Gauge | cores/bytes | Requested resources. Over-request wastes cluster capacity. Under-request = eviction risk. |
| **P9** | `kube_pod_container_resource_limits` | Gauge | cores/bytes | Resource limits. `P1 / P9{resource="memory"}` > 0.85 = danger zone. |
| **P10** | `kube_pod_status_phase` | Gauge | phase | Running/Pending/Failed/Succeeded. Pending too long = scheduling issues. |
| **P11** | `kube_pod_container_status_restarts_total` | Counter | restarts | OOMKills and CrashLoopBackoff. Any restart = data in flight was lost. |
| **P12** | `container_network_receive_bytes_total` | Counter | bytes | Network ingress. Correlate with ingest volume. Spike = large batch ingest. |
| **P13** | `container_network_transmit_bytes_total` | Counter | bytes | Network egress. Correlate with query response sizes. |
### Memory Breakdown (What Lives in the Pod)
```
Pod Memory Budget (e.g., 2Gi limit)
├── Embedding Model weights ~500MB (loaded once at startup)
├── sqlx connection pool ~50MB (20 connections × ~2.5MB each)
├── OpenSearch HTTP client pool ~20MB (keep-alive connections)
├── In-flight ingest embeddings ~variable (concurrent_requests × ~60KB)
├── In-flight query results ~variable (concurrent_queries × ~200KB)
├── Tokio runtime + thread stacks ~30MB (worker threads × 8MB stack)
├── Rate limiter buckets ~5MB (in-memory token buckets)
├── Idempotency store (24h TTL) ~10-50MB (grows with ingest volume)
└── Heap overhead + fragmentation ~100-200MB
─────────
~800MB baseline + ~variable per-request
```
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `P1 / P4` > 0.85 for 5min | **Critical** | Memory pressure. OOMKill imminent. Scale up limit or reduce concurrency. |
| `P11` increments | **Critical** | Pod restarted. Check if OOMKilled (`kubectl describe pod`). Raise memory limit. |
| `rate(P6[5m])` > 0 for 10min | **Warning** | Sustained CPU throttling. Query/ingest latency affected. Raise CPU limit. |
| `P7 / total_periods` > 0.25 | **Warning** | 25%+ of CPU periods throttled. Under-provisioned. |
| `P1` growing monotonically over 24h | **Warning** | Memory leak. Check idempotency store TTL, connection pool, or embedding cache. |
---
## 7. Availability
### Why
A knowledge base that is down cannot reduce hallucination. If the memory service is unavailable during an LLM generation call, the model falls back to parametric knowledge only — which is exactly where hallucinations come from. Availability is not just uptime; it is **the probability that a query gets a correct answer within the latency SLO**.
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **A1** | `http_requests_total` | Counter (labeled) | requests | Label: `method`, `endpoint`, `status_code`. Foundation for error rate calculation. |
| **A2** | `http_requests_duration_seconds` | Histogram (labeled) | seconds | Label: `endpoint`. Per-endpoint latency distribution. |
| **A3** | `http_5xx_total` | Counter | requests | Server errors. Any 5xx = something broke internally. |
| **A4** | `http_4xx_total` | Counter (labeled) | requests | Label: `status_code`. 401/403 = auth issues. 429 = rate limiting. 400 = bad client. |
| **A5** | `availability_ratio` | Gauge | ratio (0-1) | `1 - (A3 / A1)` over rolling window. SLO: >= 0.999 (three nines). |
| **A6** | `successful_query_ratio` | Gauge | ratio (0-1) | Queries that return 200 with >= 1 result, within 500ms. Stricter than raw availability — includes quality. |
| **A7** | `health_check_consecutive_failures` | Gauge | count | Consecutive `/health` failures. Kubernetes uses this for restart decisions (liveness probe). |
| **A8** | `dependency_up` | Gauge (labeled) | 0/1 | Label: `dependency=postgres\|opensearch\|obsidian\|embedding_model`. Which backends are reachable. |
| **A9** | `graceful_degradation_total` | Counter (labeled) | requests | Label: `degraded_component`. Requests served with partial results because a dependency was down. e.g., OpenSearch down = semantic-only results. |
| **A10** | `circuit_breaker_state` | Gauge (labeled) | 0/1/2 | Label: `backend`. 0=closed (healthy), 1=half-open (probing), 2=open (failing). Per dependency. |
### Availability Calculation
```
Successful Requests (2xx, within SLO latency)
Availability = ────────────────────────────────────────────────────
Total Requests
Three tiers of availability:
1. RAW AVAILABILITY: 1 - (5xx / total) Target: 99.9%
"Did it respond?"
2. LATENCY AVAILABILITY: requests_within_slo / total Target: 99.5%
"Did it respond fast enough?"
3. QUALITY AVAILABILITY: queries_with_results / total Target: 95%
"Did it respond with useful results?"
Monitor all three. A system can be 99.9% available (raw) but only
80% available (quality) if 20% of queries return empty results.
```
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `availability_ratio` < 0.999 over 1h | **Critical** | SLO breach. Page on-call. Check A8 for which dependency is down. |
| `http_5xx_total` rate > 10/min for 5min | **Critical** | Error spike. Check pod logs, Postgres connectivity, OpenSearch health. |
| `dependency_up{dependency="postgres"}` = 0 | **Critical** | Primary store down. All writes and most reads fail. |
| `dependency_up{dependency="opensearch"}` = 0 | **Warning** | Lexical search unavailable. Semantic-only fallback active. Quality degraded. |
| `graceful_degradation_total` rate > 5% of A1 | **Warning** | Serving partial results too often. Fix the degraded dependency. |
| `successful_query_ratio` < 0.90 | **Warning** | 10%+ of queries failing or empty. Check ingest pipeline, index health. |
---
## 8. Ingest Rate Patterns
### Why
Ingest rate is not just a throughput number. The **pattern** of writes reveals system behavior. Bursty writes from batch jobs behave differently from steady trickle from live sessions. A sudden drop in ingest rate may mean the upstream source broke. A sudden spike may mean a replay or backfill is running, which changes storage projections.
For a knowledge base, write rate directly affects retrieval quality: every new chunk is a new candidate that can dilute search precision. Knowing when and how fast writes happen lets you plan compaction, predict storage growth, and detect anomalies.
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **IR1** | `ingest_rate_1m` | Gauge | chunks/min | 1-minute rolling write rate. Shows bursts. |
| **IR2** | `ingest_rate_1h` | Gauge | chunks/hour | Hourly smoothed rate. Capacity planning baseline. |
| **IR3** | `ingest_rate_by_project` | Gauge (labeled) | chunks/hour | Label: `project`. Identifies which project dominates writes. |
| **IR4** | `ingest_rate_by_level` | Gauge (labeled) | chunks/hour | Label: `level=L0\|L1\|L2\|R`. L0 dominance = raw data flooding. L1/L2 growing = healthy knowledge promotion. |
| **IR5** | `ingest_rate_by_source` | Gauge (labeled) | chunks/hour | Label: `source=transcript\|document\|api\|batch`. Reveals upstream behavior. |
| **IR6** | `ingest_batch_size` | Histogram | chunks/batch | Size of batch ingest requests. Large batches (>100) need different backpressure. |
| **IR7** | `ingest_queue_depth` | Gauge | messages | External queue (kmsvc) pending messages. Growing = workers can't keep up. |
| **IR8** | `ingest_queue_age_seconds` | Histogram | seconds | Age of oldest message in queue. > 60s = processing lag. |
| **IR9** | `ingest_bytes_per_chunk` | Histogram | bytes | Average chunk size. Sudden increase = source sending larger payloads. |
| **IR10** | `ingest_throughput_bytes_per_second` | Gauge | bytes/sec | Sustained write bandwidth. Correlate with P12 (network ingress). |
### Rate Patterns and What They Mean
```
Pattern 1: STEADY TRICKLE (healthy)
────────────────────────────────────
chunks/min
10 │ ─ ─ ─ ─ ─ ─ ─ ─ ─ ─ ─ ─
5 │
0 └──────────────────────────── time
Constant ~8-12 chunks/min from live sessions.
Storage growth predictable. Compaction schedule stable.
Pattern 2: BURST (batch job or backfill)
────────────────────────────────────────
chunks/min
500 │ ██
250 │ ██████
0 │──────██──────██─────────── time
Sudden spike. Check: is this a planned backfill?
If unexpected: rate limit may trigger, queue depth spikes.
Action: verify source, check queue lag (IR7).
Pattern 3: DROP TO ZERO (upstream broken)
─────────────────────────────────────────
chunks/min
10 │ ─ ─ ─ ─ ┐
5 │ │
0 │ └──────────────── time
Ingest stopped. Source may be down, auth token expired,
or network partition. Silent failure — no errors, just absence.
Alert on: IR2 = 0 for > 30min during business hours.
Pattern 4: MONOTONIC GROWTH (runaway source)
─────────────────────────────────────────────
chunks/min
100 │ ╱
50 │ ╱───
10 │ ─ ─ ─ ─ ─ ─ ╱───
0 └──────────────────────────── time
Write rate increasing over days. Source producing more data.
Storage projection changes. Compaction may not keep up.
Action: review source, consider sampling or filtering.
```
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `ingest_rate_1h` = 0 for 30min (during business hours) | **Warning** | Ingest stopped. Check upstream source, auth tokens, network. |
| `ingest_rate_1m` > 200 sustained 10min | **Warning** | Burst ingest. Check if planned. Monitor queue depth (IR7). |
| `ingest_queue_depth` > 1000 for 15min | **Critical** | Workers can't keep up. Scale workers or throttle source. |
| `ingest_queue_age_seconds` p99 > 300 | **Warning** | 5+ minutes processing lag. Stale data entering the system. |
| `ingest_rate_by_level{level="L0"}` / total > 0.9 sustained 24h | **Warning** | 90% raw data, no promotion. Knowledge extraction pipeline stalled. |
---
## 9. Postgres Internal Observability
### Why
Postgres is the primary store. Every vector lives there. Every query hits it. Postgres health directly determines memory service health. But Postgres problems are **silent** — a bloated table doesn't throw errors, it just gets slower. A missing VACUUM doesn't alert, it just consumes 2x disk. An HNSW index with wrong parameters doesn't fail, it just returns worse results.
These metrics catch degradation before users notice it.
### Connection Pool and Session Metrics
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG1** | `pg_stat_activity_count` | `pg_stat_activity` | connections | Active connections by state. `active` = running query. `idle` = waiting. `idle in transaction` = **dangerous** — holds locks. |
| **PG2** | `pg_stat_activity_max_duration_seconds` | `pg_stat_activity` | seconds | Longest running query. > 30s = likely stuck or missing index. |
| **PG3** | `pg_stat_activity_waiting_count` | `pg_stat_activity` | connections | Queries waiting for locks. > 0 sustained = lock contention. |
| **PG4** | `pg_settings_max_connections` | `pg_settings` | connections | Max allowed connections. `PG1 / PG4` > 0.8 = pool exhaustion risk. |
### Query Performance
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG5** | `pg_stat_statements_mean_exec_time` | `pg_stat_statements` | ms | Mean execution time per query pattern. Tracks if vector search is degrading over time. |
| **PG6** | `pg_stat_statements_calls` | `pg_stat_statements` | count | Call count per query. Identifies hot queries. Top-1 query consuming 80% of DB time = optimization target. |
| **PG7** | `pg_stat_statements_rows` | `pg_stat_statements` | rows | Rows returned per query. Vector search returning 10k rows when limit is 50 = missing index or wrong query plan. |
| **PG8** | `pg_stat_user_tables_seq_scan` | `pg_stat_user_tables` | scans | Sequential scans on `memory_vector`. Any seq scan on a large vector table = catastrophic. HNSW index not being used. |
| **PG9** | `pg_stat_user_tables_idx_scan` | `pg_stat_user_tables` | scans | Index scans. Should be >> seq scans for vector table. |
### Table and Index Health
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG10** | `pg_stat_user_tables_n_live_tup` | `pg_stat_user_tables` | tuples | Live rows in `memory_vector`. Growth rate = storage planning. |
| **PG11** | `pg_stat_user_tables_n_dead_tup` | `pg_stat_user_tables` | tuples | Dead tuples (deleted/updated but not vacuumed). High ratio = bloat. |
| **PG12** | `pg_dead_tuple_ratio` | computed | ratio | `PG11 / (PG10 + PG11)`. > 0.2 = 20% bloat. VACUUM needed. |
| **PG13** | `pg_stat_user_tables_last_autovacuum` | `pg_stat_user_tables` | timestamp | When autovacuum last ran. > 24h ago on active table = misconfigured threshold. |
| **PG14** | `pg_stat_user_tables_last_autoanalyze` | `pg_stat_user_tables` | timestamp | When autoanalyze last ran. Stale statistics = bad query plans. |
| **PG15** | `pg_table_size_bytes` | `pg_total_relation_size()` | bytes | Total table size including indexes and TOAST. |
| **PG16** | `pg_index_size_bytes` | `pg_indexes_size()` | bytes | HNSW index size. Grows with vectors. If index > table, check parameters. |
| **PG17** | `pg_index_bloat_ratio` | `pgstattuple` | ratio | Index bloat. > 0.3 = REINDEX needed. HNSW indexes don't bloat like B-tree, but monitor anyway. |
### HNSW Index Specific
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG18** | `pg_hnsw_index_size` | `pg_relation_size()` | bytes | Size of the HNSW index on `memory_vector.embedding`. Grows as O(n × m) where m=16. |
| **PG19** | `pg_hnsw_build_time_seconds` | manual / `CREATE INDEX` | seconds | Time to rebuild HNSW index. Needed after parameter changes. At 1M vectors: ~30min. At 10M: hours. Plan maintenance windows. |
| **PG20** | `pg_hnsw_recall_estimate` | benchmark | ratio | Estimated recall of HNSW at current parameters (m=16, ef_construction=200). Run periodic benchmark with known queries. If recall < 0.95, increase ef_search or rebuild with higher m. |
### WAL and Replication (CNPG)
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG21** | `pg_wal_lsn_diff` | `pg_current_wal_lsn()` | bytes | WAL generation rate. High during bulk ingest. Correlate with IR1. |
| **PG22** | `pg_replication_lag_bytes` | `pg_stat_replication` | bytes | Replica lag in bytes. CNPG manages replicas. Lag > 100MB = replica falling behind. |
| **PG23** | `pg_replication_lag_seconds` | `pg_stat_replication` | seconds | Replica lag in time. > 10s = replica can't keep up with write rate. Read queries to replica return stale results. |
| **PG24** | `pg_wal_size_bytes` | `pg_wal` directory | bytes | Total WAL on disk. Unbounded growth = archiving broken or wal_keep_size too high. |
### Transaction and Lock Health
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG25** | `pg_stat_database_xact_commit` | `pg_stat_database` | transactions | Committed transactions/sec. Baseline throughput. |
| **PG26** | `pg_stat_database_xact_rollback` | `pg_stat_database` | transactions | Rolled back transactions. `PG26 / PG25` > 0.01 = 1% rollback rate. Check constraint violations or deadlocks. |
| **PG27** | `pg_stat_database_deadlocks` | `pg_stat_database` | deadlocks | Any deadlock = concurrent write contention. Rare in append-mostly workload. If seen, check compaction + ingest overlap. |
| **PG28** | `pg_stat_database_conflicts` | `pg_stat_database` | conflicts | Replication conflicts. Query on replica canceled due to WAL replay. Adjust `max_standby_streaming_delay`. |
| **PG29** | `pg_locks_count` | `pg_locks` | locks | Lock count by mode. `AccessExclusiveLock` blocks everything — check for DDL during traffic. |
### Cache Efficiency
| ID | Metric | Source | Unit | Why It Matters |
|----|--------|--------|------|----------------|
| **PG30** | `pg_stat_database_blks_hit` | `pg_stat_database` | blocks | Buffer cache hits. |
| **PG31** | `pg_stat_database_blks_read` | `pg_stat_database` | blocks | Disk reads (cache misses). |
| **PG32** | `pg_cache_hit_ratio` | computed | ratio | `PG30 / (PG30 + PG31)`. SLO: >= 0.99. Below 0.95 = shared_buffers too small or working set exceeds RAM. |
| **PG33** | `pg_stat_user_indexes_idx_blks_hit` | `pg_stat_user_indexes` | blocks | HNSW index cache hits. Low hit ratio = index doesn't fit in memory. Increase shared_buffers or effective_cache_size. |
### Key SQL Queries for Monitoring
```sql
-- Dead tuple ratio (bloat indicator)
SELECT relname,
n_live_tup,
n_dead_tup,
CASE WHEN n_live_tup > 0
THEN round(n_dead_tup::numeric / (n_live_tup + n_dead_tup) * 100, 2)
ELSE 0 END AS dead_pct,
last_autovacuum,
last_autoanalyze
FROM pg_stat_user_tables
WHERE relname IN ('memory_vector', 'memory_entity', 'memory_edge')
ORDER BY n_dead_tup DESC;
-- Slowest queries (requires pg_stat_statements)
SELECT query,
calls,
round(mean_exec_time::numeric, 2) AS mean_ms,
round(max_exec_time::numeric, 2) AS max_ms,
rows
FROM pg_stat_statements
WHERE dbid = (SELECT oid FROM pg_database WHERE datname = 'memory')
ORDER BY mean_exec_time DESC
LIMIT 10;
-- Sequential vs index scans (vector table must use index)
SELECT relname,
seq_scan,
idx_scan,
CASE WHEN (seq_scan + idx_scan) > 0
THEN round(idx_scan::numeric / (seq_scan + idx_scan) * 100, 2)
ELSE 100 END AS idx_scan_pct
FROM pg_stat_user_tables
WHERE relname = 'memory_vector';
-- Table and index sizes
SELECT relname,
pg_size_pretty(pg_total_relation_size(relid)) AS total_size,
pg_size_pretty(pg_relation_size(relid)) AS table_size,
pg_size_pretty(pg_indexes_size(relid)) AS index_size
FROM pg_stat_user_tables
WHERE schemaname = 'public'
ORDER BY pg_total_relation_size(relid) DESC;
-- Replication lag (CNPG replicas)
SELECT client_addr,
state,
pg_wal_lsn_diff(pg_current_wal_lsn(), replay_lsn) AS lag_bytes,
extract(epoch FROM now() - replay_lag) AS lag_seconds
FROM pg_stat_replication;
-- Cache hit ratio
SELECT datname,
round(
blks_hit::numeric / NULLIF(blks_hit + blks_read, 0) * 100, 2
) AS cache_hit_pct
FROM pg_stat_database
WHERE datname = 'memory';
-- Connection state breakdown
SELECT state, count(*)
FROM pg_stat_activity
WHERE datname = 'memory'
GROUP BY state;
```
### Alerts
| Condition | Severity | Action |
|-----------|----------|--------|
| `pg_dead_tuple_ratio` > 0.20 on `memory_vector` | **Warning** | 20% bloat. Run `VACUUM ANALYZE memory_vector;` or check autovacuum config. |
| `pg_stat_user_tables_seq_scan` on `memory_vector` increments | **Critical** | Sequential scan on vector table. HNSW index not used. Check query plan with `EXPLAIN ANALYZE`. |
| `pg_cache_hit_ratio` < 0.95 | **Critical** | Cache thrashing. Increase `shared_buffers` or scale to larger instance. |
| `pg_replication_lag_seconds` > 30 | **Warning** | Replica 30s behind. Read queries returning stale data. Check write rate, replica resources. |
| `pg_stat_database_deadlocks` > 0 | **Warning** | Deadlock detected. Check concurrent write patterns (ingest + compaction). |
| `pg_stat_activity_max_duration_seconds` > 60 | **Warning** | Query running > 60s. Likely stuck. Check for missing index or lock wait. |
| `pg_stat_activity_count{state="idle in transaction"}` > 5 for 10min | **Warning** | Idle-in-transaction connections holding locks. Connection pool leak or application bug. |
| `pg_wal_size_bytes` > 10GB | **Warning** | WAL accumulation. Check archiving, replication, or `wal_keep_size` setting. |
| `PG1 / PG4` > 0.8 | **Critical** | Connection pool near max. Add PgBouncer or increase `max_connections`. |
---
## 10. System Health (Infrastructure Summary)
### Metrics
| ID | Metric | Type | Unit | Why It Matters |
|----|--------|------|------|----------------|
| **H1** | `health_check_status` | Gauge | 0/1 | `/health` endpoint. Basic liveness. |
| **H2** | `pgvector_connection_pool_active` | Gauge | connections | Active DB connections. Near max = pool exhaustion risk. |
| **H3** | `pgvector_connection_pool_idle` | Gauge | connections | Idle connections. Zero idle + high active = under-provisioned. |
| **H4** | `opensearch_cluster_status` | Gauge | 0/1/2 | 0=red, 1=yellow, 2=green. Yellow = replica missing. Red = data loss risk. |
| **H5** | `embedding_model_loaded` | Gauge | 0/1 | Model health check. 0 = all ingest and query embeds will fail. |
| **H6** | `rate_limit_rejections_total` | Counter | requests | Rate limit hits. High = legitimate traffic being blocked, or DDoS. |
| **H7** | `auth_failures_total` | Counter (labeled) | requests | Label: `reason=expired\|invalid\|missing`. Pattern reveals attack or misconfiguration. |
---
## 11. Dashboard Layout
### Grafana Rows (top to bottom)
```
Row 1: SYSTEM HEALTH
┌──────────────┬──────────────┬──────────────┬──────────────┐
│ Health: UP │ PG Pool: │ OpenSearch: │ Embed Model │
│ (H1) │ 12/20 active│ GREEN │ LOADED │
└──────────────┴──────────────┴──────────────┴──────────────┘
Row 2: WRITE PATH (Ingest)
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Ingest Rate (I1) │ Write Latency p50/p99 │ Dedup Hit Ratio │
│ [line chart, 24h] │ (I5) [line chart, 24h] │ (I3/I1) [line, 24h] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ OpenSearch Failures (I8)│ Contradiction Queue (I12)│ Chunks Written (I10) │
│ [counter, 24h] │ [gauge, current depth] │ [counter, 24h] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 3: READ PATH (Query)
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Query Rate (Q1) │ Query Latency p50/p99 │ Empty Results (Q11) │
│ [line chart, 24h] │ (Q9) [line chart, 24h] │ [%, 24h] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Latency Breakdown │ Intent Distribution │ Score Distribution │
│ sem/lex/rrf/rerank │ (Q3) [pie chart] │ (Q12) [histogram] │
│ [stacked area, 24h] │ │ │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 4: CONTEXT (3-Tier)
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Tier Hit Distribution │ Context Latency p50/p99 │ Budget Usage │
│ (C7) [stacked bar, 7d] │ (C6) [line chart, 24h] │ (C5) [histogram] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Tier-1 Hit Rate │ Degraded Responses (C8) │ Dropped Results (C5a) │
│ (C2/C1) [gauge, target │ [counter, 24h] │ [counter, 24h] │
│ >= 0.80] │ │ │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 5: RELEVANCE (Qwen-7B Judge) ← MOST IMPORTANT ROW
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ NDCG@10 (R5) │ MRR (R6) │ Precision/Recall │
│ [line chart, 30d │ [line chart, 30d │ (R7, R8) [line, 30d │
│ rolling avg, target │ rolling avg] │ rolling avg] │
│ >= 0.85] │ │ │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Relevance Score Dist │ Judge Cost (R3a) │ Judge Agreement (R9) │
│ (R4) [bar: 0/1/2, 7d] │ [counter, daily USD] │ [gauge, weekly] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 6: POD RESOURCES
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Memory Usage (P1) │ CPU Usage (P5 rate) │ Pod Restarts (P11) │
│ [line, 24h, limit line] │ [line, 24h, limit line] │ [counter, 7d] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Memory Pressure (P1/P4) │ CPU Throttle (P6 rate) │ Network I/O (P12, P13) │
│ [gauge, target < 0.85] │ [line, 24h] │ [line, 24h] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 7: AVAILABILITY
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Availability (A5) │ Error Rate (A3/A1) │ Dependencies (A8) │
│ [gauge, target >= 99.9%]│ [line, 24h] │ [status grid: pg/os/emb]│
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Quality Avail (A6) │ Degraded Responses (A9) │ 4xx Breakdown (A4) │
│ [gauge, target >= 95%] │ [counter, 24h] │ [stacked bar, 24h] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 8: INGEST RATE PATTERNS
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Write Rate/Min (IR1) │ Write Rate/Hour (IR2) │ Queue Depth (IR7) │
│ [line, 24h, burst high] │ [line, 7d] │ [gauge, target < 1000] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Rate by Project (IR3) │ Rate by Level (IR4) │ Queue Age p99 (IR8) │
│ [stacked area, 24h] │ [stacked area, 24h] │ [line, 24h] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 9: POSTGRES INTERNALS
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Cache Hit Ratio (PG32) │ Dead Tuple Ratio (PG12) │ Connections (PG1) │
│ [gauge, target >= 99%] │ [gauge, target < 20%] │ [stacked bar by state] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Seq vs Idx Scans (PG8/9)│ Replication Lag (PG23) │ Table Sizes (PG15) │
│ [line, 7d] │ [line, 24h, target < 10s]│ [bar chart, current] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Slowest Queries (PG5) │ WAL Size (PG24) │ Deadlocks (PG27) │
│ [table, top 5] │ [line, 7d] │ [counter, 30d] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
Row 10: STORAGE
┌──────────────────────────┬──────────────────────────┬──────────────────────────┐
│ Total Vectors (W1) │ Storage Bytes │ Write Rate/Hour (W6) │
│ [gauge, current] │ (W2+W4) [line, 30d] │ [line chart, 24h] │
├──────────────────────────┼──────────────────────────┼──────────────────────────┤
│ Parity Drift (W5) │ Level Distribution (W8) │ Compaction Freed (W12) │
│ [gauge, target = 0] │ [stacked bar] │ [counter per run] │
└──────────────────────────┴──────────────────────────┴──────────────────────────┘
```
---
## 12. Implementation: Prometheus Metrics in Rust
```rust
use prometheus::{
register_counter, register_counter_vec, register_gauge, register_gauge_vec,
register_histogram, register_histogram_vec,
Counter, CounterVec, Gauge, GaugeVec, Histogram, HistogramVec,
};
use lazy_static::lazy_static;
lazy_static! {
// === INGEST ===
pub static ref INGEST_REQUESTS: Counter =
register_counter!("memory_ingest_requests_total", "Total ingest requests").unwrap();
pub static ref INGEST_CHUNKS: Counter =
register_counter!("memory_ingest_chunks_total", "Total chunks written").unwrap();
pub static ref INGEST_BYTES: Counter =
register_counter!("memory_ingest_bytes_total", "Total bytes ingested").unwrap();
pub static ref INGEST_DEDUP_HITS: Counter =
register_counter!("memory_ingest_dedup_hits_total", "Deduplicated chunks skipped").unwrap();
pub static ref INGEST_CONTRADICTIONS: Counter =
register_counter!("memory_ingest_contradictions_total", "Contradictions detected").unwrap();
pub static ref INGEST_EMBED_DURATION: Histogram =
register_histogram!("memory_ingest_embed_seconds", "Embedding latency per chunk",
vec![0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0]).unwrap();
pub static ref INGEST_PGVECTOR_DURATION: Histogram =
register_histogram!("memory_ingest_pgvector_seconds", "pgvector write latency",
vec![0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0]).unwrap();
pub static ref INGEST_OPENSEARCH_DURATION: Histogram =
register_histogram!("memory_ingest_opensearch_seconds", "OpenSearch index latency",
vec![0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0]).unwrap();
pub static ref INGEST_OPENSEARCH_FAILURES: Counter =
register_counter!("memory_ingest_opensearch_failures_total", "OpenSearch write failures").unwrap();
pub static ref REVIEW_QUEUE_DEPTH: Gauge =
register_gauge!("memory_review_queue_depth", "Pending contradiction reviews").unwrap();
// === QUERY ===
pub static ref QUERY_REQUESTS: Counter =
register_counter!("memory_query_requests_total", "Total query requests").unwrap();
pub static ref QUERY_EMPTY_RESULTS: Counter =
register_counter!("memory_query_empty_results_total", "Queries returning zero results").unwrap();
pub static ref QUERY_TOTAL_DURATION: Histogram =
register_histogram!("memory_query_total_seconds", "End-to-end query latency",
vec![0.05, 0.1, 0.2, 0.5, 1.0, 2.0, 5.0]).unwrap();
pub static ref QUERY_SEMANTIC_DURATION: Histogram =
register_histogram!("memory_query_semantic_seconds", "pgvector search latency",
vec![0.01, 0.025, 0.05, 0.1, 0.2, 0.5]).unwrap();
pub static ref QUERY_LEXICAL_DURATION: Histogram =
register_histogram!("memory_query_lexical_seconds", "OpenSearch BM25 latency",
vec![0.01, 0.025, 0.05, 0.1, 0.2, 0.5]).unwrap();
pub static ref QUERY_INTENT: CounterVec =
register_counter_vec!("memory_query_intent_total", "Query intent classification",
&["intent"]).unwrap();
pub static ref QUERY_RESULTS_COUNT: Histogram =
register_histogram!("memory_query_results_count", "Results returned per query",
vec![0.0, 1.0, 3.0, 5.0, 10.0, 20.0, 50.0]).unwrap();
pub static ref QUERY_TOP1_SCORE: Histogram =
register_histogram!("memory_query_top1_score", "Top-1 result similarity score",
vec![0.3, 0.5, 0.6, 0.7, 0.8, 0.9, 0.95, 1.0]).unwrap();
// === CONTEXT ===
pub static ref CONTEXT_REQUESTS: Counter =
register_counter!("memory_context_requests_total", "Total context lookups").unwrap();
pub static ref CONTEXT_TIER_HITS: CounterVec =
register_counter_vec!("memory_context_tier_hits_total", "Hits per tier",
&["tier"]).unwrap();
pub static ref CONTEXT_TOTAL_DURATION: Histogram =
register_histogram!("memory_context_total_seconds", "End-to-end context latency",
vec![0.1, 0.25, 0.5, 1.0, 2.0, 5.0]).unwrap();
pub static ref CONTEXT_DROPPED: Counter =
register_counter!("memory_context_dropped_results_total", "Results dropped for budget").unwrap();
// === RELEVANCE (updated daily by batch job) ===
pub static ref RELEVANCE_NDCG: Gauge =
register_gauge!("memory_relevance_ndcg_10", "NDCG@10 from Qwen-7B judge").unwrap();
pub static ref RELEVANCE_MRR: Gauge =
register_gauge!("memory_relevance_mrr", "Mean Reciprocal Rank").unwrap();
pub static ref RELEVANCE_PRECISION: Gauge =
register_gauge!("memory_relevance_precision_10", "Precision@10").unwrap();
pub static ref RELEVANCE_RECALL: Gauge =
register_gauge!("memory_relevance_recall_10", "Recall@10").unwrap();
// === STORAGE ===
pub static ref STORAGE_VECTORS: Gauge =
register_gauge!("memory_storage_vectors_total", "Total vectors in pgvector").unwrap();
pub static ref STORAGE_BYTES: Gauge =
register_gauge!("memory_storage_bytes", "Total storage bytes (pg + os)").unwrap();
pub static ref STORAGE_PARITY_DRIFT: Gauge =
register_gauge!("memory_storage_parity_drift", "pgvector vs OpenSearch doc count difference").unwrap();
pub static ref WRITE_RATE: Gauge =
register_gauge!("memory_write_rate_per_hour", "Current write rate (chunks/hour)").unwrap();
}
```
### Instrumentation Example (Ingest Handler)
```rust
pub async fn ingest_handler(req: HttpRequest, body: web::Json<IngestRequest>, state: web::Data<AppState>) -> HttpResponse {
INGEST_REQUESTS.inc();
// Auth
let auth_timer = INGEST_AUTH_DURATION.start_timer();
let (claims, token) = match validate_auth(&req, &state).await { ... };
auth_timer.observe_duration();
// Dedup
if state.idempotency_store.is_duplicate(&body.idempotency_key) {
INGEST_DEDUP_HITS.inc();
return HttpResponse::Ok().json(json!({"status": "duplicate"}));
}
// Embed
let embed_timer = INGEST_EMBED_DURATION.start_timer();
let embedding = state.embeddings.embed_one(&body.text).await?;
embed_timer.observe_duration();
// pgvector write
let pg_timer = INGEST_PGVECTOR_DURATION.start_timer();
state.vector_store.insert(&body.project, &body.text, &embedding).await?;
pg_timer.observe_duration();
// OpenSearch write
let os_timer = INGEST_OPENSEARCH_DURATION.start_timer();
match state.opensearch_client.index_document(...).await {
Ok(_) => {},
Err(e) => {
INGEST_OPENSEARCH_FAILURES.inc();
tracing::warn!("OpenSearch write failed (non-blocking): {}", e);
}
}
os_timer.observe_duration();
INGEST_CHUNKS.inc();
INGEST_BYTES.inc_by(body.text.len() as f64);
HttpResponse::Created().json(...)
}
```
---
## 13. Relevance Evaluation CronJob
```yaml
apiVersion: batch/v1
kind: CronJob
metadata:
name: memory-relevance-eval
namespace: poimen
spec:
schedule: "0 3 * * *" # Daily at 03:00 UTC
jobTemplate:
spec:
template:
spec:
containers:
- name: relevance-eval
image: forgejo.riotpiao.com/rock/poimen-memory:latest
command: ["mem", "evaluate-relevance"]
env:
- name: EVAL_SAMPLE_SIZE
value: "500"
- name: EVAL_JUDGE_MODEL
value: "qwen2.5-7b"
- name: EVAL_JUDGE_ENDPOINT
value: "http://ollama.poimen.svc:11434/api/generate"
- name: EVAL_QUERY_LOG_HOURS
value: "24"
- name: DATABASE_URL
valueFrom:
secretKeyRef:
name: memory-db-credentials
key: url
resources:
requests:
cpu: "500m"
memory: "512Mi"
limits:
cpu: "1"
memory: "1Gi"
restartPolicy: OnFailure
```
---
## 14. SLOs Summary
| Signal | Target | Window | Consequence of Miss |
|--------|--------|--------|---------------------|
| Ingest p99 latency | < 500ms | 24h rolling | Backpressure on upstream systems |
| Query p99 latency | < 500ms | 24h rolling | User-perceived slowness |
| Context p99 latency | < 2s | 24h rolling | Agent timeout, degraded assistance |
| Query empty rate | < 20% | 24h rolling | Users get no answer, lose trust |
| Tier-1 hit rate | >= 80% | 7d rolling | System not learning from failures |
| NDCG@10 | >= 0.85 | 7d rolling | Retrieval quality degraded, hallucination risk |
| MRR | >= 0.80 | 7d rolling | Relevant results buried in ranking |
| Storage parity drift | = 0 | 1h | Dual-write inconsistency, partial search |
| Review queue depth | < 100 | 24h | Unreviewed contradictions leaking through |
| Relevance judge agreement | >= 85% | Weekly | Automated evaluation unreliable |
| Raw availability | >= 99.9% | 24h rolling | Service down, LLM falls back to parametric knowledge |
| Quality availability | >= 95% | 24h rolling | Queries succeeding but returning nothing useful |
| Pod memory pressure | < 85% of limit | 5min | OOMKill imminent, in-flight requests lost |
| CPU throttle ratio | < 25% | 5min | Latency degradation across all endpoints |
| PG cache hit ratio | >= 99% | 1h | Disk thrashing, query latency spikes |
| PG dead tuple ratio | < 20% | 24h | Table bloat, slower scans, wasted disk |
| PG replication lag | < 10s | 5min | Stale reads from replica |
| Ingest rate (zero) | > 0 during business hours | 30min | Silent upstream failure, knowledge going stale |
| Ingest queue depth | < 1000 | 15min | Workers can't keep up, processing lag |
-28
View File
@@ -1,28 +0,0 @@
{
"results": [
{
"id": "chunk-abc123",
"level": "L1",
"score": 0.95,
"text": "Kubernetes uses port 8080 for API server",
"source": "transcript://session-001"
},
{
"id": "chunk-def456",
"level": "L2",
"score": 0.87,
"text": "Common debugging pattern for CrashLoopBackOff pods",
"source": "transcript://session-002"
},
{
"id": "chunk-ghi789",
"level": "R",
"score": 0.72,
"text": "See kubectl troubleshooting guide section 3.2",
"source": "obsidian://poimen-vault/kubectl.md"
}
],
"total_hits": 127,
"search_time_ms": 145,
"query": "fix kubernetes port conflict"
}
-33
View File
@@ -1,33 +0,0 @@
# Kubernetes Troubleshooting Guide
## Port Conflicts
When a port conflict occurs on port 8080, check for existing services:
```bash
kubectl get svc --all-namespaces | grep 8080
```
### Common Causes
1. Multiple services binding to same NodePort
2. Host network pods conflicting with node services
3. Ingress controller port overlap
## CrashLoopBackOff
Pods enter CrashLoopBackOff when the container exits repeatedly.
### Diagnosis Steps
1. Check pod logs: `kubectl logs <pod> --previous`
2. Check events: `kubectl describe pod <pod>`
3. Check resource limits: memory/CPU constraints
4. Check liveness probes: incorrect health check paths
### Resolution
- Increase memory limits if OOMKilled
- Fix application startup errors
- Adjust probe timing (initialDelaySeconds)
- Check environment variable configuration
-16
View File
@@ -1,16 +0,0 @@
2025-01-15T10:00:00Z INFO Starting service on port 8080
2025-01-15T10:00:01Z DEBUG Database connection pool initialized (max=20)
2025-01-15T10:00:02Z INFO Health check endpoint ready at /health
2025-01-15T10:00:05Z WARN High memory usage detected: 85% of 512Mi limit
2025-01-15T10:00:10Z ERROR Connection refused: temporal-frontend:7233
2025-01-15T10:00:15Z INFO Retry attempt 1/3 for temporal connection
2025-01-15T10:00:20Z INFO Connected to temporal-frontend.temporal.svc.cluster.local:7233
2025-01-15T10:00:25Z DEBUG Worker registered on task queue: poimen-taskqueue
2025-01-15T10:00:30Z INFO Processing ingest request: project=poimen source=transcript://session-001
2025-01-15T10:00:31Z DEBUG Entity extraction complete: 5 entities found
2025-01-15T10:00:32Z DEBUG Fact extraction complete: 3 facts found
2025-01-15T10:00:33Z INFO Contradiction check: 0 contradictions detected
2025-01-15T10:00:34Z INFO Ingest complete: chunk-abc123 (145ms)
2025-01-15T10:00:40Z WARN Slow query detected: 850ms for hybrid search
2025-01-15T10:00:45Z ERROR Pod OOMKilled: poimen-worker-abc123 (memory limit exceeded)
2025-01-15T10:00:50Z INFO Pod restarted: poimen-worker-abc123 (restart count: 1)
-4
View File
@@ -1,4 +0,0 @@
creation_rules:
- path_regex: .*\.enc\.ya?ml$
encrypted_regex: '^(stringData|data)$'
age: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
-157
View File
@@ -1,157 +0,0 @@
# Poimen Memory - Environment Configuration Guide
All downstream service URIs are read from environment variables, sourced from ConfigMap.
## How It Works
1. **ConfigMap provides URIs**: `k8s/app/config.yaml` (production, SOPS-encrypted)
2. **Deployment injects via envFrom**: `envFrom: configMapRef: poimen-memory-config`
3. **Application reads from ENV**: Code parses `LLM_ENDPOINT`, `OPENSEARCH_HOST`, `AUTHENTIK_ISSUER`, etc.
```yaml
# deployment.yaml
envFrom:
- configMapRef:
name: poimen-memory-config # All vars injected as ENV
```
## Environment Variables
### LLM Service (Entity & Fact Extraction)
- `LLM_ENDPOINT` — full URL to chat/completions endpoint
- `LLM_API_BASE` — base API URL (used for client initialization)
- `LLM_MODEL` — model identifier (ornith:35b, qwen:7b, etc.)
- `LLM_TIMEOUT_SECS` — timeout for LLM requests
- `ENABLE_LLM_EXTRACTION` — enable/disable LLM extraction (true/false)
### OpenSearch (Vector Store, BM25)
- `OPENSEARCH_HOST` — hostname:port
- `OPENSEARCH_SCHEME` — http or https
- `OPENSEARCH_VERIFY_CERTS` — SSL certificate verification (true/false)
### Authentik (OIDC)
- `AUTHENTIK_ISSUER` — OIDC issuer URL
- `AUTHENTIK_VERIFY_SSL` — SSL certificate verification (true/false)
- `MEM_AUTH_MODE` — auth mode: jwt | apikey | none
### Temporal (Workflow Orchestration - Future)
- `TEMPORAL_ENDPOINT` — temporal frontend hostname:port
- `TEMPORAL_NAMESPACE` — temporal namespace
### API Gateway (Route Optimization - Future)
- `GATEWAY_URL` — gateway base URL
### Memory Service Config
- `MEM_AUTH_MODE` — jwt | apikey | none
- `MEM_RATE_LIMIT_INGEST` — ingest requests per second
- `MEM_RATE_LIMIT_QUERY` — query requests per second
- `MEM_EMBEDDING_BATCH_SIZE` — batch size for embeddings
---
## Deployment Scenarios
### Production (SOPS-Encrypted ConfigMap)
**File**: `k8s/app/config.yaml`
Services use cluster-internal DNS:
```yaml
LLM_ENDPOINT: http://reasoning-predictor.llm-serving.svc.cluster.local:8000/v1/chat/completions
OPENSEARCH_HOST: opensearch.poimen.svc.cluster.local:9200
AUTHENTIK_ISSUER: https://authentik.auth.svc.cluster.local:9443/application/o/poimen/
TEMPORAL_ENDPOINT: temporal-frontend.temporal.svc.cluster.local:7233
GATEWAY_URL: http://api-gw.poimen.svc.cluster.local:8080
MEM_AUTH_MODE: jwt
```
**Deploy**:
```bash
# SOPS auto-decrypts based on .sops.yaml age key
kubectl apply -f k8s/app/config.yaml -k k8s/app/
```
### Local/Development (Plaintext ConfigMap)
**File**: `k8s/app/config.local.yaml`
Services via external URLs (ingress):
```yaml
LLM_ENDPOINT: https://api.riotpiao.com/v1/chat/completions
OPENSEARCH_HOST: opensearch.riotpiao.com:443
AUTHENTIK_ISSUER: https://authentik.riotpiao.com/application/o/poimen/
TEMPORAL_ENDPOINT: temporal.riotpiao.com:443
GATEWAY_URL: https://api.riotpiao.com
MEM_AUTH_MODE: none
```
**Deploy** (override production config):
```bash
# Delete prod config, apply local
kubectl delete configmap poimen-memory-config -n poimen
kubectl apply -f k8s/app/config.local.yaml
```
---
## Encrypting with SOPS
Production `config.yaml` is encrypted with SOPS (Age-based).
**Encrypt**:
```bash
sops -e k8s/app/config.yaml > k8s/app/config.yaml.enc
mv k8s/app/config.yaml.enc k8s/app/config.yaml
```
**Decrypt for editing** (SOPS auto-handles with $EDITOR):
```bash
sops k8s/app/config.yaml
```
**View decrypted** (without editing):
```bash
sops -d k8s/app/config.yaml
```
**.sops.yaml** defines encryption key:
```yaml
creation_rules:
- path_regex: k8s/app/config.yaml
key_groups:
- age:
- <age-public-key>
```
---
## Application Code Pattern
Example: Application should read URIs from ENV at startup.
```rust
// Pseudocode
let llm_endpoint = env::var("LLM_ENDPOINT")
.unwrap_or("http://localhost:11434/v1/chat/completions".to_string());
let opensearch_host = env::var("OPENSEARCH_HOST")
.unwrap_or("localhost:9200".to_string());
let auth_mode = env::var("MEM_AUTH_MODE")
.unwrap_or("none".to_string());
// Initialize clients with these URIs
let llm_client = LlmClient::new(llm_endpoint)?;
let search_client = OpenSearchClient::new(opensearch_host)?;
```
---
## Summary
| Aspect | Production | Local |
|--------|-----------|-------|
| **Config File** | `config.yaml` | `config.local.yaml` |
| **Encryption** | SOPS (Age) | Plaintext |
| **Service URIs** | Cluster-internal DNS | External HTTPS |
| **Auth Mode** | JWT (Authentik) | None (disabled) |
| **Rate Limits** | 100/1000 | 1000/10000 |
| **Deploy** | `kubectl apply -k k8s/app/` | `kubectl apply -f config.local.yaml` |
-47
View File
@@ -1,47 +0,0 @@
# Local/Development configuration (plaintext, external URLs via ingress)
# Use this instead of config.yaml for local testing
# kubectl apply -f config.local.yaml
apiVersion: v1
kind: ConfigMap
metadata:
name: poimen-memory-config
namespace: poimen
labels:
app.kubernetes.io/name: poimen-memory
app.kubernetes.io/component: config
data:
# Auth mode: jwt | apikey | none (disabled for local testing)
MEM_AUTH_MODE: "none"
# Rate limiting (higher for testing)
MEM_RATE_LIMIT_INGEST: "1000"
MEM_RATE_LIMIT_QUERY: "10000"
MEM_IDEMPOTENCY_TTL_SECS: "86400"
# Embeddings
MEM_EMBEDDING_BATCH_SIZE: "32"
# Downstream services - external URLs via ingress
# LLM Service (via api.riotpiao.com ingress)
LLM_ENDPOINT: "https://api.riotpiao.com/v1/chat/completions"
LLM_API_BASE: "https://api.riotpiao.com/v1"
LLM_MODEL: "qwen:7b"
LLM_TIMEOUT_SECS: "60"
ENABLE_LLM_EXTRACTION: "true"
# OpenSearch (via ingress)
OPENSEARCH_HOST: "opensearch.riotpiao.com:443"
OPENSEARCH_SCHEME: "https"
OPENSEARCH_VERIFY_CERTS: "true"
# Authentik (via ingress - optional for local)
AUTHENTIK_ISSUER: "https://authentik.riotpiao.com/application/o/poimen/"
AUTHENTIK_VERIFY_SSL: "true"
# Temporal (via ingress)
TEMPORAL_ENDPOINT: "temporal.riotpiao.com:443"
TEMPORAL_NAMESPACE: "poimen"
# API Gateway (via ingress)
GATEWAY_URL: "https://api.riotpiao.com"
+7 -33
View File
@@ -1,7 +1,5 @@
# Production environment configuration for poimen-memory
# All services use cluster-internal DNS names
# This file is encrypted with SOPS in production
# For local dev, use plaintext version with external URLs
# Non-sensitive environment variables for poimen-memory
# Change these without redeploying secrets.
apiVersion: v1
kind: ConfigMap
metadata:
@@ -11,39 +9,15 @@ metadata:
app.kubernetes.io/name: poimen-memory
app.kubernetes.io/component: config
data:
# Auth mode: jwt | apikey | none
MEM_AUTH_MODE: "jwt"
# Auth mode: jwt | apikey
MEM_AUTH_MODE: "none"
# Rate limiting
MEM_RATE_LIMIT_INGEST: "100"
MEM_RATE_LIMIT_QUERY: "1000"
MEM_IDEMPOTENCY_TTL_SECS: "86400"
# Embeddings
MEM_EMBEDDING_BATCH_SIZE: "32"
# Downstream services - read by application from ENV
# Internal cluster DNS (prod) / external URLs (local)
# LLM Service (entity extraction, fact extraction)
LLM_ENDPOINT: "http://reasoning-predictor.llm-serving.svc.cluster.local:8000/v1/chat/completions"
LLM_API_BASE: "http://reasoning-predictor.llm-serving.svc.cluster.local:8000/v1"
LLM_MODEL: "ornith:35b"
LLM_TIMEOUT_SECS: "30"
ENABLE_LLM_EXTRACTION: "true"
# OpenSearch (vector store, BM25 retrieval)
# OpenSearch
OPENSEARCH_HOST: "opensearch.poimen.svc.cluster.local:9200"
OPENSEARCH_SCHEME: "http"
OPENSEARCH_VERIFY_CERTS: "false"
# Authentik (OIDC provider)
AUTHENTIK_ISSUER: "https://authentik.auth.svc.cluster.local:9443/application/o/poimen/"
AUTHENTIK_VERIFY_SSL: "false"
# Temporal (workflow orchestration - future)
TEMPORAL_ENDPOINT: "temporal-frontend.temporal.svc.cluster.local:7233"
TEMPORAL_NAMESPACE: "poimen"
# API Gateway (external queue, route optimization - future)
GATEWAY_URL: "http://api-gw.poimen.svc.cluster.local:8080"
# Obsidian
OBSIDIAN_URL: "http://obsidian-server.poimen.svc.cluster.local:8080"
+8 -29
View File
@@ -1,6 +1,6 @@
# Poimen Memory API Server
# Serves HTTP endpoints for memory ingest, query, visualization.
# Connects to memory-db (pgvector) + api.riotpiao.com (LLM via Authentik JWT).
# Serves 7 HTTP endpoints for memory ingest, query, and management.
# Connects to memory-db (pgvector) for persistent storage.
apiVersion: apps/v1
kind: Deployment
metadata:
@@ -29,7 +29,7 @@ spec:
type: RuntimeDefault
containers:
- name: memory
image: forgejo.riotpiao.com/riotpiao-poimen/poimen-memory:latest
image: forgejo.riotpiao.com/rock/poimen-memory:latest
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
@@ -60,44 +60,22 @@ spec:
key: password
- name: DATABASE_URL
value: "postgresql://$(DATABASE_USER):$(DATABASE_PASSWORD)@$(DATABASE_HOST):$(DATABASE_PORT)/$(DATABASE_NAME)?sslmode=disable"
# All downstream service URIs read from ConfigMap
# (LLM_ENDPOINT, LLM_API_BASE, LLM_MODEL, OPENSEARCH_HOST, etc.)
# These are injected via envFrom below
# Authentik service account (memory-agent-oidc secret)
# Only needed if MEM_AUTH_MODE=jwt in ConfigMap
- name: AUTHENTIK_CLIENT_ID
valueFrom:
secretKeyRef:
name: memory-agent-oidc
key: CLIENT_ID
- name: AUTHENTIK_CLIENT_SECRET
valueFrom:
secretKeyRef:
name: memory-agent-oidc
key: CLIENT_SECRET
- name: TOKEN_URL
valueFrom:
secretKeyRef:
name: memory-agent-oidc
key: TOKEN_URL
# Server config
# LLM Gateway API key
- name: MEM_API_KEY
valueFrom:
secretKeyRef:
name: poimen-memory-secrets
key: llm-api-key
# Server config (from ConfigMap)
- name: MEM_PORT
value: "8080"
- name: MEM_HOME
value: "/tmp"
envFrom:
# ConfigMap with all service URIs (prod: encrypted, local: plaintext)
- configMapRef:
name: poimen-memory-config
command: ["/app/mem"]
- secretRef:
name: poimen-memory-auth
args:
- serve
- --port
@@ -130,6 +108,7 @@ spec:
- name: tmp
emptyDir:
sizeLimit: 64Mi
# Tolerate control-plane nodes
tolerations:
- key: node-role.kubernetes.io/control-plane
operator: Exists
+5 -3
View File
@@ -1,11 +1,13 @@
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
namespace: poimen
resources:
# vault-pvc.yaml removed — memory service uses pgvector, not local storage
- deployment.yaml
- service.yaml
- config.yaml # Production config (SOPS-encrypted)
- config.yaml
- obsidian.yaml
# Legacy secret managed separately
# - secrets.yaml
generators:
- secret-generator.yaml
-30
View File
@@ -1,30 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: memory-agent-auth
namespace: poimen
type: Opaque
stringData:
CLIENT_ID: ENC[AES256_GCM,data:qeGqSfcR8mUIkQRd4A==,iv:JZx+tR3Z9Mm8KLJqE8CfGZfZ0q+PdJKJLGT5bOKLjno=,tag:mYpvF5X/+cHXdm8vxqr5dA==,type:str]
CLIENT_SECRET: ENC[AES256_GCM,data:8nC3VGevMxzHb0EQeZZ+qEjppZrpNH8lrVBrYTJvVCEqb1gK6lKr4w==,iv:gZUVn+x7K3qHXYYxRTJQzVEZoQkkM1D6yPjFXCK0AWo=,tag:4I8p6LYc1vvNHxUmvVLQCQ==,type:str]
TOKEN_URL: ENC[AES256_GCM,data:jlpHwzKNFKlpMJTfPGCDgQS3kWb4pqAVEGJccJzq+xCPXo0=,iv:kX4D6ydoL6V/5V5g1Kzb8PwBZGKvQJZHmxQPRAcVLdo=,tag:sKGCwL3XcWG9sKZqKlEi8g==,type:str]
AUTHENTIK_ISSUER: ENC[AES256_GCM,data:ILkQfNBfZ/7L6s7Oy6dE/xRpB91qP23fKHC2Q0IzDxA=,iv:J+2mPqfKmfVaXI0L5cC3j8KhXjcPTMiZaZYvZZEhKFA=,tag:nG3g7FJ4jRgPXhCDzCzXEQ==,type:str]
AUTHENTIK_AUDIENCE: ENC[AES256_GCM,data:4d8nw7Mf2Yg=,iv:eEKzB8d6fC1Z+6JMRZ/tWPcQfbOGLFvLwP1B68n3OqI=,tag:WTKJmgXp8WEqJLfz7AQfIw==,type:str]
sops:
kms: []
gcp_kms: []
azure_kv: []
hc_vault: []
age:
- recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBHSmt6SzRWSENWdFdN
L1o4TDZGdlY1UzVZVld1SXU1eHR3RENKSUZGYXdVCnlQQUwwUkxQa1pQRjhE
TDM1b2pxRjE5WmRKd3oxZGpkdm1FVkxRaXcKLT4gAhagIFqyQ1hpIVg6
-----END AGE ENCRYPTED FILE-----
lastmodified: "2025-01-30T15:43:00Z"
mac: ENC[AES256_GCM,data:REDACTED,iv:REDACTED,tag:REDACTED,type:str]
pgp: []
unencrypted_suffix: _unencrypted
version: 3.8.1
+24
View File
@@ -0,0 +1,24 @@
apiVersion: ENC[AES256_GCM,data:gSI=,iv:nfXxHTEXSY6eDPOLfQWxQaX/Ge7s08QF6GqQ847cdKg=,tag:szUjeOolHuomGQQdrV7U4A==,type:str]
kind: ENC[AES256_GCM,data:HC8zcR8G,iv:wk4XliU5bPi32M0QV6OhJs3tSkirOczWJjR+1MgjxpM=,tag:jcD8R327PRv8x7wYh4Tdrg==,type:str]
metadata:
name: ENC[AES256_GCM,data:HouUGg1P3iPycnr5doLc9w==,iv:kzODDxNBix4e/kAGrF8io165crqPHewyuG8MCZhr3mM=,tag:hX9peWSY5LwM7/08S+QLuw==,type:str]
namespace: ENC[AES256_GCM,data:OCIDOqNz,iv:GhtxD5cXXTnl/7Po1rY3I+jacI9Kz4bXp+Nz2UVTOTE=,tag:F7TuNLkSluKm4TZZ1Q33VQ==,type:str]
type: ENC[AES256_GCM,data:ErqH5L3k,iv:JioZqat2ZYSO83vEnl1MY6YiCC3RttfEkGc2OumJHBY=,tag:K68wmCX8sMzcnGUz1aBpWA==,type:str]
stringData:
id_ed25519: ENC[AES256_GCM,data:YqrAUmZCvEDC2q8c4Ns+WxW2l+oC31arj1MPYwt6AkIv026kZk9ucywkOWT/Ww0VgiUWF/0t/gzkm7K5D5fSk5vP6PyWV5Zqvlo8Zqy1VBLc3V9gbQeCsA4i8+FO3zZ0k2l3HAefEqhJ4Bnj3dKDOTX4bsbE/H/4n8WCojYnOLdU4esqm5r4bOCnFv5wBkbkob6AwfqekdaBZfuPOqW2sstDSRN6km1UZafCYuMY0XQTxCKYM8Izt3px8sfBq37oA6syDpxNuEpAk0uYumHhSBHKRAnsDY61pjfCbR2xy/3IeEPr6EVf5HwV+2ElhtSB/Zfzin5GZvAgX3gu3HuFznHUYH5olLEEFvVtw8fWLh3avwwCsAlUsDKZoTV1t1bzGni1OPYK3ZfsAmQqI1lvdFlRj++e6L3vDBkVG5qowTelbSb6/TWMDpJw/CsX3bgeKEoFUt2vi9IxwdYO/onuExVrT23WeanoSmrnXRaBqr6xIV5yW5CCbBBRmRU7a6jkwtkhe8dHFTKejaqjBpzPdlZOvhzKBHlOy4eDGjeV7CkzrRw=,iv:bSGkeMli13DSDFAu1+4Kg5sqSJ8LbdpLfN5oIwzLyTM=,tag:9DYK17ET7rkfmmpwxjicog==,type:str]
known_hosts: ENC[AES256_GCM,data:FaWsLxkot5Zxh7mobbUGDFqKLOdmP5APg09nkDyqGHyDp0v0Pjc19jeizM+tq3b3aH59YGlPe/4xQf7ZXZRZjWQE+m/TcVujtLD4hfCc40wqZh1XtTtdC6Tf1p3JbqxP0uQVy1+EFVwPCimUsZfp7gtcT/Hmu1JAW9biGmvACO9+dDeHGzBH5ZRCw3+dEYmcLgGIRgzpOwJLfHr/hkvdlflhzmEHMliBIl+TpqQ38GFQmw0ia7UEJzj3ghoDj7HjrHqlBa7aBHJpaEBYVwq8cN7JaLnyO1Y4+LIU8ln/CEzeg9wxVJoMO8IBcQCCgXoC+ogEpNFVb+pdUfRl/3Ye97ZJFmdJoorvSHIR02e9n7E2G3Ox9iImwnwI76X3FokuY0zcGIkcIho6JN3/8k3Z4VLDl2qGflo7jK6QP1DEsGUGhGwPRVrpPNkMDxYQeBIqlwFzfRuF/gDZj3ZWadCYB7NwByVgTcZFqiMtQ74z6jYGMiPIpW2OCY4HGu9ecGPR02USEu39CjJUCWH9WbQZTjmK3n4yYy4X4WPMbc0IekSCC2ossBznMoFsu7q7L13arqC99j9ZOv8aJ7KgMpGpOVPoN2AURJTFhgMX8TD93AvbFNVNqA0t+Y+g0Hq5f/py8XPzj4b6l8A4QxK3Awj4gf5BbK2lNjL+Cgo2kgwPsPvNd4hvx3gbamDOPNf+lH6iGLF19QyoJPHlOD/k9hkcMRerAEOLZriTylgpj2joXOaeKVrk8qkAZBunQKNc3e0xXH1i5obqz871DbbOVvrfYhmSNG93IEDG3hvNPbIc2uXYIJBciXxEaNO2PEypaLDgnNczEoVyUn2MXTJ6XMr/fkKvVBTmDuFd7BWk1JM6gWKnKvdN2G7bBHz5D43jGmmv+Z4bcd4Z4ViO8yAMbB3kKMLxMZQiSsrOudcQM1rzip2HBYb7JhK3yQwTbqTXZo64hejYW6+KYZysSB+A4itITvQR1G50lKndLd1XXqRbfV00DSZmNPRt0dKcTRkEX2RdCOkh6Wddo7D41V4eDRoHY+rctirsDKE/IA875ooAxpj5CWsHjAzlVrSUUVB2D0IBdlCPaZ5LVSo8i8+9S4rakEdmwe5dceUTC9mNIuzFtUeAshW4J2U4UzIQGOVt0BolwWfWiO5QvuAMC7aH1kJOr9gUB69IwDASdd+PbyEvjCQigVXEeObcliJu9QZohcoQU09fI1ajmPC5x2g2Rbo8DPakQTdsHuzdNm/zMZ/yUhi7+5/jZmwI3kRdgA9EMmhl+T2qAwNbhBiW2Iim44R0ZFuWMv6SCX2Kl9KxH+HW7dS8H7FJbAG7w9kiivpHP0sBTRfSgN9ddR6veVRiZWXNJ38sNy88GXdwzRM90XJAwQ==,iv:mv3hoMwPcEmOBbsIRoKLUuEsUolotv1VtikLiItwuJg=,tag:7amQiuiaVwowLAQcNoq72A==,type:str]
sops:
age:
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBINTF2OGRmUWoxTmFEZUdv
SWJKRlFhaVkzbUtOaTVnbHFjd1UzR2RFdVFNClpFODhlQVJIQjhlOEJBL2pDVmJa
UjlZZmFHKzA4eDJEMk9KTDMyOGZ2VWcKLS0tIFRHREo1dXNRKzJVYkROQSt1WjFV
ZXZYVjAwSlZhT0ZMbG1qNDVUWnJyQ2cKfs4t6HsQG5Wiyp6QvFqvm4+/o4NAL3qu
6L9vyhl2jufrbxmR+IsEBCxYS7rh6dCbxTUFap3MD2lYIGF9hRnGjQ==
-----END AGE ENCRYPTED FILE-----
recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
lastmodified: "2026-08-28T23:27:09Z"
mac: ENC[AES256_GCM,data:wJC6YCHXq6I/bqUjfwFRvpULZ1Yt39PoWFKzzOAq6h/pHsUrWEgrkm+3+dLaPpz663b0B75BiCQjQb4igXWr38O5I+FKonRHsbsH+D+pO+dq++yNYG8T30KGaquVfnsm8ijWGWxOY9nULUXfKcYfqvsR9P7KCV7bdcWuZ5xzZ5o=,iv:w5f3h0hb2ooeNYK1QZactpmpT8mAYa94V8FBewP0MUY=,tag:qlgUutFanhjk1HIBJqmLQg==,type:str]
unencrypted_suffix: _unencrypted
version: 3.13.2
+159
View File
@@ -0,0 +1,159 @@
---
# Obsidian server deployment
# Serves local vault with web UI and API
apiVersion: apps/v1
kind: Deployment
metadata:
name: obsidian-server
namespace: poimen
labels:
app.kubernetes.io/name: obsidian-server
app.kubernetes.io/part-of: poimen-memory
spec:
replicas: 1
selector:
matchLabels:
app.kubernetes.io/name: obsidian-server
template:
metadata:
labels:
app.kubernetes.io/name: obsidian-server
app.kubernetes.io/part-of: poimen-memory
spec:
serviceAccountName: obsidian-server
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
fsGroup: 1000
seccompProfile:
type: RuntimeDefault
initContainers:
- name: git-sync-init
image: alpine/git:latest
securityContext:
runAsNonRoot: false
runAsUser: 0
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
add:
- CHOWN
- DAC_OVERRIDE
command:
- sh
- -c
- |
export GIT_SSH_COMMAND="ssh -i /root/.ssh/id_ed25519 -o StrictHostKeyChecking=no"
git config --global --add safe.directory /vault
if [ -d /vault/.git ]; then
cd /vault && git pull origin main || true
else
# Clone into temp, move contents into vault
rm -rf /tmp/repo
git clone ssh://[email protected]:2222/rock/poimen-obesdient-memory.git /tmp/repo
cp -a /tmp/repo/. /vault/
rm -rf /tmp/repo
fi
chown -R 1000:1000 /vault
volumeMounts:
- name: vault
mountPath: /vault
- name: ssh-key
mountPath: /root/.ssh
readOnly: true
containers:
- name: obsidian-server
image: ppatlabs/obsidian:latest
imagePullPolicy: IfNotPresent
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
ports:
- name: http
containerPort: 27124
protocol: TCP
env:
- name: VAULT_NAME
value: poimen-vault
- name: VAULT_PATH
value: /vault
- name: REST_API_ENABLED
value: "true"
- name: REST_API_PORT
value: "8080"
volumeMounts:
- name: vault
mountPath: /vault
- name: config
mountPath: /config
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
cpu: 500m
memory: 512Mi
livenessProbe:
httpGet:
path: /
port: http
scheme: HTTPS
initialDelaySeconds: 30
periodSeconds: 10
timeoutSeconds: 5
readinessProbe:
httpGet:
path: /
port: http
scheme: HTTPS
initialDelaySeconds: 15
periodSeconds: 5
timeoutSeconds: 5
volumes:
- name: vault
persistentVolumeClaim:
claimName: obsidian-vault
- name: config
emptyDir: {}
- name: ssh-key
secret:
secretName: obsidian-git-ssh
defaultMode: 0400
# PVC managed by homelab repo (k8s/infra/databases/obsidian-vault-pvc.yaml)
---
# Service for Obsidian server
apiVersion: v1
kind: Service
metadata:
name: obsidian-server
namespace: poimen
labels:
app.kubernetes.io/name: obsidian-server
spec:
type: ClusterIP
ports:
- name: http
port: 80
targetPort: 27124
protocol: TCP
selector:
app.kubernetes.io/name: obsidian-server
---
# ServiceAccount for Obsidian
apiVersion: v1
kind: ServiceAccount
metadata:
name: obsidian-server
namespace: poimen
labels:
app.kubernetes.io/name: obsidian-server
# Ingress managed by homelab repo (obsidian.riotpiao.com)
# See: homelab/k8s/bootstrap/ingress/ingress.yaml
-24
View File
@@ -1,24 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: poimen-memory-auth
namespace: poimen
labels:
app.kubernetes.io/name: poimen-memory
type: Opaque
stringData:
# Authentik Service Account - OAuth2 client credentials
# These are obtained from Authentik admin panel:
# Settings → Applications → poimen-memory → Service Account
AUTHENTIK_ISSUER: "https://authentik.riotpiao.com/application/o/memory"
AUTHENTIK_AUDIENCE: "poimen-memory"
AUTHENTIK_CLIENT_ID: "${AUTHENTIK_SERVICE_ACCOUNT_CLIENT_ID}"
AUTHENTIK_CLIENT_SECRET: "${AUTHENTIK_SERVICE_ACCOUNT_SECRET}"
# LLM API Key
# Generated by Authentik service account with permissions to LLM gateway
LLM_API_KEY: "${LLM_API_KEY_FROM_AUTHENTIK}"
# S3 Credentials for backups (Velero)
S3_ACCESS_KEY: "${MINIO_ACCESS_KEY}"
S3_SECRET_KEY: "${MINIO_SECRET_KEY}"
-1
View File
@@ -6,4 +6,3 @@ kind: Kustomization
resources:
- memory-db.yaml
- opensearch.yaml
- opensearch-secrets.enc.yaml
+38 -16
View File
@@ -1,6 +1,8 @@
# Dedicated CNPG Postgres for Poimen Memory (GitOps, wave 2).
# Matches homelab/k8s/infra/databases/memory-db.yaml — single source of truth.
# CNPG generates secret `memory-db-app` + service `memory-db-rw` in ns poimen.
---
# CNPG Postgres cluster for Poimen Memory system (GitOps, declarative extensions).
# 2 instances, pgvector 0.7.0 via spec.extensions (not manual CREATE EXTENSION).
# Storage: 10Gi longhorn, consistent with temporal-db.yaml.
# No manual psql needed — all via git/ArgoCD.
apiVersion: postgresql.cnpg.io/v1
kind: Cluster
metadata:
@@ -9,24 +11,15 @@ metadata:
annotations:
argocd.argoproj.io/sync-options: SkipDryRunOnMissingResource=true
spec:
instances: 3
instances: 2
imageName: ghcr.io/cloudnative-pg/postgresql:16.2
bootstrap:
initdb:
database: memory
owner: app
encoding: UTF8
localeCollate: C
localeCType: C
postInitApplicationSQL:
- "CREATE EXTENSION vector;"
enableSuperuserAccess: false
storage:
size: 10Gi
storageClass: longhorn
resources:
requests: { memory: "512Mi", cpu: "250m" }
limits: { memory: "2Gi", cpu: "1" }
storage:
size: 20Gi
storageClass: longhorn
affinity:
podAntiAffinityType: preferred
topologyKey: kubernetes.io/hostname
@@ -34,3 +27,32 @@ spec:
- key: node-role.kubernetes.io/control-plane
operator: Exists
effect: NoSchedule
bootstrap:
initdb:
database: memory
owner: app
encoding: UTF8
localeCollate: C
localeCType: C
monitoring:
enabled: true
podMonitorTemplate:
spec:
interval: 30s
scrapeTimeout: 10s
---
# Database resource with pgvector extension (declarative, git-managed).
# CNPG 1.30.0+ supports this via spec.extensions on the Database CRD.
# Ensures pgvector is installed and available for HNSW indexing.
apiVersion: postgresql.cnpg.io/v1
kind: Database
metadata:
name: memory
namespace: poimen
spec:
cluster:
name: memory-db
owner: app
extensions:
- name: vector
ensure: present
@@ -1,47 +0,0 @@
apiVersion: ENC[AES256_GCM,data:qM0=,iv:znTNMu1+efRh38Vn0GWlNZTk/6VjCJfJeaEzbM17N8c=,tag:sPSc9mwoZWYvjD1bzM+uzg==,type:str]
kind: ENC[AES256_GCM,data:pHDYbqGy,iv:8kUzizuj3tkgx8FU19FBr8lcz1DFEN2abQTJCFLPL0w=,tag:YqiHCVZ8Pwyx51YkxLSykQ==,type:str]
metadata:
name: ENC[AES256_GCM,data:yC5ph8jQnEd2Jn60tCNYJQq2,iv:QRAhTVXNt77kcbcLXDJo9Y1X3hRu1EZXADwTS3rPq/g=,tag:X80UnBGCV28GiOWNo3K/bA==,type:str]
namespace: ENC[AES256_GCM,data:7N36Xqio,iv:a8yemv8LA1WdXUyNRgTu5teZIB23ClufXh7ovd9m5GU=,tag:ukU63zxVZdD9PwppgAmaEw==,type:str]
type: ENC[AES256_GCM,data:9NGNI47z,iv:tiaioFpXheBY4BimysI3sr5OzFOEI1mG68ObCDiqAIU=,tag:wpEln7lSyAPfxpVjcWhpVg==,type:str]
stringData:
admin-password: ENC[AES256_GCM,data:HeKM7q8662fdrlJbpWh/7VJuhr7h2sRYK6/sN+eBtBo=,iv:4ur6YKAYp6+kvIkmBcx9/0DK2MvK7XDedoZhKl8gjBY=,tag:pct7MAlre1h7bm8Polvutg==,type:str]
sops:
age:
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGV2NJQzhDN0ZacXBDeklV
aGE1eGlmMkp6b1RDL2ZiblNwSk1PUkJZdFZjCi84dWpXMFNNcFYrLzkwOUFGZDZ4
SGM0NG9UMkJTME82dUU0MkxFNjVzcTAKLS0tIE5xNlg2RUdheUxyUytsblI3UTFH
UktjaHNGOUlmZGxiSlhoSkJSMW5LMkkKdNAzdge1HaAgBqbE4dCkJgZBlIAP76P+
4GOsh7RbuVDDMzUHTS4aNv2zoM5WC5pv+ZKtf8Yu7LIwiOPAp2u/7g==
-----END AGE ENCRYPTED FILE-----
recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
lastmodified: "2026-09-12T14:22:55Z"
mac: ENC[AES256_GCM,data:lO+5lWN4ZVIkg4XAG4mz6n2SxqNfU6KdahoZqj9nZ33maX/9OT7aunwl3eIoE8JlN4vN1UU/s0l1ioT0+PxdGtlQfhisZ0ypzA3z8Nxkcw18XzQaMf99A0Icw1OEGRRx/T6Bf8+l0ZI4HIH+KZmlUg2lAfGK+WTxqr5xJefw5XA=,iv:kWNox9QX7Jv9muHjBo6yuwRjBRuhawaKJ+5+O9E57z4=,tag:qZ+5ceNo2C8cPIt0PBtqiw==,type:str]
unencrypted_suffix: _unencrypted
version: 3.13.2
---
apiVersion: ENC[AES256_GCM,data:jew=,iv:bzrjT8rJssrSv4xZCn9ihNtyelKteybg/XZVJRUawvo=,tag:qN50XwBiN2lnWH1CS35W/g==,type:str]
kind: ENC[AES256_GCM,data:clwtkPLP,iv:Y2sF8dpOJslo4OHeRprK/wcuvUzdOW30Bz3M0Kh8yE4=,tag:TkQo8zZZow/zdAlxViWQlA==,type:str]
metadata:
name: ENC[AES256_GCM,data:UiOyRh0x7Yor3qudRqgwtrv5bBHkHnFMK0Smxw==,iv:a3hB+wBIMjD0Xj7p3ZIqDf3/la1xlzbCYRRc/LV80ig=,tag:ivw42aZepKgOa+cVm0URZg==,type:str]
namespace: ENC[AES256_GCM,data:a37Jp+qA,iv:cDVuBJ/aFo4EcZTC/N9NGk8UdrCROHKiirWBWlrSDMQ=,tag:/LNyMpSaEhgQYIE8PJcXBg==,type:str]
type: ENC[AES256_GCM,data:ql+XYM25,iv:OCfk13+9Ft4Vq6Tq3R6v54zK1tz7imzyr/g8ytcpBEk=,tag:PpJBFsUSpMo3ktGynr8AUw==,type:str]
stringData:
password: ENC[AES256_GCM,data:lT7f3cY+VLqRcfuYnf9lnI5QVqfmmt5pc9tb2EEuRbE=,iv:2Dnwssmj5ddcr4UypVVkm4UydeLc1L/ExsaRsbu/CVw=,tag:6++s/j3MOBIirkKS66Z46Q==,type:str]
sops:
age:
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGV2NJQzhDN0ZacXBDeklV
aGE1eGlmMkp6b1RDL2ZiblNwSk1PUkJZdFZjCi84dWpXMFNNcFYrLzkwOUFGZDZ4
SGM0NG9UMkJTME82dUU0MkxFNjVzcTAKLS0tIE5xNlg2RUdheUxyUytsblI3UTFH
UktjaHNGOUlmZGxiSlhoSkJSMW5LMkkKdNAzdge1HaAgBqbE4dCkJgZBlIAP76P+
4GOsh7RbuVDDMzUHTS4aNv2zoM5WC5pv+ZKtf8Yu7LIwiOPAp2u/7g==
-----END AGE ENCRYPTED FILE-----
recipient: age1e5fq3hwxy78psus2nfvmtmua36g0u3suk78ephw6246l974d2utsvn0hla
lastmodified: "2026-09-12T14:22:55Z"
mac: ENC[AES256_GCM,data:lO+5lWN4ZVIkg4XAG4mz6n2SxqNfU6KdahoZqj9nZ33maX/9OT7aunwl3eIoE8JlN4vN1UU/s0l1ioT0+PxdGtlQfhisZ0ypzA3z8Nxkcw18XzQaMf99A0Icw1OEGRRx/T6Bf8+l0ZI4HIH+KZmlUg2lAfGK+WTxqr5xJefw5XA=,iv:kWNox9QX7Jv9muHjBo6yuwRjBRuhawaKJ+5+O9E57z4=,tag:qZ+5ceNo2C8cPIt0PBtqiw==,type:str]
unencrypted_suffix: _unencrypted
version: 3.13.2
+35 -8
View File
@@ -65,7 +65,8 @@ data:
# Cluster settings
cluster.name: poimen-memory
node.name: ${HOSTNAME}
discovery.type: single-node
cluster.initial_master_nodes: opensearch-0
discovery.seed_hosts: opensearch-0.opensearch.poimen.svc.cluster.local
# Network
network.host: 0.0.0.0
@@ -126,12 +127,16 @@ spec:
spec:
serviceAccountName: opensearch
hostNetwork: false
securityContext:
fsGroup: 1000
tolerations:
- key: node-role.kubernetes.io/control-plane
operator: Exists
effect: NoSchedule
initContainers:
- name: sysctl
image: busybox:1.28
command:
- sysctl
- -w
- vm.max_map_count=262144
securityContext:
privileged: true
containers:
- name: opensearch
@@ -395,6 +400,18 @@ spec:
---
# Secret: OpenSearch Dashboards password
apiVersion: v1
kind: Secret
metadata:
name: opensearch-dashboards-secret
namespace: poimen
type: Opaque
stringData:
password: "admin" # ⚠️ Change in production
---
# ServiceAccount for OpenSearch Dashboards
apiVersion: v1
kind: ServiceAccount
@@ -402,4 +419,14 @@ metadata:
name: opensearch-dashboards
namespace: poimen
# Secrets moved to opensearch-secrets.enc.yaml (SOPS-encrypted)
---
# Secret for OpenSearch Admin Password
apiVersion: v1
kind: Secret
metadata:
name: opensearch-secrets
namespace: poimen
type: Opaque
stringData:
admin-password: "OpenSearch@Admin123!"

Some files were not shown because too many files have changed in this diff Show More