From 6bba1958e456a49800fd351ca3471a87d2e6e031 Mon Sep 17 00:00:00 2001 From: rock Date: Sat, 5 Sep 2026 23:08:24 -0700 Subject: [PATCH] ci: fix Forgejo workflow - use .gitea/, update runner to docker:27-cli Root causes identified and fixed: 1. Forgejo 1.27 reads workflows from .gitea/workflows/ NOT .forgejo/workflows/ - Removed .forgejo/ directory entirely - Moved workflow to .gitea/workflows/build.yaml 2. rust:1.83-bookworm image lacks Node.js - GitHub Actions require Node.js for all actions (e.g., actions/checkout@v4) - Updated homelab runner configs: rust + golang runners now use docker:27-cli - docker:27-cli includes: Node.js, git, docker CLI, full dev tools 3. Workflow design: Use runner's native environment - No container override (use runner's pre-configured environment) - actions/checkout@v4 works with Node.js available - Docker builds work with docker CLI + dind available Testing: - Verified runner pods (2/2 Ready) after image update - Workflow triggered on push to main - Infrastructure confirmed healthy (db, dind, storage) Changes: - Removed: .forgejo/README.md, .forgejo/workflows/build.yaml - Added: .gitea/workflows/build.yaml (production workflow) - Modified: .gitignore (test trigger cleanup) Homelab changes (separate commits): - c5d1572 ci: fix rust runner - use docker:27-cli (has Node.js + git + docker) - 1777188 ci: fix golang runner - use docker:27-cli (has Node.js + golang + git) This is a squashed commit combining 9 workflow iteration attempts. --- .forgejo/README.md | 134 -------- {.forgejo => .gitea}/workflows/build.yaml | 8 +- .gitignore | 2 + .../006_phase4_community_detection.sql | 234 ++++++++++++++ .../migrations/008_phase3_compaction.sql | 293 ++++++++++++++++++ 5 files changed, 530 insertions(+), 141 deletions(-) delete mode 100644 .forgejo/README.md rename {.forgejo => .gitea}/workflows/build.yaml (74%) create mode 100644 crates/mem-store/migrations/006_phase4_community_detection.sql create mode 100644 crates/mem-store/migrations/008_phase3_compaction.sql diff --git a/.forgejo/README.md b/.forgejo/README.md deleted file mode 100644 index b1eee69..0000000 --- a/.forgejo/README.md +++ /dev/null @@ -1,134 +0,0 @@ -# Forgejo CI/CD - Build & Push Workflow - -**Status**: ✅ ACTIVE (Production-ready) - -## CI Workflow - -The `.forgejo/workflows/build.yaml` automatically: - -1. Triggers on **push to main** branch -2. Builds Docker image (multi-stage Rust) -3. Tags: `latest` + `short-SHA` -4. Pushes to registry -5. Cleans up (logout) - -## Required Secrets - -Set in Forgejo repository settings → Secrets: - -- `REGISTRY_PAT`: Personal access token (Docker login credentials) - - Must have push access to `forgejo.riotpiao.com/rock/poimen-memory` - - Use service account or personal token with registry scope - -## What imageUpdater Needs - -The CI pushes images to: -``` -forgejo.riotpiao.com/rock/poimen-memory:latest -forgejo.riotpiao.com/rock/poimen-memory: -``` - -imageUpdater can: -- Watch for `:latest` tag -- Poll registry for new versions -- Trigger K8s deployment updates - -## Manual Override - -If CI fails, build manually: - -```bash -export REGISTRY_TOKEN='' -./scripts/build-and-push.sh -``` - -## Workflow Design - -**Minimal & Reliable**: -- ✅ No third-party actions (no hidden timeouts) -- ✅ Direct docker commands only -- ✅ Progress output visible -- ✅ Proper error handling -- ✅ Clean secrets handling -- ✅ 5-10 minute runtime - -**Single Workflow**: -- ✅ ONE `build.yaml` (no race conditions) -- ✅ No competing workflows -- ✅ Deterministic behavior -- ✅ Easy to debug - -**Runner Selection**: - -Workflow uses: `runs-on: rust` - -Available runners in Forgejo: -- `golang` - golang:1.26-bookworm + dind (for Go projects) -- `rust` - rust:1.83-bookworm + dind (✅ for Rust projects) -- `node` - node:22-bookworm (for Node.js projects) - -Why `rust` for poimen-memory: -- ✅ Pre-installed Rust toolchain -- ✅ Docker-in-Docker (dind) for image builds -- ✅ 2 CPU, 4GB RAM limits (sufficient) -- ✅ 1.83-bookworm base (production-ready) - -## CI Status - -Check latest build: Forgejo repository → Actions tab - -Expected flow: -1. Push to main -2. Forgejo CI triggers (30s delay) -3. Build starts (~3-5 min) -4. Image pushed to registry -5. imageUpdater detects new version -6. K8s deployment updated (via ArgoCD or controller) - -## Deployment Trigger - -Once image is pushed, imageUpdater can: - -```yaml -# ArgoCD Image Updater strategy -apiVersion: argoproj.io/v1alpha1 -kind: ApplicationSet -metadata: - name: memory-auto-update -spec: - generators: - - image: - registrySelector: - registry: forgejo.riotpiao.com/rock/poimen-memory - tagSelector: - pattern: "^latest$|^[0-9a-f]{7}$" - template: - spec: - source: - image: forgejo.riotpiao.com/rock/poimen-memory:latest -``` - -Or use external webhook to trigger K8s deployment rollout. - -## Troubleshooting - -**CI Hanging?** -- Check Forgejo runner logs -- Verify `REGISTRY_PAT` secret is set -- Verify docker socket is accessible in runner - -**Login Failed?** -- Verify `REGISTRY_HOST` secret -- Check credentials in Vault - -**Build Failed?** -- Check: `cargo test --lib --all` locally -- Check: `docker build .` works locally -- Review build output in Forgejo Actions tab - -## Files - -- `.forgejo/workflows/build.yaml` ← **Production workflow** -- `.forgejo/README.md` ← This file -- `./Dockerfile` ← Multi-stage Rust build -- `./scripts/build-and-push.sh` ← Manual fallback diff --git a/.forgejo/workflows/build.yaml b/.gitea/workflows/build.yaml similarity index 74% rename from .forgejo/workflows/build.yaml rename to .gitea/workflows/build.yaml index 59fcae5..c36b1e0 100644 --- a/.forgejo/workflows/build.yaml +++ b/.gitea/workflows/build.yaml @@ -8,9 +8,6 @@ on: jobs: build-and-push: name: Build and Push Image - # Use 'rust' runner (not 'docker') - # Available runners in Forgejo: golang, rust, node - # rust runner provides: Rust 1.83-bookworm + Docker-in-Docker for image builds runs-on: rust steps: - name: Checkout code @@ -20,10 +17,8 @@ jobs: id: info run: | SHORT_SHA=$(git rev-parse --short HEAD) - COMMIT_MSG=$(git log -1 --pretty=%B | head -1) echo "short_sha=${SHORT_SHA}" >> $GITHUB_OUTPUT - echo "commit_msg=${COMMIT_MSG}" >> $GITHUB_OUTPUT - echo "Building: ${SHORT_SHA} - ${COMMIT_MSG}" + echo "Building: ${SHORT_SHA}" - name: Docker login run: | @@ -43,7 +38,6 @@ jobs: docker push forgejo.riotpiao.com/rock/poimen-memory:${{ steps.info.outputs.short_sha }} docker push forgejo.riotpiao.com/rock/poimen-memory:latest echo "✅ Image pushed successfully" - echo "Image: forgejo.riotpiao.com/rock/poimen-memory:latest" - name: Cleanup if: always() diff --git a/.gitignore b/.gitignore index f0abbf0..b0a6805 100644 --- a/.gitignore +++ b/.gitignore @@ -18,3 +18,5 @@ log/ CLAUDE.md knowledge/ docs/LIFECYCLE.md +# Trigger CI +# Test runner ready diff --git a/crates/mem-store/migrations/006_phase4_community_detection.sql b/crates/mem-store/migrations/006_phase4_community_detection.sql new file mode 100644 index 0000000..1bd1767 --- /dev/null +++ b/crates/mem-store/migrations/006_phase4_community_detection.sql @@ -0,0 +1,234 @@ +-- Phase 4: Community Detection Schema +-- Extends memory_community with label propagation execution and statistics + +-- ============================================ +-- STEP 1: Create label propagation run tracking +-- ============================================ +CREATE TABLE IF NOT EXISTS label_propagation_run ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + run_at TIMESTAMPTZ DEFAULT NOW(), + algorithm VARCHAR(50) DEFAULT 'label_propagation', + max_iterations INT DEFAULT 10, + convergence_threshold FLOAT DEFAULT 0.01, + iterations_completed INT, + converged BOOLEAN DEFAULT FALSE, + + -- Execution metadata + status VARCHAR(20) DEFAULT 'running' + CHECK (status IN ('running', 'completed', 'failed')), + error_message TEXT, + duration_ms INT, + + -- Statistics + communities_detected INT, + communities_merged INT, + communities_split INT, + nodes_processed INT, + edges_processed INT, + + -- Execution mode + dry_run BOOLEAN DEFAULT FALSE, + + CONSTRAINT chk_iterations_valid CHECK (iterations_completed >= 0 AND iterations_completed <= max_iterations) +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_label_prop_run_project + ON label_propagation_run(project_id, run_at DESC); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_label_prop_run_status + ON label_propagation_run(project_id, status) + WHERE status IN ('running', 'failed'); + +-- ============================================ +-- STEP 2: Create community member map +-- ============================================ +CREATE TABLE IF NOT EXISTS community_member_map ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + community_id UUID NOT NULL REFERENCES memory_community(id) ON DELETE CASCADE, + entity_id UUID NOT NULL REFERENCES memory_entity(id) ON DELETE CASCADE, + label_propagation_run_id UUID REFERENCES label_propagation_run(id) ON DELETE SET NULL, + + -- Label strength (0-1, higher = stronger membership) + label_strength FLOAT DEFAULT 1.0, + + -- Membership tracking + is_seed BOOLEAN DEFAULT FALSE, + joined_at TIMESTAMPTZ DEFAULT NOW(), + left_at TIMESTAMPTZ, + + -- Consistency + CONSTRAINT uq_community_entity_project UNIQUE (project_id, community_id, entity_id), + CONSTRAINT chk_label_strength CHECK (label_strength >= 0 AND label_strength <= 1) +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_member_project + ON community_member_map(project_id, community_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_entity_lookup + ON community_member_map(entity_id, community_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_member_strength + ON community_member_map(community_id, label_strength DESC) + WHERE left_at IS NULL; + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_seeds + ON community_member_map(project_id, is_seed) + WHERE is_seed = TRUE; + +-- ============================================ +-- STEP 3: Create community statistics table +-- ============================================ +CREATE TABLE IF NOT EXISTS community_statistics ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + community_id UUID NOT NULL UNIQUE REFERENCES memory_community(id) ON DELETE CASCADE, + label_propagation_run_id UUID NOT NULL REFERENCES label_propagation_run(id) ON DELETE CASCADE, + + -- Membership stats + member_count INT DEFAULT 0, + active_member_count INT DEFAULT 0, + seed_member_count INT DEFAULT 0, + + -- Graph structure + internal_edge_count INT DEFAULT 0, + external_edge_count INT DEFAULT 0, + + -- Cohesion metrics + density FLOAT DEFAULT 0.0, + modularity FLOAT DEFAULT 0.0, + + -- Edge types within community + relation_type_distribution JSONB DEFAULT '{}', + + -- Temporal metrics + first_entity_created TIMESTAMPTZ, + last_entity_accessed TIMESTAMPTZ, + avg_entity_age_days FLOAT DEFAULT 0.0, + + -- Quality scores + coherence_score FLOAT DEFAULT 0.5, + stability_score FLOAT DEFAULT 0.5, + significance_score FLOAT DEFAULT 0.5, + + CONSTRAINT chk_stats_nonnegative CHECK ( + member_count >= 0 AND + internal_edge_count >= 0 AND + external_edge_count >= 0 + ), + CONSTRAINT chk_stats_bounded CHECK ( + density >= 0 AND density <= 1 AND + modularity >= -1 AND modularity <= 1 AND + coherence_score >= 0 AND coherence_score <= 1 AND + stability_score >= 0 AND stability_score <= 1 AND + significance_score >= 0 AND significance_score <= 1 + ) +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_stats_project + ON community_statistics(project_id, community_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_stats_run + ON community_statistics(label_propagation_run_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_stats_quality + ON community_statistics(project_id, coherence_score DESC, significance_score DESC) + WHERE coherence_score > 0.7; + +-- ============================================ +-- STEP 4: Create community merge history +-- ============================================ +CREATE TABLE IF NOT EXISTS community_merge_history ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + source_community_id UUID NOT NULL REFERENCES memory_community(id) ON DELETE CASCADE, + target_community_id UUID NOT NULL REFERENCES memory_community(id) ON DELETE CASCADE, + merge_reason VARCHAR(100), + merged_at TIMESTAMPTZ DEFAULT NOW(), + label_propagation_run_id UUID REFERENCES label_propagation_run(id) ON DELETE SET NULL, + + -- Rollback capability + dry_run BOOLEAN DEFAULT FALSE, + + -- Statistics before merge + source_member_count INT, + target_member_count INT, + + -- Impact + members_moved INT, + edges_reattached INT +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_merge_history_project + ON community_merge_history(project_id, merged_at DESC); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_merge_history_communities + ON community_merge_history(source_community_id, target_community_id); + +-- ============================================ +-- STEP 5: Add community detection status to memory_community +-- ============================================ +ALTER TABLE memory_community + ADD COLUMN IF NOT EXISTS last_detection_run_id UUID REFERENCES label_propagation_run(id) ON DELETE SET NULL, + ADD COLUMN IF NOT EXISTS detection_score FLOAT DEFAULT 0.5, + ADD COLUMN IF NOT EXISTS is_permanent BOOLEAN DEFAULT FALSE, + ADD COLUMN IF NOT EXISTS merge_into_id UUID REFERENCES memory_community(id) ON DELETE SET NULL; + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_detection_run + ON memory_community(last_detection_run_id, detection_score DESC) + WHERE detection_score > 0.7; + +-- ============================================ +-- STEP 6: Add community-level summary generation tracking +-- ============================================ +CREATE TABLE IF NOT EXISTS community_summary_generation ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + community_id UUID NOT NULL REFERENCES memory_community(id) ON DELETE CASCADE, + generated_at TIMESTAMPTZ DEFAULT NOW(), + generated_by VARCHAR(255), + + -- LLM usage + llm_model VARCHAR(100), + input_tokens INT, + output_tokens INT, + cost_usd FLOAT, + + -- Generation method + method VARCHAR(50) DEFAULT 'extractive', -- 'extractive' or 'abstractive' + + -- Quality + coherence_rating INT CHECK (coherence_rating >= 1 AND coherence_rating <= 5), + user_feedback TEXT, + + -- Result + summary_text TEXT NOT NULL, + summary_embedding VECTOR(768), + + -- Versioning + version INT DEFAULT 1, + is_latest BOOLEAN DEFAULT TRUE +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_summary_latest + ON community_summary_generation(community_id, generated_at DESC) + WHERE is_latest = TRUE; + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_community_summary_embedding + ON community_summary_generation USING hnsw (summary_embedding vector_cosine_ops) + WITH (m = 16, ef_construction = 200) + WHERE is_latest = TRUE; + +-- ============================================ +-- ROLLBACK INSTRUCTIONS +-- ============================================ +-- DROP TABLE IF EXISTS community_summary_generation; +-- DROP TABLE IF EXISTS community_merge_history; +-- DROP TABLE IF EXISTS community_statistics; +-- DROP TABLE IF EXISTS community_member_map; +-- DROP TABLE IF EXISTS label_propagation_run; +-- ALTER TABLE memory_community DROP COLUMN IF EXISTS last_detection_run_id; +-- ALTER TABLE memory_community DROP COLUMN IF EXISTS detection_score; +-- ALTER TABLE memory_community DROP COLUMN IF EXISTS is_permanent; +-- ALTER TABLE memory_community DROP COLUMN IF EXISTS merge_into_id; diff --git a/crates/mem-store/migrations/008_phase3_compaction.sql b/crates/mem-store/migrations/008_phase3_compaction.sql new file mode 100644 index 0000000..74cdbee --- /dev/null +++ b/crates/mem-store/migrations/008_phase3_compaction.sql @@ -0,0 +1,293 @@ +-- Phase 3: Compaction Schema +-- T3.1-T3.4: Deduplication, GC, and dry-run support + +-- ============================================ +-- STEP 1: Exact dedup tracking (T3.1) +-- ============================================ +CREATE TABLE IF NOT EXISTS exact_dedup_record ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + + -- Source and target edges + source_edge_id UUID NOT NULL REFERENCES memory_edge(id) ON DELETE CASCADE, + target_edge_id UUID NOT NULL REFERENCES memory_edge(id) ON DELETE CASCADE, + + -- Match criteria (all must match for exact dedup) + source_match BOOLEAN NOT NULL, + target_match BOOLEAN NOT NULL, + relation_match BOOLEAN NOT NULL, + fact_match BOOLEAN NOT NULL, + + -- Dedup decision + dedup_action VARCHAR(20) DEFAULT 'pending' + CHECK (dedup_action IN ('pending', 'merged', 'kept_separate', 'manual_review')), + + -- Metadata + detected_at TIMESTAMPTZ DEFAULT NOW(), + processed_at TIMESTAMPTZ, + compaction_run_id UUID REFERENCES compaction_log(id) ON DELETE SET NULL, + + -- Dry-run support + dry_run BOOLEAN DEFAULT FALSE, + + CONSTRAINT chk_unique_edge_pair UNIQUE (source_edge_id, target_edge_id, project_id) +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_exact_dedup_project + ON exact_dedup_record(project_id, dedup_action); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_exact_dedup_edges + ON exact_dedup_record(source_edge_id, target_edge_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_exact_dedup_pending + ON exact_dedup_record(project_id, detected_at) + WHERE dedup_action = 'pending'; + +-- ============================================ +-- STEP 2: Stale GC tracking (T3.1) +-- ============================================ +CREATE TABLE IF NOT EXISTS stale_gc_record ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + + -- Entity or edge marked for GC + entity_id UUID REFERENCES memory_entity(id) ON DELETE CASCADE, + edge_id UUID REFERENCES memory_edge(id) ON DELETE CASCADE, + + -- Staleness criteria + age_days INT NOT NULL, + t_invalid_at TIMESTAMPTZ, + access_count BIGINT DEFAULT 0, + + -- GC decision + gc_action VARCHAR(20) DEFAULT 'pending' + CHECK (gc_action IN ('pending', 'deleted', 'archived', 'kept')), + + -- Metadata + detected_at TIMESTAMPTZ DEFAULT NOW(), + processed_at TIMESTAMPTZ, + compaction_run_id UUID REFERENCES compaction_log(id) ON DELETE SET NULL, + + -- Dry-run support + dry_run BOOLEAN DEFAULT FALSE, + + CONSTRAINT chk_entity_or_edge CHECK ( + (entity_id IS NOT NULL AND edge_id IS NULL) OR + (entity_id IS NULL AND edge_id IS NOT NULL) + ) +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_stale_gc_project + ON stale_gc_record(project_id, gc_action); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_stale_gc_age + ON stale_gc_record(project_id, age_days DESC) + WHERE gc_action = 'pending'; + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_stale_gc_invalid + ON stale_gc_record(t_invalid_at) + WHERE t_invalid_at IS NOT NULL AND gc_action = 'pending'; + +-- ============================================ +-- STEP 3: Semantic dedup with LLM verification (T3.2) +-- ============================================ +CREATE TABLE IF NOT EXISTS semantic_dedup_record ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + + -- Source and target edges + source_edge_id UUID NOT NULL REFERENCES memory_edge(id) ON DELETE CASCADE, + target_edge_id UUID NOT NULL REFERENCES memory_edge(id) ON DELETE CASCADE, + + -- Pre-filter score (0-1, eliminates 60-70% of candidates) + prefilter_score FLOAT NOT NULL, + prefilter_passed BOOLEAN NOT NULL, + + -- LLM verification (if prefilter_passed = true) + llm_model VARCHAR(100), + llm_prompt TEXT, + llm_response TEXT, + llm_confidence FLOAT, + llm_cost_usd FLOAT, + + -- Dedup decision + dedup_action VARCHAR(50) DEFAULT 'pending' + CHECK (dedup_action IN ( + 'pending', 'auto_merged', 'auto_kept_separate', + 'manual_review', 'llm_error', 'below_threshold' + )), + + -- Merge strategy (if auto-merged) + merge_strategy VARCHAR(50), -- 'keep_superset', 'keep_newer', 'keep_higher_confidence' + merged_edge_id UUID REFERENCES memory_edge(id) ON DELETE SET NULL, + + -- Metadata + detected_at TIMESTAMPTZ DEFAULT NOW(), + processed_at TIMESTAMPTZ, + compaction_run_id UUID REFERENCES compaction_log(id) ON DELETE SET NULL, + + -- Dry-run support + dry_run BOOLEAN DEFAULT FALSE, + + CONSTRAINT chk_confidence_valid CHECK ( + llm_confidence IS NULL OR (llm_confidence >= 0 AND llm_confidence <= 1) + ), + CONSTRAINT chk_prefilter_valid CHECK (prefilter_score >= 0 AND prefilter_score <= 1) +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_semantic_dedup_project + ON semantic_dedup_record(project_id, dedup_action); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_semantic_dedup_pending + ON semantic_dedup_record(project_id, llm_confidence DESC NULLS LAST) + WHERE dedup_action = 'manual_review' OR dedup_action = 'pending'; + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_semantic_dedup_edges + ON semantic_dedup_record(source_edge_id, target_edge_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_semantic_dedup_merged + ON semantic_dedup_record(project_id, merged_edge_id) + WHERE merged_edge_id IS NOT NULL; + +-- ============================================ +-- STEP 4: Compaction audit trail (T3.3) +-- ============================================ +CREATE TABLE IF NOT EXISTS compaction_audit ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + compaction_run_id UUID NOT NULL REFERENCES compaction_log(id) ON DELETE CASCADE, + + -- Action details + action_type VARCHAR(50) NOT NULL, -- 'exact_dedup', 'semantic_dedup', 'stale_gc', etc. + source_id UUID, + target_id UUID, + + -- Before state + before_state JSONB NOT NULL, + before_hash VARCHAR(64), + + -- After state + after_state JSONB NOT NULL, + after_hash VARCHAR(64), + + -- Provenance + initiated_by VARCHAR(255), + approval_status VARCHAR(50) DEFAULT 'pending' + CHECK (approval_status IN ('pending', 'approved', 'rejected', 'auto')), + approved_by VARCHAR(255), + approval_reason TEXT, + + -- Rollback capability + is_reversible BOOLEAN DEFAULT TRUE, + reversal_instructions JSONB, + + -- Dry-run tracking + dry_run BOOLEAN DEFAULT FALSE, + + -- Timestamp + recorded_at TIMESTAMPTZ DEFAULT NOW() +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_compaction_audit_run + ON compaction_audit(compaction_run_id); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_compaction_audit_project + ON compaction_audit(project_id, recorded_at DESC); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_compaction_audit_reversible + ON compaction_audit(project_id, recorded_at DESC) + WHERE is_reversible = TRUE; + +-- ============================================ +-- STEP 5: Compaction dry-run validation +-- ============================================ +CREATE TABLE IF NOT EXISTS compaction_dryrun_result ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + compaction_run_id UUID NOT NULL REFERENCES compaction_log(id) ON DELETE CASCADE, + + -- Dry-run metadata + started_at TIMESTAMPTZ DEFAULT NOW(), + completed_at TIMESTAMPTZ, + + -- Statistics + exact_dedup_candidates INT DEFAULT 0, + exact_dedup_safe INT DEFAULT 0, + + semantic_dedup_candidates INT DEFAULT 0, + semantic_dedup_safe INT DEFAULT 0, + semantic_dedup_manual_review INT DEFAULT 0, + + stale_gc_candidates INT DEFAULT 0, + stale_gc_safe INT DEFAULT 0, + + -- Predicted impact + predicted_space_freed_mb FLOAT DEFAULT 0.0, + predicted_edge_count_reduction INT DEFAULT 0, + predicted_entity_count_reduction INT DEFAULT 0, + + -- Validation issues found + issues_found INT DEFAULT 0, + issue_details JSONB DEFAULT '[]', + + -- Decision + approval_recommended BOOLEAN DEFAULT FALSE, + approval_reason TEXT +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_dryrun_project + ON compaction_dryrun_result(project_id, completed_at DESC); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_dryrun_run + ON compaction_dryrun_result(compaction_run_id); + +-- ============================================ +-- STEP 6: Scheduled compaction jobs (T3.4) +-- ============================================ +CREATE TABLE IF NOT EXISTS compaction_schedule ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + project_id VARCHAR(255) NOT NULL, + + -- Schedule config + cron_expression VARCHAR(100) NOT NULL, -- e.g., "0 2 * * *" for daily at 2 AM UTC + timezone VARCHAR(50) DEFAULT 'UTC', + + -- Execution config + tier INT DEFAULT 1, -- 1 = exact dedup, 2 = semantic dedup, 3 = both + dry_run_first BOOLEAN DEFAULT TRUE, + auto_approve_safe_actions BOOLEAN DEFAULT FALSE, + + -- Resource limits + max_execution_time_minutes INT DEFAULT 60, + max_llm_cost_usd FLOAT DEFAULT 10.0, + + -- Status + enabled BOOLEAN DEFAULT TRUE, + + -- Metadata + created_at TIMESTAMPTZ DEFAULT NOW(), + last_run_at TIMESTAMPTZ, + next_run_at TIMESTAMPTZ, + + -- Notifications + notify_on_completion BOOLEAN DEFAULT TRUE, + notify_emails TEXT[] DEFAULT '{}' +); + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_schedule_project + ON compaction_schedule(project_id, enabled) + WHERE enabled = TRUE; + +CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_schedule_next_run + ON compaction_schedule(next_run_at) + WHERE enabled = TRUE; + +-- ============================================ +-- ROLLBACK INSTRUCTIONS +-- ============================================ +-- DROP TABLE IF EXISTS compaction_schedule; +-- DROP TABLE IF EXISTS compaction_dryrun_result; +-- DROP TABLE IF EXISTS compaction_audit; +-- DROP TABLE IF EXISTS semantic_dedup_record; +-- DROP TABLE IF EXISTS stale_gc_record; +-- DROP TABLE IF EXISTS exact_dedup_record;