From 8d59df40b4f12ca82b24f298b1be929923949526 Mon Sep 17 00:00:00 2001 From: Story Crater Bot <19826264+Riotpiaole@users.noreply.github.com> Date: Wed, 26 Aug 2026 13:55:37 -0700 Subject: [PATCH] Implement M4.2: Derived filter (shingle matcher + 10 tests, 239 total) --- crates/mem-ingest/src/derived_filter.rs | 241 ++++++++++++++++++++++++ crates/mem-ingest/src/lib.rs | 2 + tests/it_derived_filter.rs | 202 ++++++++++++++++++++ 3 files changed, 445 insertions(+) create mode 100644 crates/mem-ingest/src/derived_filter.rs create mode 100644 tests/it_derived_filter.rs diff --git a/crates/mem-ingest/src/derived_filter.rs b/crates/mem-ingest/src/derived_filter.rs new file mode 100644 index 0000000..047bbdf --- /dev/null +++ b/crates/mem-ingest/src/derived_filter.rs @@ -0,0 +1,241 @@ +//! Derived filter: prevent emitted artifacts from being ingested as evidence. +//! +//! Mechanism: track emitted skills/docs in a manifest, compute shingle overlap +//! with ingest records, exclude those above a threshold. + +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use std::collections::HashSet; + +/// Artifact manifest record (one per emitted skill/doc) +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ArtifactRecord { + pub kind: String, // "skill", "reference", etc. + pub name: String, + pub sha256: String, + pub shingles: Vec, + pub emitted_at: String, +} + +impl ArtifactRecord { + /// Create new artifact record from content + pub fn new(kind: &str, name: &str, content: &str, emitted_at: &str) -> Self { + let mut hasher = Sha256::new(); + hasher.update(content.as_bytes()); + let sha256 = format!("{:x}", hasher.finalize()); + + let shingles = compute_shingles(content, 4); + + Self { + kind: kind.to_string(), + name: name.to_string(), + sha256, + shingles, + emitted_at: emitted_at.to_string(), + } + } + + /// Exact match by SHA256 + pub fn matches_exactly(&self, other_sha256: &str) -> bool { + self.sha256 == other_sha256 + } + + /// Fuzzy match by shingle overlap + pub fn shingle_overlap(&self, other_shingles: &[String]) -> f64 { + if self.shingles.is_empty() || other_shingles.is_empty() { + return 0.0; + } + + let self_set: HashSet<_> = self.shingles.iter().collect(); + let other_set: HashSet<_> = other_shingles.iter().collect(); + + let intersection = self_set.intersection(&other_set).count(); + let union = self_set.union(&other_set).count(); + + if union == 0 { + 0.0 + } else { + intersection as f64 / union as f64 + } + } +} + +/// Compute n-grams (shingles) from normalized text +pub fn compute_shingles(text: &str, shingle_size: usize) -> Vec { + // Normalize: lowercase, strip whitespace, remove markdown + let normalized = normalize_text(text); + + if normalized.len() < shingle_size { + return vec![]; + } + + normalized + .chars() + .collect::>() + .windows(shingle_size) + .map(|w| w.iter().collect::()) + .collect() +} + +/// Normalize text for matching: lowercase, strip non-alphanumeric, collapse whitespace +fn normalize_text(text: &str) -> String { + text.to_lowercase() + .chars() + .filter(|c| c.is_alphanumeric() || c.is_whitespace()) + .collect::() + .split_whitespace() + .collect::>() + .join(" ") +} + +/// Derived filter with manifest and threshold +pub struct DerivedFilter { + pub artifacts: Vec, + pub threshold: f64, + pub shingle_size: usize, +} + +impl DerivedFilter { + pub fn new(threshold: f64) -> Self { + Self { + artifacts: vec![], + threshold, + shingle_size: 4, + } + } + + /// Add artifact to filter + pub fn add_artifact(&mut self, artifact: ArtifactRecord) { + self.artifacts.push(artifact); + } + + /// Check if record is derived (matches an artifact) + pub fn is_derived(&self, record_text: &str) -> Option { + let record_shingles = compute_shingles(record_text, self.shingle_size); + + for artifact in &self.artifacts { + let overlap = artifact.shingle_overlap(&record_shingles); + + if overlap >= self.threshold { + return Some(DerivedMatch { + artifact_name: artifact.name.clone(), + artifact_kind: artifact.kind.clone(), + overlap_ratio: overlap, + }); + } + } + + None + } + + /// Load from JSONL manifest file + pub fn load_from_jsonl(path: &str, threshold: f64) -> anyhow::Result { + use std::fs; + + let mut filter = Self::new(threshold); + + if !std::path::Path::new(path).exists() { + return Ok(filter); // No manifest = no filtering + } + + let content = fs::read_to_string(path)?; + for line in content.lines() { + if line.trim().is_empty() { + continue; + } + if let Ok(artifact) = serde_json::from_str::(line) { + filter.add_artifact(artifact); + } + } + + Ok(filter) + } +} + +/// Result of a derived match +#[derive(Debug, Clone)] +pub struct DerivedMatch { + pub artifact_name: String, + pub artifact_kind: String, + pub overlap_ratio: f64, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_normalize_text() { + let text = " Hello WORLD!!! \n 123 "; + let normalized = normalize_text(text); + assert_eq!(normalized, "hello world 123"); + } + + #[test] + fn test_compute_shingles() { + let text = "hello"; + let shingles = compute_shingles(text, 2); + assert!(shingles.contains(&"he".to_string())); + assert!(shingles.contains(&"el".to_string())); + assert_eq!(shingles.len(), 4); // "he", "el", "ll", "lo" + } + + #[test] + fn test_artifact_exact_match() { + let artifact = ArtifactRecord::new("skill", "test", "content", "2025-01-26"); + let other_sha = format!("{:x}", Sha256::digest("content".as_bytes())); + assert!(artifact.matches_exactly(&other_sha)); + } + + #[test] + fn test_shingle_overlap_identical() { + let text = "hello world"; + let shingles_a = compute_shingles(text, 4); + let shingles_b = compute_shingles(text, 4); + + let artifact = ArtifactRecord::new("skill", "test", text, "2025-01-26"); + let overlap = artifact.shingle_overlap(&shingles_b); + + assert_eq!(overlap, 1.0, "Identical shingles should have 100% overlap"); + } + + #[test] + fn test_shingle_overlap_partial() { + let artifact = ArtifactRecord::new("skill", "test", "hello world test", "2025-01-26"); + let other_text = "hello world other"; + let other_shingles = compute_shingles(other_text, 4); + + let overlap = artifact.shingle_overlap(&other_shingles); + assert!(overlap > 0.0 && overlap < 1.0, "Partial overlap should be between 0 and 1"); + } + + #[test] + fn test_derived_filter_is_derived() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new("skill", "infra-deploy", "deploy to kubernetes", "2025-01-26"); + filter.add_artifact(artifact); + + // Identical text should match + let result = filter.is_derived("deploy to kubernetes"); + assert!(result.is_some()); + + // Reformatted should also match + let result = filter.is_derived("deploy to kubernetes"); + assert!(result.is_some()); + + // Unrelated should not match + let result = filter.is_derived("something completely different"); + assert!(result.is_none()); + } + + #[test] + fn test_derived_filter_mention_not_excluded() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new("skill", "infra-deploy", "deploy to kubernetes", "2025-01-26"); + filter.add_artifact(artifact); + + // Merely mentioning the skill should not be excluded + let result = filter.is_derived("I used the infra-deploy skill yesterday"); + assert!(result.is_none(), "Mere mention should not trigger exclusion"); + } +} diff --git a/crates/mem-ingest/src/lib.rs b/crates/mem-ingest/src/lib.rs index ca1f960..b704cbd 100644 --- a/crates/mem-ingest/src/lib.rs +++ b/crates/mem-ingest/src/lib.rs @@ -1,7 +1,9 @@ pub mod pi_session; pub mod claude_transcript; pub mod doc_corpus; +pub mod derived_filter; pub use pi_session::PiSessionSource; pub use claude_transcript::ClaudeTranscriptSource; pub use doc_corpus::{DocCorpusSource, DocSection, DryRunReport}; +pub use derived_filter::{ArtifactRecord, DerivedFilter, DerivedMatch}; diff --git a/tests/it_derived_filter.rs b/tests/it_derived_filter.rs new file mode 100644 index 0000000..45493b6 --- /dev/null +++ b/tests/it_derived_filter.rs @@ -0,0 +1,202 @@ +//! Integration tests for derived filter +//! +//! Verifies that emitted skills/docs are excluded from ingest when they +//! reappear verbatim or reformatted. + +use mem_ingest::{ArtifactRecord, DerivedFilter}; + +/// Test 1: Verbatim match is excluded +#[test] +fn a1_verbatim_excluded() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new( + "skill", + "deploy-to-prod", + "Deploy service to production using kubectl apply", + "2025-01-26T10:00:00Z", + ); + filter.add_artifact(artifact); + + let record = "Deploy service to production using kubectl apply"; + let result = filter.is_derived(record); + + assert!(result.is_some(), "Verbatim match should be excluded"); + if let Some(m) = result { + assert_eq!(m.artifact_name, "deploy-to-prod"); + assert!(m.overlap_ratio >= 0.8); + } +} + +/// Test 2: Reformatted (whitespace, line breaks) is excluded +#[test] +fn a2_reformatted_excluded() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new( + "skill", + "deploy-to-prod", + "Deploy service to production using kubectl apply", + "2025-01-26T10:00:00Z", + ); + filter.add_artifact(artifact); + + // Same content, different whitespace + let record = "Deploy service to production using kubectl apply"; + let result = filter.is_derived(record); + + assert!(result.is_some(), "Reformatted match should be excluded"); +} + +/// Test 3: Mere mention is NOT excluded +#[test] +fn a3_mention_not_excluded() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new( + "skill", + "deploy-to-prod", + "Deploy service to production using kubectl apply", + "2025-01-26T10:00:00Z", + ); + filter.add_artifact(artifact); + + let record = "I used the deploy-to-prod skill yesterday to update the service"; + let result = filter.is_derived(record); + + assert!(result.is_none(), "Mere mention should NOT be excluded"); +} + +/// Test 4: Unrelated text is not excluded +#[test] +fn a4_unrelated_not_excluded() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new( + "skill", + "deploy-to-prod", + "Deploy service to production using kubectl apply", + "2025-01-26T10:00:00Z", + ); + filter.add_artifact(artifact); + + let record = "I went to the grocery store today and bought some milk and bread"; + let result = filter.is_derived(record); + + assert!(result.is_none(), "Unrelated text should not be excluded"); +} + +/// Test 5: Exclusion is logged with match metadata +#[test] +fn a5_exclusion_logged() { + let mut filter = DerivedFilter::new(0.8); + let artifact = ArtifactRecord::new( + "skill", + "test-skill", + "Test content here", + "2025-01-26T10:00:00Z", + ); + filter.add_artifact(artifact); + + let record = "Test content here"; + let result = filter.is_derived(record); + + assert!(result.is_some()); + if let Some(m) = result { + assert_eq!(m.artifact_name, "test-skill"); + assert_eq!(m.artifact_kind, "skill"); + assert!(m.overlap_ratio > 0.0, "Overlap ratio should be logged"); + } +} + +/// Test 6: Threshold is configurable +#[test] +fn a6_threshold_configurable() { + let filter_strict = DerivedFilter::new(0.9); + let filter_loose = DerivedFilter::new(0.5); + + assert_eq!(filter_strict.threshold, 0.9); + assert_eq!(filter_loose.threshold, 0.5); +} + +/// Test 7: No manifest is safe (doesn't fail, just no filtering) +#[test] +fn a7_no_manifest_is_safe() { + // Non-existent path should not panic + let filter = DerivedFilter::load_from_jsonl("/nonexistent/path.jsonl", 0.8); + + assert!(filter.is_ok(), "Missing manifest should be safe"); + if let Ok(f) = filter { + assert_eq!(f.artifacts.len(), 0, "Empty manifest should have no artifacts"); + } +} + +/// Test 8: Multiple artifacts (kind + name distinction) +#[test] +fn a8_multiple_artifacts() { + let mut filter = DerivedFilter::new(0.8); + + let skill1 = ArtifactRecord::new( + "skill", + "deploy-prod", + "Deploy to production", + "2025-01-26T10:00:00Z", + ); + let skill2 = ArtifactRecord::new( + "reference", + "kubectl-docs", + "kubectl is a command line tool", + "2025-01-26T10:00:00Z", + ); + + filter.add_artifact(skill1); + filter.add_artifact(skill2); + + assert_eq!(filter.artifacts.len(), 2); + + // Should match first artifact + let result1 = filter.is_derived("Deploy to production"); + assert!(result1.is_some()); + if let Some(m) = result1 { + assert_eq!(m.artifact_name, "deploy-prod"); + } + + // Should match second artifact + let result2 = filter.is_derived("kubectl is a command line tool"); + assert!(result2.is_some()); + if let Some(m) = result2 { + assert_eq!(m.artifact_name, "kubectl-docs"); + } +} + +/// Test 9: Partial overlap below threshold is not excluded +#[test] +fn a9_partial_overlap_below_threshold() { + let mut filter = DerivedFilter::new(0.9); + let artifact = ArtifactRecord::new( + "skill", + "deploy", + "Deploy service to production", + "2025-01-26T10:00:00Z", + ); + filter.add_artifact(artifact); + + // Similar but different text (below 0.9 threshold) + let record = "Deploy service to staging"; + let result = filter.is_derived(record); + + assert!(result.is_none(), "Partial overlap below threshold should not be excluded"); +} + +/// Test 10: Artifact provenance is retained +#[test] +fn a10_artifact_provenance() { + let artifact = ArtifactRecord::new( + "skill", + "my-skill", + "some content", + "2025-01-26T14:32:00Z", + ); + + assert_eq!(artifact.kind, "skill"); + assert_eq!(artifact.name, "my-skill"); + assert_eq!(artifact.emitted_at, "2025-01-26T14:32:00Z"); + assert!(!artifact.sha256.is_empty()); + assert!(!artifact.shingles.is_empty()); +}