Files
poimen-memory/crates/mem-ingest/src/entity_extractor.rs
T

264 lines
8.4 KiB
Rust
Raw Normal View History

//! Entity extraction: LLM-based with reflection verification + fallback
//!
//! Three-stage extraction:
//! 1. Initial LLM extraction (entities + types + summaries)
//! 2. Reflection verification (confirm entities exist in text)
//! 3. Fallback to wiki_links if LLM fails
//!
//! CRAP: 18 (LLM complexity + hallucination risk; Mitigations: reflection + fallback)
//! SOLID: Trait-based (Open/Closed), DependencyInversion (LLM abstraction)
//! DRY: Shares EntityType from Phase 1
use anyhow::Result;
use async_trait::async_trait;
use mem_core::entity::{Entity, EntityType};
use serde::{Deserialize, Serialize};
/// Extracted entity from LLM (intermediate representation)
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ExtractedEntity {
pub name: String,
pub entity_type: EntityType,
pub summary: String,
pub confidence: f32,
}
impl ExtractedEntity {
/// Convert to domain model (Phase 1 type)
pub fn to_domain(&self, project_id: &str) -> Entity {
Entity::new(project_id, &self.name, self.entity_type)
.with_summary(&self.summary)
}
}
/// Entity extractor trait - pluggable implementations
/// Three implementations: LLM, WikiLink fallback, Composite
#[async_trait]
pub trait EntityExtractor: Send + Sync {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>>;
}
/// LLM-based extractor with reflection verification (stage 1 + 2)
pub struct LlmEntityExtractor {
model_name: String,
enable_reflection: bool,
}
impl LlmEntityExtractor {
pub fn new(model_name: &str) -> Self {
Self {
model_name: model_name.to_string(),
enable_reflection: true,
}
}
/// Parse extraction response JSON
/// Format: { "entities": [{ "name": "...", "type": "...", "summary": "..." }, ...] }
fn parse_extraction(response: &str) -> Result<Vec<ExtractedEntity>> {
#[derive(Deserialize)]
struct Response {
entities: Vec<ExtractedEntity>,
}
let parsed: Response = serde_json::from_str(response)?;
Ok(parsed.entities)
}
/// Parse reflection response JSON
/// Format: { "verified": [{ "name": "...", "present": true/false }, ...] }
fn parse_reflection(response: &str) -> Result<Vec<(String, bool)>> {
#[derive(Deserialize)]
struct Verified {
name: String,
present: bool,
}
#[derive(Deserialize)]
struct ReflectionResponse {
verified: Vec<Verified>,
}
let parsed: ReflectionResponse = serde_json::from_str(response)?;
Ok(parsed.verified.into_iter().map(|v| (v.name, v.present)).collect())
}
/// Mock LLM call - replace with real API in production
/// TODO (Phase 2.6): Integrate with api.riotpiao.com/v1/chat/completions
/// TODO (Phase 2.6): Add JWT authentication from Authentik OIDC
async fn simulate_llm(&self, _prompt: &str) -> Result<String> {
// Production: call api.riotpiao.com with Bearer JWT token
// Mock response for testing
Ok(r#"{
"entities": [
{"name": "Rock", "type": "person", "summary": "SRE engineer", "confidence": 0.95},
{"name": "Kubernetes", "type": "tool", "summary": "Container orchestration", "confidence": 0.98}
]
}"#
.to_string())
}
}
#[async_trait]
impl EntityExtractor for LlmEntityExtractor {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>> {
// Stage 1: Extract entities
let prompt = format!(
r#"Extract named entities from this text.
For each entity provide:
- name: Canonical name (proper capitalization)
- type: One of [person, tool, concept, location, event, organization]
- summary: One sentence
CRITICAL: Only extract entities EXPLICITLY mentioned. No inference.
Text:
"{}"
Respond in JSON:
{{"entities": [{{"name": "...", "type": "...", "summary": "..."}}, ...]}}
"#,
text
);
let extraction_response = self.simulate_llm(&prompt).await?;
let mut entities = Self::parse_extraction(&extraction_response)?;
// Stage 2: Reflection verification (filter hallucinations)
if self.enable_reflection {
let reflection_prompt = format!(
r#"Verify these entities are explicitly in the text:
Text:
"{}"
Entities:
{:?}
Respond in JSON:
{{"verified": [{{"name": "...", "present": true/false}}, ...]}}
"#,
text, entities
);
let reflection = self.simulate_llm(&reflection_prompt).await?;
let verified = Self::parse_reflection(&reflection)?;
// Filter: keep only entities marked present
entities.retain(|e| verified.iter().any(|(name, present)| name == &e.name && *present));
// Adjust confidence for reflected entities (slight penalty for needing verification)
for entity in &mut entities {
entity.confidence *= 0.95;
}
}
Ok(entities)
}
}
/// Fallback extractor: Use wiki_links if LLM fails (stage 3)
pub struct WikiLinkFallbackExtractor;
#[async_trait]
impl EntityExtractor for WikiLinkFallbackExtractor {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>> {
// Extract [[wiki_link]] patterns from text
let mut entities = vec![];
let re = regex::Regex::new(r"\[\[([^\]]+)\]\]")?;
for cap in re.captures_iter(text) {
if let Some(name) = cap.get(1) {
let name_str = name.as_str();
entities.push(ExtractedEntity {
name: name_str.to_string(),
entity_type: EntityType::Unknown,
summary: format!("Mentioned in episode"),
confidence: 0.7, // Lower confidence for fallback
});
}
}
Ok(entities)
}
}
/// Composite extractor: LLM first, fallback to wiki_links (all stages)
pub struct CompositeEntityExtractor {
primary: Box<dyn EntityExtractor>,
fallback: Box<dyn EntityExtractor>,
}
impl CompositeEntityExtractor {
pub fn new(primary: Box<dyn EntityExtractor>, fallback: Box<dyn EntityExtractor>) -> Self {
Self { primary, fallback }
}
/// Default: LLM with wiki_links fallback
pub fn default_llm() -> Self {
Self::new(
Box::new(LlmEntityExtractor::new("reasoning")),
Box::new(WikiLinkFallbackExtractor),
)
}
}
#[async_trait]
impl EntityExtractor for CompositeEntityExtractor {
async fn extract(&self, text: &str) -> Result<Vec<ExtractedEntity>> {
match self.primary.extract(text).await {
Ok(entities) if !entities.is_empty() => {
tracing::debug!("LLM extraction succeeded: {} entities", entities.len());
Ok(entities)
}
Ok(_) => {
tracing::warn!("LLM extraction returned empty, using fallback");
self.fallback.extract(text).await
}
Err(e) => {
tracing::warn!("LLM extraction failed: {}, using fallback", e);
self.fallback.extract(text).await
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[tokio::test]
async fn test_wiki_link_extraction() {
let extractor = WikiLinkFallbackExtractor;
let text = "Rock uses [[Kubernetes]] and [[ArgoCD]] for GitOps";
let entities = extractor.extract(text).await.unwrap();
assert_eq!(entities.len(), 2);
assert!(entities.iter().any(|e| e.name == "Kubernetes"));
assert!(entities.iter().any(|e| e.name == "ArgoCD"));
}
#[tokio::test]
async fn test_extracted_entity_to_domain() {
let extracted = ExtractedEntity {
name: "Test Entity".to_string(),
entity_type: EntityType::Tool,
summary: "A test entity".to_string(),
confidence: 0.95,
};
let domain = extracted.to_domain("proj1");
assert_eq!(domain.name, "Test Entity");
assert_eq!(domain.entity_type, EntityType::Tool);
}
#[tokio::test]
async fn test_composite_fallback() {
let primary = Box::new(WikiLinkFallbackExtractor);
let fallback = Box::new(WikiLinkFallbackExtractor);
let composite = CompositeEntityExtractor::new(primary, fallback);
let text = "[[Entity1]] and [[Entity2]]";
let entities = composite.extract(text).await.unwrap();
assert!(entities.len() > 0);
}
}