""" Semantic Delta Diffing Engine for Guideline Versions. Compares recommendations across guideline editions (V_prev vs V_curr) and classifies them into UNCHANGED, AMENDED, NEW, and DEPRECATED. Ensures that only changed content triggers downstream extraction and graph updates. """ import difflib import re from dataclasses import dataclass from enum import Enum from typing import Any, Dict, List, Optional, Set, Tuple from src.chunking.clinical_chunker import ClinicalRecommendationBlock from src.graph.guideline_kg import GuidelineKnowledgeGraph class DeltaAction(str, Enum): UNCHANGED = "UNCHANGED" AMENDED = "AMENDED" NEW = "NEW" DEPRECATED = "DEPRECATED" @dataclass class DeltaDiffResult: guideline_id: str from_edition: Optional[int] to_edition: int unchanged: List[Tuple[ClinicalRecommendationBlock, str]] # (block, old_node_id) amended: List[Tuple[ClinicalRecommendationBlock, str]] # (block, old_node_id) new_blocks: List[ClinicalRecommendationBlock] deprecated_node_ids: List[str] @property def summary(self) -> Dict[str, int]: return { "unchanged_count": len(self.unchanged), "amended_count": len(self.amended), "new_count": len(self.new_blocks), "deprecated_count": len(self.deprecated_node_ids), } class GuidelineDeltaEngine: """Computes semantic delta between previous guideline KG and new recommendation blocks.""" def __init__(self, similarity_threshold_exact: float = 0.92, similarity_threshold_amended: float = 0.50): self.similarity_threshold_exact = similarity_threshold_exact self.similarity_threshold_amended = similarity_threshold_amended def compute_delta( self, incoming_blocks: List[ClinicalRecommendationBlock], incoming_edition: int, previous_kg: Optional[GuidelineKnowledgeGraph] = None, ) -> DeltaDiffResult: guideline_id = previous_kg.guideline_id if previous_kg else "UNKNOWN" # Case 1: First Ingestion (No previous KG) if previous_kg is None or len(previous_kg.graph) == 0: return DeltaDiffResult( guideline_id=guideline_id, from_edition=None, to_edition=incoming_edition, unchanged=[], amended=[], new_blocks=incoming_blocks, deprecated_node_ids=[], ) # Case 2: Incremental Ingestion against previous active recommendations active_prev_recs = previous_kg.get_active_recommendations() matched_prev_ids: Set[str] = set() unchanged: List[Tuple[ClinicalRecommendationBlock, str]] = [] amended: List[Tuple[ClinicalRecommendationBlock, str]] = [] new_blocks: List[ClinicalRecommendationBlock] = [] from_edition = None if active_prev_recs: from_edition = active_prev_recs[0].get("edition", incoming_edition - 1) for block in incoming_blocks: # 1. Explicit New tag check (e.g. "[New 2015]") if block.revision_tag and "new" in block.revision_tag.lower(): new_blocks.append(block) continue # 2. Match against active previous recommendations best_match_id = None best_score = 0.0 best_old_data = None norm_new_stmt = self._normalize_text(block.statement) norm_new_q = self._normalize_text(block.clinical_question) for prev_rec in active_prev_recs: p_id = prev_rec["node_id"] if p_id in matched_prev_ids: continue # already paired norm_old_stmt = self._normalize_text(prev_rec.get("statement", "")) norm_old_q = self._normalize_text(prev_rec.get("clinical_question", "")) # Compute text similarity stmt_sim = self._calculate_similarity(norm_new_stmt, norm_old_stmt) # Question similarity bonus if identical or very close q_sim = self._calculate_similarity(norm_new_q, norm_old_q) total_score = 0.75 * stmt_sim + 0.25 * q_sim if total_score > best_score: best_score = total_score best_match_id = p_id best_old_data = prev_rec # 3. Classify based on similarity score if best_score >= self.similarity_threshold_exact: # Same text & question. Check evidence grade: old_grade = best_old_data.get("evidence_grade", "") if old_grade == block.evidence_grade: unchanged.append((block, best_match_id)) else: # Grade changed (e.g. Grade C -> Grade B) amended.append((block, best_match_id)) matched_prev_ids.add(best_match_id) elif best_score >= self.similarity_threshold_amended: # Wording updated, criteria refined, or amended amended.append((block, best_match_id)) matched_prev_ids.add(best_match_id) else: # No close match in previous version new_blocks.append(block) # 4. Check for deprecated recommendations (in previous version but omitted in current) deprecated_ids = [ prev_rec["node_id"] for prev_rec in active_prev_recs if prev_rec["node_id"] not in matched_prev_ids ] return DeltaDiffResult( guideline_id=guideline_id, from_edition=from_edition, to_edition=incoming_edition, unchanged=unchanged, amended=amended, new_blocks=new_blocks, deprecated_node_ids=deprecated_ids, ) def _calculate_similarity(self, s1: str, s2: str) -> float: if not s1 or not s2: return 0.0 matcher = difflib.SequenceMatcher(None, s1, s2) return matcher.ratio() def _normalize_text(self, text: str) -> str: text = text.lower() text = re.sub(r"[^\w\s]", " ", text) return " ".join(text.split())