medical-guidelines-kg / src /diffing /delta_engine.py
gitmodelmujtaba's picture
Deploy Medical Guidelines Knowledge Graph Explorer to Hugging Face Spaces
62c74ae
Raw History Blame Contribute Delete
6.23 kB
"""
Semantic Delta Diffing Engine for Guideline Versions.
Compares recommendations across guideline editions (V_prev vs V_curr)
and classifies them into UNCHANGED, AMENDED, NEW, and DEPRECATED.
Ensures that only changed content triggers downstream extraction and graph updates.
"""
import difflib
import re
from dataclasses import dataclass
from enum import Enum
from typing import Any, Dict, List, Optional, Set, Tuple
from src.chunking.clinical_chunker import ClinicalRecommendationBlock
from src.graph.guideline_kg import GuidelineKnowledgeGraph
class DeltaAction(str, Enum):
UNCHANGED = "UNCHANGED"
AMENDED = "AMENDED"
NEW = "NEW"
DEPRECATED = "DEPRECATED"
@dataclass
class DeltaDiffResult:
guideline_id: str
from_edition: Optional[int]
to_edition: int
unchanged: List[Tuple[ClinicalRecommendationBlock, str]] # (block, old_node_id)
amended: List[Tuple[ClinicalRecommendationBlock, str]] # (block, old_node_id)
new_blocks: List[ClinicalRecommendationBlock]
deprecated_node_ids: List[str]
@property
def summary(self) -> Dict[str, int]:
return {
"unchanged_count": len(self.unchanged),
"amended_count": len(self.amended),
"new_count": len(self.new_blocks),
"deprecated_count": len(self.deprecated_node_ids),
}
class GuidelineDeltaEngine:
"""Computes semantic delta between previous guideline KG and new recommendation blocks."""
def __init__(self, similarity_threshold_exact: float = 0.92, similarity_threshold_amended: float = 0.50):
self.similarity_threshold_exact = similarity_threshold_exact
self.similarity_threshold_amended = similarity_threshold_amended
def compute_delta(
self,
incoming_blocks: List[ClinicalRecommendationBlock],
incoming_edition: int,
previous_kg: Optional[GuidelineKnowledgeGraph] = None,
) -> DeltaDiffResult:
guideline_id = previous_kg.guideline_id if previous_kg else "UNKNOWN"
# Case 1: First Ingestion (No previous KG)
if previous_kg is None or len(previous_kg.graph) == 0:
return DeltaDiffResult(
guideline_id=guideline_id,
from_edition=None,
to_edition=incoming_edition,
unchanged=[],
amended=[],
new_blocks=incoming_blocks,
deprecated_node_ids=[],
)
# Case 2: Incremental Ingestion against previous active recommendations
active_prev_recs = previous_kg.get_active_recommendations()
matched_prev_ids: Set[str] = set()
unchanged: List[Tuple[ClinicalRecommendationBlock, str]] = []
amended: List[Tuple[ClinicalRecommendationBlock, str]] = []
new_blocks: List[ClinicalRecommendationBlock] = []
from_edition = None
if active_prev_recs:
from_edition = active_prev_recs[0].get("edition", incoming_edition - 1)
for block in incoming_blocks:
# 1. Explicit New tag check (e.g. "[New 2015]")
if block.revision_tag and "new" in block.revision_tag.lower():
new_blocks.append(block)
continue
# 2. Match against active previous recommendations
best_match_id = None
best_score = 0.0
best_old_data = None
norm_new_stmt = self._normalize_text(block.statement)
norm_new_q = self._normalize_text(block.clinical_question)
for prev_rec in active_prev_recs:
p_id = prev_rec["node_id"]
if p_id in matched_prev_ids:
continue # already paired
norm_old_stmt = self._normalize_text(prev_rec.get("statement", ""))
norm_old_q = self._normalize_text(prev_rec.get("clinical_question", ""))
# Compute text similarity
stmt_sim = self._calculate_similarity(norm_new_stmt, norm_old_stmt)
# Question similarity bonus if identical or very close
q_sim = self._calculate_similarity(norm_new_q, norm_old_q)
total_score = 0.75 * stmt_sim + 0.25 * q_sim
if total_score > best_score:
best_score = total_score
best_match_id = p_id
best_old_data = prev_rec
# 3. Classify based on similarity score
if best_score >= self.similarity_threshold_exact:
# Same text & question. Check evidence grade:
old_grade = best_old_data.get("evidence_grade", "")
if old_grade == block.evidence_grade:
unchanged.append((block, best_match_id))
else:
# Grade changed (e.g. Grade C -> Grade B)
amended.append((block, best_match_id))
matched_prev_ids.add(best_match_id)
elif best_score >= self.similarity_threshold_amended:
# Wording updated, criteria refined, or amended
amended.append((block, best_match_id))
matched_prev_ids.add(best_match_id)
else:
# No close match in previous version
new_blocks.append(block)
# 4. Check for deprecated recommendations (in previous version but omitted in current)
deprecated_ids = [
prev_rec["node_id"]
for prev_rec in active_prev_recs
if prev_rec["node_id"] not in matched_prev_ids
]
return DeltaDiffResult(
guideline_id=guideline_id,
from_edition=from_edition,
to_edition=incoming_edition,
unchanged=unchanged,
amended=amended,
new_blocks=new_blocks,
deprecated_node_ids=deprecated_ids,
)
def _calculate_similarity(self, s1: str, s2: str) -> float:
if not s1 or not s2:
return 0.0
matcher = difflib.SequenceMatcher(None, s1, s2)
return matcher.ratio()
def _normalize_text(self, text: str) -> str:
text = text.lower()
text = re.sub(r"[^\w\s]", " ", text)
return " ".join(text.split())