Download src/diffing/delta_engine.py from gitmodelmujtaba/medical-guidelines-kg: direct link, hf CLI and curl.
- Browser
- Download file 6.23 kB
-
https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/diffing/delta_engine.py
- Command line
-
hf download hf://spaces/gitmodelmujtaba/medical-guidelines-kg/src/diffing/delta_engine.py
-
curl -L -o delta_engine.py https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/diffing/delta_engine.py
6.23 kB
| """ | |
| Semantic Delta Diffing Engine for Guideline Versions. | |
| Compares recommendations across guideline editions (V_prev vs V_curr) | |
| and classifies them into UNCHANGED, AMENDED, NEW, and DEPRECATED. | |
| Ensures that only changed content triggers downstream extraction and graph updates. | |
| """ | |
| import difflib | |
| import re | |
| from dataclasses import dataclass | |
| from enum import Enum | |
| from typing import Any, Dict, List, Optional, Set, Tuple | |
| from src.chunking.clinical_chunker import ClinicalRecommendationBlock | |
| from src.graph.guideline_kg import GuidelineKnowledgeGraph | |
| class DeltaAction(str, Enum): | |
| UNCHANGED = "UNCHANGED" | |
| AMENDED = "AMENDED" | |
| NEW = "NEW" | |
| DEPRECATED = "DEPRECATED" | |
| class DeltaDiffResult: | |
| guideline_id: str | |
| from_edition: Optional[int] | |
| to_edition: int | |
| unchanged: List[Tuple[ClinicalRecommendationBlock, str]] # (block, old_node_id) | |
| amended: List[Tuple[ClinicalRecommendationBlock, str]] # (block, old_node_id) | |
| new_blocks: List[ClinicalRecommendationBlock] | |
| deprecated_node_ids: List[str] | |
| def summary(self) -> Dict[str, int]: | |
| return { | |
| "unchanged_count": len(self.unchanged), | |
| "amended_count": len(self.amended), | |
| "new_count": len(self.new_blocks), | |
| "deprecated_count": len(self.deprecated_node_ids), | |
| } | |
| class GuidelineDeltaEngine: | |
| """Computes semantic delta between previous guideline KG and new recommendation blocks.""" | |
| def __init__(self, similarity_threshold_exact: float = 0.92, similarity_threshold_amended: float = 0.50): | |
| self.similarity_threshold_exact = similarity_threshold_exact | |
| self.similarity_threshold_amended = similarity_threshold_amended | |
| def compute_delta( | |
| self, | |
| incoming_blocks: List[ClinicalRecommendationBlock], | |
| incoming_edition: int, | |
| previous_kg: Optional[GuidelineKnowledgeGraph] = None, | |
| ) -> DeltaDiffResult: | |
| guideline_id = previous_kg.guideline_id if previous_kg else "UNKNOWN" | |
| # Case 1: First Ingestion (No previous KG) | |
| if previous_kg is None or len(previous_kg.graph) == 0: | |
| return DeltaDiffResult( | |
| guideline_id=guideline_id, | |
| from_edition=None, | |
| to_edition=incoming_edition, | |
| unchanged=[], | |
| amended=[], | |
| new_blocks=incoming_blocks, | |
| deprecated_node_ids=[], | |
| ) | |
| # Case 2: Incremental Ingestion against previous active recommendations | |
| active_prev_recs = previous_kg.get_active_recommendations() | |
| matched_prev_ids: Set[str] = set() | |
| unchanged: List[Tuple[ClinicalRecommendationBlock, str]] = [] | |
| amended: List[Tuple[ClinicalRecommendationBlock, str]] = [] | |
| new_blocks: List[ClinicalRecommendationBlock] = [] | |
| from_edition = None | |
| if active_prev_recs: | |
| from_edition = active_prev_recs[0].get("edition", incoming_edition - 1) | |
| for block in incoming_blocks: | |
| # 1. Explicit New tag check (e.g. "[New 2015]") | |
| if block.revision_tag and "new" in block.revision_tag.lower(): | |
| new_blocks.append(block) | |
| continue | |
| # 2. Match against active previous recommendations | |
| best_match_id = None | |
| best_score = 0.0 | |
| best_old_data = None | |
| norm_new_stmt = self._normalize_text(block.statement) | |
| norm_new_q = self._normalize_text(block.clinical_question) | |
| for prev_rec in active_prev_recs: | |
| p_id = prev_rec["node_id"] | |
| if p_id in matched_prev_ids: | |
| continue # already paired | |
| norm_old_stmt = self._normalize_text(prev_rec.get("statement", "")) | |
| norm_old_q = self._normalize_text(prev_rec.get("clinical_question", "")) | |
| # Compute text similarity | |
| stmt_sim = self._calculate_similarity(norm_new_stmt, norm_old_stmt) | |
| # Question similarity bonus if identical or very close | |
| q_sim = self._calculate_similarity(norm_new_q, norm_old_q) | |
| total_score = 0.75 * stmt_sim + 0.25 * q_sim | |
| if total_score > best_score: | |
| best_score = total_score | |
| best_match_id = p_id | |
| best_old_data = prev_rec | |
| # 3. Classify based on similarity score | |
| if best_score >= self.similarity_threshold_exact: | |
| # Same text & question. Check evidence grade: | |
| old_grade = best_old_data.get("evidence_grade", "") | |
| if old_grade == block.evidence_grade: | |
| unchanged.append((block, best_match_id)) | |
| else: | |
| # Grade changed (e.g. Grade C -> Grade B) | |
| amended.append((block, best_match_id)) | |
| matched_prev_ids.add(best_match_id) | |
| elif best_score >= self.similarity_threshold_amended: | |
| # Wording updated, criteria refined, or amended | |
| amended.append((block, best_match_id)) | |
| matched_prev_ids.add(best_match_id) | |
| else: | |
| # No close match in previous version | |
| new_blocks.append(block) | |
| # 4. Check for deprecated recommendations (in previous version but omitted in current) | |
| deprecated_ids = [ | |
| prev_rec["node_id"] | |
| for prev_rec in active_prev_recs | |
| if prev_rec["node_id"] not in matched_prev_ids | |
| ] | |
| return DeltaDiffResult( | |
| guideline_id=guideline_id, | |
| from_edition=from_edition, | |
| to_edition=incoming_edition, | |
| unchanged=unchanged, | |
| amended=amended, | |
| new_blocks=new_blocks, | |
| deprecated_node_ids=deprecated_ids, | |
| ) | |
| def _calculate_similarity(self, s1: str, s2: str) -> float: | |
| if not s1 or not s2: | |
| return 0.0 | |
| matcher = difflib.SequenceMatcher(None, s1, s2) | |
| return matcher.ratio() | |
| def _normalize_text(self, text: str) -> str: | |
| text = text.lower() | |
| text = re.sub(r"[^\w\s]", " ", text) | |
| return " ".join(text.split()) | |