Download src/validation/jsonl_validator.py from gitmodelmujtaba/medical-guidelines-kg: direct link, hf CLI and curl.
- Browser
- Download file 7.16 kB
-
https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/validation/jsonl_validator.py
- Command line
-
hf download hf://spaces/gitmodelmujtaba/medical-guidelines-kg/src/validation/jsonl_validator.py
-
curl -L -o jsonl_validator.py https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/validation/jsonl_validator.py
7.16 kB
| """ | |
| JSONL Knowledge Graph Validator. | |
| Validates the generated JSONL triplets for clinical guidelines: | |
| - Schema adherence & field integrity | |
| - Standard medical ontology grounding (SNOMED CT, RxNorm, LOINC) | |
| - Bi-temporal consistency (valid_from, valid_to, status) | |
| - Strict W3C PROV-O provenance completeness (hash, page #, verbatim quote, models) | |
| - Clinical modality & negation alignment | |
| """ | |
| import json | |
| import os | |
| import re | |
| from dataclasses import dataclass, field | |
| from typing import Any, Dict, List, Optional, Set | |
| class ValidationError: | |
| line_number: int | |
| triplet_id: str | |
| field_name: str | |
| error_message: str | |
| severity: str = "ERROR" # "ERROR" or "WARNING" | |
| class ValidationReport: | |
| file_path: str | |
| total_lines: int | |
| valid_lines: int | |
| invalid_lines: int | |
| errors: List[ValidationError] | |
| warnings: List[ValidationError] | |
| vocabulary_breakdown: Dict[str, int] = field(default_factory=dict) | |
| modality_breakdown: Dict[str, int] = field(default_factory=dict) | |
| provenance_completeness_rate: float = 100.0 | |
| def is_valid(self) -> bool: | |
| return self.invalid_lines == 0 and len(self.errors) == 0 | |
| class JSONLKGValidator: | |
| """Validates clinical knowledge graph JSONL files.""" | |
| ALLOWED_VOCABULARIES = {"SNOMED-CT", "RxNorm", "LOINC", "CLINICAL_LOCAL"} | |
| ALLOWED_MODALITIES = {"RECOMMENDED", "CONTRAINDICATED", "CONSIDER", "CAUTION", "MUST_NOT", "MONITORED"} | |
| ALLOWED_STATUSES = {"Active", "Superseded", "Deprecated"} | |
| DATE_REGEX = re.compile(r"^\d{4}(?:-\d{2})?$") | |
| def validate_file(self, jsonl_path: str) -> ValidationReport: | |
| if not os.path.exists(jsonl_path): | |
| raise FileNotFoundError(f"JSONL file not found: {jsonl_path}") | |
| errors: List[ValidationError] = [] | |
| warnings: List[ValidationError] = [] | |
| vocab_counts: Dict[str, int] = {} | |
| modality_counts: Dict[str, int] = {} | |
| total_lines = 0 | |
| valid_lines = 0 | |
| with open(jsonl_path, "r", encoding="utf-8") as f: | |
| for line_idx, line in enumerate(f, start=1): | |
| line = line.strip() | |
| if not line: | |
| continue | |
| total_lines += 1 | |
| # 1. JSON parse check | |
| try: | |
| record = json.loads(line) | |
| except json.JSONDecodeError as e: | |
| errors.append( | |
| ValidationError(line_idx, "UNKNOWN", "JSON_SYNTAX", f"Malformed JSON: {e}") | |
| ) | |
| continue | |
| # 2. Schema and fields check | |
| line_errors, line_warnings = self._validate_record(record, line_idx) | |
| if line_errors: | |
| errors.extend(line_errors) | |
| else: | |
| valid_lines += 1 | |
| warnings.extend(line_warnings) | |
| # Track metrics | |
| modality = record.get("modality", "UNKNOWN") | |
| modality_counts[modality] = modality_counts.get(modality, 0) + 1 | |
| subj_vocab = record.get("subject", {}).get("vocabulary", "UNKNOWN") | |
| obj_vocab = record.get("object", {}).get("vocabulary", "UNKNOWN") | |
| vocab_counts[subj_vocab] = vocab_counts.get(subj_vocab, 0) + 1 | |
| vocab_counts[obj_vocab] = vocab_counts.get(obj_vocab, 0) + 1 | |
| prov_complete = (valid_lines / total_lines * 100.0) if total_lines > 0 else 0.0 | |
| return ValidationReport( | |
| file_path=jsonl_path, | |
| total_lines=total_lines, | |
| valid_lines=valid_lines, | |
| invalid_lines=total_lines - valid_lines, | |
| errors=errors, | |
| warnings=warnings, | |
| vocabulary_breakdown=vocab_counts, | |
| modality_breakdown=modality_counts, | |
| provenance_completeness_rate=round(prov_complete, 2), | |
| ) | |
| def _validate_record(self, record: Dict[str, Any], line_num: int) -> tuple[List[ValidationError], List[ValidationError]]: | |
| errs: List[ValidationError] = [] | |
| warns: List[ValidationError] = [] | |
| t_id = record.get("triplet_id", f"LINE_{line_num}") | |
| # Required top-level keys | |
| required_keys = [ | |
| "triplet_id", "guideline_id", "edition", "publication_date", | |
| "section", "clinical_question", "recommendation_statement", | |
| "evidence_grade", "modality", "assertion", "subject", "predicate", | |
| "object", "temporal", "provenance", "lineage" | |
| ] | |
| for k in required_keys: | |
| if k not in record or record[k] is None: | |
| errs.append(ValidationError(line_num, t_id, k, f"Missing required top-level field: {k}")) | |
| # Modality check | |
| modality = record.get("modality") | |
| if modality not in self.ALLOWED_MODALITIES: | |
| errs.append(ValidationError(line_num, t_id, "modality", f"Invalid modality: {modality}")) | |
| # Subject & Object concept validation | |
| for field_name in ["subject", "object"]: | |
| concept = record.get(field_name, {}) | |
| if not isinstance(concept, dict): | |
| errs.append(ValidationError(line_num, t_id, field_name, f"{field_name} must be an object")) | |
| continue | |
| for ckey in ["mention", "preferred_term", "concept_id", "vocabulary", "category"]: | |
| if not concept.get(ckey): | |
| errs.append(ValidationError(line_num, t_id, f"{field_name}.{ckey}", f"Missing {field_name}.{ckey}")) | |
| vocab = concept.get("vocabulary") | |
| if vocab not in self.ALLOWED_VOCABULARIES: | |
| warns.append(ValidationError(line_num, t_id, f"{field_name}.vocabulary", f"Non-standard vocabulary: {vocab}", severity="WARNING")) | |
| # Temporal check | |
| temporal = record.get("temporal", {}) | |
| v_from = temporal.get("valid_from", "") | |
| if not self.DATE_REGEX.match(v_from): | |
| errs.append(ValidationError(line_num, t_id, "temporal.valid_from", f"Invalid valid_from date format: {v_from}")) | |
| status = temporal.get("status") | |
| if status not in self.ALLOWED_STATUSES: | |
| errs.append(ValidationError(line_num, t_id, "temporal.status", f"Invalid status: {status}")) | |
| if status == "Superseded" and not temporal.get("valid_to"): | |
| errs.append(ValidationError(line_num, t_id, "temporal.valid_to", "Superseded triplet must have non-null valid_to")) | |
| # Provenance check (W3C PROV-O compliance) | |
| prov = record.get("provenance", {}) | |
| if not prov.get("source_file"): | |
| errs.append(ValidationError(line_num, t_id, "provenance.source_file", "Missing source_file")) | |
| if not prov.get("file_sha256"): | |
| errs.append(ValidationError(line_num, t_id, "provenance.file_sha256", "Missing file SHA-256 hash")) | |
| if not prov.get("verbatim_quote"): | |
| errs.append(ValidationError(line_num, t_id, "provenance.verbatim_quote", "Missing verbatim quotation")) | |
| if prov.get("page_number", 0) < 1: | |
| errs.append(ValidationError(line_num, t_id, "provenance.page_number", "Invalid page number")) | |
| return errs, warns | |