Spaces:
Sleeping
Sleeping
File size: 10,225 Bytes
c99d3a1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 | """
Human+ PDF Processor
Extracts biomarkers from lab PDF using Gemini vision.
Returns structured list of biomarker dicts ready for UI rendering.
"""
import json
import re
from typing import Optional
from core.ai_client import call_gemini, pdf_bytes_to_part, TEMP_EXTRACTION
# ββ Biomarker extraction prompt ββββββββββββββββββββββββββββββββββββββ
_EXTRACTION_PROMPT = """
Kamu adalah AI lab analyst dari Human+ Bali.
Tugasmu: Ekstrak SEMUA biomarker dari hasil lab PDF ini.
PENTING β Kembalikan HANYA valid JSON array, tanpa teks lain, tanpa markdown code block.
Format setiap item:
{
"name": "Nama biomarker (gunakan nama standar, contoh: hs-CRP, Vitamin D, HbA1c)",
"value": "Nilai numerik sebagai string (contoh: '2.4', '22', '5.6')",
"unit": "Satuan (contoh: mg/L, ng/mL, %)",
"reference_range": "Range referensi dari lab (contoh: 0.0-5.0 mg/L)",
"lab_flag": "normal | high | low | critical_high | critical_low"
}
Prioritas biomarker (wajib cari jika ada):
- Inflammation: hs-CRP, Homocysteine, Ferritin
- Metabolic: HbA1c, Fasting Glucose, Fasting Insulin
- Hormones: Total Testosterone, Free Testosterone, Vitamin D, DHEA-S, Cortisol, TSH
- Nutrients: Vitamin B12, Magnesium, Zinc, Folate, Vitamin B6
- Cardiovascular: ApoB, Triglycerides, HDL, LDL, Total Cholesterol
- Blood: Hemoglobin, Hematocrit, WBC, Platelets
- Liver: ALT, AST, GGT, Bilirubin
- Kidney: Creatinine, eGFR, Uric Acid
Jika nilai tidak terbaca atau tidak ada, SKIP biomarker tersebut.
Kembalikan HANYA JSON array, tidak ada teks lain.
"""
# ββ Human+ optimal ranges (untuk status mapping) βββββββββββββββββββββ
_OPTIMAL_RANGES: dict[str, dict] = {
"hs-crp": {"optimal": (None, 1.0), "warning": (1.0, 3.0)},
"homocysteine": {"optimal": (None, 8.0), "warning": (8.0, 12.0)},
"ferritin_m": {"optimal": (50, 150), "warning": (30, 50)}, # pria
"ferritin_f": {"optimal": (30, 100), "warning": (15, 30)}, # wanita
"hba1c": {"optimal": (4.8, 5.2), "warning": (5.2, 5.6)},
"fasting glucose": {"optimal": (75, 86), "warning": (70, 100)},
"fasting insulin": {"optimal": (None, 7.0), "warning": (7.0, 10.0)},
"total testosterone": {"optimal": (600, None), "warning": (400, 600)},
"vitamin d": {"optimal": (50, 80), "warning": (30, 50)},
"dhea-s": {"optimal": None, "warning": None}, # age-dependent
"vitamin b12": {"optimal": (500, None), "warning": (300, 500)},
"magnesium": {"optimal": (2.2, None), "warning": (1.8, 2.2)},
"zinc": {"optimal": (90, 120), "warning": (70, 90)},
"folate": {"optimal": (10, None), "warning": (5, 10)},
"apob": {"optimal": (None, 80), "warning": (80, 100)},
"triglycerides": {"optimal": (None, 90), "warning": (90, 150)},
"hdl": {"optimal": (60, None), "warning": (40, 60)},
"tsh": {"optimal": (0.5, 2.0), "warning": (2.0, 4.0)},
}
class PDFProcessorError(Exception):
"""Raised when PDF extraction fails."""
pass
def extract_biomarkers(pdf_bytes: bytes) -> list[dict]:
"""
Extract biomarkers dari PDF bytes menggunakan Gemini.
Args:
pdf_bytes: raw PDF file bytes
Returns:
list of biomarker dicts dengan keys:
- name, value, unit, status, reference (Human+ optimal range)
- raw_value, lab_flag, reference_range (dari lab asli)
Raises:
PDFProcessorError: jika extraction gagal
"""
# 1. Call Gemini dengan PDF
raw_json = _call_gemini_extraction(pdf_bytes)
# 2. Parse JSON response
raw_biomarkers = _parse_json_response(raw_json)
# 3. Map ke Human+ status (optimal/warning/danger)
enriched = [_enrich_biomarker(b) for b in raw_biomarkers]
# 4. Filter yang invalid, sort by status priority
valid = [b for b in enriched if b is not None]
return _sort_by_priority(valid)
def get_raw_extraction_text(pdf_bytes: bytes) -> str:
"""
Return raw Gemini extraction text (untuk ditampilkan di expander debug).
Useful untuk troubleshooting jika parsing gagal.
"""
return _call_gemini_extraction(pdf_bytes)
# ββ Private helpers ββββββββββββββββββββββββββββββββββββββββββββββββββ
def _call_gemini_extraction(pdf_bytes: bytes) -> str:
"""Send PDF to Gemini and return raw text response."""
try:
pdf_part = pdf_bytes_to_part(pdf_bytes)
response = call_gemini(
contents=[_EXTRACTION_PROMPT, pdf_part],
temperature=TEMP_EXTRACTION,
)
return response
except Exception as e:
raise PDFProcessorError(f"Gemini extraction gagal: {e}") from e
def _parse_json_response(raw: str) -> list[dict]:
"""
Parse JSON dari Gemini response.
Gemini kadang return ```json ... ``` atau teks tambahan β kita strip dulu.
"""
# Strip markdown code fences jika ada
cleaned = re.sub(r"```(?:json)?", "", raw).strip().rstrip("```").strip()
# Coba parse langsung
try:
data = json.loads(cleaned)
if isinstance(data, list):
return data
# Kadang Gemini wrap dalam object
if isinstance(data, dict):
for key in ["biomarkers", "results", "data", "markers"]:
if key in data and isinstance(data[key], list):
return data[key]
except json.JSONDecodeError:
pass
# Fallback: cari array JSON di dalam teks
match = re.search(r"\[[\s\S]*\]", cleaned)
if match:
try:
return json.loads(match.group())
except json.JSONDecodeError:
pass
raise PDFProcessorError(
"Gagal parse JSON dari Gemini. "
"PDF mungkin tidak terbaca dengan baik atau format tidak standard."
)
def _enrich_biomarker(raw: dict) -> Optional[dict]:
"""
Enrich raw biomarker dict dengan Human+ status mapping.
Input keys: name, value, unit, reference_range, lab_flag
Output adds: status ('optimal'|'warning'|'danger'|'neutral'), reference (Human+)
"""
try:
name = str(raw.get("name", "")).strip()
value = str(raw.get("value", "")).strip()
unit = str(raw.get("unit", "")).strip()
if not name or not value:
return None
# Parse numeric value
try:
numeric_value = float(re.sub(r"[^\d.\-]", "", value))
except (ValueError, TypeError):
numeric_value = None
# Map status
status, human_plus_ref = _map_to_human_plus_status(
name, numeric_value, raw.get("lab_flag", "normal")
)
# Gunakan Human+ target jika tersedia, fallback ke reference_range dari lab
display_reference = human_plus_ref if human_plus_ref else raw.get("reference_range", "")
return {
# For UI rendering
"name": name,
"value": value,
"unit": unit,
"status": status,
"reference": display_reference,
# Raw lab data (untuk expander / debug)
"raw_value": numeric_value,
"lab_flag": raw.get("lab_flag", "normal"),
"reference_range": raw.get("reference_range", ""),
"human_plus_ref": human_plus_ref, # Human+ target khusus (bisa berbeda)
}
except Exception:
return None
def _map_to_human_plus_status(
name: str,
value: Optional[float],
lab_flag: str,
) -> tuple[str, str]:
"""
Map biomarker to Human+ optimal status.
Returns:
(status, human_plus_reference_string)
status: 'optimal' | 'warning' | 'danger' | 'neutral'
"""
# Normalisasi: hapus teks dalam kurung, lowercase
# contoh: "hs-CRP (High-sensitivity C-Reactive Protein)" β "hs-crp"
name_lower = re.sub(r"\s*\(.*?\)", "", name.lower()).strip()
# Find matching range config
range_cfg = None
for key, cfg in _OPTIMAL_RANGES.items():
if key == name_lower or key in name_lower or name_lower in key:
range_cfg = cfg
break
# No config found β fallback ke lab_flag
if range_cfg is None or value is None:
if lab_flag in ("normal",):
return "neutral", ""
elif lab_flag in ("high", "low"):
return "warning", ""
elif lab_flag in ("critical_high", "critical_low"):
return "danger", ""
return "neutral", ""
# Check optimal range
optimal = range_cfg.get("optimal")
warning = range_cfg.get("warning")
if optimal:
lo, hi = optimal
in_optimal = (
(lo is None or value >= lo) and
(hi is None or value <= hi)
)
if in_optimal:
ref_str = _format_range_str(lo, hi)
return "optimal", ref_str
if warning:
lo, hi = warning
in_warning = (
(lo is None or value >= lo) and
(hi is None or value <= hi)
)
if in_warning:
opt_lo, opt_hi = (optimal or (None, None))
ref_str = _format_range_str(opt_lo, opt_hi)
return "warning", ref_str
# Below all ranges or above all ranges β danger
opt_lo, opt_hi = (optimal or (None, None))
ref_str = _format_range_str(opt_lo, opt_hi)
return "danger", ref_str
def _format_range_str(lo: Optional[float], hi: Optional[float]) -> str:
"""Format optimal range sebagai human-readable string."""
if lo is not None and hi is not None:
return f"{lo}β{hi}"
elif lo is not None:
return f"> {lo}"
elif hi is not None:
return f"< {hi}"
return ""
def _sort_by_priority(biomarkers: list[dict]) -> list[dict]:
"""Sort: danger first, then warning, then optimal, then neutral."""
priority = {"danger": 0, "warning": 1, "optimal": 2, "neutral": 3}
return sorted(biomarkers, key=lambda b: priority.get(b["status"], 3))
|