import base64 import hashlib import json import os import re import tempfile import uuid import zipfile from datetime import UTC, datetime from io import BytesIO from pathlib import Path import cv2 import gradio as gr import numpy as np import requests from detection import ( DETECTION_DPI, detect_rectangles, draw_detections, get_model_version, pt_to_px, px_to_pt, render_pdf_page, ) from PIL import Image # Fix gradio_client 5.9.1 crash when schema is a bool try: import gradio_client.utils as _gcu _orig_get_type = _gcu.get_type def _safe_get_type(schema): if not isinstance(schema, dict): return "Any" return _orig_get_type(schema) _gcu.get_type = _safe_get_type _orig_j2p = _gcu._json_schema_to_python_type def _safe_j2p(schema, defs=None): if not isinstance(schema, dict): return "Any" try: return _orig_j2p(schema, defs) except (TypeError, AttributeError): return "Any" _gcu._json_schema_to_python_type = _safe_j2p _orig_j2p_pub = _gcu.json_schema_to_python_type def _safe_j2p_pub(schema): try: return _orig_j2p_pub(schema) except (TypeError, AttributeError): return "Any" _gcu.json_schema_to_python_type = _safe_j2p_pub except Exception: pass try: import fitz PDF_SUPPORT = True except ImportError: PDF_SUPPORT = False try: import pandas as pd PANDAS_SUPPORT = True except ImportError: PANDAS_SUPPORT = False def _pil_to_bgr(pil_img): return cv2.cvtColor(np.array(pil_img.convert("RGB")), cv2.COLOR_RGB2BGR) def _render_pdf_page(pdf_path, page_number, dpi=100): if not PDF_SUPPORT: return None try: doc = fitz.open(pdf_path) idx = max(0, min(page_number - 1, len(doc) - 1)) page = doc[idx] mat = fitz.Matrix(dpi / 72, dpi / 72) pix = page.get_pixmap(matrix=mat, colorspace=fitz.csRGB) return np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width, 3) except Exception as exc: print(f"[app] PDF render error: {exc}") return None def _count_pdf_pages(pdf_path): if not PDF_SUPPORT: return 1 try: return len(fitz.open(pdf_path)) except Exception: return 1 def _load_image_for_detect(image_path, pdf_path, pdf_page): if image_path: img = cv2.imread(image_path) if img is None: img = _pil_to_bgr(Image.open(image_path)) return img if pdf_path and PDF_SUPPORT: rgb = _render_pdf_page(pdf_path, pdf_page, dpi=DETECTION_DPI) return cv2.cvtColor(rgb, cv2.COLOR_RGB2BGR) if rgb is not None else None return None # --------------------------------------------------------------------------- # Canvas helpers — HTML5 canvas replaces Plotly for the Review & Edit tab # --------------------------------------------------------------------------- _CANVAS_MAX_PX = 2400 # max dimension sent to canvas (large enough to stay sharp when zoomed) def _build_canvas_json(img_rgb, annotations, selected=-1): """ Return JSON string sent to the HTML canvas via rv_canvas_render_txt. Pass img_rgb=None for annotations-only updates (JS keeps its cached image). Annotation coords are always in ORIGINAL image pixels; JS scales for display. """ anns_data = [ { "x1": float(a["x1"]), "y1": float(a["y1"]), "x2": float(a["x2"]), "y2": float(a["y2"]), "conf": float(a.get("conf", 1.0)), "status": a.get("status", "pending"), "source": a.get("source", "unknown"), } for a in annotations ] if img_rgb is None: return json.dumps( { "image": None, "imageW": 0, "imageH": 0, "annotations": anns_data, "selected": selected, } ) H, W = img_rgb.shape[:2] # Scale down for transfer — keep annotation coords in original pixels scale = min(1.0, _CANVAS_MAX_PX / max(W, H)) if scale < 1.0: nw, nh = int(W * scale), int(H * scale) pil = Image.fromarray(img_rgb).resize((nw, nh), Image.LANCZOS) else: pil = Image.fromarray(img_rgb) buf = BytesIO() pil.save(buf, format="JPEG", quality=80) b64 = base64.b64encode(buf.getvalue()).decode() return json.dumps( { "image": f"data:image/jpeg;base64,{b64}", "imageW": W, "imageH": H, # original dimensions so JS scales coords correctly "annotations": anns_data, "selected": selected, } ) # The detections grid is a secondary "click row to select" aid — the canvas is # the primary selector. Gradio's Dataframe (Svelte Table) crashes # ("RangeError: Too many properties to enumerate") on dense pages with 100+ # rows, which wedges the whole UI. Cap the rows it renders; all boxes still # show on the canvas and are selectable there. _REVIEW_DF_MAX_ROWS = 50 def _build_review_df(annotations): if not PANDAS_SUPPORT or not annotations: return [] rows = [ { "#": i + 1, "x1": round(ann["x1"]), "y1": round(ann["y1"]), "x2": round(ann["x2"]), "y2": round(ann["y2"]), "conf": f"{ann.get('conf', 1.0):.2f}", "status": ann.get("status", "pending"), } for i, ann in enumerate(annotations) ] if len(rows) > _REVIEW_DF_MAX_ROWS: # Keep the grid small enough to never crash; select the rest on the canvas. rows = rows[:_REVIEW_DF_MAX_ROWS] return pd.DataFrame(rows) # --------------------------------------------------------------------------- # Ground Truth Ledger helpers # --------------------------------------------------------------------------- def _now(): return datetime.now(UTC).isoformat() def _make_box(ann): return {"x1": ann["x1"], "y1": ann["y1"], "x2": ann["x2"], "y2": ann["y2"], "conf": ann.get("conf", 1.0)} def _ledger_summary(ledger): counts = {"accepted": 0, "rejected": 0, "modified": 0, "created": 0} for e in ledger: t = e.get("event_type", "") if t in counts: counts[t] += 1 return ( f"{len(ledger)} events — " f"{counts['accepted']} accepted, {counts['rejected']} rejected, " f"{counts['modified']} modified, {counts['created']} created" ) def _append_events(ledger, events): return list(ledger) + events def _make_event(page, ann, event_type, before=None, after=None): return { "event_id": str(uuid.uuid4()), "timestamp": _now(), "page_number": page, "annotation_id": ann.get("id", -1), "source": ann.get("source", "unknown"), "model_version": get_model_version(), "event_type": event_type, "before": before, "after": after, } # --------------------------------------------------------------------------- # Export helpers # --------------------------------------------------------------------------- def _image_hash(img_rgb): return hashlib.sha256(img_rgb.tobytes()).hexdigest()[:16] def export_events_jsonl(ledger): if not ledger: return None tmp = tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=False, encoding="utf-8") for event in ledger: tmp.write(json.dumps(event) + "\n") tmp.close() return tmp.name def export_yolo_dataset_zip(ledger, rv_pdf): if not ledger or not rv_pdf or not PDF_SUPPORT: return None by_page = {} for ev in ledger: pg = ev.get("page_number", 1) by_page.setdefault(pg, []).append(ev) tile_size = 640 tmp_dir = Path(tempfile.mkdtemp()) img_dir = tmp_dir / "images" lbl_dir = tmp_dir / "labels" img_dir.mkdir() lbl_dir.mkdir() written = 0 for page_num, events in by_page.items(): img_rgb = _render_pdf_page(rv_pdf, page_num, dpi=100) if img_rgb is None: continue H, W = img_rgb.shape[:2] img_hash = _image_hash(img_rgb) for ev in events: ev_type = ev.get("event_type", "") box = ev.get("after") or ev.get("before") if box is None: continue cx = (box["x1"] + box["x2"]) / 2 cy = (box["y1"] + box["y2"]) / 2 tx = max(0, min(int(cx - tile_size / 2), W - tile_size)) ty = max(0, min(int(cy - tile_size / 2), H - tile_size)) tx2 = min(tx + tile_size, W) ty2 = min(ty + tile_size, H) crop = img_rgb[ty:ty2, tx:tx2] pad_h = tile_size - crop.shape[0] pad_w = tile_size - crop.shape[1] if pad_h > 0 or pad_w > 0: crop = np.pad(crop, ((0, pad_h), (0, pad_w), (0, 0)), mode="constant", constant_values=255) stem = f"p{page_num}_{img_hash}_{ev['event_id'][:8]}" Image.fromarray(crop).save(img_dir / f"{stem}.jpg", quality=90) lbl_path = lbl_dir / f"{stem}.txt" if ev_type in ("accepted", "created", "modified"): bx1 = max(box["x1"] - tx, 0) by1 = max(box["y1"] - ty, 0) bx2 = min(box["x2"] - tx, tile_size) by2 = min(box["y2"] - ty, tile_size) bw = bx2 - bx1 bh = by2 - by1 if bw > 0 and bh > 0: bcx = (bx1 + bx2) / 2 / tile_size bcy = (by1 + by2) / 2 / tile_size lbl_path.write_text(f"0 {bcx:.6f} {bcy:.6f} {bw / tile_size:.6f} {bh / tile_size:.6f}\n") else: lbl_path.write_text("") else: lbl_path.write_text("") written += 1 if written == 0: return None (tmp_dir / "data.yaml").write_text("path: .\ntrain: images\nval: images\nnc: 1\nnames: ['facade_frame']\n") zip_path = tmp_dir.parent / "facade_annotations.zip" with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf: for f in tmp_dir.rglob("*"): if f.is_file(): zf.write(f, f.relative_to(tmp_dir)) return str(zip_path) # --------------------------------------------------------------------------- # Detect tab handlers # --------------------------------------------------------------------------- def run_detection( image_path, pdf_file, pdf_page, page_number, conf_threshold, min_area, max_area, epsilon_factor, threshold ): pdf_path = (pdf_file if isinstance(pdf_file, str) else pdf_file.name) if pdf_file is not None else None img = _load_image_for_detect(image_path, pdf_path, int(pdf_page)) if img is None: return None, {}, [], "No image provided." from detection import _get_model using_yolo = _get_model() is not None boxes = detect_rectangles( img, conf=float(conf_threshold), min_area=int(min_area), max_area=int(max_area), epsilon_factor=float(epsilon_factor), threshold=int(threshold), ) annotated_rgb = cv2.cvtColor(draw_detections(img, boxes), cv2.COLOR_BGR2RGB) shapes = [ {"page_number": int(page_number), "x1": b["x1"], "y1": b["y1"], "x2": b["x2"], "y2": b["y2"], "conf": b["conf"]} for b in boxes ] payload = {"frame_template_id": 1, "shapes": shapes, "color": None, "quantity_multiplier": 1} engine = "YOLO" if using_yolo else "OpenCV (YOLO model not found)" return annotated_rgb, payload, shapes, f"[{engine}] Detected {len(boxes)} panel(s)." # --------------------------------------------------------------------------- # Machine API endpoints (called by the backend via gradio_client) # --------------------------------------------------------------------------- def detect_image_api(image_file, conf=0.25): """ Detect frames on a single page image (PNG/JPG). Primary backend entrypoint: the caller renders the PDF page itself and uploads only the image. Returns {"width_px", "height_px", "model_version", "boxes": [{x1,y1,x2,y2,conf}]} with coordinates in the uploaded image's pixel space. """ if image_file is None: return {"error": "image file is required"} path = image_file if isinstance(image_file, str) else image_file.name img = cv2.imread(path) if img is None: try: img = _pil_to_bgr(Image.open(path)) except Exception as exc: return {"error": f"could not read image: {exc}"} boxes = detect_rectangles(img, conf=float(conf)) h, w = img.shape[:2] return { "width_px": int(w), "height_px": int(h), "model_version": get_model_version(), "boxes": boxes, } def detect_page_api(pdf_file, page_number=1, dpi=DETECTION_DPI, conf=0.25): """ Render one PDF page at the given DPI and detect frames on it. Testing/fallback path — prefer detect_image_api to avoid re-uploading large PDFs per page. """ if pdf_file is None: return {"error": "pdf file is required"} path = pdf_file if isinstance(pdf_file, str) else pdf_file.name rgb = render_pdf_page(path, int(page_number), dpi=int(dpi)) if rgb is None: return {"error": f"could not render page {page_number}"} bgr = cv2.cvtColor(rgb, cv2.COLOR_RGB2BGR) boxes = detect_rectangles(bgr, conf=float(conf)) h, w = rgb.shape[:2] return { "page_number": int(page_number), "dpi": int(dpi), "width_px": int(w), "height_px": int(h), "model_version": get_model_version(), "boxes": boxes, } # --------------------------------------------------------------------------- # Per-page label JSON import/export (canonical ground-truth format) # --------------------------------------------------------------------------- _LABEL_DISPLAY_DPI = 100 # the Review tab canvas coordinate space # Autosave: when running from the repo checkout (not the HF Space), every # annotation change is checkpointed to disk so a browser refresh/crash loses # nothing. Files go to a staging dir, NOT the canonical labels dir — the # canonical ground truth is only ever written via merge_labels.py. def _resolve_autosave_dir(): env = os.environ.get("FACADE_AUTOSAVE_DIR") if env: p = Path(env) p.mkdir(parents=True, exist_ok=True) return p try: working_set = Path(__file__).resolve().parents[2] / "data" / "working_set" except IndexError: # app sits near filesystem root (HF Space: /app/app.py) return None if working_set.is_dir(): p = working_set / "autosave" p.mkdir(parents=True, exist_ok=True) return p return None # standalone checkout — autosave off _AUTOSAVE_DIR = _resolve_autosave_dir() def _slugify_doc(stem): # must match scripts/annotation/common.py::doc_slug return re.sub(r"[^A-Za-z0-9._-]+", "_", stem).strip("_")[:60] def _source_tag(ann, model_version): src = ann.get("source", "unknown") if src == "yolo": return f"yolo:{model_version}" if src in ("human_added", "human"): return "human:gradio" return src def _page_label_data(rv_pdf, page, anns, audit_status): """Build one page's canonical label dict from canvas annotations. Coordinates are converted from the 100-DPI canvas space to PDF points so labels are DPI-independent and merge with other sources downstream. The per-box "status" key is an extra field for exact import round-trips. """ mv = get_model_version() boxes, rejected = [], [] for ann in anns: entry = { "id": ann.get("uid") or str(uuid.uuid4()), "x1_pt": round(px_to_pt(ann["x1"], _LABEL_DISPLAY_DPI), 3), "y1_pt": round(px_to_pt(ann["y1"], _LABEL_DISPLAY_DPI), 3), "x2_pt": round(px_to_pt(ann["x2"], _LABEL_DISPLAY_DPI), 3), "y2_pt": round(px_to_pt(ann["y2"], _LABEL_DISPLAY_DPI), 3), "source": _source_tag(ann, mv), "conf": round(float(ann.get("conf", 1.0)), 4), "status": ann.get("status", "pending"), "history": ann.get("history", []), } (rejected if ann.get("status") == "rejected" else boxes).append(entry) return { "doc": _slugify_doc(Path(rv_pdf).stem), "pdf": Path(rv_pdf).name, "page": int(page), "page_type": "", "audit_status": audit_status, "audited_at": datetime.now(UTC).isoformat(), "boxes": boxes, "rejected_boxes": rejected, } def _autosave(rv_pdf, rv_annotations): """Checkpoint every annotated page to the autosave dir. Never raises — a failed checkpoint must not break the edit that triggered it.""" if _AUTOSAVE_DIR is None or rv_pdf is None: return try: out_dir = _AUTOSAVE_DIR / _slugify_doc(Path(rv_pdf).stem) out_dir.mkdir(parents=True, exist_ok=True) for page, anns in (rv_annotations or {}).items(): if not anns: continue # A page still holding unreviewed machine boxes is not an audit unreviewed = any( a.get("status", "pending") == "pending" and not str(a.get("source", "")).startswith("human") for a in anns ) data = _page_label_data(rv_pdf, page, anns, "machine" if unreviewed else "audited") (out_dir / f"p{int(page):03d}.json").write_text(json.dumps(data, indent=2), encoding="utf-8") except Exception as exc: print(f"[autosave] checkpoint failed: {exc}") def on_rv_export_labels(rv_pdf, rv_annotations): """Export every page with annotations as canonical label JSONs (zipped).""" if rv_pdf is None: return None, "No PDF loaded." pages = {p: anns for p, anns in (rv_annotations or {}).items() if anns} if not pages: return None, "No annotations to export." stem = Path(rv_pdf).stem tmp_dir = Path(tempfile.mkdtemp(prefix="labels_")) written = [] for page, anns in sorted(pages.items()): data = _page_label_data(rv_pdf, page, anns, "audited") out = tmp_dir / f"p{int(page):03d}.json" out.write_text(json.dumps(data, indent=2), encoding="utf-8") written.append(out) zip_path = tmp_dir / f"{stem}_labels.zip" with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf: for f in written: zf.write(f, f.name) return str(zip_path), f"Exported labels for {len(written)} page(s)." def on_rv_import_labels(label_files, rv_pdf, rv_page, rv_annotations): """Import canonical label JSON(s) into the Review tab. Boxes from audited/verified files arrive approved; machine bootstrap files arrive pending; rejected_boxes arrive rejected (so they round-trip). """ if not label_files: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "No label files provided." rv_annotations = dict(rv_annotations or {}) loaded_pages = [] for lf in label_files: path = lf if isinstance(lf, str) else lf.name try: data = json.loads(Path(path).read_text(encoding="utf-8")) except Exception as exc: return ( _build_canvas_json(None, rv_annotations.get(rv_page, [])), _build_review_df(rv_annotations.get(rv_page, [])), rv_annotations, f"Failed to parse {Path(path).name}: {exc}", ) page = int(data.get("page", 1)) default_status = "approved" if data.get("audit_status") in ("audited", "verified") else "pending" anns = [] for status, group in ((default_status, data.get("boxes", [])), ("rejected", data.get("rejected_boxes", []))): for box in group: anns.append( { "id": len(anns), "uid": box.get("id"), "x1": pt_to_px(box["x1_pt"], _LABEL_DISPLAY_DPI), "y1": pt_to_px(box["y1_pt"], _LABEL_DISPLAY_DPI), "x2": pt_to_px(box["x2_pt"], _LABEL_DISPLAY_DPI), "y2": pt_to_px(box["y2_pt"], _LABEL_DISPLAY_DPI), "conf": float(box.get("conf", 1.0)), # autosave files carry per-box status for exact resume "status": "rejected" if status == "rejected" else box.get("status", status), "source": box.get("source", "unknown"), "history": box.get("history", []), } ) rv_annotations[page] = anns loaded_pages.append(page) _autosave(rv_pdf, rv_annotations) current = rv_annotations.get(rv_page, []) img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100) if rv_pdf is not None and rv_page in loaded_pages else None return ( _build_canvas_json(img_rgb, current), _build_review_df(current), rv_annotations, f"Imported labels for page(s) {sorted(loaded_pages)}.", ) # --------------------------------------------------------------------------- # Send to Backend handler # --------------------------------------------------------------------------- def send_to_backend( rv_annotations, backend_url, document_id, frame_template_id, x_tenant_id, x_user_id, color, quantity_multiplier ): shapes = [] for page_num, anns in sorted((rv_annotations or {}).items()): for ann in anns: if ann.get("status") == "approved": shapes.append( { "page_number": page_num, "x1": ann["x1"], "y1": ann["y1"], "x2": ann["x2"], "y2": ann["y2"], } ) if not shapes: return {"error": "No approved annotations. Go to Review & Edit, detect + approve boxes first."} if not backend_url: return {"error": "Backend URL is required."} url = f"{backend_url.rstrip('/')}/api/v1/tenants/documents/{int(document_id)}/frame-instances/from-annotation" payload = { "frame_template_id": int(frame_template_id), "shapes": shapes, "quantity_multiplier": int(quantity_multiplier), } if color and color.strip(): payload["color"] = color.strip() headers = { "Content-Type": "application/json", "x-tenant-id": str(int(x_tenant_id)), "x-user-id": str(int(x_user_id)), } try: resp = requests.post(url, json=payload, headers=headers, timeout=15) try: body = resp.json() except Exception: body = resp.text return {"status_code": resp.status_code, "response": body} except requests.exceptions.RequestException as exc: return {"error": str(exc)} # --------------------------------------------------------------------------- # Review & Edit tab handlers # --------------------------------------------------------------------------- def on_rv_pdf_upload(pdf_file): if pdf_file is None or not PDF_SUPPORT: return (_build_canvas_json(None, []), _build_review_df([]), "No PDF loaded", None, 1, 1, {}, []) pdf_path = pdf_file if isinstance(pdf_file, str) else pdf_file.name total = _count_pdf_pages(pdf_path) img_rgb = _render_pdf_page(pdf_path, 1, dpi=100) return (_build_canvas_json(img_rgb, []), _build_review_df([]), f"Page 1 / {total}", pdf_path, 1, total, {}, []) def _navigate(rv_pdf, rv_page, rv_total, rv_annotations, delta): if rv_pdf is None: return _build_canvas_json(None, []), _build_review_df([]), "No PDF loaded", rv_page new_page = max(1, min(rv_total, rv_page + delta)) anns = rv_annotations.get(new_page, []) img_rgb = _render_pdf_page(rv_pdf, new_page, dpi=100) return ( _build_canvas_json(img_rgb, anns), _build_review_df(anns), f"Page {new_page} / {rv_total} — {len(anns)} box(es)", new_page, ) def on_rv_prev(rv_pdf, rv_page, rv_total, rv_annotations): return _navigate(rv_pdf, rv_page, rv_total, rv_annotations, -1) def on_rv_next(rv_pdf, rv_page, rv_total, rv_annotations): return _navigate(rv_pdf, rv_page, rv_total, rv_annotations, +1) def on_rv_goto_page(page_str, rv_pdf, rv_total, rv_annotations): try: page_num = max(1, min(int(rv_total), int(str(page_str).strip()))) except (ValueError, TypeError): return gr.update(), gr.update(), gr.update(), gr.update() if rv_pdf is None: return _build_canvas_json(None, []), _build_review_df([]), "No PDF loaded", 1 anns = rv_annotations.get(page_num, []) img_rgb = _render_pdf_page(rv_pdf, page_num, dpi=100) return ( _build_canvas_json(img_rgb, anns), _build_review_df(anns), f"Page {page_num} / {rv_total} — {len(anns)} box(es)", page_num, ) def on_canvas_events(events_json, rv_pdf, rv_page, rv_annotations, rv_ledger): """Process annotation changes sent from the JS canvas.""" if not events_json or not events_json.strip(): return gr.update(), rv_annotations, rv_ledger, gr.update(), -1 try: data = json.loads(events_json) except Exception: return gr.update(), rv_annotations, rv_ledger, gr.update(), -1 raw_anns = data.get("annotations", []) selected = int(data.get("selected", -1)) action = data.get("action", "sync") old_anns = list(rv_annotations.get(rv_page, [])) new_anns = [] for i, a in enumerate(raw_anns): new_anns.append( { "id": i, "x1": float(a["x1"]), "y1": float(a["y1"]), "x2": float(a["x2"]), "y2": float(a["y2"]), "conf": float(a.get("conf", 1.0)), "status": a.get("status", "pending"), "source": a.get("source", "human_added"), } ) rv_annotations = dict(rv_annotations) rv_annotations[rv_page] = new_anns events = [] if action == "created" and new_anns: ann = new_anns[-1] events.append(_make_event(rv_page, ann, "created", before=None, after=_make_box(ann))) elif action in ("moved", "resized") and 0 <= selected < len(new_anns): if selected < len(old_anns): events.append( _make_event( rv_page, new_anns[selected], "modified", before=_make_box(old_anns[selected]), after=_make_box(new_anns[selected]), ) ) rv_ledger = _append_events(rv_ledger, events) if action in ("created", "moved", "resized", "deleted"): _autosave(rv_pdf, rv_annotations) canvas_json = _build_canvas_json(None, new_anns, selected=selected) return canvas_json, rv_annotations, rv_ledger, _build_review_df(new_anns), selected def on_rv_detect(rv_pdf, rv_page, rv_annotations, rv_conf): if rv_pdf is None: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "No PDF loaded." img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100) if img_rgb is None: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "Failed to render page." from detection import _get_model using_yolo = _get_model() is not None # Detect on a DETECTION_DPI render (the model's working resolution), then # scale boxes back into the 100-DPI display space the canvas uses. detect_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=DETECTION_DPI) if detect_rgb is None: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "Failed to render page." detect_bgr = cv2.cvtColor(detect_rgb, cv2.COLOR_RGB2BGR) boxes = detect_rectangles(detect_bgr, conf=float(rv_conf)) scale = 100.0 / DETECTION_DPI annotations = [ { "id": i, "x1": b["x1"] * scale, "y1": b["y1"] * scale, "x2": b["x2"] * scale, "y2": b["y2"] * scale, "conf": b["conf"], "status": "pending", "source": "yolo" if using_yolo else "opencv", } for i, b in enumerate(boxes) ] rv_annotations = dict(rv_annotations) rv_annotations[rv_page] = annotations _autosave(rv_pdf, rv_annotations) engine = "YOLO" if using_yolo else "OpenCV" return ( _build_canvas_json(img_rgb, annotations), _build_review_df(annotations), rv_annotations, f"[{engine}] Found {len(boxes)} panel(s) on page {rv_page}.", ) def _box_iou(a, b): ix1, iy1 = max(a["x1"], b["x1"]), max(a["y1"], b["y1"]) ix2, iy2 = min(a["x2"], b["x2"]), min(a["y2"], b["y2"]) inter = max(0.0, ix2 - ix1) * max(0.0, iy2 - iy1) if inter <= 0: return 0.0 area_a = (a["x2"] - a["x1"]) * (a["y2"] - a["y1"]) area_b = (b["x2"] - b["x1"]) * (b["y2"] - b["y1"]) return inter / (area_a + area_b - inter) def on_rv_find_more(rv_pdf, rv_page, rv_annotations, rv_conf): """Low-conf re-detect that MERGES: keeps every existing box (approved, rejected, human, pending) and only adds detections that don't overlap one. Lets the model fill audit gaps without the user redrawing by hand.""" anns = rv_annotations.get(rv_page, []) if rv_pdf is None: return gr.update(), _build_review_df(anns), rv_annotations, "No PDF loaded." detect_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=DETECTION_DPI) if detect_rgb is None: return gr.update(), _build_review_df(anns), rv_annotations, "Failed to render page." from detection import _get_model using_yolo = _get_model() is not None detect_bgr = cv2.cvtColor(detect_rgb, cv2.COLOR_RGB2BGR) conf = min(float(rv_conf), 0.10) # always cast a wide net regardless of the slider boxes = detect_rectangles(detect_bgr, conf=conf) scale = 100.0 / DETECTION_DPI rv_annotations = dict(rv_annotations) new_anns = list(anns) added = 0 for b in boxes: cand = { "x1": b["x1"] * scale, "y1": b["y1"] * scale, "x2": b["x2"] * scale, "y2": b["y2"] * scale, } # an existing box of ANY status wins — including rejected, so boxes the # user already threw out don't come back if any(_box_iou(cand, a) > 0.30 for a in anns): continue new_anns.append( dict( cand, id=len(new_anns), conf=b["conf"], status="pending", source="yolo" if using_yolo else "opencv", ) ) added += 1 rv_annotations[rv_page] = new_anns _autosave(rv_pdf, rv_annotations) return ( _build_canvas_json(None, new_anns), _build_review_df(new_anns), rv_annotations, f"Find More @conf {conf:.2f}: +{added} new box(es) (existing {len(anns)} kept).", ) def on_rv_detect_all(rv_pdf, rv_page, rv_total, rv_annotations, rv_conf): if rv_pdf is None: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "No PDF loaded." if not PDF_SUPPORT: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "PDF support unavailable." from detection import _get_model using_yolo = _get_model() is not None all_anns = dict(rv_annotations) total_found = 0 detect_dpi = DETECTION_DPI # the model's working resolution; lower DPI misses most frames try: doc = fitz.open(rv_pdf) mat = fitz.Matrix(detect_dpi / 72, detect_dpi / 72) for page_num in range(1, rv_total + 1): try: page = doc[page_num - 1] pix = page.get_pixmap(matrix=mat, colorspace=fitz.csRGB) img_rgb = np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width, 3) except Exception: continue img_bgr = cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR) boxes = detect_rectangles(img_bgr, conf=float(rv_conf)) # Scale coords back to 100-DPI space so they match the display image scale = 100.0 / detect_dpi all_anns[page_num] = [ { "id": i, "x1": b["x1"] * scale, "y1": b["y1"] * scale, "x2": b["x2"] * scale, "y2": b["y2"] * scale, "conf": b["conf"], "status": "pending", "source": "yolo" if using_yolo else "opencv", } for i, b in enumerate(boxes) ] total_found += len(boxes) doc.close() except Exception as exc: return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, f"Error: {exc}" img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100) anns = all_anns.get(rv_page, []) _autosave(rv_pdf, all_anns) engine = "YOLO" if using_yolo else "OpenCV" return ( _build_canvas_json(img_rgb, anns), _build_review_df(anns), all_anns, f"[{engine}] Detected {total_found} panel(s) across {rv_total} page(s).", ) def on_fill_edit_form(selected_idx, rv_page, rv_annotations): anns = rv_annotations.get(rv_page, []) idx = int(selected_idx) if selected_idx is not None else -1 if idx < 0 or idx >= len(anns): return None, None, None, None a = anns[idx] return a["x1"], a["y1"], a["x2"], a["y2"] def on_rv_update_box(rv_pdf, rv_page, rv_annotations, ledger, selected_idx, x1, y1, x2, y2): anns = rv_annotations.get(rv_page, []) if rv_pdf is None or not anns or selected_idx is None or int(selected_idx) < 0: return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first." idx = int(selected_idx) if idx >= len(anns): return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection." try: x1, y1, x2, y2 = float(x1), float(y1), float(x2), float(y2) except (TypeError, ValueError): return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid coordinates." if x2 <= x1 or y2 <= y1: return gr.update(), _build_review_df(anns), rv_annotations, ledger, "x2>x1 and y2>y1 required." rv_annotations = dict(rv_annotations) new_anns = list(anns) old = new_anns[idx] before = _make_box(old) new_anns[idx] = dict(old, x1=x1, y1=y1, x2=x2, y2=y2) rv_annotations[rv_page] = new_anns event = _make_event(rv_page, old, "modified", before=before, after=_make_box(new_anns[idx])) ledger = _append_events(ledger, [event]) _autosave(rv_pdf, rv_annotations) img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100) return ( _build_canvas_json(img_rgb, new_anns, selected=idx), _build_review_df(new_anns), rv_annotations, ledger, f"Updated box #{idx + 1}. {_ledger_summary(ledger)}", ) def on_rv_approve_all(rv_pdf, rv_page, rv_annotations, ledger): if rv_pdf is None or not rv_annotations.get(rv_page): return gr.update(), _build_review_df([]), rv_annotations, ledger, "No boxes on this page." rv_annotations = dict(rv_annotations) old_anns = rv_annotations[rv_page] new_anns = [dict(a, status="approved") for a in old_anns] rv_annotations[rv_page] = new_anns events = [ _make_event(rv_page, a, "accepted", before=_make_box(a), after=_make_box(a)) for a in old_anns if a.get("status") != "approved" ] ledger = _append_events(ledger, events) _autosave(rv_pdf, rv_annotations) return ( _build_canvas_json(None, new_anns), _build_review_df(new_anns), rv_annotations, ledger, f"Approved {len(new_anns)} box(es). {_ledger_summary(ledger)}", ) def on_rv_reject_all(rv_pdf, rv_page, rv_annotations, ledger): if rv_pdf is None or not rv_annotations.get(rv_page): return gr.update(), _build_review_df([]), rv_annotations, ledger, "No boxes on this page." rv_annotations = dict(rv_annotations) old_anns = rv_annotations[rv_page] new_anns = [dict(a, status="rejected") for a in old_anns] rv_annotations[rv_page] = new_anns events = [ _make_event(rv_page, a, "rejected", before=_make_box(a), after=None) for a in old_anns if a.get("status") != "rejected" ] ledger = _append_events(ledger, events) _autosave(rv_pdf, rv_annotations) return ( _build_canvas_json(None, new_anns), _build_review_df(new_anns), rv_annotations, ledger, f"Rejected {len(new_anns)} box(es). {_ledger_summary(ledger)}", ) def on_rv_approve_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx): anns = rv_annotations.get(rv_page, []) if rv_pdf is None or not anns or selected_idx is None or selected_idx < 0: return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first." idx = int(selected_idx) if idx >= len(anns): return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection." rv_annotations = dict(rv_annotations) new_anns = list(anns) old = new_anns[idx] new_anns[idx] = dict(old, status="approved") rv_annotations[rv_page] = new_anns event = _make_event(rv_page, old, "accepted", before=_make_box(old), after=_make_box(new_anns[idx])) ledger = _append_events(ledger, [event]) _autosave(rv_pdf, rv_annotations) return ( _build_canvas_json(None, new_anns, selected=idx), _build_review_df(new_anns), rv_annotations, ledger, f"Approved box #{idx + 1}. {_ledger_summary(ledger)}", ) def on_rv_reject_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx): anns = rv_annotations.get(rv_page, []) if rv_pdf is None or not anns or selected_idx is None or selected_idx < 0: return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first." idx = int(selected_idx) if idx >= len(anns): return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection." rv_annotations = dict(rv_annotations) new_anns = list(anns) old = new_anns[idx] new_anns[idx] = dict(old, status="rejected") rv_annotations[rv_page] = new_anns event = _make_event(rv_page, old, "rejected", before=_make_box(old), after=_make_box(new_anns[idx])) ledger = _append_events(ledger, [event]) _autosave(rv_pdf, rv_annotations) return ( _build_canvas_json(None, new_anns, selected=idx), _build_review_df(new_anns), rv_annotations, ledger, f"Rejected box #{idx + 1}. {_ledger_summary(ledger)}", ) def on_rv_delete_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx): anns = rv_annotations.get(rv_page, []) if rv_pdf is None or not anns or selected_idx is None or selected_idx < 0: return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first.", selected_idx idx = int(selected_idx) if idx >= len(anns): return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection.", selected_idx rv_annotations = dict(rv_annotations) removed = anns[idx] new_anns = [a for i, a in enumerate(anns) if i != idx] for j, a in enumerate(new_anns): a["id"] = j rv_annotations[rv_page] = new_anns event = _make_event(rv_page, removed, "rejected", before=_make_box(removed), after=None) ledger = _append_events(ledger, [event]) _autosave(rv_pdf, rv_annotations) new_sel = min(idx, len(new_anns) - 1) if new_anns else -1 return ( _build_canvas_json(None, new_anns, selected=new_sel), _build_review_df(new_anns), rv_annotations, ledger, f"Deleted box #{idx + 1}. {_ledger_summary(ledger)}", new_sel, ) def on_rv_copy_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx): anns = rv_annotations.get(rv_page, []) if not anns or selected_idx is None or selected_idx < 0: return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first." idx = int(selected_idx) if idx >= len(anns): return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection." rv_annotations = dict(rv_annotations) src = anns[idx] new_ann = dict( src, id=len(anns), x1=src["x1"] + 15, y1=src["y1"] + 15, x2=src["x2"] + 15, y2=src["y2"] + 15, status="pending" ) new_anns = [*list(anns), new_ann] rv_annotations[rv_page] = new_anns new_sel = len(new_anns) - 1 event = _make_event(rv_page, new_ann, "created", before=None, after=_make_box(new_ann)) ledger = _append_events(ledger, [event]) _autosave(rv_pdf, rv_annotations) return ( _build_canvas_json(None, new_anns, selected=new_sel), _build_review_df(new_anns), rv_annotations, ledger, f"Copied to box #{new_sel + 1}. {_ledger_summary(ledger)}", ) def on_row_select(evt: gr.SelectData): return evt.index[0] if evt and evt.index else -1 def on_export_events(ledger): if not ledger: return None return export_events_jsonl(ledger) def on_export_dataset(ledger, rv_pdf): if not ledger: return None return export_yolo_dataset_zip(ledger, rv_pdf) # --------------------------------------------------------------------------- # Gradio UI # --------------------------------------------------------------------------- _model_version = get_model_version() _CANVAS_HTML = """