Annotation / app.py
mat12322's picture
Deploy Takeoff AI Annotator
8a66464 verified
Raw
History Blame Contribute Delete
77.5 kB
import base64
import hashlib
import json
import os
import re
import tempfile
import uuid
import zipfile
from datetime import UTC, datetime
from io import BytesIO
from pathlib import Path
import cv2
import gradio as gr
import numpy as np
import requests
from detection import (
DETECTION_DPI,
detect_rectangles,
draw_detections,
get_model_version,
pt_to_px,
px_to_pt,
render_pdf_page,
)
from PIL import Image
# Fix gradio_client 5.9.1 crash when schema is a bool
try:
import gradio_client.utils as _gcu
_orig_get_type = _gcu.get_type
def _safe_get_type(schema):
if not isinstance(schema, dict):
return "Any"
return _orig_get_type(schema)
_gcu.get_type = _safe_get_type
_orig_j2p = _gcu._json_schema_to_python_type
def _safe_j2p(schema, defs=None):
if not isinstance(schema, dict):
return "Any"
try:
return _orig_j2p(schema, defs)
except (TypeError, AttributeError):
return "Any"
_gcu._json_schema_to_python_type = _safe_j2p
_orig_j2p_pub = _gcu.json_schema_to_python_type
def _safe_j2p_pub(schema):
try:
return _orig_j2p_pub(schema)
except (TypeError, AttributeError):
return "Any"
_gcu.json_schema_to_python_type = _safe_j2p_pub
except Exception:
pass
try:
import fitz
PDF_SUPPORT = True
except ImportError:
PDF_SUPPORT = False
try:
import pandas as pd
PANDAS_SUPPORT = True
except ImportError:
PANDAS_SUPPORT = False
def _pil_to_bgr(pil_img):
return cv2.cvtColor(np.array(pil_img.convert("RGB")), cv2.COLOR_RGB2BGR)
def _render_pdf_page(pdf_path, page_number, dpi=100):
if not PDF_SUPPORT:
return None
try:
doc = fitz.open(pdf_path)
idx = max(0, min(page_number - 1, len(doc) - 1))
page = doc[idx]
mat = fitz.Matrix(dpi / 72, dpi / 72)
pix = page.get_pixmap(matrix=mat, colorspace=fitz.csRGB)
return np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width, 3)
except Exception as exc:
print(f"[app] PDF render error: {exc}")
return None
def _count_pdf_pages(pdf_path):
if not PDF_SUPPORT:
return 1
try:
return len(fitz.open(pdf_path))
except Exception:
return 1
def _load_image_for_detect(image_path, pdf_path, pdf_page):
if image_path:
img = cv2.imread(image_path)
if img is None:
img = _pil_to_bgr(Image.open(image_path))
return img
if pdf_path and PDF_SUPPORT:
rgb = _render_pdf_page(pdf_path, pdf_page, dpi=DETECTION_DPI)
return cv2.cvtColor(rgb, cv2.COLOR_RGB2BGR) if rgb is not None else None
return None
# ---------------------------------------------------------------------------
# Canvas helpers β€” HTML5 canvas replaces Plotly for the Review & Edit tab
# ---------------------------------------------------------------------------
_CANVAS_MAX_PX = 2400 # max dimension sent to canvas (large enough to stay sharp when zoomed)
def _build_canvas_json(img_rgb, annotations, selected=-1):
"""
Return JSON string sent to the HTML canvas via rv_canvas_render_txt.
Pass img_rgb=None for annotations-only updates (JS keeps its cached image).
Annotation coords are always in ORIGINAL image pixels; JS scales for display.
"""
anns_data = [
{
"x1": float(a["x1"]),
"y1": float(a["y1"]),
"x2": float(a["x2"]),
"y2": float(a["y2"]),
"conf": float(a.get("conf", 1.0)),
"status": a.get("status", "pending"),
"source": a.get("source", "unknown"),
}
for a in annotations
]
if img_rgb is None:
return json.dumps(
{
"image": None,
"imageW": 0,
"imageH": 0,
"annotations": anns_data,
"selected": selected,
}
)
H, W = img_rgb.shape[:2]
# Scale down for transfer β€” keep annotation coords in original pixels
scale = min(1.0, _CANVAS_MAX_PX / max(W, H))
if scale < 1.0:
nw, nh = int(W * scale), int(H * scale)
pil = Image.fromarray(img_rgb).resize((nw, nh), Image.LANCZOS)
else:
pil = Image.fromarray(img_rgb)
buf = BytesIO()
pil.save(buf, format="JPEG", quality=80)
b64 = base64.b64encode(buf.getvalue()).decode()
return json.dumps(
{
"image": f"data:image/jpeg;base64,{b64}",
"imageW": W,
"imageH": H, # original dimensions so JS scales coords correctly
"annotations": anns_data,
"selected": selected,
}
)
# The detections grid is a secondary "click row to select" aid β€” the canvas is
# the primary selector. Gradio's Dataframe (Svelte Table) crashes
# ("RangeError: Too many properties to enumerate") on dense pages with 100+
# rows, which wedges the whole UI. Cap the rows it renders; all boxes still
# show on the canvas and are selectable there.
_REVIEW_DF_MAX_ROWS = 50
def _build_review_df(annotations):
if not PANDAS_SUPPORT or not annotations:
return []
rows = [
{
"#": i + 1,
"x1": round(ann["x1"]),
"y1": round(ann["y1"]),
"x2": round(ann["x2"]),
"y2": round(ann["y2"]),
"conf": f"{ann.get('conf', 1.0):.2f}",
"status": ann.get("status", "pending"),
}
for i, ann in enumerate(annotations)
]
if len(rows) > _REVIEW_DF_MAX_ROWS:
# Keep the grid small enough to never crash; select the rest on the canvas.
rows = rows[:_REVIEW_DF_MAX_ROWS]
return pd.DataFrame(rows)
# ---------------------------------------------------------------------------
# Ground Truth Ledger helpers
# ---------------------------------------------------------------------------
def _now():
return datetime.now(UTC).isoformat()
def _make_box(ann):
return {"x1": ann["x1"], "y1": ann["y1"], "x2": ann["x2"], "y2": ann["y2"], "conf": ann.get("conf", 1.0)}
def _ledger_summary(ledger):
counts = {"accepted": 0, "rejected": 0, "modified": 0, "created": 0}
for e in ledger:
t = e.get("event_type", "")
if t in counts:
counts[t] += 1
return (
f"{len(ledger)} events β€” "
f"{counts['accepted']} accepted, {counts['rejected']} rejected, "
f"{counts['modified']} modified, {counts['created']} created"
)
def _append_events(ledger, events):
return list(ledger) + events
def _make_event(page, ann, event_type, before=None, after=None):
return {
"event_id": str(uuid.uuid4()),
"timestamp": _now(),
"page_number": page,
"annotation_id": ann.get("id", -1),
"source": ann.get("source", "unknown"),
"model_version": get_model_version(),
"event_type": event_type,
"before": before,
"after": after,
}
# ---------------------------------------------------------------------------
# Export helpers
# ---------------------------------------------------------------------------
def _image_hash(img_rgb):
return hashlib.sha256(img_rgb.tobytes()).hexdigest()[:16]
def export_events_jsonl(ledger):
if not ledger:
return None
tmp = tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=False, encoding="utf-8")
for event in ledger:
tmp.write(json.dumps(event) + "\n")
tmp.close()
return tmp.name
def export_yolo_dataset_zip(ledger, rv_pdf):
if not ledger or not rv_pdf or not PDF_SUPPORT:
return None
by_page = {}
for ev in ledger:
pg = ev.get("page_number", 1)
by_page.setdefault(pg, []).append(ev)
tile_size = 640
tmp_dir = Path(tempfile.mkdtemp())
img_dir = tmp_dir / "images"
lbl_dir = tmp_dir / "labels"
img_dir.mkdir()
lbl_dir.mkdir()
written = 0
for page_num, events in by_page.items():
img_rgb = _render_pdf_page(rv_pdf, page_num, dpi=100)
if img_rgb is None:
continue
H, W = img_rgb.shape[:2]
img_hash = _image_hash(img_rgb)
for ev in events:
ev_type = ev.get("event_type", "")
box = ev.get("after") or ev.get("before")
if box is None:
continue
cx = (box["x1"] + box["x2"]) / 2
cy = (box["y1"] + box["y2"]) / 2
tx = max(0, min(int(cx - tile_size / 2), W - tile_size))
ty = max(0, min(int(cy - tile_size / 2), H - tile_size))
tx2 = min(tx + tile_size, W)
ty2 = min(ty + tile_size, H)
crop = img_rgb[ty:ty2, tx:tx2]
pad_h = tile_size - crop.shape[0]
pad_w = tile_size - crop.shape[1]
if pad_h > 0 or pad_w > 0:
crop = np.pad(crop, ((0, pad_h), (0, pad_w), (0, 0)), mode="constant", constant_values=255)
stem = f"p{page_num}_{img_hash}_{ev['event_id'][:8]}"
Image.fromarray(crop).save(img_dir / f"{stem}.jpg", quality=90)
lbl_path = lbl_dir / f"{stem}.txt"
if ev_type in ("accepted", "created", "modified"):
bx1 = max(box["x1"] - tx, 0)
by1 = max(box["y1"] - ty, 0)
bx2 = min(box["x2"] - tx, tile_size)
by2 = min(box["y2"] - ty, tile_size)
bw = bx2 - bx1
bh = by2 - by1
if bw > 0 and bh > 0:
bcx = (bx1 + bx2) / 2 / tile_size
bcy = (by1 + by2) / 2 / tile_size
lbl_path.write_text(f"0 {bcx:.6f} {bcy:.6f} {bw / tile_size:.6f} {bh / tile_size:.6f}\n")
else:
lbl_path.write_text("")
else:
lbl_path.write_text("")
written += 1
if written == 0:
return None
(tmp_dir / "data.yaml").write_text("path: .\ntrain: images\nval: images\nnc: 1\nnames: ['facade_frame']\n")
zip_path = tmp_dir.parent / "facade_annotations.zip"
with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf:
for f in tmp_dir.rglob("*"):
if f.is_file():
zf.write(f, f.relative_to(tmp_dir))
return str(zip_path)
# ---------------------------------------------------------------------------
# Detect tab handlers
# ---------------------------------------------------------------------------
def run_detection(
image_path, pdf_file, pdf_page, page_number, conf_threshold, min_area, max_area, epsilon_factor, threshold
):
pdf_path = (pdf_file if isinstance(pdf_file, str) else pdf_file.name) if pdf_file is not None else None
img = _load_image_for_detect(image_path, pdf_path, int(pdf_page))
if img is None:
return None, {}, [], "No image provided."
from detection import _get_model
using_yolo = _get_model() is not None
boxes = detect_rectangles(
img,
conf=float(conf_threshold),
min_area=int(min_area),
max_area=int(max_area),
epsilon_factor=float(epsilon_factor),
threshold=int(threshold),
)
annotated_rgb = cv2.cvtColor(draw_detections(img, boxes), cv2.COLOR_BGR2RGB)
shapes = [
{"page_number": int(page_number), "x1": b["x1"], "y1": b["y1"], "x2": b["x2"], "y2": b["y2"], "conf": b["conf"]}
for b in boxes
]
payload = {"frame_template_id": 1, "shapes": shapes, "color": None, "quantity_multiplier": 1}
engine = "YOLO" if using_yolo else "OpenCV (YOLO model not found)"
return annotated_rgb, payload, shapes, f"[{engine}] Detected {len(boxes)} panel(s)."
# ---------------------------------------------------------------------------
# Machine API endpoints (called by the backend via gradio_client)
# ---------------------------------------------------------------------------
def detect_image_api(image_file, conf=0.25):
"""
Detect frames on a single page image (PNG/JPG). Primary backend entrypoint:
the caller renders the PDF page itself and uploads only the image.
Returns {"width_px", "height_px", "model_version", "boxes": [{x1,y1,x2,y2,conf}]}
with coordinates in the uploaded image's pixel space.
"""
if image_file is None:
return {"error": "image file is required"}
path = image_file if isinstance(image_file, str) else image_file.name
img = cv2.imread(path)
if img is None:
try:
img = _pil_to_bgr(Image.open(path))
except Exception as exc:
return {"error": f"could not read image: {exc}"}
boxes = detect_rectangles(img, conf=float(conf))
h, w = img.shape[:2]
return {
"width_px": int(w),
"height_px": int(h),
"model_version": get_model_version(),
"boxes": boxes,
}
def detect_page_api(pdf_file, page_number=1, dpi=DETECTION_DPI, conf=0.25):
"""
Render one PDF page at the given DPI and detect frames on it.
Testing/fallback path β€” prefer detect_image_api to avoid re-uploading
large PDFs per page.
"""
if pdf_file is None:
return {"error": "pdf file is required"}
path = pdf_file if isinstance(pdf_file, str) else pdf_file.name
rgb = render_pdf_page(path, int(page_number), dpi=int(dpi))
if rgb is None:
return {"error": f"could not render page {page_number}"}
bgr = cv2.cvtColor(rgb, cv2.COLOR_RGB2BGR)
boxes = detect_rectangles(bgr, conf=float(conf))
h, w = rgb.shape[:2]
return {
"page_number": int(page_number),
"dpi": int(dpi),
"width_px": int(w),
"height_px": int(h),
"model_version": get_model_version(),
"boxes": boxes,
}
# ---------------------------------------------------------------------------
# Per-page label JSON import/export (canonical ground-truth format)
# ---------------------------------------------------------------------------
_LABEL_DISPLAY_DPI = 100 # the Review tab canvas coordinate space
# Autosave: when running from the repo checkout (not the HF Space), every
# annotation change is checkpointed to disk so a browser refresh/crash loses
# nothing. Files go to a staging dir, NOT the canonical labels dir β€” the
# canonical ground truth is only ever written via merge_labels.py.
def _resolve_autosave_dir():
env = os.environ.get("FACADE_AUTOSAVE_DIR")
if env:
p = Path(env)
p.mkdir(parents=True, exist_ok=True)
return p
try:
working_set = Path(__file__).resolve().parents[2] / "data" / "working_set"
except IndexError: # app sits near filesystem root (HF Space: /app/app.py)
return None
if working_set.is_dir():
p = working_set / "autosave"
p.mkdir(parents=True, exist_ok=True)
return p
return None # standalone checkout β€” autosave off
_AUTOSAVE_DIR = _resolve_autosave_dir()
def _slugify_doc(stem):
# must match scripts/annotation/common.py::doc_slug
return re.sub(r"[^A-Za-z0-9._-]+", "_", stem).strip("_")[:60]
def _source_tag(ann, model_version):
src = ann.get("source", "unknown")
if src == "yolo":
return f"yolo:{model_version}"
if src in ("human_added", "human"):
return "human:gradio"
return src
def _page_label_data(rv_pdf, page, anns, audit_status):
"""Build one page's canonical label dict from canvas annotations.
Coordinates are converted from the 100-DPI canvas space to PDF points so
labels are DPI-independent and merge with other sources downstream. The
per-box "status" key is an extra field for exact import round-trips.
"""
mv = get_model_version()
boxes, rejected = [], []
for ann in anns:
entry = {
"id": ann.get("uid") or str(uuid.uuid4()),
"x1_pt": round(px_to_pt(ann["x1"], _LABEL_DISPLAY_DPI), 3),
"y1_pt": round(px_to_pt(ann["y1"], _LABEL_DISPLAY_DPI), 3),
"x2_pt": round(px_to_pt(ann["x2"], _LABEL_DISPLAY_DPI), 3),
"y2_pt": round(px_to_pt(ann["y2"], _LABEL_DISPLAY_DPI), 3),
"source": _source_tag(ann, mv),
"conf": round(float(ann.get("conf", 1.0)), 4),
"status": ann.get("status", "pending"),
"history": ann.get("history", []),
}
(rejected if ann.get("status") == "rejected" else boxes).append(entry)
return {
"doc": _slugify_doc(Path(rv_pdf).stem),
"pdf": Path(rv_pdf).name,
"page": int(page),
"page_type": "",
"audit_status": audit_status,
"audited_at": datetime.now(UTC).isoformat(),
"boxes": boxes,
"rejected_boxes": rejected,
}
def _autosave(rv_pdf, rv_annotations):
"""Checkpoint every annotated page to the autosave dir. Never raises β€”
a failed checkpoint must not break the edit that triggered it."""
if _AUTOSAVE_DIR is None or rv_pdf is None:
return
try:
out_dir = _AUTOSAVE_DIR / _slugify_doc(Path(rv_pdf).stem)
out_dir.mkdir(parents=True, exist_ok=True)
for page, anns in (rv_annotations or {}).items():
if not anns:
continue
# A page still holding unreviewed machine boxes is not an audit
unreviewed = any(
a.get("status", "pending") == "pending" and not str(a.get("source", "")).startswith("human")
for a in anns
)
data = _page_label_data(rv_pdf, page, anns, "machine" if unreviewed else "audited")
(out_dir / f"p{int(page):03d}.json").write_text(json.dumps(data, indent=2), encoding="utf-8")
except Exception as exc:
print(f"[autosave] checkpoint failed: {exc}")
def on_rv_export_labels(rv_pdf, rv_annotations):
"""Export every page with annotations as canonical label JSONs (zipped)."""
if rv_pdf is None:
return None, "No PDF loaded."
pages = {p: anns for p, anns in (rv_annotations or {}).items() if anns}
if not pages:
return None, "No annotations to export."
stem = Path(rv_pdf).stem
tmp_dir = Path(tempfile.mkdtemp(prefix="labels_"))
written = []
for page, anns in sorted(pages.items()):
data = _page_label_data(rv_pdf, page, anns, "audited")
out = tmp_dir / f"p{int(page):03d}.json"
out.write_text(json.dumps(data, indent=2), encoding="utf-8")
written.append(out)
zip_path = tmp_dir / f"{stem}_labels.zip"
with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf:
for f in written:
zf.write(f, f.name)
return str(zip_path), f"Exported labels for {len(written)} page(s)."
def on_rv_import_labels(label_files, rv_pdf, rv_page, rv_annotations):
"""Import canonical label JSON(s) into the Review tab.
Boxes from audited/verified files arrive approved; machine bootstrap files
arrive pending; rejected_boxes arrive rejected (so they round-trip).
"""
if not label_files:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "No label files provided."
rv_annotations = dict(rv_annotations or {})
loaded_pages = []
for lf in label_files:
path = lf if isinstance(lf, str) else lf.name
try:
data = json.loads(Path(path).read_text(encoding="utf-8"))
except Exception as exc:
return (
_build_canvas_json(None, rv_annotations.get(rv_page, [])),
_build_review_df(rv_annotations.get(rv_page, [])),
rv_annotations,
f"Failed to parse {Path(path).name}: {exc}",
)
page = int(data.get("page", 1))
default_status = "approved" if data.get("audit_status") in ("audited", "verified") else "pending"
anns = []
for status, group in ((default_status, data.get("boxes", [])), ("rejected", data.get("rejected_boxes", []))):
for box in group:
anns.append(
{
"id": len(anns),
"uid": box.get("id"),
"x1": pt_to_px(box["x1_pt"], _LABEL_DISPLAY_DPI),
"y1": pt_to_px(box["y1_pt"], _LABEL_DISPLAY_DPI),
"x2": pt_to_px(box["x2_pt"], _LABEL_DISPLAY_DPI),
"y2": pt_to_px(box["y2_pt"], _LABEL_DISPLAY_DPI),
"conf": float(box.get("conf", 1.0)),
# autosave files carry per-box status for exact resume
"status": "rejected" if status == "rejected" else box.get("status", status),
"source": box.get("source", "unknown"),
"history": box.get("history", []),
}
)
rv_annotations[page] = anns
loaded_pages.append(page)
_autosave(rv_pdf, rv_annotations)
current = rv_annotations.get(rv_page, [])
img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100) if rv_pdf is not None and rv_page in loaded_pages else None
return (
_build_canvas_json(img_rgb, current),
_build_review_df(current),
rv_annotations,
f"Imported labels for page(s) {sorted(loaded_pages)}.",
)
# ---------------------------------------------------------------------------
# Send to Backend handler
# ---------------------------------------------------------------------------
def send_to_backend(
rv_annotations, backend_url, document_id, frame_template_id, x_tenant_id, x_user_id, color, quantity_multiplier
):
shapes = []
for page_num, anns in sorted((rv_annotations or {}).items()):
for ann in anns:
if ann.get("status") == "approved":
shapes.append(
{
"page_number": page_num,
"x1": ann["x1"],
"y1": ann["y1"],
"x2": ann["x2"],
"y2": ann["y2"],
}
)
if not shapes:
return {"error": "No approved annotations. Go to Review & Edit, detect + approve boxes first."}
if not backend_url:
return {"error": "Backend URL is required."}
url = f"{backend_url.rstrip('/')}/api/v1/tenants/documents/{int(document_id)}/frame-instances/from-annotation"
payload = {
"frame_template_id": int(frame_template_id),
"shapes": shapes,
"quantity_multiplier": int(quantity_multiplier),
}
if color and color.strip():
payload["color"] = color.strip()
headers = {
"Content-Type": "application/json",
"x-tenant-id": str(int(x_tenant_id)),
"x-user-id": str(int(x_user_id)),
}
try:
resp = requests.post(url, json=payload, headers=headers, timeout=15)
try:
body = resp.json()
except Exception:
body = resp.text
return {"status_code": resp.status_code, "response": body}
except requests.exceptions.RequestException as exc:
return {"error": str(exc)}
# ---------------------------------------------------------------------------
# Review & Edit tab handlers
# ---------------------------------------------------------------------------
def on_rv_pdf_upload(pdf_file):
if pdf_file is None or not PDF_SUPPORT:
return (_build_canvas_json(None, []), _build_review_df([]), "No PDF loaded", None, 1, 1, {}, [])
pdf_path = pdf_file if isinstance(pdf_file, str) else pdf_file.name
total = _count_pdf_pages(pdf_path)
img_rgb = _render_pdf_page(pdf_path, 1, dpi=100)
return (_build_canvas_json(img_rgb, []), _build_review_df([]), f"Page 1 / {total}", pdf_path, 1, total, {}, [])
def _navigate(rv_pdf, rv_page, rv_total, rv_annotations, delta):
if rv_pdf is None:
return _build_canvas_json(None, []), _build_review_df([]), "No PDF loaded", rv_page
new_page = max(1, min(rv_total, rv_page + delta))
anns = rv_annotations.get(new_page, [])
img_rgb = _render_pdf_page(rv_pdf, new_page, dpi=100)
return (
_build_canvas_json(img_rgb, anns),
_build_review_df(anns),
f"Page {new_page} / {rv_total} β€” {len(anns)} box(es)",
new_page,
)
def on_rv_prev(rv_pdf, rv_page, rv_total, rv_annotations):
return _navigate(rv_pdf, rv_page, rv_total, rv_annotations, -1)
def on_rv_next(rv_pdf, rv_page, rv_total, rv_annotations):
return _navigate(rv_pdf, rv_page, rv_total, rv_annotations, +1)
def on_rv_goto_page(page_str, rv_pdf, rv_total, rv_annotations):
try:
page_num = max(1, min(int(rv_total), int(str(page_str).strip())))
except (ValueError, TypeError):
return gr.update(), gr.update(), gr.update(), gr.update()
if rv_pdf is None:
return _build_canvas_json(None, []), _build_review_df([]), "No PDF loaded", 1
anns = rv_annotations.get(page_num, [])
img_rgb = _render_pdf_page(rv_pdf, page_num, dpi=100)
return (
_build_canvas_json(img_rgb, anns),
_build_review_df(anns),
f"Page {page_num} / {rv_total} β€” {len(anns)} box(es)",
page_num,
)
def on_canvas_events(events_json, rv_pdf, rv_page, rv_annotations, rv_ledger):
"""Process annotation changes sent from the JS canvas."""
if not events_json or not events_json.strip():
return gr.update(), rv_annotations, rv_ledger, gr.update(), -1
try:
data = json.loads(events_json)
except Exception:
return gr.update(), rv_annotations, rv_ledger, gr.update(), -1
raw_anns = data.get("annotations", [])
selected = int(data.get("selected", -1))
action = data.get("action", "sync")
old_anns = list(rv_annotations.get(rv_page, []))
new_anns = []
for i, a in enumerate(raw_anns):
new_anns.append(
{
"id": i,
"x1": float(a["x1"]),
"y1": float(a["y1"]),
"x2": float(a["x2"]),
"y2": float(a["y2"]),
"conf": float(a.get("conf", 1.0)),
"status": a.get("status", "pending"),
"source": a.get("source", "human_added"),
}
)
rv_annotations = dict(rv_annotations)
rv_annotations[rv_page] = new_anns
events = []
if action == "created" and new_anns:
ann = new_anns[-1]
events.append(_make_event(rv_page, ann, "created", before=None, after=_make_box(ann)))
elif action in ("moved", "resized") and 0 <= selected < len(new_anns):
if selected < len(old_anns):
events.append(
_make_event(
rv_page,
new_anns[selected],
"modified",
before=_make_box(old_anns[selected]),
after=_make_box(new_anns[selected]),
)
)
rv_ledger = _append_events(rv_ledger, events)
if action in ("created", "moved", "resized", "deleted"):
_autosave(rv_pdf, rv_annotations)
canvas_json = _build_canvas_json(None, new_anns, selected=selected)
return canvas_json, rv_annotations, rv_ledger, _build_review_df(new_anns), selected
def on_rv_detect(rv_pdf, rv_page, rv_annotations, rv_conf):
if rv_pdf is None:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "No PDF loaded."
img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100)
if img_rgb is None:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "Failed to render page."
from detection import _get_model
using_yolo = _get_model() is not None
# Detect on a DETECTION_DPI render (the model's working resolution), then
# scale boxes back into the 100-DPI display space the canvas uses.
detect_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=DETECTION_DPI)
if detect_rgb is None:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "Failed to render page."
detect_bgr = cv2.cvtColor(detect_rgb, cv2.COLOR_RGB2BGR)
boxes = detect_rectangles(detect_bgr, conf=float(rv_conf))
scale = 100.0 / DETECTION_DPI
annotations = [
{
"id": i,
"x1": b["x1"] * scale,
"y1": b["y1"] * scale,
"x2": b["x2"] * scale,
"y2": b["y2"] * scale,
"conf": b["conf"],
"status": "pending",
"source": "yolo" if using_yolo else "opencv",
}
for i, b in enumerate(boxes)
]
rv_annotations = dict(rv_annotations)
rv_annotations[rv_page] = annotations
_autosave(rv_pdf, rv_annotations)
engine = "YOLO" if using_yolo else "OpenCV"
return (
_build_canvas_json(img_rgb, annotations),
_build_review_df(annotations),
rv_annotations,
f"[{engine}] Found {len(boxes)} panel(s) on page {rv_page}.",
)
def _box_iou(a, b):
ix1, iy1 = max(a["x1"], b["x1"]), max(a["y1"], b["y1"])
ix2, iy2 = min(a["x2"], b["x2"]), min(a["y2"], b["y2"])
inter = max(0.0, ix2 - ix1) * max(0.0, iy2 - iy1)
if inter <= 0:
return 0.0
area_a = (a["x2"] - a["x1"]) * (a["y2"] - a["y1"])
area_b = (b["x2"] - b["x1"]) * (b["y2"] - b["y1"])
return inter / (area_a + area_b - inter)
def on_rv_find_more(rv_pdf, rv_page, rv_annotations, rv_conf):
"""Low-conf re-detect that MERGES: keeps every existing box (approved,
rejected, human, pending) and only adds detections that don't overlap one.
Lets the model fill audit gaps without the user redrawing by hand."""
anns = rv_annotations.get(rv_page, [])
if rv_pdf is None:
return gr.update(), _build_review_df(anns), rv_annotations, "No PDF loaded."
detect_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=DETECTION_DPI)
if detect_rgb is None:
return gr.update(), _build_review_df(anns), rv_annotations, "Failed to render page."
from detection import _get_model
using_yolo = _get_model() is not None
detect_bgr = cv2.cvtColor(detect_rgb, cv2.COLOR_RGB2BGR)
conf = min(float(rv_conf), 0.10) # always cast a wide net regardless of the slider
boxes = detect_rectangles(detect_bgr, conf=conf)
scale = 100.0 / DETECTION_DPI
rv_annotations = dict(rv_annotations)
new_anns = list(anns)
added = 0
for b in boxes:
cand = {
"x1": b["x1"] * scale,
"y1": b["y1"] * scale,
"x2": b["x2"] * scale,
"y2": b["y2"] * scale,
}
# an existing box of ANY status wins β€” including rejected, so boxes the
# user already threw out don't come back
if any(_box_iou(cand, a) > 0.30 for a in anns):
continue
new_anns.append(
dict(
cand,
id=len(new_anns),
conf=b["conf"],
status="pending",
source="yolo" if using_yolo else "opencv",
)
)
added += 1
rv_annotations[rv_page] = new_anns
_autosave(rv_pdf, rv_annotations)
return (
_build_canvas_json(None, new_anns),
_build_review_df(new_anns),
rv_annotations,
f"Find More @conf {conf:.2f}: +{added} new box(es) (existing {len(anns)} kept).",
)
def on_rv_detect_all(rv_pdf, rv_page, rv_total, rv_annotations, rv_conf):
if rv_pdf is None:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "No PDF loaded."
if not PDF_SUPPORT:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, "PDF support unavailable."
from detection import _get_model
using_yolo = _get_model() is not None
all_anns = dict(rv_annotations)
total_found = 0
detect_dpi = DETECTION_DPI # the model's working resolution; lower DPI misses most frames
try:
doc = fitz.open(rv_pdf)
mat = fitz.Matrix(detect_dpi / 72, detect_dpi / 72)
for page_num in range(1, rv_total + 1):
try:
page = doc[page_num - 1]
pix = page.get_pixmap(matrix=mat, colorspace=fitz.csRGB)
img_rgb = np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width, 3)
except Exception:
continue
img_bgr = cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR)
boxes = detect_rectangles(img_bgr, conf=float(rv_conf))
# Scale coords back to 100-DPI space so they match the display image
scale = 100.0 / detect_dpi
all_anns[page_num] = [
{
"id": i,
"x1": b["x1"] * scale,
"y1": b["y1"] * scale,
"x2": b["x2"] * scale,
"y2": b["y2"] * scale,
"conf": b["conf"],
"status": "pending",
"source": "yolo" if using_yolo else "opencv",
}
for i, b in enumerate(boxes)
]
total_found += len(boxes)
doc.close()
except Exception as exc:
return _build_canvas_json(None, []), _build_review_df([]), rv_annotations, f"Error: {exc}"
img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100)
anns = all_anns.get(rv_page, [])
_autosave(rv_pdf, all_anns)
engine = "YOLO" if using_yolo else "OpenCV"
return (
_build_canvas_json(img_rgb, anns),
_build_review_df(anns),
all_anns,
f"[{engine}] Detected {total_found} panel(s) across {rv_total} page(s).",
)
def on_fill_edit_form(selected_idx, rv_page, rv_annotations):
anns = rv_annotations.get(rv_page, [])
idx = int(selected_idx) if selected_idx is not None else -1
if idx < 0 or idx >= len(anns):
return None, None, None, None
a = anns[idx]
return a["x1"], a["y1"], a["x2"], a["y2"]
def on_rv_update_box(rv_pdf, rv_page, rv_annotations, ledger, selected_idx, x1, y1, x2, y2):
anns = rv_annotations.get(rv_page, [])
if rv_pdf is None or not anns or selected_idx is None or int(selected_idx) < 0:
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first."
idx = int(selected_idx)
if idx >= len(anns):
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection."
try:
x1, y1, x2, y2 = float(x1), float(y1), float(x2), float(y2)
except (TypeError, ValueError):
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid coordinates."
if x2 <= x1 or y2 <= y1:
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "x2>x1 and y2>y1 required."
rv_annotations = dict(rv_annotations)
new_anns = list(anns)
old = new_anns[idx]
before = _make_box(old)
new_anns[idx] = dict(old, x1=x1, y1=y1, x2=x2, y2=y2)
rv_annotations[rv_page] = new_anns
event = _make_event(rv_page, old, "modified", before=before, after=_make_box(new_anns[idx]))
ledger = _append_events(ledger, [event])
_autosave(rv_pdf, rv_annotations)
img_rgb = _render_pdf_page(rv_pdf, rv_page, dpi=100)
return (
_build_canvas_json(img_rgb, new_anns, selected=idx),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Updated box #{idx + 1}. {_ledger_summary(ledger)}",
)
def on_rv_approve_all(rv_pdf, rv_page, rv_annotations, ledger):
if rv_pdf is None or not rv_annotations.get(rv_page):
return gr.update(), _build_review_df([]), rv_annotations, ledger, "No boxes on this page."
rv_annotations = dict(rv_annotations)
old_anns = rv_annotations[rv_page]
new_anns = [dict(a, status="approved") for a in old_anns]
rv_annotations[rv_page] = new_anns
events = [
_make_event(rv_page, a, "accepted", before=_make_box(a), after=_make_box(a))
for a in old_anns
if a.get("status") != "approved"
]
ledger = _append_events(ledger, events)
_autosave(rv_pdf, rv_annotations)
return (
_build_canvas_json(None, new_anns),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Approved {len(new_anns)} box(es). {_ledger_summary(ledger)}",
)
def on_rv_reject_all(rv_pdf, rv_page, rv_annotations, ledger):
if rv_pdf is None or not rv_annotations.get(rv_page):
return gr.update(), _build_review_df([]), rv_annotations, ledger, "No boxes on this page."
rv_annotations = dict(rv_annotations)
old_anns = rv_annotations[rv_page]
new_anns = [dict(a, status="rejected") for a in old_anns]
rv_annotations[rv_page] = new_anns
events = [
_make_event(rv_page, a, "rejected", before=_make_box(a), after=None)
for a in old_anns
if a.get("status") != "rejected"
]
ledger = _append_events(ledger, events)
_autosave(rv_pdf, rv_annotations)
return (
_build_canvas_json(None, new_anns),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Rejected {len(new_anns)} box(es). {_ledger_summary(ledger)}",
)
def on_rv_approve_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx):
anns = rv_annotations.get(rv_page, [])
if rv_pdf is None or not anns or selected_idx is None or selected_idx < 0:
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first."
idx = int(selected_idx)
if idx >= len(anns):
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection."
rv_annotations = dict(rv_annotations)
new_anns = list(anns)
old = new_anns[idx]
new_anns[idx] = dict(old, status="approved")
rv_annotations[rv_page] = new_anns
event = _make_event(rv_page, old, "accepted", before=_make_box(old), after=_make_box(new_anns[idx]))
ledger = _append_events(ledger, [event])
_autosave(rv_pdf, rv_annotations)
return (
_build_canvas_json(None, new_anns, selected=idx),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Approved box #{idx + 1}. {_ledger_summary(ledger)}",
)
def on_rv_reject_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx):
anns = rv_annotations.get(rv_page, [])
if rv_pdf is None or not anns or selected_idx is None or selected_idx < 0:
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first."
idx = int(selected_idx)
if idx >= len(anns):
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection."
rv_annotations = dict(rv_annotations)
new_anns = list(anns)
old = new_anns[idx]
new_anns[idx] = dict(old, status="rejected")
rv_annotations[rv_page] = new_anns
event = _make_event(rv_page, old, "rejected", before=_make_box(old), after=_make_box(new_anns[idx]))
ledger = _append_events(ledger, [event])
_autosave(rv_pdf, rv_annotations)
return (
_build_canvas_json(None, new_anns, selected=idx),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Rejected box #{idx + 1}. {_ledger_summary(ledger)}",
)
def on_rv_delete_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx):
anns = rv_annotations.get(rv_page, [])
if rv_pdf is None or not anns or selected_idx is None or selected_idx < 0:
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first.", selected_idx
idx = int(selected_idx)
if idx >= len(anns):
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection.", selected_idx
rv_annotations = dict(rv_annotations)
removed = anns[idx]
new_anns = [a for i, a in enumerate(anns) if i != idx]
for j, a in enumerate(new_anns):
a["id"] = j
rv_annotations[rv_page] = new_anns
event = _make_event(rv_page, removed, "rejected", before=_make_box(removed), after=None)
ledger = _append_events(ledger, [event])
_autosave(rv_pdf, rv_annotations)
new_sel = min(idx, len(new_anns) - 1) if new_anns else -1
return (
_build_canvas_json(None, new_anns, selected=new_sel),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Deleted box #{idx + 1}. {_ledger_summary(ledger)}",
new_sel,
)
def on_rv_copy_selected(rv_pdf, rv_page, rv_annotations, ledger, selected_idx):
anns = rv_annotations.get(rv_page, [])
if not anns or selected_idx is None or selected_idx < 0:
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Select a box first."
idx = int(selected_idx)
if idx >= len(anns):
return gr.update(), _build_review_df(anns), rv_annotations, ledger, "Invalid selection."
rv_annotations = dict(rv_annotations)
src = anns[idx]
new_ann = dict(
src, id=len(anns), x1=src["x1"] + 15, y1=src["y1"] + 15, x2=src["x2"] + 15, y2=src["y2"] + 15, status="pending"
)
new_anns = [*list(anns), new_ann]
rv_annotations[rv_page] = new_anns
new_sel = len(new_anns) - 1
event = _make_event(rv_page, new_ann, "created", before=None, after=_make_box(new_ann))
ledger = _append_events(ledger, [event])
_autosave(rv_pdf, rv_annotations)
return (
_build_canvas_json(None, new_anns, selected=new_sel),
_build_review_df(new_anns),
rv_annotations,
ledger,
f"Copied to box #{new_sel + 1}. {_ledger_summary(ledger)}",
)
def on_row_select(evt: gr.SelectData):
return evt.index[0] if evt and evt.index else -1
def on_export_events(ledger):
if not ledger:
return None
return export_events_jsonl(ledger)
def on_export_dataset(ledger, rv_pdf):
if not ledger:
return None
return export_yolo_dataset_zip(ledger, rv_pdf)
# ---------------------------------------------------------------------------
# Gradio UI
# ---------------------------------------------------------------------------
_model_version = get_model_version()
_CANVAS_HTML = """
<div id="rv-canvas-outer" style="width:100%;min-height:600px;background:#0f0f0f;border-radius:8px;overflow:hidden;position:relative;user-select:none;-webkit-user-select:none;">
<label style="position:absolute;top:8px;left:8px;z-index:10;background:rgba(0,0,0,0.6);color:#fff;font-family:monospace;font-size:12px;padding:4px 8px;border-radius:4px;cursor:pointer;">
<input type="checkbox" id="rv-hide-rejected" style="vertical-align:middle;margin-right:4px;">hide rejected <kbd style="opacity:0.6;">H</kbd>
</label>
<canvas id="rv-ann-canvas" style="display:block;width:100%;cursor:crosshair;touch-action:none;"></canvas>
<div id="rv-canvas-msg" style="position:absolute;top:50%;left:50%;transform:translate(-50%,-50%);color:#555;font-family:monospace;font-size:13px;pointer-events:none;text-align:center;line-height:1.8;">
Upload a PDF to begin<br><span style="color:#333;font-size:11px;">Scroll to zoom &bull; right-click + drag to pan &bull; double-click to reset view<br>Drag on canvas to draw &bull; click box to select &bull; drag to move &bull; drag corner to resize</span>
</div>
</div>
<div id="rv-ann-tooltip" style="position:fixed;display:none;background:rgba(20,20,20,0.93);color:#fff;border-radius:6px;padding:8px 12px;font-family:monospace;font-size:12px;pointer-events:none;z-index:9999;box-shadow:0 2px 10px rgba(0,0,0,0.5);line-height:1.7;max-width:280px;"></div>
"""
_CANVAS_JS = """
<script>
(function() {
'use strict';
var COLORS = {pending:'#F97316', approved:'#22C55E', rejected:'#EF4444'};
var HANDLE = 9;
var canvas = null, ctx = null;
var cachedImg = null, cachedImgSrc = '';
var imgW = 0, imgH = 0;
var anns = [], selected = -1;
var drag = null;
var lastRender = '';
// View transform: zoom is multiplied on top of the fit-to-width base scale;
// pan offsets are in canvas pixels. zoom 1 + pan 0 == whole page fits the column.
var zoom = 1, panX = 0, panY = 0, spaceDown = false;
// ── Boot ────────────────────────────────────────────────────────────────
var _iv = setInterval(function() {
if (document.getElementById('rv-ann-canvas')) {
clearInterval(_iv);
canvas = document.getElementById('rv-ann-canvas');
ctx = canvas.getContext('2d');
canvas.addEventListener('mousedown', onDown);
canvas.addEventListener('mousemove', onMove);
canvas.addEventListener('mouseup', onUp);
canvas.addEventListener('mouseleave', function(e) { hideTooltip(); if (drag) { onUp(e); } });
canvas.addEventListener('wheel', onWheel, { passive: false });
canvas.addEventListener('dblclick', function() { resetView(); draw(); });
canvas.addEventListener('auxclick', function(e) { e.preventDefault(); });
canvas.addEventListener('contextmenu', function(e) { e.preventDefault(); });
var hr = document.getElementById('rv-hide-rejected');
if (hr) hr.addEventListener('change', draw);
new ResizeObserver(refit).observe(canvas.parentElement);
setInterval(checkRender, 150);
}
}, 200);
// ── Python β†’ JS sync ────────────────────────────────────────────────────
function checkRender() {
var el = document.querySelector('#rv-canvas-render textarea');
if (!el || !el.value || el.value === lastRender) return;
lastRender = el.value;
try { applyRender(JSON.parse(el.value)); } catch(e) {}
}
function applyRender(d) {
anns = (d.annotations || []).map(function(a) { return Object.assign({}, a); });
selected = (d.selected !== undefined && d.selected !== null) ? parseInt(d.selected, 10) : -1;
if (isNaN(selected)) selected = -1;
var msg = document.getElementById('rv-canvas-msg');
if (d.image && d.image !== cachedImgSrc) {
imgW = d.imageW; imgH = d.imageH;
resetView();
var im = new Image();
im.onload = function() { cachedImg = im; cachedImgSrc = d.image; refit(); };
im.onerror = function() { console.warn('Canvas: failed to load image'); };
im.src = d.image;
if (msg) msg.style.display = 'none';
} else {
draw();
if (msg && cachedImg) msg.style.display = 'none';
}
}
function refit() {
if (!canvas || !imgW || !imgH) return;
var w = canvas.parentElement.clientWidth || 800;
canvas.width = Math.round(w);
canvas.height = Math.round(imgH * (w / imgW));
clampPan();
draw();
}
function s() { return canvas && imgW ? (canvas.width / imgW) * zoom : 1; }
// Rejected boxes stay in the data (training hard-negatives) but can be
// hidden from the canvas; hidden boxes are also skipped by hit-testing.
function rejectedHidden() {
var cb = document.getElementById('rv-hide-rejected');
return !!(cb && cb.checked);
}
function isHidden(ann) { return rejectedHidden() && ann.status === 'rejected'; }
// ── Zoom / pan ───────────────────────────────────────────────────────────
function resetView() { zoom = 1; panX = 0; panY = 0; }
function clampPan() {
if (!canvas || !imgW) return;
var sc = s();
panX = Math.min(0, Math.max(canvas.width - imgW * sc, panX));
panY = Math.min(0, Math.max(canvas.height - imgH * sc, panY));
}
function onWheel(e) {
if (!imgW) return;
e.preventDefault();
var r = canvas.getBoundingClientRect();
var mx = (e.clientX - r.left) * (canvas.width / r.width);
var my = (e.clientY - r.top) * (canvas.height / r.height);
var nz = Math.min(16, Math.max(1, zoom * Math.pow(1.0015, -e.deltaY)));
panX = mx - (mx - panX) * (nz / zoom);
panY = my - (my - panY) * (nz / zoom);
zoom = nz;
if (zoom <= 1.001) resetView();
clampPan();
draw();
}
// ── Draw ─────────────────────────────────────────────────────────────────
function draw() {
if (!canvas) return;
var sc = s();
ctx.clearRect(0, 0, canvas.width, canvas.height);
if (cachedImg) ctx.drawImage(cachedImg, panX, panY, imgW * sc, imgH * sc);
anns.forEach(function(ann, i) {
if (isHidden(ann)) return;
var x1 = ann.x1*sc+panX, y1 = ann.y1*sc+panY, x2 = ann.x2*sc+panX, y2 = ann.y2*sc+panY;
var c = COLORS[ann.status] || COLORS.pending;
var sel = (i === selected);
// Fill
ctx.fillStyle = c + (sel ? '55' : '22');
ctx.fillRect(x1, y1, x2-x1, y2-y1);
// Border
ctx.strokeStyle = c;
ctx.lineWidth = sel ? 3 : 2;
ctx.strokeRect(x1, y1, x2-x1, y2-y1);
// Resize handles on selected box
if (sel) {
[[x1,y1],[x2,y1],[x2,y2],[x1,y2]].forEach(function(pt) {
ctx.fillStyle = '#fff';
ctx.fillRect(pt[0]-HANDLE/2, pt[1]-HANDLE/2, HANDLE, HANDLE);
ctx.strokeStyle = c; ctx.lineWidth = 2;
ctx.strokeRect(pt[0]-HANDLE/2, pt[1]-HANDLE/2, HANDLE, HANDLE);
});
}
});
// Draw-in-progress preview (drag coords are in image space)
if (drag && drag.mode === 'draw') {
var ox = drag.ox*sc+panX, oy = drag.oy*sc+panY, dcx = drag.cx*sc+panX, dcy = drag.cy*sc+panY;
ctx.strokeStyle = '#2563EB'; ctx.lineWidth = 2;
ctx.setLineDash([5, 4]);
ctx.strokeRect(ox, oy, dcx - ox, dcy - oy);
ctx.setLineDash([]);
ctx.fillStyle = 'rgba(37,99,235,0.1)';
ctx.fillRect(ox, oy, dcx - ox, dcy - oy);
}
// Zoom badge
if (zoom > 1.001) {
var badge = Math.round(zoom * 100) + '% \\u00B7 right-drag = pan \\u00B7 dbl-click = reset';
ctx.font = '12px monospace';
var bw = ctx.measureText(badge).width + 16;
ctx.fillStyle = 'rgba(0,0,0,0.6)';
ctx.fillRect(canvas.width - bw - 8, 8, bw, 22);
ctx.fillStyle = '#fff';
ctx.fillText(badge, canvas.width - bw, 23);
}
}
// ── Tooltip ──────────────────────────────────────────────────────────────
var hoveredIdx = -1;
function showTooltip(ann, i, clientX, clientY) {
var tip = document.getElementById('rv-ann-tooltip');
if (!tip) return;
tip.innerHTML =
'<b>Box #' + (i + 1) + '</b><br>' +
'conf: ' + ((ann.conf || 1) * 1).toFixed(2) + '<br>' +
'status: ' + ann.status + '<br>' +
'(' + Math.round(ann.x1) + ', ' + Math.round(ann.y1) + ') &rarr; ' +
'(' + Math.round(ann.x2) + ', ' + Math.round(ann.y2) + ')';
tip.style.display = 'block';
tip.style.left = (clientX + 14) + 'px';
tip.style.top = (clientY + 14) + 'px';
}
function hideTooltip() {
var tip = document.getElementById('rv-ann-tooltip');
if (tip) tip.style.display = 'none';
hoveredIdx = -1;
}
// ── Hit testing (image-space coords) ─────────────────────────────────────
function hitHandle(ann, ix, iy) {
var h = HANDLE / s();
var pts = [
[ann.x1, ann.y1, 'nw'], [ann.x2, ann.y1, 'ne'],
[ann.x2, ann.y2, 'se'], [ann.x1, ann.y2, 'sw'],
];
for (var i = 0; i < pts.length; i++) {
if (Math.abs(ix - pts[i][0]) <= h && Math.abs(iy - pts[i][1]) <= h)
return pts[i][2];
}
return null;
}
function hitBox(ann, ix, iy) {
return ix >= ann.x1 && ix <= ann.x2 && iy >= ann.y1 && iy <= ann.y2;
}
// Mouse position in IMAGE pixels (pan/zoom already removed)
function pos(e) {
var r = canvas.getBoundingClientRect();
var cx = (e.clientX - r.left) * (canvas.width / r.width);
var cy = (e.clientY - r.top) * (canvas.height / r.height);
var sc = s();
return { x: (cx - panX) / sc, y: (cy - panY) / sc };
}
// ── Mouse ────────────────────────────────────────────────────────────────
function onDown(e) {
hideTooltip();
// Pan: right or middle mouse drag, or hold Space + left-drag
if (e.button === 2 || e.button === 1 || spaceDown) {
e.preventDefault();
drag = { mode:'pan', sx:e.clientX, sy:e.clientY, px:panX, py:panY };
canvas.style.cursor = 'grabbing';
return;
}
if (e.button !== 0) return;
var p = pos(e);
// Shift+click: stamp a copy of the selected box centered at the cursor
// (the original stays selected so you can keep stamping down a row)
if (e.shiftKey && selected >= 0 && selected < anns.length) {
var tpl = anns[selected];
var hw = (tpl.x2 - tpl.x1) / 2, hh = (tpl.y2 - tpl.y1) / 2;
anns.push(Object.assign({}, tpl, {
x1: p.x - hw, y1: p.y - hh, x2: p.x + hw, y2: p.y + hh,
status: 'pending', source: 'human_added'
}));
draw();
sync('created');
return;
}
// Resize handles on selected box take priority
if (selected >= 0 && selected < anns.length) {
var dir = hitHandle(anns[selected], p.x, p.y);
if (dir) {
drag = { mode:'resize', dir:dir, idx:selected,
orig:Object.assign({}, anns[selected]), ox:p.x, oy:p.y };
return;
}
}
// Click inside any box to select + move
for (var i = anns.length - 1; i >= 0; i--) {
if (isHidden(anns[i])) continue;
if (hitBox(anns[i], p.x, p.y)) {
selected = i;
drag = { mode:'move', idx:i,
orig:Object.assign({}, anns[i]), ox:p.x, oy:p.y };
draw();
sync('select');
return;
}
}
// Empty area β€” start drawing new box
selected = -1;
drag = { mode:'draw', ox:p.x, oy:p.y, cx:p.x, cy:p.y };
draw();
}
function onMove(e) {
if (drag && drag.mode === 'pan') {
var r = canvas.getBoundingClientRect();
panX = drag.px + (e.clientX - drag.sx) * (canvas.width / r.width);
panY = drag.py + (e.clientY - drag.sy) * (canvas.height / r.height);
clampPan();
draw();
return;
}
var p = pos(e);
// Tooltip on hover (only when not dragging)
if (!drag) {
var hovered = -1;
for (var j = anns.length - 1; j >= 0; j--) {
if (isHidden(anns[j])) continue;
if (hitBox(anns[j], p.x, p.y)) { hovered = j; break; }
}
if (hovered >= 0) {
showTooltip(anns[hovered], hovered, e.clientX, e.clientY);
} else {
hideTooltip();
}
return;
}
var dx = p.x - drag.ox, dy = p.y - drag.oy;
if (drag.mode === 'move') {
var o = drag.orig;
anns[drag.idx] = Object.assign({}, o,
{ x1:o.x1+dx, y1:o.y1+dy, x2:o.x2+dx, y2:o.y2+dy });
} else if (drag.mode === 'resize') {
var o = drag.orig, a = Object.assign({}, o), d = drag.dir;
if (d.indexOf('e') >= 0) a.x2 = o.x2 + dx;
if (d.indexOf('w') >= 0) a.x1 = o.x1 + dx;
if (d.indexOf('s') >= 0) a.y2 = o.y2 + dy;
if (d.indexOf('n') >= 0) a.y1 = o.y1 + dy;
anns[drag.idx] = a;
} else if (drag.mode === 'draw') {
drag.cx = p.x; drag.cy = p.y;
}
draw();
}
function onUp(e) {
if (!drag) return;
if (drag.mode === 'pan') {
drag = null;
canvas.style.cursor = spaceDown ? 'grab' : 'crosshair';
return;
}
var action = drag.mode;
if (drag.mode === 'draw') {
var p = pos(e);
var x1c = Math.min(drag.ox, p.x), y1c = Math.min(drag.oy, p.y);
var x2c = Math.max(drag.ox, p.x), y2c = Math.max(drag.oy, p.y);
if ((x2c - x1c) * s() > 5 && (y2c - y1c) * s() > 5) {
anns.push({ x1:x1c, y1:y1c, x2:x2c, y2:y2c,
conf:1.0, status:'pending', source:'human_added' });
selected = anns.length - 1;
action = 'created';
} else {
drag = null; draw(); return;
}
} else {
action = drag.mode === 'resize' ? 'resized' : 'moved';
}
drag = null;
draw();
sync(action);
}
// ── JS-side copy (C key / canvas copy shortcut) ──────────────────────────
function copySelected() {
if (selected < 0 || selected >= anns.length) return;
var src = anns[selected];
anns.push(Object.assign({}, src, {
x1: src.x1+15, y1: src.y1+15, x2: src.x2+15, y2: src.y2+15, status:'pending'
}));
selected = anns.length - 1;
draw();
sync('created');
}
function deleteSelected() {
if (selected < 0 || selected >= anns.length) return;
anns.splice(selected, 1);
selected = anns.length > 0 ? Math.min(selected, anns.length - 1) : -1;
draw();
sync('deleted');
}
// ── JS β†’ Python sync ────────────────────────────────────────────────────
function sync(action) {
var tb = document.querySelector('#rv-canvas-events textarea');
if (!tb) return;
var payload = JSON.stringify({ annotations:anns, selected:selected, action:action || 'sync' });
tb.value = payload;
tb.dispatchEvent(new Event('input', { bubbles:true }));
tb.dispatchEvent(new Event('change', { bubbles:true }));
}
// ── Keyboard ─────────────────────────────────────────────────────────────
function triggerBtn(id) {
var el = document.getElementById(id);
if (!el) return;
var btn = el.tagName === 'BUTTON' ? el : el.querySelector('button');
if (btn) btn.click();
}
document.addEventListener('keydown', function(e) {
var tag = e.target.tagName;
if (tag === 'INPUT' || tag === 'TEXTAREA' || tag === 'SELECT') return;
switch (e.key) {
case 'c': case 'C': copySelected(); break;
case 'Delete': case 'Backspace': deleteSelected(); break;
case 'a': case 'A': triggerBtn('rv-approve-sel'); break;
case 'r': case 'R': triggerBtn('rv-reject-sel'); break;
case 'h': case 'H': {
var cb = document.getElementById('rv-hide-rejected');
if (cb) { cb.checked = !cb.checked; draw(); }
break;
}
case 'ArrowLeft': e.preventDefault(); triggerBtn('rv-prev'); break;
case 'ArrowRight': e.preventDefault(); triggerBtn('rv-next'); break;
case ' ': e.preventDefault();
if (!spaceDown && canvas) { spaceDown = true; if (!drag) canvas.style.cursor = 'grab'; }
break;
case '0': resetView(); draw(); break;
}
});
document.addEventListener('keyup', function(e) {
if (e.key === ' ') {
spaceDown = false;
if (canvas && !drag) canvas.style.cursor = 'crosshair';
}
});
})();
</script>
"""
# _CANVAS_JS must go through head= β€” gr.HTML inserts via innerHTML, which never
# executes <script> tags, so the canvas engine would silently not run.
with gr.Blocks(title="Takeoff AI Annotator", head=_CANVAS_JS) as demo:
gr.Markdown(f"# Takeoff AI Annotator \n_Model: `{_model_version}`_")
shapes_state = gr.State([])
rv_pdf_state = gr.State(None)
rv_page_state = gr.State(1)
rv_total_state = gr.State(1)
rv_annotations_state = gr.State({})
rv_ledger_state = gr.State([])
rv_selected_state = gr.State(-1)
with gr.Tabs():
# ------------------------------------------------------------------ #
# Tab 1: Detect #
# ------------------------------------------------------------------ #
with gr.Tab("Detect"):
with gr.Row():
with gr.Column():
image_input = gr.Image(label="Upload Image (PNG/JPG)", type="filepath")
if PDF_SUPPORT:
pdf_input = gr.File(label="Or Upload PDF", file_types=[".pdf"])
pdf_page_input = gr.Number(label="PDF Page Number", value=1, minimum=1, precision=0)
else:
pdf_input = gr.State(None)
pdf_page_input = gr.State(1)
gr.Markdown("_PDF support unavailable._")
page_number_input = gr.Number(
label="Page Number (sent to backend)",
value=1,
minimum=1,
precision=0,
)
with gr.Accordion("Detection Parameters", open=True):
conf_slider = gr.Slider(0.05, 0.95, value=0.25, step=0.05, label="Confidence Threshold (YOLO)")
gr.Markdown("_Sliders below are OpenCV fallback only._")
min_area_slider = gr.Slider(100, 10_000, value=500, step=100, label="Min Area (pxΒ²)")
max_area_slider = gr.Slider(1_000, 2_000_000, value=500_000, step=1_000, label="Max Area (pxΒ²)")
epsilon_slider = gr.Slider(0.005, 0.1, value=0.02, step=0.005, label="Epsilon Factor")
threshold_slider = gr.Slider(0, 255, value=127, step=1, label="Binarization Threshold")
detect_btn = gr.Button("Detect Panels", variant="primary")
status_output = gr.Textbox(label="Status", interactive=False)
with gr.Column():
annotated_output = gr.Image(label="Detected Rectangles")
json_output = gr.JSON(label="Payload Preview")
detect_btn.click(
fn=run_detection,
inputs=[
image_input,
pdf_input,
pdf_page_input,
page_number_input,
conf_slider,
min_area_slider,
max_area_slider,
epsilon_slider,
threshold_slider,
],
outputs=[annotated_output, json_output, shapes_state, status_output],
)
# ------------------------------------------------------------------ #
# Tab 2: Review & Edit #
# ------------------------------------------------------------------ #
with gr.Tab("Review & Edit"):
rv_pdf_input = gr.File(label="Upload PDF", file_types=[".pdf"])
with gr.Row():
rv_prev_btn = gr.Button("β—€", scale=1, min_width=48, elem_id="rv-prev")
rv_page_lbl = gr.Textbox(value="Upload a PDF to begin", show_label=False, interactive=False, scale=4)
rv_next_btn = gr.Button("β–Ά", scale=1, min_width=48, elem_id="rv-next")
rv_page_jump = gr.Textbox(value="", show_label=False, scale=1, min_width=80, placeholder="Go to page…")
rv_conf_slider = gr.Slider(0.05, 0.95, value=0.25, step=0.05, label="Conf", scale=3)
rv_detect_btn = gr.Button("Detect Page", variant="primary", scale=2)
rv_find_more_btn = gr.Button("πŸ”Ž Find More", scale=2)
rv_detect_all_btn = gr.Button("Detect All", scale=2)
with gr.Row():
with gr.Column(scale=7):
gr.HTML(_CANVAS_HTML)
rv_canvas_render_txt = gr.Textbox(visible=False, elem_id="rv-canvas-render")
rv_canvas_events_txt = gr.Textbox(visible=False, elem_id="rv-canvas-events")
with gr.Column(scale=3):
rv_status = gr.Textbox(show_label=False, interactive=False, placeholder="Status", lines=1)
gr.Markdown("**Selected box**")
rv_selected_lbl = gr.Textbox(value="β€” none β€”", show_label=False, interactive=False)
with gr.Row():
rv_approve_sel_btn = gr.Button(
"βœ“ Approve", variant="secondary", elem_id="rv-approve-sel", scale=2
)
rv_reject_sel_btn = gr.Button("βœ— Reject", variant="stop", elem_id="rv-reject-sel", scale=2)
with gr.Row():
rv_copy_sel_btn = gr.Button("⧉ Copy", scale=1)
rv_delete_sel_btn = gr.Button("πŸ—‘ Delete", variant="stop", elem_id="rv-delete-sel", scale=2)
gr.Markdown("**Edit coords** _(type values β†’ Update)_")
with gr.Row():
edit_x1 = gr.Number(label="x1", precision=0, min_width=80)
edit_y1 = gr.Number(label="y1", precision=0, min_width=80)
with gr.Row():
edit_x2 = gr.Number(label="x2", precision=0, min_width=80)
edit_y2 = gr.Number(label="y2", precision=0, min_width=80)
update_box_btn = gr.Button("↩ Update Box", variant="primary")
gr.Markdown("**Page actions**")
with gr.Row():
rv_approve_all_btn = gr.Button("βœ“ Approve All", variant="secondary")
rv_reject_all_btn = gr.Button("βœ— Reject All", variant="stop")
rv_df = gr.Dataframe(
headers=["#", "x1", "y1", "x2", "y2", "conf", "status"],
label="Detections β€” click row to select",
interactive=False,
wrap=False,
max_height=220,
)
with gr.Row():
export_events_btn = gr.Button("↓ Events", scale=1)
export_dataset_btn = gr.Button("↓ Dataset", scale=1)
export_labels_btn = gr.Button("↓ Labels", scale=1)
export_file_out = gr.File(label="Download", visible=True)
import_labels_in = gr.File(
label="Import label JSON(s)",
file_count="multiple",
file_types=[".json"],
)
gr.Markdown(
"`A` approve &nbsp; `R` reject &nbsp; `H` hide/show rejected &nbsp; `C` copy &nbsp; `Del` delete &nbsp; "
"`← β†’` pages &nbsp; β€” draw box by dragging on empty canvas \n"
"⚑ **Shift+click = stamp a copy** of the selected box at the cursor (select one good box, "
"then Shift+click each identical frame) &nbsp; πŸ”Ž **Find More** = re-detect at low confidence, "
"adds only boxes you don't already have \n"
"πŸ” **Scroll wheel = zoom** &nbsp; **right-click + drag = pan** &nbsp; "
"double-click or `0` = reset view"
+ (
f" \nπŸ’Ύ **Autosave on** β€” every change is checkpointed to `{_AUTOSAVE_DIR}`; "
"a refresh/crash loses nothing (re-import from there to resume)."
if _AUTOSAVE_DIR
else ""
)
)
# ── Wire events ─────────────────────────────────────────────────
rv_pdf_input.upload(
fn=on_rv_pdf_upload,
inputs=[rv_pdf_input],
outputs=[
rv_canvas_render_txt,
rv_df,
rv_page_lbl,
rv_pdf_state,
rv_page_state,
rv_total_state,
rv_annotations_state,
rv_ledger_state,
],
)
rv_prev_btn.click(
fn=on_rv_prev,
inputs=[rv_pdf_state, rv_page_state, rv_total_state, rv_annotations_state],
outputs=[rv_canvas_render_txt, rv_df, rv_page_lbl, rv_page_state],
)
rv_next_btn.click(
fn=on_rv_next,
inputs=[rv_pdf_state, rv_page_state, rv_total_state, rv_annotations_state],
outputs=[rv_canvas_render_txt, rv_df, rv_page_lbl, rv_page_state],
)
rv_page_jump.submit(
fn=on_rv_goto_page,
inputs=[rv_page_jump, rv_pdf_state, rv_total_state, rv_annotations_state],
outputs=[rv_canvas_render_txt, rv_df, rv_page_lbl, rv_page_state],
)
rv_canvas_events_txt.change(
fn=on_canvas_events,
inputs=[rv_canvas_events_txt, rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state],
outputs=[rv_canvas_render_txt, rv_annotations_state, rv_ledger_state, rv_df, rv_selected_state],
).then(
fn=lambda idx: f"Box #{idx + 1}" if idx >= 0 else "β€” none β€”",
inputs=[rv_selected_state],
outputs=[rv_selected_lbl],
).then(
fn=on_fill_edit_form,
inputs=[rv_selected_state, rv_page_state, rv_annotations_state],
outputs=[edit_x1, edit_y1, edit_x2, edit_y2],
)
rv_detect_btn.click(
fn=on_rv_detect,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_conf_slider],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_status],
)
rv_find_more_btn.click(
fn=on_rv_find_more,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_conf_slider],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_status],
)
rv_detect_all_btn.click(
fn=on_rv_detect_all,
inputs=[rv_pdf_state, rv_page_state, rv_total_state, rv_annotations_state, rv_conf_slider],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_status],
)
rv_approve_all_btn.click(
fn=on_rv_approve_all,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_ledger_state, rv_status],
)
rv_reject_all_btn.click(
fn=on_rv_reject_all,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_ledger_state, rv_status],
)
rv_df.select(
fn=on_row_select,
inputs=None,
outputs=[rv_selected_state],
).then(
fn=lambda idx: f"Box #{idx + 1}" if idx >= 0 else "β€” none β€”",
inputs=[rv_selected_state],
outputs=[rv_selected_lbl],
).then(
fn=on_fill_edit_form,
inputs=[rv_selected_state, rv_page_state, rv_annotations_state],
outputs=[edit_x1, edit_y1, edit_x2, edit_y2],
)
rv_approve_sel_btn.click(
fn=on_rv_approve_selected,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state, rv_selected_state],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_ledger_state, rv_status],
)
rv_reject_sel_btn.click(
fn=on_rv_reject_selected,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state, rv_selected_state],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_ledger_state, rv_status],
)
rv_copy_sel_btn.click(
fn=on_rv_copy_selected,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state, rv_selected_state],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_ledger_state, rv_status],
)
rv_delete_sel_btn.click(
fn=on_rv_delete_selected,
inputs=[rv_pdf_state, rv_page_state, rv_annotations_state, rv_ledger_state, rv_selected_state],
outputs=[
rv_canvas_render_txt,
rv_df,
rv_annotations_state,
rv_ledger_state,
rv_status,
rv_selected_state,
],
).then(
fn=lambda idx: f"Box #{idx + 1}" if idx >= 0 else "β€” none β€”",
inputs=[rv_selected_state],
outputs=[rv_selected_lbl],
)
update_box_btn.click(
fn=on_rv_update_box,
inputs=[
rv_pdf_state,
rv_page_state,
rv_annotations_state,
rv_ledger_state,
rv_selected_state,
edit_x1,
edit_y1,
edit_x2,
edit_y2,
],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_ledger_state, rv_status],
)
export_events_btn.click(
fn=on_export_events,
inputs=[rv_ledger_state],
outputs=[export_file_out],
)
export_dataset_btn.click(
fn=on_export_dataset,
inputs=[rv_ledger_state, rv_pdf_state],
outputs=[export_file_out],
)
export_labels_btn.click(
fn=on_rv_export_labels,
inputs=[rv_pdf_state, rv_annotations_state],
outputs=[export_file_out, rv_status],
)
import_labels_in.upload(
fn=on_rv_import_labels,
inputs=[import_labels_in, rv_pdf_state, rv_page_state, rv_annotations_state],
outputs=[rv_canvas_render_txt, rv_df, rv_annotations_state, rv_status],
)
# ------------------------------------------------------------------ #
# Tab 3: Send to Backend #
# ------------------------------------------------------------------ #
with gr.Tab("Send to Backend"):
gr.Markdown(
"Sends **approved** annotations from the **Review & Edit** tab to your backend. "
"Detect pages, approve the boxes you want, then click **Send to Backend**."
)
with gr.Row():
with gr.Column():
backend_url_input = gr.Textbox(label="Backend Base URL", value="http://localhost:8000")
document_id_input = gr.Number(label="Document ID", value=1, precision=0)
frame_template_id_input = gr.Number(label="Frame Template ID", value=1, precision=0)
tenant_id_input = gr.Number(label="x-tenant-id", value=1, precision=0)
user_id_input = gr.Number(label="x-user-id", value=1, precision=0)
color_input = gr.Textbox(label="Color (optional)", placeholder="#FF5500")
quantity_input = gr.Number(label="Quantity Multiplier", value=1, minimum=1, precision=0)
send_btn = gr.Button("Send to Backend", variant="primary")
with gr.Column():
response_output = gr.JSON(label="Backend Response")
send_btn.click(
fn=send_to_backend,
inputs=[
rv_annotations_state,
backend_url_input,
document_id_input,
frame_template_id_input,
tenant_id_input,
user_id_input,
color_input,
quantity_input,
],
outputs=[response_output],
)
# ------------------------------------------------------------------ #
# Machine API (hidden) β€” called by backends via gradio_client: #
# client.predict(handle_file(png), 0.25, api_name="/detect_image") #
# ------------------------------------------------------------------ #
with gr.Row(visible=False):
api_image_in = gr.File(label="api_image", file_types=["image"])
api_pdf_in = gr.File(label="api_pdf", file_types=[".pdf"])
api_page_in = gr.Number(value=1, precision=0)
api_dpi_in = gr.Number(value=DETECTION_DPI, precision=0)
api_conf_in = gr.Number(value=0.25)
api_json_out = gr.JSON()
api_detect_image_btn = gr.Button("api detect_image")
api_detect_page_btn = gr.Button("api detect_page")
api_detect_image_btn.click(
fn=detect_image_api,
inputs=[api_image_in, api_conf_in],
outputs=[api_json_out],
api_name="detect_image",
)
api_detect_page_btn.click(
fn=detect_page_api,
inputs=[api_pdf_in, api_page_in, api_dpi_in, api_conf_in],
outputs=[api_json_out],
api_name="detect_page",
)
if __name__ == "__main__":
demo.launch()