""" Provider-agnostic input validation shared by every front end onto the grounding pipeline (the local FastAPI app in server.py, and the Hugging Face Spaces/ZeroGPU app in scripts/spaces_app/). None of this depends on which model actually does the grounding -- it's the (1) is this a real paper / (2) does it link a real GitHub repo gate that runs before any model call, so it belongs to neither backend specifically. """ from __future__ import annotations import re import subprocess from pathlib import Path from typing import Optional GITHUB_URL_RE = re.compile(r"https?://github\.com/([\w.-]+)/([\w.-]+?)(?=[)\]\s,.;\"'<>]|$)") ARXIV_ABS_RE = re.compile(r"^https?://arxiv\.org/abs/(\d{4}\.\d{4,5}(v\d+)?)/?$") def normalize_paper_url(url: str) -> str: m = ARXIV_ABS_RE.match(url.strip()) if m: return f"https://arxiv.org/pdf/{m.group(1)}" return url.strip() def looks_like_a_paper(pdf_path: Path) -> tuple[bool, str]: """Heuristic, not a classifier: a real academic paper has an Abstract near the start and a References/Bibliography section later, and is a plausible page count. Good enough to reject "someone uploaded a random PDF" without needing an ML model for it.""" proc = subprocess.run(["pdfinfo", str(pdf_path)], capture_output=True, text=True) if proc.returncode != 0: return False, "not a readable PDF" pages = 0 for line in proc.stdout.splitlines(): if line.startswith("Pages:"): pages = int(line.split(":", 1)[1].strip()) if pages < 4 or pages > 80: return False, f"page count ({pages}) doesn't look like a paper" text_proc = subprocess.run(["pdftotext", str(pdf_path), "-"], capture_output=True, text=True) text = text_proc.stdout.lower() first_page = "\n".join(text.splitlines()[:80]) if "abstract" not in first_page: return False, "no 'Abstract' found near the start of the document" if "references" not in text and "bibliography" not in text: return False, "no 'References'/'Bibliography' section found" return True, "looks like a paper" def _degap(text: str) -> str: """pdftotext preserves the PDF's visual line wrapping, which routinely splits a URL across a line break (e.g. 'https:' / '//github.com/...' have shown up on two separate lines in real papers). A real URL never contains whitespace, so it's always safe to strip it out before searching for one -- this can't create a false match, only recover a real one that line-wrapping broke.""" return re.sub(r"\s+", "", text) def _candidate_repo_urls(text: str) -> list[str]: """All github.com/owner/repo matches in `text`, each followed by a variant with one trailing digit stripped. A citation superscript or footnote marker (e.g. 'UniComp¹') routinely gets flattened by pdftotext into a bare digit glued directly onto the repo name with no separating whitespace or punctuation for the regex to stop at -- so 'UniComp1' as extracted may really be 'UniComp' + a footnote number. Only offering the stripped variant as a fallback (checked in extract_github_repo, not preferred blindly) avoids ever preferring a guess over what was actually printed.""" out = [] for m in GITHUB_URL_RE.finditer(_degap(text)): owner, repo = m.group(1), m.group(2) out.append(f"https://github.com/{owner}/{repo}") if repo[-1:].isdigit() and len(repo) > 1: out.append(f"https://github.com/{owner}/{repo[:-1]}") return out def repo_exists(repo_url: str) -> bool: proc = subprocess.run(["git", "ls-remote", repo_url], capture_output=True, text=True, timeout=20) return proc.returncode == 0 def extract_github_repo(pdf_path: Path) -> Optional[str]: """Prefer a repo link on page 1 (where authors usually put their own code link, in the abstract or a footnote) over one found anywhere else in the document (which might just be a cited related-work repo). Within each page, prefer a candidate that's actually a real, reachable repo over one that merely matches the URL shape -- catches the trailing-footnote-digit case above without ever silently proceeding with a repo nobody actually linked.""" page1 = subprocess.run( ["pdftotext", "-f", "1", "-l", "1", str(pdf_path), "-"], capture_output=True, text=True, ).stdout for url in _candidate_repo_urls(page1): if repo_exists(url): return url full = subprocess.run(["pdftotext", str(pdf_path), "-"], capture_output=True, text=True).stdout for url in _candidate_repo_urls(full): if repo_exists(url): return url return None