mfontana355's picture
Upload folder using huggingface_hub
3b09ba9 verified
Raw History Blame Contribute Delete
4.7 kB
"""
Provider-agnostic input validation shared by every front end onto the
grounding pipeline (the local FastAPI app in server.py, and the
Hugging Face Spaces/ZeroGPU app in scripts/spaces_app/). None of this
depends on which model actually does the grounding -- it's the (1) is
this a real paper / (2) does it link a real GitHub repo gate that runs
before any model call, so it belongs to neither backend specifically.
"""
from __future__ import annotations
import re
import subprocess
from pathlib import Path
from typing import Optional
GITHUB_URL_RE = re.compile(r"https?://github\.com/([\w.-]+)/([\w.-]+?)(?=[)\]\s,.;\"'<>]|$)")
ARXIV_ABS_RE = re.compile(r"^https?://arxiv\.org/abs/(\d{4}\.\d{4,5}(v\d+)?)/?$")
def normalize_paper_url(url: str) -> str:
m = ARXIV_ABS_RE.match(url.strip())
if m:
return f"https://arxiv.org/pdf/{m.group(1)}"
return url.strip()
def looks_like_a_paper(pdf_path: Path) -> tuple[bool, str]:
"""Heuristic, not a classifier: a real academic paper has an Abstract
near the start and a References/Bibliography section later, and is a
plausible page count. Good enough to reject "someone uploaded a random
PDF" without needing an ML model for it."""
proc = subprocess.run(["pdfinfo", str(pdf_path)], capture_output=True, text=True)
if proc.returncode != 0:
return False, "not a readable PDF"
pages = 0
for line in proc.stdout.splitlines():
if line.startswith("Pages:"):
pages = int(line.split(":", 1)[1].strip())
if pages < 4 or pages > 80:
return False, f"page count ({pages}) doesn't look like a paper"
text_proc = subprocess.run(["pdftotext", str(pdf_path), "-"], capture_output=True, text=True)
text = text_proc.stdout.lower()
first_page = "\n".join(text.splitlines()[:80])
if "abstract" not in first_page:
return False, "no 'Abstract' found near the start of the document"
if "references" not in text and "bibliography" not in text:
return False, "no 'References'/'Bibliography' section found"
return True, "looks like a paper"
def _degap(text: str) -> str:
"""pdftotext preserves the PDF's visual line wrapping, which routinely
splits a URL across a line break (e.g. 'https:' / '//github.com/...'
have shown up on two separate lines in real papers). A real URL never
contains whitespace, so it's always safe to strip it out before
searching for one -- this can't create a false match, only recover a
real one that line-wrapping broke."""
return re.sub(r"\s+", "", text)
def _candidate_repo_urls(text: str) -> list[str]:
"""All github.com/owner/repo matches in `text`, each followed by a
variant with one trailing digit stripped. A citation superscript or
footnote marker (e.g. 'UniComp¹') routinely gets flattened by
pdftotext into a bare digit glued directly onto the repo name with no
separating whitespace or punctuation for the regex to stop at -- so
'UniComp1' as extracted may really be 'UniComp' + a footnote number.
Only offering the stripped variant as a fallback (checked in
extract_github_repo, not preferred blindly) avoids ever preferring a
guess over what was actually printed."""
out = []
for m in GITHUB_URL_RE.finditer(_degap(text)):
owner, repo = m.group(1), m.group(2)
out.append(f"https://github.com/{owner}/{repo}")
if repo[-1:].isdigit() and len(repo) > 1:
out.append(f"https://github.com/{owner}/{repo[:-1]}")
return out
def repo_exists(repo_url: str) -> bool:
proc = subprocess.run(["git", "ls-remote", repo_url], capture_output=True, text=True, timeout=20)
return proc.returncode == 0
def extract_github_repo(pdf_path: Path) -> Optional[str]:
"""Prefer a repo link on page 1 (where authors usually put their own
code link, in the abstract or a footnote) over one found anywhere else
in the document (which might just be a cited related-work repo).
Within each page, prefer a candidate that's actually a real, reachable
repo over one that merely matches the URL shape -- catches the
trailing-footnote-digit case above without ever silently proceeding
with a repo nobody actually linked."""
page1 = subprocess.run(
["pdftotext", "-f", "1", "-l", "1", str(pdf_path), "-"],
capture_output=True, text=True,
).stdout
for url in _candidate_repo_urls(page1):
if repo_exists(url):
return url
full = subprocess.run(["pdftotext", str(pdf_path), "-"], capture_output=True, text=True).stdout
for url in _candidate_repo_urls(full):
if repo_exists(url):
return url
return None