Download scripts/webapp/validation.py from mfontana355/paper-code-grounding: direct link, hf CLI and curl.
- Browser
- Download file 4.7 kB
-
https://huggingface.co/spaces/mfontana355/paper-code-grounding/resolve/main/scripts/webapp/validation.py
- Command line
-
hf download hf://spaces/mfontana355/paper-code-grounding/scripts/webapp/validation.py
-
curl -L -o validation.py https://huggingface.co/spaces/mfontana355/paper-code-grounding/resolve/main/scripts/webapp/validation.py
4.7 kB
| """ | |
| Provider-agnostic input validation shared by every front end onto the | |
| grounding pipeline (the local FastAPI app in server.py, and the | |
| Hugging Face Spaces/ZeroGPU app in scripts/spaces_app/). None of this | |
| depends on which model actually does the grounding -- it's the (1) is | |
| this a real paper / (2) does it link a real GitHub repo gate that runs | |
| before any model call, so it belongs to neither backend specifically. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import subprocess | |
| from pathlib import Path | |
| from typing import Optional | |
| GITHUB_URL_RE = re.compile(r"https?://github\.com/([\w.-]+)/([\w.-]+?)(?=[)\]\s,.;\"'<>]|$)") | |
| ARXIV_ABS_RE = re.compile(r"^https?://arxiv\.org/abs/(\d{4}\.\d{4,5}(v\d+)?)/?$") | |
| def normalize_paper_url(url: str) -> str: | |
| m = ARXIV_ABS_RE.match(url.strip()) | |
| if m: | |
| return f"https://arxiv.org/pdf/{m.group(1)}" | |
| return url.strip() | |
| def looks_like_a_paper(pdf_path: Path) -> tuple[bool, str]: | |
| """Heuristic, not a classifier: a real academic paper has an Abstract | |
| near the start and a References/Bibliography section later, and is a | |
| plausible page count. Good enough to reject "someone uploaded a random | |
| PDF" without needing an ML model for it.""" | |
| proc = subprocess.run(["pdfinfo", str(pdf_path)], capture_output=True, text=True) | |
| if proc.returncode != 0: | |
| return False, "not a readable PDF" | |
| pages = 0 | |
| for line in proc.stdout.splitlines(): | |
| if line.startswith("Pages:"): | |
| pages = int(line.split(":", 1)[1].strip()) | |
| if pages < 4 or pages > 80: | |
| return False, f"page count ({pages}) doesn't look like a paper" | |
| text_proc = subprocess.run(["pdftotext", str(pdf_path), "-"], capture_output=True, text=True) | |
| text = text_proc.stdout.lower() | |
| first_page = "\n".join(text.splitlines()[:80]) | |
| if "abstract" not in first_page: | |
| return False, "no 'Abstract' found near the start of the document" | |
| if "references" not in text and "bibliography" not in text: | |
| return False, "no 'References'/'Bibliography' section found" | |
| return True, "looks like a paper" | |
| def _degap(text: str) -> str: | |
| """pdftotext preserves the PDF's visual line wrapping, which routinely | |
| splits a URL across a line break (e.g. 'https:' / '//github.com/...' | |
| have shown up on two separate lines in real papers). A real URL never | |
| contains whitespace, so it's always safe to strip it out before | |
| searching for one -- this can't create a false match, only recover a | |
| real one that line-wrapping broke.""" | |
| return re.sub(r"\s+", "", text) | |
| def _candidate_repo_urls(text: str) -> list[str]: | |
| """All github.com/owner/repo matches in `text`, each followed by a | |
| variant with one trailing digit stripped. A citation superscript or | |
| footnote marker (e.g. 'UniComp¹') routinely gets flattened by | |
| pdftotext into a bare digit glued directly onto the repo name with no | |
| separating whitespace or punctuation for the regex to stop at -- so | |
| 'UniComp1' as extracted may really be 'UniComp' + a footnote number. | |
| Only offering the stripped variant as a fallback (checked in | |
| extract_github_repo, not preferred blindly) avoids ever preferring a | |
| guess over what was actually printed.""" | |
| out = [] | |
| for m in GITHUB_URL_RE.finditer(_degap(text)): | |
| owner, repo = m.group(1), m.group(2) | |
| out.append(f"https://github.com/{owner}/{repo}") | |
| if repo[-1:].isdigit() and len(repo) > 1: | |
| out.append(f"https://github.com/{owner}/{repo[:-1]}") | |
| return out | |
| def repo_exists(repo_url: str) -> bool: | |
| proc = subprocess.run(["git", "ls-remote", repo_url], capture_output=True, text=True, timeout=20) | |
| return proc.returncode == 0 | |
| def extract_github_repo(pdf_path: Path) -> Optional[str]: | |
| """Prefer a repo link on page 1 (where authors usually put their own | |
| code link, in the abstract or a footnote) over one found anywhere else | |
| in the document (which might just be a cited related-work repo). | |
| Within each page, prefer a candidate that's actually a real, reachable | |
| repo over one that merely matches the URL shape -- catches the | |
| trailing-footnote-digit case above without ever silently proceeding | |
| with a repo nobody actually linked.""" | |
| page1 = subprocess.run( | |
| ["pdftotext", "-f", "1", "-l", "1", str(pdf_path), "-"], | |
| capture_output=True, text=True, | |
| ).stdout | |
| for url in _candidate_repo_urls(page1): | |
| if repo_exists(url): | |
| return url | |
| full = subprocess.run(["pdftotext", str(pdf_path), "-"], capture_output=True, text=True).stdout | |
| for url in _candidate_repo_urls(full): | |
| if repo_exists(url): | |
| return url | |
| return None | |