Spaces:
Running on Zero
Running on Zero
Download scripts/build_papers.py from wang2226/beyond-tokens-decoding: direct link, hf CLI and curl.
- Browser
- Download file 4.64 kB
-
https://huggingface.co/spaces/wang2226/beyond-tokens-decoding/resolve/main/scripts/build_papers.py
- Command line
-
hf download hf://spaces/wang2226/beyond-tokens-decoding/scripts/build_papers.py
-
curl -L -o build_papers.py https://huggingface.co/spaces/wang2226/beyond-tokens-decoding/resolve/main/scripts/build_papers.py
4.64 kB
| """Build data/papers.json and data/citation.bib from the Awesome-LLM-Decoding README. | |
| Usage: python scripts/build_papers.py [path-or-url] | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import re | |
| import sys | |
| import urllib.request | |
| from pathlib import Path | |
| README_URL = "https://raw.githubusercontent.com/wang2226/Awesome-LLM-Decoding/main/README.md" | |
| ROOT = Path(__file__).resolve().parents[1] | |
| TITLE = re.compile(r"^- \*\*(.+?)\*\*") | |
| AUTHORS = re.compile(r"^\s*\*(.+?)\*\.?\s*$") | |
| LINK = re.compile(r"\[\[(\w+)\]\((\S+?)\)\]") | |
| BADGE = re.compile(r"img\.shields\.io/badge/([^)\s]+)-(blue|brown|red)\)") | |
| HEADING = re.compile(r"^(#{2,5})\s+(.*)$") | |
| def unbadge(text: str) -> str: | |
| """shields.io escaping: '--' is '-', '__' is '_', '_' is a space.""" | |
| return text.replace("--", "\0").replace("__", "\1").replace("_", " ").replace("\0", "-").replace("\1", "_") | |
| def year_of(venue: str, pdf: str) -> int | None: | |
| m = re.search(r"(20\d\d)", venue or "") | |
| if m: | |
| return int(m.group(1)) | |
| m = re.search(r"arxiv\.org/(?:abs|pdf)/(\d\d)(\d\d)\.", pdf or "") | |
| if m: | |
| return 2000 + int(m.group(1)) | |
| m = re.search(r"aclanthology\.org/(20\d\d)\.", pdf or "") | |
| return int(m.group(1)) if m else None | |
| def parse(readme: str) -> tuple[list[dict], str]: | |
| lines = readme.splitlines() | |
| try: | |
| start = next(i for i, l in enumerate(lines) if l.strip() == "## Papers") | |
| except StopIteration: | |
| start = 0 | |
| papers, path, cur = [], {}, None | |
| def flush(): | |
| if cur: | |
| papers.append(cur) | |
| for line in lines[start:]: | |
| h = HEADING.match(line) | |
| if h: | |
| level, name = len(h.group(1)), h.group(2).strip() | |
| if level == 2 and name != "Papers": | |
| break | |
| flush() | |
| cur = None | |
| path = {k: v for k, v in path.items() if k < level} | |
| path[level] = name | |
| continue | |
| t = TITLE.match(line) | |
| if t: | |
| flush() | |
| sections = [path[k] for k in sorted(path) if k >= 3] | |
| if sections and sections[0] == "Keywords Convention": | |
| cur = None | |
| continue | |
| cur = {"section": " › ".join(sections), "title": t.group(1).strip(), "authors": "", "abbr": "", | |
| "venue": "", "model": "", "pdf": "", "code": "", "year": None} | |
| continue | |
| if cur is None: | |
| continue | |
| a = AUTHORS.match(line) | |
| if a and not cur["authors"] and "img.shields" not in line: | |
| cur["authors"] = a.group(1).strip().rstrip(".") | |
| for kind, url in LINK.findall(line): | |
| if kind in ("pdf", "code") and not cur[kind]: | |
| cur[kind] = url | |
| for text, color in BADGE.findall(line): | |
| key = {"blue": "abbr", "brown": "venue", "red": "model"}[color] | |
| value = unbadge(text) | |
| cur[key] = f"{cur[key]}, {value}" if cur[key] else value | |
| flush() | |
| for p in papers: | |
| p["year"] = year_of(p["venue"], p["pdf"]) | |
| top = p["section"].split(" › ") | |
| if top[0] == "Paradigms" and len(top) > 1: | |
| p["group"] = top[1].lower() # contrastive / guided / parallel | |
| elif top[0] == "Survey": | |
| p["group"] = "survey" | |
| else: | |
| p["group"] = "application" | |
| m = re.search(r"## Citation\s*```bibtex\s*(.*?)```", readme, flags=re.S) | |
| return papers, (m.group(1).strip() if m else "") | |
| def main() -> None: | |
| src = sys.argv[1] if len(sys.argv) > 1 else README_URL | |
| if src.startswith("http"): | |
| with urllib.request.urlopen(src, timeout=30) as r: | |
| readme = r.read().decode("utf-8") | |
| else: | |
| readme = Path(src).read_text(encoding="utf-8") | |
| papers, bib = parse(readme) | |
| out = ROOT / "data" / "papers.json" | |
| out.parent.mkdir(parents=True, exist_ok=True) | |
| out.write_text(json.dumps(papers, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") | |
| if bib: | |
| (ROOT / "data" / "citation.bib").write_text(bib + "\n", encoding="utf-8") | |
| counts: dict[str, int] = {} | |
| for p in papers: | |
| counts[p["group"]] = counts.get(p["group"], 0) + 1 | |
| print(f"{len(papers)} papers -> {out.relative_to(ROOT)} {counts}") | |
| # Report entries worth a second look. | |
| seen: dict[str, str] = {} | |
| for p in papers: | |
| if not p["authors"]: | |
| print(" no authors:", p["title"][:70]) | |
| if p["pdf"] and p["pdf"] in seen and seen[p["pdf"]] != p["title"]: | |
| print(f" shared pdf link: {p['title'][:50]!r} and {seen[p['pdf']][:50]!r}") | |
| seen.setdefault(p["pdf"], p["title"]) | |
| if __name__ == "__main__": | |
| main() | |