beyond-tokens-decoding / scripts /build_papers.py
wang2226's picture
Beyond Tokens decoding playground: contrastive, guided and parallel decoding
371d90c verified
Raw History Blame Contribute Delete
4.64 kB
"""Build data/papers.json and data/citation.bib from the Awesome-LLM-Decoding README.
Usage: python scripts/build_papers.py [path-or-url]
"""
from __future__ import annotations
import json
import re
import sys
import urllib.request
from pathlib import Path
README_URL = "https://raw.githubusercontent.com/wang2226/Awesome-LLM-Decoding/main/README.md"
ROOT = Path(__file__).resolve().parents[1]
TITLE = re.compile(r"^- \*\*(.+?)\*\*")
AUTHORS = re.compile(r"^\s*\*(.+?)\*\.?\s*$")
LINK = re.compile(r"\[\[(\w+)\]\((\S+?)\)\]")
BADGE = re.compile(r"img\.shields\.io/badge/([^)\s]+)-(blue|brown|red)\)")
HEADING = re.compile(r"^(#{2,5})\s+(.*)$")
def unbadge(text: str) -> str:
"""shields.io escaping: '--' is '-', '__' is '_', '_' is a space."""
return text.replace("--", "\0").replace("__", "\1").replace("_", " ").replace("\0", "-").replace("\1", "_")
def year_of(venue: str, pdf: str) -> int | None:
m = re.search(r"(20\d\d)", venue or "")
if m:
return int(m.group(1))
m = re.search(r"arxiv\.org/(?:abs|pdf)/(\d\d)(\d\d)\.", pdf or "")
if m:
return 2000 + int(m.group(1))
m = re.search(r"aclanthology\.org/(20\d\d)\.", pdf or "")
return int(m.group(1)) if m else None
def parse(readme: str) -> tuple[list[dict], str]:
lines = readme.splitlines()
try:
start = next(i for i, l in enumerate(lines) if l.strip() == "## Papers")
except StopIteration:
start = 0
papers, path, cur = [], {}, None
def flush():
if cur:
papers.append(cur)
for line in lines[start:]:
h = HEADING.match(line)
if h:
level, name = len(h.group(1)), h.group(2).strip()
if level == 2 and name != "Papers":
break
flush()
cur = None
path = {k: v for k, v in path.items() if k < level}
path[level] = name
continue
t = TITLE.match(line)
if t:
flush()
sections = [path[k] for k in sorted(path) if k >= 3]
if sections and sections[0] == "Keywords Convention":
cur = None
continue
cur = {"section": " › ".join(sections), "title": t.group(1).strip(), "authors": "", "abbr": "",
"venue": "", "model": "", "pdf": "", "code": "", "year": None}
continue
if cur is None:
continue
a = AUTHORS.match(line)
if a and not cur["authors"] and "img.shields" not in line:
cur["authors"] = a.group(1).strip().rstrip(".")
for kind, url in LINK.findall(line):
if kind in ("pdf", "code") and not cur[kind]:
cur[kind] = url
for text, color in BADGE.findall(line):
key = {"blue": "abbr", "brown": "venue", "red": "model"}[color]
value = unbadge(text)
cur[key] = f"{cur[key]}, {value}" if cur[key] else value
flush()
for p in papers:
p["year"] = year_of(p["venue"], p["pdf"])
top = p["section"].split(" › ")
if top[0] == "Paradigms" and len(top) > 1:
p["group"] = top[1].lower() # contrastive / guided / parallel
elif top[0] == "Survey":
p["group"] = "survey"
else:
p["group"] = "application"
m = re.search(r"## Citation\s*```bibtex\s*(.*?)```", readme, flags=re.S)
return papers, (m.group(1).strip() if m else "")
def main() -> None:
src = sys.argv[1] if len(sys.argv) > 1 else README_URL
if src.startswith("http"):
with urllib.request.urlopen(src, timeout=30) as r:
readme = r.read().decode("utf-8")
else:
readme = Path(src).read_text(encoding="utf-8")
papers, bib = parse(readme)
out = ROOT / "data" / "papers.json"
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(json.dumps(papers, indent=1, ensure_ascii=False) + "\n", encoding="utf-8")
if bib:
(ROOT / "data" / "citation.bib").write_text(bib + "\n", encoding="utf-8")
counts: dict[str, int] = {}
for p in papers:
counts[p["group"]] = counts.get(p["group"], 0) + 1
print(f"{len(papers)} papers -> {out.relative_to(ROOT)} {counts}")
# Report entries worth a second look.
seen: dict[str, str] = {}
for p in papers:
if not p["authors"]:
print(" no authors:", p["title"][:70])
if p["pdf"] and p["pdf"] in seen and seen[p["pdf"]] != p["title"]:
print(f" shared pdf link: {p['title'][:50]!r} and {seen[p['pdf']][:50]!r}")
seen.setdefault(p["pdf"], p["title"])
if __name__ == "__main__":
main()