"""Build data/papers.json and data/citation.bib from the Awesome-LLM-Decoding README. Usage: python scripts/build_papers.py [path-or-url] """ from __future__ import annotations import json import re import sys import urllib.request from pathlib import Path README_URL = "https://raw.githubusercontent.com/wang2226/Awesome-LLM-Decoding/main/README.md" ROOT = Path(__file__).resolve().parents[1] TITLE = re.compile(r"^- \*\*(.+?)\*\*") AUTHORS = re.compile(r"^\s*\*(.+?)\*\.?\s*$") LINK = re.compile(r"\[\[(\w+)\]\((\S+?)\)\]") BADGE = re.compile(r"img\.shields\.io/badge/([^)\s]+)-(blue|brown|red)\)") HEADING = re.compile(r"^(#{2,5})\s+(.*)$") def unbadge(text: str) -> str: """shields.io escaping: '--' is '-', '__' is '_', '_' is a space.""" return text.replace("--", "\0").replace("__", "\1").replace("_", " ").replace("\0", "-").replace("\1", "_") def year_of(venue: str, pdf: str) -> int | None: m = re.search(r"(20\d\d)", venue or "") if m: return int(m.group(1)) m = re.search(r"arxiv\.org/(?:abs|pdf)/(\d\d)(\d\d)\.", pdf or "") if m: return 2000 + int(m.group(1)) m = re.search(r"aclanthology\.org/(20\d\d)\.", pdf or "") return int(m.group(1)) if m else None def parse(readme: str) -> tuple[list[dict], str]: lines = readme.splitlines() try: start = next(i for i, l in enumerate(lines) if l.strip() == "## Papers") except StopIteration: start = 0 papers, path, cur = [], {}, None def flush(): if cur: papers.append(cur) for line in lines[start:]: h = HEADING.match(line) if h: level, name = len(h.group(1)), h.group(2).strip() if level == 2 and name != "Papers": break flush() cur = None path = {k: v for k, v in path.items() if k < level} path[level] = name continue t = TITLE.match(line) if t: flush() sections = [path[k] for k in sorted(path) if k >= 3] if sections and sections[0] == "Keywords Convention": cur = None continue cur = {"section": " › ".join(sections), "title": t.group(1).strip(), "authors": "", "abbr": "", "venue": "", "model": "", "pdf": "", "code": "", "year": None} continue if cur is None: continue a = AUTHORS.match(line) if a and not cur["authors"] and "img.shields" not in line: cur["authors"] = a.group(1).strip().rstrip(".") for kind, url in LINK.findall(line): if kind in ("pdf", "code") and not cur[kind]: cur[kind] = url for text, color in BADGE.findall(line): key = {"blue": "abbr", "brown": "venue", "red": "model"}[color] value = unbadge(text) cur[key] = f"{cur[key]}, {value}" if cur[key] else value flush() for p in papers: p["year"] = year_of(p["venue"], p["pdf"]) top = p["section"].split(" › ") if top[0] == "Paradigms" and len(top) > 1: p["group"] = top[1].lower() # contrastive / guided / parallel elif top[0] == "Survey": p["group"] = "survey" else: p["group"] = "application" m = re.search(r"## Citation\s*```bibtex\s*(.*?)```", readme, flags=re.S) return papers, (m.group(1).strip() if m else "") def main() -> None: src = sys.argv[1] if len(sys.argv) > 1 else README_URL if src.startswith("http"): with urllib.request.urlopen(src, timeout=30) as r: readme = r.read().decode("utf-8") else: readme = Path(src).read_text(encoding="utf-8") papers, bib = parse(readme) out = ROOT / "data" / "papers.json" out.parent.mkdir(parents=True, exist_ok=True) out.write_text(json.dumps(papers, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") if bib: (ROOT / "data" / "citation.bib").write_text(bib + "\n", encoding="utf-8") counts: dict[str, int] = {} for p in papers: counts[p["group"]] = counts.get(p["group"], 0) + 1 print(f"{len(papers)} papers -> {out.relative_to(ROOT)} {counts}") # Report entries worth a second look. seen: dict[str, str] = {} for p in papers: if not p["authors"]: print(" no authors:", p["title"][:70]) if p["pdf"] and p["pdf"] in seen and seen[p["pdf"]] != p["title"]: print(f" shared pdf link: {p['title'][:50]!r} and {seen[p['pdf']][:50]!r}") seen.setdefault(p["pdf"], p["title"]) if __name__ == "__main__": main()