Before the next submission: a revised version links only under the same verified ORCID; a DOI is dead only when doi.org has no such DOI (an outage or another registration agency is "unchecked", never a finding); the authors' surnames are blinded wherever the panel reads them; the per-reference citation lines sit inside the untrusted submission block and the RQC evidence block is declared untrusted; "unanimous" names the reviewers' shared vote; the code digest fetches at most 60 files in 180 s; a forced re-draft keeps the first decision time and the audit log records a draft, never a send.
384705a Download code_digest.py from ICSAC-Institute/editorial-system: direct link, hf CLI and curl.
- Browser
- Download file 19.8 kB
-
https://huggingface.co/spaces/ICSAC-Institute/editorial-system/resolve/main/code_digest.py
- Command line
-
hf download hf://spaces/ICSAC-Institute/editorial-system/code_digest.py
-
curl -L -o code_digest.py https://huggingface.co/spaces/ICSAC-Institute/editorial-system/resolve/main/code_digest.py
19.8 kB
| """Code package digest for the panel. | |
| A panel can call an analysis "reproducible" without having opened the | |
| code, even when the shipped scripts implement a different measure from the one the | |
| paper defines. The panel cannot judge a claim about code it has not seen, so | |
| this module fetches the author's DECLARED code/data archive, read-only, and | |
| puts in front of the reviewers: | |
| * the file listing (name, size) of the archive, | |
| * the head of its README, | |
| * the function/definition inventory of every source file, and | |
| * the lines that define or assign the constructs the paper names (its | |
| acronyms and named measures, e.g. XRM, XRM-2, NPT), with file:line. | |
| What it never does: execute anything, write the archive to disk, follow more | |
| than a size cap of bytes, or touch a link the author did not declare. ICSAC's | |
| "never host authors' code" policy (2026-09-28) is about hosting; this is a | |
| bounded read, in memory, discarded after the digest is written. Sources: | |
| Zenodo records (files listed by the API, zip members read in memory), GitHub | |
| repositories (the zipball), and nothing else -- an unknown host is named in | |
| the digest as "not fetched" so the panel knows the claim stands unchecked. | |
| The digest is untrusted author content: it goes INSIDE the submission block | |
| of the panel prompt, under the same defensive preamble as the manuscript, | |
| and through the blind-review span removal like the abstract. Fail OPEN: any | |
| error yields a one-line digest that says what could not be fetched. | |
| """ | |
| from __future__ import annotations | |
| import io | |
| import json | |
| import os | |
| import re | |
| import sys | |
| import time | |
| import urllib.error | |
| import urllib.parse | |
| import urllib.request | |
| import zipfile | |
| from collections import Counter | |
| _REPO_ROOT = os.path.dirname(os.path.abspath(__file__)) | |
| if _REPO_ROOT not in sys.path: | |
| sys.path.insert(0, _REPO_ROOT) | |
| import config # noqa: E402 | |
| USER_AGENT = "icsac-editorial-system/1.0 (help@icsacinstitute.org)" | |
| TIMEOUT = 45 | |
| MAX_ARCHIVE_BYTES = int(getattr(config, "CODE_DIGEST_MAX_BYTES", 25 * 1024 * 1024)) | |
| MAX_MEMBER_BYTES = 400 * 1024 # a source file bigger than this is listed, not read | |
| MAX_FILES_LISTED = 200 | |
| MAX_SOURCE_FILES_READ = 60 | |
| MAX_FILES_FETCHED = 60 # Zenodo files fetched one by one; the rest are listed only | |
| FETCH_WALL_SEC = 180 # and never for longer than this in total | |
| MAX_DIGEST_CHARS = int(getattr(config, "CODE_DIGEST_MAX_CHARS", 14000)) | |
| README_LINES = 40 | |
| HITS_PER_TERM = 12 | |
| MAX_TERMS = 12 | |
| DEFS_PER_FILE = 15 | |
| SOURCE_EXT = {".py", ".r", ".R", ".jl", ".m", ".js", ".ts", ".java", ".c", ".cpp", ".h", | |
| ".rs", ".go", ".scala", ".sh", ".ipynb", ".sql", ".do", ".sas", ".nb", ".f90"} | |
| TEXT_EXT = SOURCE_EXT | {".md", ".txt", ".rst", ".cfg", ".toml", ".yaml", ".yml", ".json", ".csv", ".tsv"} | |
| _ACRONYM_RE = re.compile(r"\b(?:Ξ|Ξ-)?[A-Z][A-Za-z]?-?[A-Z]{2,}(?:-[A-Z]{2,})?\b") | |
| _ACRONYM_STOP = {"DOI", "PDF", "URL", "HTTP", "HTTPS", "ORCID", "ISSN", "ISBN", "USA", "UK", "EU", | |
| "AI", "LLM", "GPT", "API", "CSV", "JSON", "MIT", "GNU", "GPL", "CC", "BY", "SA", | |
| "ND", "NC", "IEEE", "ACM", "PNAS", "PLOS", "MDPI", "ETC", "ET", "AL", "II", "III", | |
| "IV", "VI", "VII", "AND", "OR", "THE", "FOR", "NOT", "ARXIV", "ZENODO", "GITHUB", | |
| "README", "LICENSE", "COVID", "USD", "EUR", "UTC", "GMT", "ISO", "ANSI", "ASCII", | |
| "UTF", "IBM", "NASA", "NIH", "NSF", "ERC", "PI", "RQ", "FAQ", "TL", "DR", "OK", | |
| "NB", "PS", "CC-BY", "CC-BY-SA", "SD", "SE", "CI"} | |
| _DEF_RE = re.compile( | |
| r"^\s*(?:def\s+(\w+)|class\s+(\w+)|function\s+(\w+)|(\w+)\s*(?:<-|=)\s*function\b" | |
| r"|(?:export\s+)?(?:async\s+)?function\s+(\w+)|(?:const|let|var)\s+(\w+)\s*=\s*(?:\(|async|function))", | |
| re.M) | |
| # βββ link discovery βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def links_from(submission: dict, full_text: str = "") -> list[str]: | |
| """Every code/data location the author declared: the form's URL, related | |
| identifiers of supplement type, and repository links in the paper text.""" | |
| out: list[str] = [] | |
| cd = submission.get("code_data") or {} | |
| if isinstance(cd, dict) and cd.get("url"): | |
| out.append(str(cd["url"]).strip()) | |
| for r in submission.get("related_identifiers") or []: | |
| if not isinstance(r, dict): | |
| continue | |
| rel = (r.get("relation") or "").lower() | |
| if rel in ("issupplementto", "issupplementedby", "references", "isreferencedby", | |
| "isderivedfrom", "issourceof", "requires", "isrequiredby"): | |
| ident = (r.get("identifier") or "").strip() | |
| if ident: | |
| out.append(ident) | |
| if full_text: | |
| try: | |
| import code_availability | |
| out.extend(code_availability.find_links(full_text)) | |
| except Exception: | |
| pass | |
| seen, uniq = set(), [] | |
| for u in out: | |
| key = _canonical(u) | |
| if key and key not in seen: | |
| seen.add(key) | |
| uniq.append(u) | |
| return uniq | |
| def _canonical(u: str) -> str: | |
| u = u.strip().rstrip("./") | |
| u = re.sub(r"^https?://(?:dx\.)?doi\.org/", "", u, flags=re.I) | |
| return u.lower() | |
| def classify(link: str) -> tuple[str, str]: | |
| """('zenodo', record_id) | ('github', 'owner/repo') | ('other', link).""" | |
| s = _canonical(link) | |
| m = re.search(r"10\.5281/zenodo\.(\d+)", s) | |
| if m: | |
| return "zenodo", m.group(1) | |
| m = re.search(r"zenodo\.org/(?:records?|record)/(\d+)", s) | |
| if m: | |
| return "zenodo", m.group(1) | |
| m = re.search(r"github\.com/([\w.-]+)/([\w.-]+)", s) | |
| if m: | |
| return "github", f"{m.group(1)}/{m.group(2).removesuffix('.git')}" | |
| return "other", link | |
| # βββ fetching (read-only, in memory, capped) βββββββββββββββββββββββββ | |
| class DigestFetchError(RuntimeError): | |
| pass | |
| def _get(url: str, accept: str = "*/*", cap: int = MAX_ARCHIVE_BYTES) -> bytes: | |
| req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT, "Accept": accept}) | |
| try: | |
| with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: | |
| length = resp.headers.get("Content-Length") | |
| if length and int(length) > cap: | |
| raise DigestFetchError(f"{url}: {int(length)} bytes exceeds the {cap}-byte cap") | |
| buf = resp.read(cap + 1) | |
| except urllib.error.HTTPError as e: | |
| raise DigestFetchError(f"{url}: HTTP {e.code}") | |
| except (urllib.error.URLError, TimeoutError, OSError) as e: | |
| raise DigestFetchError(f"{url}: {e}") | |
| if len(buf) > cap: | |
| raise DigestFetchError(f"{url}: exceeds the {cap}-byte cap") | |
| return buf | |
| def fetch_zenodo(record_id: str) -> tuple[dict, list[dict]]: | |
| """(record metadata, [{name, size, data|None}]) -- data is None past the cap.""" | |
| meta = json.loads(_get(f"https://zenodo.org/api/records/{record_id}", "application/json", | |
| cap=4 * 1024 * 1024).decode("utf-8", errors="replace")) | |
| files = [] | |
| budget = MAX_ARCHIVE_BYTES | |
| fetched = 0 | |
| started = time.monotonic() | |
| for f in meta.get("files") or []: | |
| name = f.get("key") or "" | |
| size = int(f.get("size") or 0) | |
| url = ((f.get("links") or {}).get("self")) or "" | |
| row = {"name": name, "size": size, "data": None} | |
| if url and size <= budget and (_is_archive(name) or _ext(name) in TEXT_EXT): | |
| # A file-count and wall-time cap as well as the byte budget: a data | |
| # record with hundreds of small text files meant hundreds of | |
| # sequential requests before the panel (audit 2026-10-01 N6). | |
| if fetched >= MAX_FILES_FETCHED or time.monotonic() - started > FETCH_WALL_SEC: | |
| row["error"] = "not read (file-count or time cap)" | |
| else: | |
| fetched += 1 | |
| try: | |
| row["data"] = _get(url, cap=size + 1) | |
| budget -= size | |
| except DigestFetchError as e: | |
| row["error"] = str(e) | |
| files.append(row) | |
| return meta, files | |
| def fetch_github(owner_repo: str) -> tuple[dict, list[dict]]: | |
| meta = json.loads(_get(f"https://api.github.com/repos/{owner_repo}", | |
| "application/vnd.github+json", cap=1024 * 1024).decode("utf-8", errors="replace")) | |
| branch = meta.get("default_branch") or "main" | |
| data = _get(f"https://codeload.github.com/{owner_repo}/zip/refs/heads/{branch}") | |
| return meta, [{"name": f"{owner_repo}@{branch}.zip", "size": len(data), "data": data}] | |
| def _ext(name: str) -> str: | |
| return os.path.splitext(name)[1].lower() if not name.endswith(".R") else ".R" | |
| def _is_archive(name: str) -> bool: | |
| return name.lower().endswith((".zip",)) | |
| def expand(files: list[dict]) -> list[dict]: | |
| """Flatten zip archives into member rows [{path, size, text|None}] without | |
| writing anything to disk. Members past MAX_MEMBER_BYTES are listed only.""" | |
| rows: list[dict] = [] | |
| for f in files: | |
| data = f.get("data") | |
| if data is None: | |
| rows.append({"path": f["name"], "size": f["size"], "text": None, | |
| "note": f.get("error") or "not read (size cap or type)"}) | |
| continue | |
| if _is_archive(f["name"]): | |
| try: | |
| zf = zipfile.ZipFile(io.BytesIO(data)) | |
| except zipfile.BadZipFile: | |
| rows.append({"path": f["name"], "size": f["size"], "text": None, "note": "not a zip file"}) | |
| continue | |
| read_budget = MAX_ARCHIVE_BYTES | |
| for m in zf.infolist(): | |
| if m.is_dir() or "__MACOSX/" in m.filename or "/." in "/" + m.filename: | |
| continue | |
| row = {"path": m.filename, "size": m.file_size, "text": None} | |
| if _ext(m.filename) in TEXT_EXT and m.file_size <= MAX_MEMBER_BYTES and m.file_size <= read_budget: | |
| try: | |
| with zf.open(m) as fh: | |
| raw = fh.read(MAX_MEMBER_BYTES + 1) | |
| read_budget -= len(raw) | |
| row["text"] = raw.decode("utf-8", errors="replace") | |
| except Exception as e: # corrupt member, encrypted, bomb guard | |
| row["note"] = f"unreadable: {type(e).__name__}" | |
| rows.append(row) | |
| else: | |
| text = data.decode("utf-8", errors="replace") if _ext(f["name"]) in TEXT_EXT else None | |
| rows.append({"path": f["name"], "size": f["size"], "text": text}) | |
| return rows | |
| # βββ analysis βββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def constructs_from_text(full_text: str, extra: list[str] | None = None) -> list[str]: | |
| """The paper's named constructs: acronyms and hyphenated acronym forms, | |
| frequency-ranked, minus boilerplate. 'ΞXRM' and 'D-XRM' both fold to | |
| terms the grep will find.""" | |
| counts: Counter = Counter() | |
| for m in _ACRONYM_RE.finditer(full_text or ""): | |
| tok = m.group(0).replace("Ξ-", "").replace("Ξ", "") | |
| core = tok.split("-")[-1] if "-" in tok else tok | |
| if core.upper() in _ACRONYM_STOP or tok.upper() in _ACRONYM_STOP or len(core) < 2: | |
| continue | |
| counts[tok] += 1 | |
| terms = [t for t, n in counts.most_common(MAX_TERMS * 2) if n >= 3][:MAX_TERMS] | |
| for t in extra or []: | |
| if t and t not in terms: | |
| terms.insert(0, t) | |
| return terms[:MAX_TERMS] | |
| def _notebook_code(text: str) -> str: | |
| try: | |
| nb = json.loads(text) | |
| cells = [c for c in nb.get("cells", []) if c.get("cell_type") == "code"] | |
| return "\n".join("".join(c.get("source", [])) for c in cells) | |
| except Exception: | |
| return "" | |
| def _source_lines(row: dict) -> list[str]: | |
| text = row.get("text") or "" | |
| if row["path"].lower().endswith(".ipynb"): | |
| text = _notebook_code(text) | |
| return text.splitlines() | |
| def definitions(rows: list[dict]) -> list[tuple[str, list[str]]]: | |
| out = [] | |
| for r in rows: | |
| if _ext(r["path"]) not in SOURCE_EXT or not r.get("text"): | |
| continue | |
| names = [] | |
| for m in _DEF_RE.finditer("\n".join(_source_lines(r))): | |
| name = next((g for g in m.groups() if g), None) | |
| if name and name not in names: | |
| names.append(name) | |
| if names: | |
| out.append((r["path"], names[:DEFS_PER_FILE])) | |
| return out | |
| def term_hits(rows: list[dict], terms: list[str]) -> dict[str, list[str]]: | |
| """For each term, the source lines that DEFINE it first: a def/class whose | |
| name carries the term, then an assignment whose left-hand identifier | |
| carries it (`xrm = a * b + c`, `xrm_value = compute_xrm(`), | |
| then any other line naming it, with prints, labels, docstrings and | |
| comments last; `path:line: text`, capped per term. Identifier matching is | |
| case-insensitive with `-` folded to `_` so XRM-2 finds xrm_2.""" | |
| hits: dict[str, list[str]] = {} | |
| for term in terms: | |
| ident = re.escape(term.replace("-", "_").lower()) | |
| exact = re.compile(r"(?<![A-Za-z0-9_])" + re.escape(term) + r"(?![A-Za-z0-9])") | |
| in_ident = re.compile(r"(?<![a-z0-9])" + ident + r"(?![a-z0-9])|_" + ident + r"(?![a-z0-9])|" + ident + r"_", re.I) | |
| defn = re.compile(r"^\s*(?:def|class|function)\s+\w*" + ident + r"\w*\s*[(:]", re.I) | |
| assign = re.compile(r"^\s*(?:[\w.\[\]\"']*" + ident + r"[\w.\[\]\"']*)\s*(?:<-|=(?!=))", re.I) | |
| noise = re.compile(r"^\s*(?:#|//|\"\"\"|''')|\bprint\s*\(|\.set_[xy]?label|\blogging\.|\blog\.(?:info|debug)|^\s*f?[\"']", re.I) | |
| ranked: list[tuple[int, str]] = [] | |
| for r in rows: | |
| if _ext(r["path"]) not in SOURCE_EXT or not r.get("text"): | |
| continue | |
| for i, line in enumerate(_source_lines(r), 1): | |
| code = line.split("#")[0] | |
| if not (exact.search(line) or in_ident.search(code.replace("-", "_"))): | |
| continue | |
| entry = f"{r['path']}:{i}: {line.strip()[:160]}" | |
| if defn.search(code.replace("-", "_")): | |
| rank = 0 | |
| elif assign.search(code.replace("-", "_")): | |
| rank = 1 | |
| elif noise.search(line): | |
| rank = 3 | |
| else: | |
| rank = 2 | |
| ranked.append((rank, entry)) | |
| ranked.sort(key=lambda t: t[0]) | |
| chosen = [e for _, e in ranked[:HITS_PER_TERM]] | |
| if chosen: | |
| hits[term] = chosen | |
| return hits | |
| def render(source_label: str, meta_line: str, rows: list[dict], terms: list[str]) -> str: | |
| lines = [f"Source: {source_label}", meta_line, ""] | |
| lines.append(f"Files ({len(rows)}):") | |
| for r in rows[:MAX_FILES_LISTED]: | |
| note = f" [{r['note']}]" if r.get("note") else "" | |
| lines.append(f" {r['path']} ({r['size']} bytes){note}") | |
| if len(rows) > MAX_FILES_LISTED: | |
| lines.append(f" ... {len(rows) - MAX_FILES_LISTED} more files not listed") | |
| readme = next((r for r in rows if os.path.basename(r["path"]).lower().startswith("readme") and r.get("text")), None) | |
| if readme: | |
| lines += ["", f"README head ({readme['path']}, first {README_LINES} lines):"] | |
| lines += [" " + l[:200] for l in readme["text"].splitlines()[:README_LINES]] | |
| defs = definitions(rows) | |
| if defs: | |
| lines += ["", "Definitions per source file:"] | |
| for path, names in defs[:MAX_SOURCE_FILES_READ]: | |
| lines.append(f" {path}: {', '.join(names)}") | |
| hits = term_hits(rows, terms) | |
| lines += ["", f"Paper constructs searched in the code: {', '.join(terms) or '(none found in the text)'}"] | |
| for term in terms: | |
| if term in hits: | |
| lines.append(f" {term}:") | |
| lines += [" " + h for h in hits[term]] | |
| else: | |
| lines.append(f" {term}: not found in any source file") | |
| text = "\n".join(lines) | |
| if len(text) > MAX_DIGEST_CHARS: | |
| text = text[:MAX_DIGEST_CHARS] + "\n[digest truncated at the size cap]" | |
| return text | |
| # βββ entry point ββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def build(submission: dict, full_text: str, *, log=print, extra_terms: list[str] | None = None) -> tuple[str, dict]: | |
| """Returns (digest_markdown, meta). Never raises. meta records what was | |
| fetched, how many bytes, and every link that was not.""" | |
| meta = {"enabled": bool(getattr(config, "CODE_DIGEST_ENABLED", True)), "links": [], | |
| "fetched": None, "bytes": 0, "files": 0, "skipped": [], "error": None} | |
| if not meta["enabled"]: | |
| return "(code package digest disabled)", meta | |
| links = links_from(submission, full_text) | |
| meta["links"] = links | |
| if not links: | |
| return "(none: the submission declares no code or data location)", meta | |
| terms = constructs_from_text(full_text, extra_terms) | |
| for link in links: | |
| kind, ref = classify(link) | |
| try: | |
| if kind == "zenodo": | |
| rec, files = fetch_zenodo(ref) | |
| title = ((rec.get("metadata") or {}).get("title")) or "" | |
| label = f"Zenodo record {ref} (declared by the author)" | |
| meta_line = f"Record title: {title[:200]}" | |
| elif kind == "github": | |
| rec, files = fetch_github(ref) | |
| label = f"GitHub repository {ref} (declared by the author)" | |
| meta_line = f"Default branch: {rec.get('default_branch')}; description: {(rec.get('description') or '')[:200]}" | |
| else: | |
| meta["skipped"].append({"link": link, "why": "host not fetched (only Zenodo and GitHub are read)"}) | |
| continue | |
| except DigestFetchError as e: | |
| meta["skipped"].append({"link": link, "why": str(e)[:200]}) | |
| log(f" code digest: {link}: {e}") | |
| continue | |
| except Exception as e: | |
| meta["skipped"].append({"link": link, "why": f"{type(e).__name__}: {e}"[:200]}) | |
| log(f" code digest: {link}: {type(e).__name__}: {e}") | |
| continue | |
| rows = expand(files) | |
| meta.update({"fetched": link, "bytes": sum(f["size"] for f in files if f.get("data") is not None), | |
| "files": len(rows), "terms": terms}) | |
| digest = render(label, meta_line, rows, terms) | |
| if meta["skipped"]: | |
| digest += "\n\nNot fetched: " + "; ".join(f"{s['link']} ({s['why']})" for s in meta["skipped"]) | |
| others = [l for l in links if l != link] | |
| if others: | |
| digest += "\n\nOther declared locations (not read; one package per digest): " + ", ".join(others) | |
| log(f" code digest: {link}: {len(rows)} files, {meta['bytes']} bytes read, {len(terms)} constructs") | |
| return digest, meta | |
| why = "; ".join(f"{s['link']} ({s['why']})" for s in meta["skipped"]) or "no fetchable link" | |
| meta["error"] = why | |
| return f"(declared but not fetched: {why})", meta | |
| if __name__ == "__main__": | |
| # `code_digest.py <sub_dir>` -- print the digest for a submission on disk. | |
| if len(sys.argv) != 2: | |
| print("usage: code_digest.py <submission dir>"); sys.exit(2) | |
| d = sys.argv[1] | |
| sub = json.load(open(os.path.join(d, "submission.json"))) | |
| import submission_intake | |
| text = submission_intake.extract_pdf_text(os.path.join(d, "paper.pdf")) | |
| md, m = build(sub, text) | |
| print(md) | |
| print("\n--- meta ---\n" + json.dumps({k: v for k, v in m.items() if k != "terms"}, indent=2)) | |