File size: 6,060 Bytes
e640531 405fe3a e640531 405fe3a e640531 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 | """Code and data availability check (policy of 2026-09-28).
ICSAC does not host authors' code or data. A paper that says its code or data
is available must say where: the submission form's code and data link, a
supplement related identifier, or a repository link in the paper's body. A paper that makes the claim with
no link anywhere is held for a revise-and-resubmit before the panel runs; the
curator signs that decision like every other one.
The check reads claims, not words. "Data" appears in nearly every paper, so
only an availability statement counts, and a statement that there is nothing
to share ("not applicable", "no new data were generated") does not. The
reference list is cut off first: links and phrases there belong to cited work.
"""
from __future__ import annotations
import re
CHECK_VERSION = 1
# A line that is only a references heading starts the reference list.
_REFS_HEADING = re.compile(
r"^[ \t]*(?:\d+\.?[ \t]*)?(?:references|bibliography|works cited|literature cited)[ \t]*:?[ \t]*$",
re.I | re.M,
)
_CLAIM_PATTERNS = [
# "Data availability", "Code availability", "Data and code availability"
r"\b(?:data|code|software)(?:\s*(?:,|and|&)\s*(?:data|code|software|materials?))*\s+availability\b",
# "The code is available", "data will be deposited", "source code and
# validated data files are available" (up to four words in between)
r"\b(?:data|code|software|scripts?|notebooks?|datasets?)\b(?:\s+[\w-]+){0,4}?\s+(?:are|is|will\s+be)\s+"
r"(?:publicly\s+|freely\s+|openly\s+|also\s+)?(?:available|provided|released|deposited|included|shared)\b",
r"\bavailable\s+(?:up)?on\s+(?:reasonable\s+)?request\b",
r"\b(?:accompanying|supplementary|supplemental)\s+(?:archive|code|data|datasets?|materials?|files?|"
r"repository|notebooks?|scripts?|software)\b",
r"\b(?:in|see)\s+the\s+supplement\b",
r"\bprovided\s+(?:code|scripts?|notebooks?)\b",
r"\b(?:archive|repository)\s+(?:contains|includes|provides)\b",
]
_CLAIM_RE = re.compile("|".join(f"(?:{p})" for p in _CLAIM_PATTERNS), re.I)
# The first pattern is a heading ("Data availability"); its answer follows it.
_HEADING_RE = re.compile(_CLAIM_PATTERNS[0], re.I)
_SENTENCE_END = re.compile(r"[.!?](?:\s|$)")
# Said right after an availability heading, these mean there is nothing to link.
_NOTHING_TO_SHARE = re.compile(
r"not\s+applicable|\bn/?a\b|no\s+(?:new\s+)?(?:data|code|datasets?)\b|"
r"(?:were|was)\s+not\s+(?:generated|used|created|collected)|"
r"does\s+not\s+(?:use|involve|generate|include)\s+(?:any\s+)?(?:data|code)",
re.I,
)
# Public places code and data live. The link only has to be public; this list
# is how the check recognises one in the paper's own text. A link anywhere in
# the body counts (front matter often carries the repository line).
_LINK_RE = re.compile(
r"(?:https?://|www\.)[^\s)\]>\"',;]*"
r"(?:github\.com|gitlab\.com|bitbucket\.org|codeberg\.org|zenodo\.org|osf\.io|figshare\.com|"
r"dataverse|kaggle\.com|colab\.research\.google\.com|huggingface\.co|softwareheritage\.org|"
r"datadryad\.org|codeocean\.com|sourceforge\.net)[^\s)\]>\"',;]*"
r"|\b10\.(?:5281/zenodo\.\d+|7910/DVN/[\w]+|6084/m9\.figshare\.[\d.]+|17605/OSF\.IO/\w+|5061/dryad\.\w+)",
re.I,
)
def body_text(full_text: str) -> str:
"""The paper without its reference list (the last references heading on)."""
matches = list(_REFS_HEADING.finditer(full_text or ""))
if not matches:
return full_text or ""
return full_text[: matches[-1].start()]
def _sentence(text: str, start: int, end: int, cap: int = 240) -> str:
left = max(text.rfind(". ", 0, start), text.rfind("\n\n", 0, start))
left = 0 if left < 0 else left + 1
right_candidates = [i for i in (text.find(". ", end), text.find("\n\n", end)) if i >= 0]
right = min(right_candidates) + 1 if right_candidates else len(text)
s = re.sub(r"\s+", " ", text[left:right]).strip()
return s if len(s) <= cap else s[: cap - 1].rstrip() + "…"
def find_claims(text: str) -> list[str]:
"""Availability statements in `text`, one quoted sentence each."""
out: list[str] = []
for m in _CLAIM_RE.finditer(text):
after = text[m.end(): m.end() + 120]
if not _HEADING_RE.fullmatch(m.group(0)):
# "Code is available. No new data were generated." The second
# sentence is about data; it does not take back the code claim.
end = _SENTENCE_END.search(after)
if end:
after = after[: end.start()]
if _NOTHING_TO_SHARE.search(after):
continue
quote = _sentence(text, m.start(), m.end())
if quote not in out:
out.append(quote)
return out
def find_links(text: str) -> list[str]:
return sorted({m.group(0).rstrip(".") for m in _LINK_RE.finditer(text)})
def assess(submission: dict, full_text: str) -> dict:
"""The record the worker saves as code_availability.json.
missing_link is True only when the paper claims code or data and no link
exists anywhere: not on the form, not as a supplement related identifier,
and not in the paper's body.
"""
body = body_text(full_text)
claims = find_claims(body)
code_data = submission.get("code_data") or {}
form_url = (code_data.get("url") or "").strip() or None
# Either direction counts: forms before 2026-09-28 only offered "is supplement to".
related = [r.get("identifier", "") for r in (submission.get("related_identifiers") or [])
if isinstance(r, dict) and r.get("relation") in ("isSupplementedBy", "isSupplementTo")
and r.get("identifier")]
in_paper = find_links(body) if claims else []
missing = bool(claims) and not form_url and not related and not in_paper
return {
"check_version": CHECK_VERSION,
"declared_available": code_data.get("available"),
"claims": claims[:5],
"links": {"form": form_url, "related": related, "in_paper": in_paper},
"missing_link": missing,
}
|