File size: 6,060 Bytes
e640531
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
405fe3a
 
 
e640531
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
405fe3a
 
 
 
 
 
 
 
e640531
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
"""Code and data availability check (policy of 2026-09-28).

ICSAC does not host authors' code or data. A paper that says its code or data
is available must say where: the submission form's code and data link, a
supplement related identifier, or a repository link in the paper's body. A paper that makes the claim with
no link anywhere is held for a revise-and-resubmit before the panel runs; the
curator signs that decision like every other one.

The check reads claims, not words. "Data" appears in nearly every paper, so
only an availability statement counts, and a statement that there is nothing
to share ("not applicable", "no new data were generated") does not. The
reference list is cut off first: links and phrases there belong to cited work.
"""
from __future__ import annotations

import re

CHECK_VERSION = 1

# A line that is only a references heading starts the reference list.
_REFS_HEADING = re.compile(
    r"^[ \t]*(?:\d+\.?[ \t]*)?(?:references|bibliography|works cited|literature cited)[ \t]*:?[ \t]*$",
    re.I | re.M,
)

_CLAIM_PATTERNS = [
    # "Data availability", "Code availability", "Data and code availability"
    r"\b(?:data|code|software)(?:\s*(?:,|and|&)\s*(?:data|code|software|materials?))*\s+availability\b",
    # "The code is available", "data will be deposited", "source code and
    # validated data files are available" (up to four words in between)
    r"\b(?:data|code|software|scripts?|notebooks?|datasets?)\b(?:\s+[\w-]+){0,4}?\s+(?:are|is|will\s+be)\s+"
    r"(?:publicly\s+|freely\s+|openly\s+|also\s+)?(?:available|provided|released|deposited|included|shared)\b",
    r"\bavailable\s+(?:up)?on\s+(?:reasonable\s+)?request\b",
    r"\b(?:accompanying|supplementary|supplemental)\s+(?:archive|code|data|datasets?|materials?|files?|"
    r"repository|notebooks?|scripts?|software)\b",
    r"\b(?:in|see)\s+the\s+supplement\b",
    r"\bprovided\s+(?:code|scripts?|notebooks?)\b",
    r"\b(?:archive|repository)\s+(?:contains|includes|provides)\b",
]
_CLAIM_RE = re.compile("|".join(f"(?:{p})" for p in _CLAIM_PATTERNS), re.I)
# The first pattern is a heading ("Data availability"); its answer follows it.
_HEADING_RE = re.compile(_CLAIM_PATTERNS[0], re.I)
_SENTENCE_END = re.compile(r"[.!?](?:\s|$)")

# Said right after an availability heading, these mean there is nothing to link.
_NOTHING_TO_SHARE = re.compile(
    r"not\s+applicable|\bn/?a\b|no\s+(?:new\s+)?(?:data|code|datasets?)\b|"
    r"(?:were|was)\s+not\s+(?:generated|used|created|collected)|"
    r"does\s+not\s+(?:use|involve|generate|include)\s+(?:any\s+)?(?:data|code)",
    re.I,
)

# Public places code and data live. The link only has to be public; this list
# is how the check recognises one in the paper's own text. A link anywhere in
# the body counts (front matter often carries the repository line).
_LINK_RE = re.compile(
    r"(?:https?://|www\.)[^\s)\]>\"',;]*"
    r"(?:github\.com|gitlab\.com|bitbucket\.org|codeberg\.org|zenodo\.org|osf\.io|figshare\.com|"
    r"dataverse|kaggle\.com|colab\.research\.google\.com|huggingface\.co|softwareheritage\.org|"
    r"datadryad\.org|codeocean\.com|sourceforge\.net)[^\s)\]>\"',;]*"
    r"|\b10\.(?:5281/zenodo\.\d+|7910/DVN/[\w]+|6084/m9\.figshare\.[\d.]+|17605/OSF\.IO/\w+|5061/dryad\.\w+)",
    re.I,
)


def body_text(full_text: str) -> str:
    """The paper without its reference list (the last references heading on)."""
    matches = list(_REFS_HEADING.finditer(full_text or ""))
    if not matches:
        return full_text or ""
    return full_text[: matches[-1].start()]


def _sentence(text: str, start: int, end: int, cap: int = 240) -> str:
    left = max(text.rfind(". ", 0, start), text.rfind("\n\n", 0, start))
    left = 0 if left < 0 else left + 1
    right_candidates = [i for i in (text.find(". ", end), text.find("\n\n", end)) if i >= 0]
    right = min(right_candidates) + 1 if right_candidates else len(text)
    s = re.sub(r"\s+", " ", text[left:right]).strip()
    return s if len(s) <= cap else s[: cap - 1].rstrip() + "…"


def find_claims(text: str) -> list[str]:
    """Availability statements in `text`, one quoted sentence each."""
    out: list[str] = []
    for m in _CLAIM_RE.finditer(text):
        after = text[m.end(): m.end() + 120]
        if not _HEADING_RE.fullmatch(m.group(0)):
            # "Code is available. No new data were generated." The second
            # sentence is about data; it does not take back the code claim.
            end = _SENTENCE_END.search(after)
            if end:
                after = after[: end.start()]
        if _NOTHING_TO_SHARE.search(after):
            continue
        quote = _sentence(text, m.start(), m.end())
        if quote not in out:
            out.append(quote)
    return out


def find_links(text: str) -> list[str]:
    return sorted({m.group(0).rstrip(".") for m in _LINK_RE.finditer(text)})


def assess(submission: dict, full_text: str) -> dict:
    """The record the worker saves as code_availability.json.

    missing_link is True only when the paper claims code or data and no link
    exists anywhere: not on the form, not as a supplement related identifier,
    and not in the paper's body.
    """
    body = body_text(full_text)
    claims = find_claims(body)
    code_data = submission.get("code_data") or {}
    form_url = (code_data.get("url") or "").strip() or None
    # Either direction counts: forms before 2026-09-28 only offered "is supplement to".
    related = [r.get("identifier", "") for r in (submission.get("related_identifiers") or [])
               if isinstance(r, dict) and r.get("relation") in ("isSupplementedBy", "isSupplementTo")
               and r.get("identifier")]
    in_paper = find_links(body) if claims else []
    missing = bool(claims) and not form_url and not related and not in_paper
    return {
        "check_version": CHECK_VERSION,
        "declared_available": code_data.get("available"),
        "claims": claims[:5],
        "links": {"form": form_url, "related": related, "in_paper": in_paper},
        "missing_link": missing,
    }