File size: 10,475 Bytes
30ea0e9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
"""Per-challenge configuration.

Everything that distinguishes one challenge from another arrives through
environment variables β€” on a deployed Space they are written by
``bootstrap/init_challenge.py`` from the repo's ``challenge.yaml``. The app
itself never reads challenge.yaml, so a running Space can be reconfigured by
editing its variables alone.
"""
import json
from functools import lru_cache
from typing import Literal

from huggingface_hub import get_token
from pydantic import Field, model_validator
from pydantic_settings import BaseSettings, SettingsConfigDict


class Settings(BaseSettings):
    model_config = SettingsConfigDict(env_file=".env", extra="ignore")

    # ── Collaboration identity (required, no defaults β€” fail loud) ──
    org: str = Field(alias="ORG")
    collab_slug: str = Field(alias="COLLAB_SLUG")
    # Default derived as {org}/{collab_slug}-main-bucket (see validator).
    central_bucket: str = Field("", alias="CENTRAL_BUCKET")
    # Private bucket for the audit log and quota ledger; the Space is its only
    # writer. Org members can read private buckets in their org, so when it
    # lives in ORG (single-org mode) audit rows omit caller_ip/user_agent;
    # put it in a separate admin org to keep those and a private eval set
    # away from participants.
    audit_bucket: str = Field(alias="AUDIT_BUCKET")
    # Durable 24h job-quota ledger, stored in the private audit bucket under a
    # separate prefix (decoupled from the audit log so purges don't reset
    # quotas). Lets the per-agent / per-user job caps survive Space restarts.
    job_quota_ledger_path: str = Field(
        "quota/job_ledger.jsonl", alias="JOB_QUOTA_LEDGER_PATH"
    )

    # Admin token: writes the central bucket, appends the audit log, and (if
    # jobs are enabled) launches benchmark jobs on org credits. It is ONLY a
    # launch credential β€” never injected into a job container.
    hf_token: str | None = Field(None, alias="HF_TOKEN")

    # ── Scoring (what makes a result a result) ──
    # The numeric frontmatter field results are ranked on.
    score_field: str = Field("score", alias="SCORE_FIELD")
    # Human-readable unit for docs and the API self-description.
    score_unit: str = Field("points", alias="SCORE_UNIT")
    # desc = higher is better, asc = lower is better.
    score_order: Literal["desc", "asc"] = Field("desc", alias="SCORE_ORDER")
    # CSV of required result-frontmatter fields. The score field is always
    # required and validated as a positive number; `status` is always
    # validated against agent-run|negative when present in this list.
    required_result_fields: str = Field(
        "score,method,status,description", alias="REQUIRED_RESULT_FIELDS"
    )

    sync_max_bytes: int = Field(5 * 1024**3, alias="SYNC_MAX_BYTES")
    sync_max_files: int = Field(10_000, alias="SYNC_MAX_FILES")

    bucket_write_per_minute: int = Field(60, alias="BUCKET_WRITE_PER_MINUTE")
    bucket_write_burst: int = Field(20, alias="BUCKET_WRITE_BURST")
    raw_message_per_minute: int = Field(5, alias="RAW_MESSAGE_PER_MINUTE")
    raw_message_per_hour: int = Field(30, alias="RAW_MESSAGE_PER_HOUR")
    registration_per_minute: int = Field(3, alias="REGISTRATION_PER_MINUTE")

    dedup_lru_size: int = Field(10_000, alias="DEDUP_LRU_SIZE")

    # Read model & discovery endpoints. The listing TTL bounds staleness for
    # out-of-band admin edits only β€” API writes are cached write-through.
    listing_ttl_s: float = Field(30.0, alias="LISTING_TTL_S")
    content_cache_max_bytes: int = Field(64 * 1024**2, alias="CONTENT_CACHE_MAX_BYTES")
    expand_max_limit: int = Field(200, alias="EXPAND_MAX_LIMIT")
    mention_fanout_cap: int = Field(10, alias="MENTION_FANOUT_CAP")

    # How long the organizer-broadcast gate caches the challenge org's
    # member→role map (rarely changes; a miss is one members-API call).
    org_roles_ttl_s: float = Field(300.0, alias="ORG_ROLES_TTL_S")

    # ── Channels (topic rooms, CHANNELS_DESIGN.md) ──
    # Creation is organizer-only (the broadcast admin gate), so it needs no
    # dedicated rate limit β€” the shared raw-message limiter bounds it.
    # Newest messages included per subscribed channel in the digest block.
    digest_channel_recent: int = Field(3, alias="DIGEST_CHANNEL_RECENT")

    # ── Watch / long-poll (WATCH_DESIGN.md) ──
    # `wait=<seconds>` parks a read until something new lands for the caller.
    # 55s is the ceiling because edge proxies kill idle connections around 60s;
    # the knob exists for self-hosted deployments without the *.hf.space proxy.
    # `wait` is always clamped into [0, max], never rejected.
    longpoll_max_wait_s: float = Field(55.0, alias="LONGPOLL_MAX_WAIT_S")
    # The waiter registry is in-process (single uvicorn worker β€” see the
    # Dockerfile CMD): per owner the OLDEST waiter is evicted, self-healing an
    # abandoned long-poll so the newest connection is the live one; past the
    # global cap new waiters are never registered and instead paced
    # server-side (a jittered hold, not an instant empty answer, so degrading
    # lowers load instead of inviting a hot loop).
    longpoll_max_waiters_per_owner: int = Field(4, alias="LONGPOLL_MAX_WAITERS_PER_OWNER")
    longpoll_max_waiters_total: int = Field(256, alias="LONGPOLL_MAX_WAITERS_TOTAL")
    # A broadcast (or a busy channel) wakes many waiters at once; resolving them
    # in the same tick makes every agent re-poll simultaneously β€” a request
    # spike into this Space that can trip the *.hf.space edge rate limit. When a
    # wake targets more than this many waiters, its releases are spread
    # uniformly over [0, spread_s] so re-polls arrive staggered. spread_s=0
    # disables (restores instant wakes); tune spread_s up to flatten the peak
    # req/s further at the cost of a little broadcast-delivery latency.
    longpoll_wake_spread_s: float = Field(8.0, alias="LONGPOLL_WAKE_SPREAD_S")
    longpoll_wake_spread_threshold: int = Field(20, alias="LONGPOLL_WAKE_SPREAD_THRESHOLD")

    # ── Benchmark jobs (optional; POST /v1/jobs:run is 404 when off) ──
    jobs_enabled: bool = Field(False, alias="JOBS_ENABLED")
    # Harness contract: a directory at {central_bucket}/{harness_prefix}
    # containing {harness_entrypoint}. The job runs
    #   python3 /harness/{entrypoint} --submission-dir /submission
    #       --state-dir /state [--private-dir /private] {extra args}
    # and must write /state/summary.json with at least {score_field: number}.
    harness_prefix: str = Field("shared_resources/harness", alias="HARNESS_PREFIX")
    harness_entrypoint: str = Field("run.py", alias="JOB_HARNESS_ENTRYPOINT")
    # JSON list of extra CLI args appended to the harness command.
    job_extra_args: str = Field("[]", alias="JOB_EXTRA_ARGS")
    job_image: str = Field("python:3.12", alias="JOB_IMAGE")
    job_flavor: str = Field("a10g-small", alias="JOB_FLAVOR")
    job_timeout_minutes: int = Field(40, alias="JOB_TIMEOUT_MINUTES")
    # How long the watcher polls past the platform cap before forcing a cancel.
    job_watch_poll_s: int = Field(20, alias="JOB_WATCH_POLL_S")
    job_watch_grace_s: int = Field(180, alias="JOB_WATCH_GRACE_S")
    job_log_tail_lines: int = Field(2000, alias="JOB_LOG_TAIL_LINES")

    # Per-window job quotas (24h sliding window).
    job_per_agent_per_day: int = Field(10, alias="JOB_PER_AGENT_PER_DAY")
    job_per_user_per_day: int = Field(30, alias="JOB_PER_USER_PER_DAY")

    # ── Automated verification on new SOTA (optional; requires jobs) ──
    # The private eval set lives in the audit bucket (never the org-readable
    # central bucket) under {private_dataset_prefix}/ and is mounted read-only
    # at /private; rw job state also lands in the audit bucket because private
    # data may echo into the job output.
    verifier_enabled: bool = Field(False, alias="VERIFIER_ENABLED")
    verifier_agent: str = Field("", alias="VERIFIER_AGENT")
    private_dataset_prefix: str = Field("eval_dataset", alias="PRIVATE_DATASET_PREFIX")
    verification_runs_prefix: str = Field(
        "verification_runs", alias="VERIFICATION_RUNS_PREFIX"
    )
    # Relative tolerance on the re-run score: |rerun - reported| / reported.
    score_tol: float = Field(0.05, alias="VERIFIER_SCORE_TOL")
    # Optional guardrail: a summary.json field that must stay <= guard_cap
    # (e.g. a quality metric like perplexity). Empty = no guardrail.
    guard_field: str = Field("", alias="VERIFIER_GUARD_FIELD")
    guard_cap: float = Field(0.0, alias="VERIFIER_GUARD_CAP")

    @model_validator(mode="after")
    def _derive_defaults(self) -> "Settings":
        if not self.central_bucket:
            self.central_bucket = f"{self.org}/{self.collab_slug}-main-bucket"
        return self

    @property
    def audit_bucket_in_org(self) -> bool:
        """True when participants (challenge-org members) can read the audit bucket."""
        return self.audit_bucket.split("/")[0] == self.org

    @property
    def agent_bucket_prefix(self) -> str:
        return f"{self.collab_slug}-"

    @property
    def required_result_field_list(self) -> list[str]:
        fields = [f.strip() for f in self.required_result_fields.split(",") if f.strip()]
        if self.score_field not in fields:
            fields.insert(0, self.score_field)
        return fields

    @property
    def job_extra_arg_list(self) -> list[str]:
        try:
            args = json.loads(self.job_extra_args)
        except json.JSONDecodeError:
            raise ValueError(f"JOB_EXTRA_ARGS is not valid JSON: {self.job_extra_args!r}")
        if not isinstance(args, list):
            raise ValueError("JOB_EXTRA_ARGS must be a JSON list of strings")
        return [str(a) for a in args]

    def agent_bucket(self, agent_id: str) -> str:
        return f"{self.org}/{self.collab_slug}-{agent_id}"

    def better(self, a: float, b: float) -> bool:
        """True iff score ``a`` beats score ``b`` under the configured order."""
        return a > b if self.score_order == "desc" else a < b

    def resolved_token(self) -> str:
        if self.hf_token:
            return self.hf_token
        cached = get_token()
        if not cached:
            raise RuntimeError(
                "no HF token available; set HF_TOKEN or run `hf auth login`"
            )
        return cached


@lru_cache
def get_settings() -> Settings:
    return Settings()