Spaces:
Running
Running
Download src/grades.py from TransLegal/grading-answers: direct link, hf CLI and curl.
- Browser
- Download file 17.5 kB
-
https://huggingface.co/spaces/TransLegal/grading-answers/resolve/main/src/grades.py
- Command line
-
hf download hf://spaces/TransLegal/grading-answers/src/grades.py
-
curl -L -o grades.py https://huggingface.co/spaces/TransLegal/grading-answers/resolve/main/src/grades.py
17.5 kB
| """Sparse grading rows shared by the Space, migrate script, and hourly sync. | |
| Live HF-mode grades are sparse CSVs keyed by ``(term, field_of_law, question_id)``. | |
| When the same key appears twice, the row with the later ``modified_at`` wins. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| from dataclasses import dataclass | |
| import pandas as pd | |
| GRADE_COLUMNS: tuple[str, ...] = ( | |
| "term", | |
| "field_of_law", | |
| "question_id", | |
| "verdict", | |
| "created_at", | |
| "modified_at", | |
| ) | |
| GRADE_KEY: tuple[str, ...] = ("term", "field_of_law", "question_id") | |
| _MIN_TS = pd.Timestamp.min.tz_localize("UTC") | |
| def empty_grades() -> pd.DataFrame: | |
| """Return a zero-row frame with the sparse grade schema.""" | |
| return pd.DataFrame(columns=list(GRADE_COLUMNS)) | |
| def answers_csv_path(path: str) -> str: | |
| """Map ``*_answers.parquet`` (or already-csv) paths to the CSV filename.""" | |
| if path.endswith(".parquet"): | |
| return path[: -len(".parquet")] + ".csv" | |
| return path | |
| def normalize_grades(df: pd.DataFrame | None) -> pd.DataFrame: | |
| """Keep graded rows only, with UTC timestamps and one row per key.""" | |
| if df is None or df.empty: | |
| return empty_grades() | |
| out = df.copy() | |
| if "created_at" not in out.columns: | |
| out["created_at"] = pd.NaT | |
| if "modified_at" not in out.columns: | |
| out["modified_at"] = pd.NaT | |
| if "time_stamp" in out.columns: | |
| legacy = pd.to_datetime(out["time_stamp"], errors="coerce", utc=True) | |
| out["created_at"] = pd.to_datetime( | |
| out["created_at"], errors="coerce", utc=True | |
| ).fillna(legacy) | |
| out["modified_at"] = pd.to_datetime( | |
| out["modified_at"], errors="coerce", utc=True | |
| ).fillna(legacy) | |
| for col in GRADE_KEY + ("verdict",): | |
| if col not in out.columns: | |
| out[col] = None | |
| out = out[out["verdict"].notna() & (out["verdict"].astype(str).str.strip() != "")] | |
| if out.empty: | |
| return empty_grades() | |
| out = out.copy() | |
| out["created_at"] = pd.to_datetime(out["created_at"], errors="coerce", utc=True) | |
| out["modified_at"] = pd.to_datetime(out["modified_at"], errors="coerce", utc=True) | |
| out["modified_at"] = out["modified_at"].fillna(out["created_at"]) | |
| return _dedupe_latest(out)[list(GRADE_COLUMNS)].reset_index(drop=True) | |
| def merge_grades(left: pd.DataFrame, right: pd.DataFrame) -> pd.DataFrame: | |
| """Union grades by key; newer ``modified_at`` wins on conflict.""" | |
| frames = [normalize_grades(left), normalize_grades(right)] | |
| frames = [frame for frame in frames if not frame.empty] | |
| if not frames: | |
| return empty_grades() | |
| combined = pd.concat(frames, ignore_index=True) | |
| return _dedupe_latest(combined).reset_index(drop=True) | |
| def newer_or_missing_rows(bucket: pd.DataFrame, dataset: pd.DataFrame) -> pd.DataFrame: | |
| """Bucket rows whose key is absent in the dataset or strictly newer.""" | |
| bucket_n = normalize_grades(bucket) | |
| dataset_n = normalize_grades(dataset) | |
| if bucket_n.empty or dataset_n.empty: | |
| return bucket_n | |
| right = dataset_n[list(GRADE_KEY) + ["modified_at"]].rename( | |
| columns={"modified_at": "dataset_modified_at"} | |
| ) | |
| left = bucket_n.merge(right, on=list(GRADE_KEY), how="left") | |
| missing = left["dataset_modified_at"].isna() | |
| newer = left["modified_at"] > left["dataset_modified_at"] | |
| picked = left.loc[missing | newer, list(GRADE_COLUMNS)] | |
| if picked.empty: | |
| return empty_grades() | |
| return picked.reset_index(drop=True) | |
| def apply_bucket_upsert( | |
| dataset: pd.DataFrame, bucket: pd.DataFrame | |
| ) -> tuple[pd.DataFrame, int]: | |
| """Apply new or newer bucket rows onto a dataset snapshot. | |
| Returns the merged grades and how many bucket rows were inserted or updated | |
| (including Pass→Fail when ``modified_at`` is newer). | |
| """ | |
| dataset_n = normalize_grades(dataset) | |
| updates = newer_or_missing_rows(bucket, dataset_n) | |
| if updates.empty: | |
| return dataset_n, 0 | |
| return merge_grades(dataset_n, updates), len(updates) | |
| def apply_grades_to_template( | |
| template: pd.DataFrame, grades: pd.DataFrame | |
| ) -> pd.DataFrame: | |
| """Left-join sparse grades onto the swipe-deck template.""" | |
| drop = [ | |
| col | |
| for col in ("verdict", "time_stamp", "created_at", "modified_at") | |
| if col in template.columns | |
| ] | |
| base = template.drop(columns=drop) | |
| sparse = normalize_grades(grades) | |
| merged = base.merge(sparse, on=list(GRADE_KEY), how="left") | |
| merged["time_stamp"] = merged["modified_at"] | |
| return merged | |
| def to_csv_bytes(grades: pd.DataFrame) -> bytes: | |
| """Serialize sparse grades as UTF-8 CSV (ISO-8601 timestamps).""" | |
| to_write = normalize_grades(grades).copy() | |
| for col in ("created_at", "modified_at"): | |
| to_write[col] = to_write[col].map(_isoformat) | |
| return to_write.to_csv(index=False).encode("utf-8") | |
| class GradeSyncFile: | |
| """One user answers file that should be committed to the dataset.""" | |
| path: str | |
| merged: pd.DataFrame | |
| n_updates: int | |
| def plan_grade_sync( | |
| bucket_by_path: dict[str, pd.DataFrame], | |
| dataset_by_path: dict[str, pd.DataFrame], | |
| ) -> list[GradeSyncFile]: | |
| """Return only files where the bucket has new or newer rows.""" | |
| planned: list[GradeSyncFile] = [] | |
| for path in sorted(bucket_by_path): | |
| dataset_df = dataset_by_path.get(path, empty_grades()) | |
| merged, n_updates = apply_bucket_upsert(dataset_df, bucket_by_path[path]) | |
| if n_updates: | |
| planned.append(GradeSyncFile(path=path, merged=merged, n_updates=n_updates)) | |
| return planned | |
| def _dedupe_latest(df: pd.DataFrame) -> pd.DataFrame: | |
| ranked = df.copy() | |
| ranked["_rank"] = ranked["modified_at"].fillna(_MIN_TS) | |
| ranked = ranked.sort_values("_rank") | |
| ranked = ranked.drop_duplicates(subset=list(GRADE_KEY), keep="last") | |
| return ranked.drop(columns=["_rank"]) | |
| def _isoformat(value: object) -> str: | |
| if pd.isna(value): | |
| return "" | |
| return pd.Timestamp(value).isoformat() | |
| PROGRESS_COLUMNS: tuple[str, ...] = ( | |
| "username", | |
| "last_skipped", | |
| "last_graded", | |
| "last_viewed", | |
| "furthest_viewed", | |
| "last_activity", | |
| ) | |
| PROGRESS_INDEX_COLUMNS: tuple[str, ...] = ( | |
| "last_skipped", | |
| "last_graded", | |
| "last_viewed", | |
| "furthest_viewed", | |
| ) | |
| def user_progress_csv_path(jurisdiction: str) -> str: | |
| """Bucket/dataset path for the per-jurisdiction navigation table.""" | |
| return f"{jurisdiction}/users/user_progress.csv" | |
| def empty_progress() -> pd.DataFrame: | |
| """Return a zero-row navigation table.""" | |
| return pd.DataFrame(columns=list(PROGRESS_COLUMNS)) | |
| def empty_progress_record(username: str) -> dict[str, object]: | |
| """Return an empty navigation row for ``username``.""" | |
| return { | |
| "username": username, | |
| "last_skipped": None, | |
| "last_graded": None, | |
| "last_viewed": None, | |
| "furthest_viewed": None, | |
| "last_activity": None, | |
| } | |
| def progress_pointers_from_graded_count(n_grade: int) -> dict[str, int | None]: | |
| """1-based demo pointers from how many cards a user has graded.""" | |
| if n_grade <= 0: | |
| return { | |
| "last_skipped": None, | |
| "last_graded": None, | |
| "last_viewed": None, | |
| "furthest_viewed": None, | |
| } | |
| last = int(n_grade) | |
| if n_grade >= 5: | |
| skipped = max(1, n_grade // 4) | |
| if skipped == last: | |
| skipped = max(1, last - 1) | |
| elif n_grade >= 2: | |
| skipped = 1 | |
| else: | |
| skipped = None | |
| return { | |
| "last_skipped": skipped, | |
| "last_graded": last, | |
| "last_viewed": last, | |
| "furthest_viewed": last, | |
| } | |
| def progress_for_jumps( | |
| stored: dict[str, object], | |
| *, | |
| freeze: dict[str, object] | None = None, | |
| displayed_last_viewed: object = None, | |
| use_displayed_last_viewed: bool = False, | |
| ) -> dict[str, object]: | |
| """Sidebar jump values. ``freeze`` wins (view-as snapshot); else live progress.""" | |
| source = freeze if freeze is not None else stored | |
| last_viewed = source.get("last_viewed") | |
| if freeze is None and use_displayed_last_viewed: | |
| last_viewed = displayed_last_viewed | |
| return { | |
| "last_skipped": source.get("last_skipped"), | |
| "last_graded": source.get("last_graded"), | |
| "last_viewed": last_viewed, | |
| "furthest_viewed": source.get("furthest_viewed"), | |
| } | |
| def _coerce_card_index(value: object) -> int | None: | |
| if value is None: | |
| return None | |
| if isinstance(value, bool): | |
| return None | |
| if isinstance(value, int): | |
| return value if value > 0 else None | |
| if isinstance(value, float): | |
| if pd.isna(value): | |
| return None | |
| number = int(value) | |
| return number if number > 0 else None | |
| text = str(value).strip() | |
| if not text: | |
| return None | |
| try: | |
| number = int(text) | |
| except ValueError: | |
| return None | |
| return number if number > 0 else None | |
| def _coerce_activity(value: object) -> pd.Timestamp | None: | |
| if value is None or isinstance(value, bool): | |
| return None | |
| if isinstance(value, str) and not value.strip(): | |
| return None | |
| stamp = pd.to_datetime(value, utc=True, errors="coerce") | |
| if pd.isna(stamp): | |
| return None | |
| return pd.Timestamp(stamp) | |
| def _activity_csv_cell(value: object) -> str: | |
| stamp = _coerce_activity(value) | |
| return "" if stamp is None else stamp.isoformat() | |
| def format_last_active(value: object) -> str: | |
| """Admin-card label for a stored ``last_activity`` timestamp.""" | |
| stamp = _coerce_activity(value) | |
| if stamp is None: | |
| return "Last active: —" | |
| month_names = ( | |
| "Jan", | |
| "Feb", | |
| "Mar", | |
| "Apr", | |
| "May", | |
| "Jun", | |
| "Jul", | |
| "Aug", | |
| "Sep", | |
| "Oct", | |
| "Nov", | |
| "Dec", | |
| ) | |
| return ( | |
| f"Last active: {stamp.hour:02d}:{stamp.minute:02d}, " | |
| f"{month_names[stamp.month - 1]} {stamp.day:02d} {stamp.year:04d}" | |
| ) | |
| def progress_activity_by_user(df: pd.DataFrame) -> dict[str, object]: | |
| """Map username to ``last_activity`` (or ``None``) for the admin console.""" | |
| table = normalize_progress(df) | |
| activity: dict[str, object] = {} | |
| for _, row in table.iterrows(): | |
| activity[str(row["username"])] = _coerce_activity(row["last_activity"]) | |
| return activity | |
| def normalize_progress(df: pd.DataFrame | None) -> pd.DataFrame: | |
| """One row per username; invalid card indices become empty.""" | |
| if df is None or df.empty: | |
| return empty_progress() | |
| out = df.copy() | |
| for col in PROGRESS_COLUMNS: | |
| if col not in out.columns: | |
| out[col] = None | |
| out["username"] = out["username"].map( | |
| lambda v: "" if pd.isna(v) else str(v).strip() | |
| ) | |
| out = out[out["username"] != ""] | |
| if out.empty: | |
| return empty_progress() | |
| for col in PROGRESS_INDEX_COLUMNS: | |
| out[col] = out[col].map(_coerce_card_index) | |
| out["last_activity"] = out["last_activity"].map(_coerce_activity) | |
| out = out.drop_duplicates(subset=["username"], keep="last") | |
| out = out.sort_values("username").reset_index(drop=True) | |
| return out[list(PROGRESS_COLUMNS)] | |
| def get_progress_record(df: pd.DataFrame, username: str) -> dict[str, object]: | |
| """Return the navigation row for ``username``, or an empty record.""" | |
| table = normalize_progress(df) | |
| match = table[table["username"] == username] | |
| if match.empty: | |
| return empty_progress_record(username) | |
| row = match.iloc[0] | |
| record: dict[str, object] = {"username": username} | |
| for col in PROGRESS_INDEX_COLUMNS: | |
| value = row[col] | |
| record[col] = None if pd.isna(value) else int(value) | |
| record["last_activity"] = _coerce_activity(row["last_activity"]) | |
| return record | |
| def apply_progress_event( | |
| df: pd.DataFrame, | |
| username: str, | |
| *, | |
| last_skipped: int | None = None, | |
| last_graded: int | None = None, | |
| last_viewed: int | None = None, | |
| furthest_viewed: int | None = None, | |
| at: object | None = None, | |
| ) -> pd.DataFrame: | |
| """Update one user's navigation row. ``furthest_viewed`` only grows.""" | |
| table = normalize_progress(df) | |
| record = get_progress_record(table, username) | |
| if last_skipped is not None: | |
| record["last_skipped"] = last_skipped | |
| if last_graded is not None: | |
| record["last_graded"] = last_graded | |
| if last_viewed is not None: | |
| record["last_viewed"] = last_viewed | |
| if furthest_viewed is not None: | |
| previous = record["furthest_viewed"] | |
| if isinstance(previous, int): | |
| record["furthest_viewed"] = max(previous, furthest_viewed) | |
| else: | |
| record["furthest_viewed"] = furthest_viewed | |
| stamp = _coerce_activity(at) | |
| record["last_activity"] = stamp if stamp is not None else pd.Timestamp.now(tz="UTC") | |
| rest = table[table["username"] != username] | |
| added = pd.DataFrame([record]) | |
| if rest.empty: | |
| return normalize_progress(added) | |
| updated = pd.concat([rest, added], ignore_index=True) | |
| return normalize_progress(updated) | |
| def rename_progress_user( | |
| df: pd.DataFrame, old_name: str, new_name: str | |
| ) -> pd.DataFrame: | |
| """Rename one navigation row. No-op when the names match.""" | |
| table = normalize_progress(df) | |
| if not old_name or old_name == new_name: | |
| return table | |
| table.loc[table["username"] == old_name, "username"] = new_name | |
| return normalize_progress(table) | |
| def drop_progress_user(df: pd.DataFrame, username: str) -> pd.DataFrame: | |
| """Remove one user's navigation row.""" | |
| table = normalize_progress(df) | |
| return table[table["username"] != username].reset_index(drop=True) | |
| def progress_tables_equal(left: pd.DataFrame, right: pd.DataFrame) -> bool: | |
| """True when normalized navigation tables serialize identically.""" | |
| return to_progress_csv_bytes(left) == to_progress_csv_bytes(right) | |
| def to_progress_csv_bytes(df: pd.DataFrame) -> bytes: | |
| """Serialize the navigation table as UTF-8 CSV (empty cells if never set).""" | |
| out = normalize_progress(df).copy() | |
| for col in PROGRESS_INDEX_COLUMNS: | |
| out[col] = out[col].map( | |
| lambda v: "" if v is None or pd.isna(v) else str(int(v)) | |
| ) | |
| out["last_activity"] = out["last_activity"].map(_activity_csv_cell) | |
| return out.to_csv(index=False).encode("utf-8") | |
| class ProgressSyncFile: | |
| """One jurisdiction navigation table that should be copied to the dataset.""" | |
| path: str | |
| merged: pd.DataFrame | |
| def plan_progress_sync( | |
| bucket_by_path: dict[str, pd.DataFrame], | |
| dataset_by_path: dict[str, pd.DataFrame], | |
| ) -> list[ProgressSyncFile]: | |
| """Return progress files whose bucket snapshot differs from the dataset.""" | |
| planned: list[ProgressSyncFile] = [] | |
| for path in sorted(bucket_by_path): | |
| bucket = normalize_progress(bucket_by_path[path]) | |
| dataset = normalize_progress(dataset_by_path.get(path)) | |
| if progress_tables_equal(bucket, dataset): | |
| continue | |
| planned.append(ProgressSyncFile(path=path, merged=bucket)) | |
| return planned | |
| def parse_card_number(raw: object, total_cards: int) -> tuple[int | None, str | None]: | |
| """Parse a 1-based card number, or return ``(None, error)``.""" | |
| if isinstance(raw, bool) or isinstance(raw, float): | |
| return None, "Card number must be digits only." | |
| if isinstance(raw, int): | |
| text = str(raw) | |
| elif raw is None: | |
| text = "" | |
| else: | |
| text = str(raw).strip() | |
| if not text: | |
| return None, "Enter a card number." | |
| if not re.fullmatch(r"[0-9]+", text): | |
| return None, "Card number must be digits only." | |
| number = int(text) | |
| if number <= 0: | |
| return None, "Card number must be at least 1." | |
| if total_cards <= 0: | |
| return None, "There are no cards to jump to." | |
| if number > total_cards: | |
| return None, f"Card number must be at most {total_cards}." | |
| return number, None | |
| def landing_card_index(last_viewed: object, total_cards: int) -> int: | |
| """Return the 0-based deck index to open after login.""" | |
| if total_cards <= 0: | |
| return 0 | |
| number = _coerce_card_index(last_viewed) | |
| if number is None: | |
| return 0 | |
| return min(number, total_cards) - 1 | |
| def is_graded_verdict(verdict: object) -> bool: | |
| """True for a Pass or fail reason. Empty/None means ungraded (skip/view).""" | |
| if verdict is None: | |
| return False | |
| if isinstance(verdict, float) and pd.isna(verdict): | |
| return False | |
| return bool(str(verdict).strip()) | |
| def count_graded_verdicts(verdicts: dict[object, object]) -> int: | |
| """Count Pass/fail entries. Skip and view do not belong in this map.""" | |
| return sum(1 for value in verdicts.values() if is_graded_verdict(value)) | |
| def other_fail_verdict(raw: object) -> str: | |
| """Return custom Other text, or ``Fail`` when the field is empty.""" | |
| if raw is None: | |
| return "Fail" | |
| if isinstance(raw, float) and pd.isna(raw): | |
| return "Fail" | |
| text = str(raw).strip() | |
| return text if text else "Fail" | |
| def other_reason_prefill( | |
| existing: object, | |
| *, | |
| pass_label: str, | |
| fail_reasons: tuple[str, ...], | |
| ) -> str: | |
| """Return stored Other-fail text, or ``""`` for Pass / named fail reasons.""" | |
| if existing is None: | |
| return "" | |
| if isinstance(existing, float) and pd.isna(existing): | |
| return "" | |
| label = str(existing).strip() | |
| if not label or label == pass_label or label in fail_reasons: | |
| return "" | |
| return label | |