"""Sparse grading rows shared by the Space, migrate script, and hourly sync. Live HF-mode grades are sparse CSVs keyed by ``(term, field_of_law, question_id)``. When the same key appears twice, the row with the later ``modified_at`` wins. """ from __future__ import annotations import re from dataclasses import dataclass import pandas as pd GRADE_COLUMNS: tuple[str, ...] = ( "term", "field_of_law", "question_id", "verdict", "created_at", "modified_at", ) GRADE_KEY: tuple[str, ...] = ("term", "field_of_law", "question_id") _MIN_TS = pd.Timestamp.min.tz_localize("UTC") def empty_grades() -> pd.DataFrame: """Return a zero-row frame with the sparse grade schema.""" return pd.DataFrame(columns=list(GRADE_COLUMNS)) def answers_csv_path(path: str) -> str: """Map ``*_answers.parquet`` (or already-csv) paths to the CSV filename.""" if path.endswith(".parquet"): return path[: -len(".parquet")] + ".csv" return path def normalize_grades(df: pd.DataFrame | None) -> pd.DataFrame: """Keep graded rows only, with UTC timestamps and one row per key.""" if df is None or df.empty: return empty_grades() out = df.copy() if "created_at" not in out.columns: out["created_at"] = pd.NaT if "modified_at" not in out.columns: out["modified_at"] = pd.NaT if "time_stamp" in out.columns: legacy = pd.to_datetime(out["time_stamp"], errors="coerce", utc=True) out["created_at"] = pd.to_datetime( out["created_at"], errors="coerce", utc=True ).fillna(legacy) out["modified_at"] = pd.to_datetime( out["modified_at"], errors="coerce", utc=True ).fillna(legacy) for col in GRADE_KEY + ("verdict",): if col not in out.columns: out[col] = None out = out[out["verdict"].notna() & (out["verdict"].astype(str).str.strip() != "")] if out.empty: return empty_grades() out = out.copy() out["created_at"] = pd.to_datetime(out["created_at"], errors="coerce", utc=True) out["modified_at"] = pd.to_datetime(out["modified_at"], errors="coerce", utc=True) out["modified_at"] = out["modified_at"].fillna(out["created_at"]) return _dedupe_latest(out)[list(GRADE_COLUMNS)].reset_index(drop=True) def merge_grades(left: pd.DataFrame, right: pd.DataFrame) -> pd.DataFrame: """Union grades by key; newer ``modified_at`` wins on conflict.""" frames = [normalize_grades(left), normalize_grades(right)] frames = [frame for frame in frames if not frame.empty] if not frames: return empty_grades() combined = pd.concat(frames, ignore_index=True) return _dedupe_latest(combined).reset_index(drop=True) def newer_or_missing_rows(bucket: pd.DataFrame, dataset: pd.DataFrame) -> pd.DataFrame: """Bucket rows whose key is absent in the dataset or strictly newer.""" bucket_n = normalize_grades(bucket) dataset_n = normalize_grades(dataset) if bucket_n.empty or dataset_n.empty: return bucket_n right = dataset_n[list(GRADE_KEY) + ["modified_at"]].rename( columns={"modified_at": "dataset_modified_at"} ) left = bucket_n.merge(right, on=list(GRADE_KEY), how="left") missing = left["dataset_modified_at"].isna() newer = left["modified_at"] > left["dataset_modified_at"] picked = left.loc[missing | newer, list(GRADE_COLUMNS)] if picked.empty: return empty_grades() return picked.reset_index(drop=True) def apply_bucket_upsert( dataset: pd.DataFrame, bucket: pd.DataFrame ) -> tuple[pd.DataFrame, int]: """Apply new or newer bucket rows onto a dataset snapshot. Returns the merged grades and how many bucket rows were inserted or updated (including Pass→Fail when ``modified_at`` is newer). """ dataset_n = normalize_grades(dataset) updates = newer_or_missing_rows(bucket, dataset_n) if updates.empty: return dataset_n, 0 return merge_grades(dataset_n, updates), len(updates) def apply_grades_to_template( template: pd.DataFrame, grades: pd.DataFrame ) -> pd.DataFrame: """Left-join sparse grades onto the swipe-deck template.""" drop = [ col for col in ("verdict", "time_stamp", "created_at", "modified_at") if col in template.columns ] base = template.drop(columns=drop) sparse = normalize_grades(grades) merged = base.merge(sparse, on=list(GRADE_KEY), how="left") merged["time_stamp"] = merged["modified_at"] return merged def to_csv_bytes(grades: pd.DataFrame) -> bytes: """Serialize sparse grades as UTF-8 CSV (ISO-8601 timestamps).""" to_write = normalize_grades(grades).copy() for col in ("created_at", "modified_at"): to_write[col] = to_write[col].map(_isoformat) return to_write.to_csv(index=False).encode("utf-8") @dataclass(frozen=True) class GradeSyncFile: """One user answers file that should be committed to the dataset.""" path: str merged: pd.DataFrame n_updates: int def plan_grade_sync( bucket_by_path: dict[str, pd.DataFrame], dataset_by_path: dict[str, pd.DataFrame], ) -> list[GradeSyncFile]: """Return only files where the bucket has new or newer rows.""" planned: list[GradeSyncFile] = [] for path in sorted(bucket_by_path): dataset_df = dataset_by_path.get(path, empty_grades()) merged, n_updates = apply_bucket_upsert(dataset_df, bucket_by_path[path]) if n_updates: planned.append(GradeSyncFile(path=path, merged=merged, n_updates=n_updates)) return planned def _dedupe_latest(df: pd.DataFrame) -> pd.DataFrame: ranked = df.copy() ranked["_rank"] = ranked["modified_at"].fillna(_MIN_TS) ranked = ranked.sort_values("_rank") ranked = ranked.drop_duplicates(subset=list(GRADE_KEY), keep="last") return ranked.drop(columns=["_rank"]) def _isoformat(value: object) -> str: if pd.isna(value): return "" return pd.Timestamp(value).isoformat() PROGRESS_COLUMNS: tuple[str, ...] = ( "username", "last_skipped", "last_graded", "last_viewed", "furthest_viewed", "last_activity", ) PROGRESS_INDEX_COLUMNS: tuple[str, ...] = ( "last_skipped", "last_graded", "last_viewed", "furthest_viewed", ) def user_progress_csv_path(jurisdiction: str) -> str: """Bucket/dataset path for the per-jurisdiction navigation table.""" return f"{jurisdiction}/users/user_progress.csv" def empty_progress() -> pd.DataFrame: """Return a zero-row navigation table.""" return pd.DataFrame(columns=list(PROGRESS_COLUMNS)) def empty_progress_record(username: str) -> dict[str, object]: """Return an empty navigation row for ``username``.""" return { "username": username, "last_skipped": None, "last_graded": None, "last_viewed": None, "furthest_viewed": None, "last_activity": None, } def progress_pointers_from_graded_count(n_grade: int) -> dict[str, int | None]: """1-based demo pointers from how many cards a user has graded.""" if n_grade <= 0: return { "last_skipped": None, "last_graded": None, "last_viewed": None, "furthest_viewed": None, } last = int(n_grade) if n_grade >= 5: skipped = max(1, n_grade // 4) if skipped == last: skipped = max(1, last - 1) elif n_grade >= 2: skipped = 1 else: skipped = None return { "last_skipped": skipped, "last_graded": last, "last_viewed": last, "furthest_viewed": last, } def progress_for_jumps( stored: dict[str, object], *, freeze: dict[str, object] | None = None, displayed_last_viewed: object = None, use_displayed_last_viewed: bool = False, ) -> dict[str, object]: """Sidebar jump values. ``freeze`` wins (view-as snapshot); else live progress.""" source = freeze if freeze is not None else stored last_viewed = source.get("last_viewed") if freeze is None and use_displayed_last_viewed: last_viewed = displayed_last_viewed return { "last_skipped": source.get("last_skipped"), "last_graded": source.get("last_graded"), "last_viewed": last_viewed, "furthest_viewed": source.get("furthest_viewed"), } def _coerce_card_index(value: object) -> int | None: if value is None: return None if isinstance(value, bool): return None if isinstance(value, int): return value if value > 0 else None if isinstance(value, float): if pd.isna(value): return None number = int(value) return number if number > 0 else None text = str(value).strip() if not text: return None try: number = int(text) except ValueError: return None return number if number > 0 else None def _coerce_activity(value: object) -> pd.Timestamp | None: if value is None or isinstance(value, bool): return None if isinstance(value, str) and not value.strip(): return None stamp = pd.to_datetime(value, utc=True, errors="coerce") if pd.isna(stamp): return None return pd.Timestamp(stamp) def _activity_csv_cell(value: object) -> str: stamp = _coerce_activity(value) return "" if stamp is None else stamp.isoformat() def format_last_active(value: object) -> str: """Admin-card label for a stored ``last_activity`` timestamp.""" stamp = _coerce_activity(value) if stamp is None: return "Last active: —" month_names = ( "Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec", ) return ( f"Last active: {stamp.hour:02d}:{stamp.minute:02d}, " f"{month_names[stamp.month - 1]} {stamp.day:02d} {stamp.year:04d}" ) def progress_activity_by_user(df: pd.DataFrame) -> dict[str, object]: """Map username to ``last_activity`` (or ``None``) for the admin console.""" table = normalize_progress(df) activity: dict[str, object] = {} for _, row in table.iterrows(): activity[str(row["username"])] = _coerce_activity(row["last_activity"]) return activity def normalize_progress(df: pd.DataFrame | None) -> pd.DataFrame: """One row per username; invalid card indices become empty.""" if df is None or df.empty: return empty_progress() out = df.copy() for col in PROGRESS_COLUMNS: if col not in out.columns: out[col] = None out["username"] = out["username"].map( lambda v: "" if pd.isna(v) else str(v).strip() ) out = out[out["username"] != ""] if out.empty: return empty_progress() for col in PROGRESS_INDEX_COLUMNS: out[col] = out[col].map(_coerce_card_index) out["last_activity"] = out["last_activity"].map(_coerce_activity) out = out.drop_duplicates(subset=["username"], keep="last") out = out.sort_values("username").reset_index(drop=True) return out[list(PROGRESS_COLUMNS)] def get_progress_record(df: pd.DataFrame, username: str) -> dict[str, object]: """Return the navigation row for ``username``, or an empty record.""" table = normalize_progress(df) match = table[table["username"] == username] if match.empty: return empty_progress_record(username) row = match.iloc[0] record: dict[str, object] = {"username": username} for col in PROGRESS_INDEX_COLUMNS: value = row[col] record[col] = None if pd.isna(value) else int(value) record["last_activity"] = _coerce_activity(row["last_activity"]) return record def apply_progress_event( df: pd.DataFrame, username: str, *, last_skipped: int | None = None, last_graded: int | None = None, last_viewed: int | None = None, furthest_viewed: int | None = None, at: object | None = None, ) -> pd.DataFrame: """Update one user's navigation row. ``furthest_viewed`` only grows.""" table = normalize_progress(df) record = get_progress_record(table, username) if last_skipped is not None: record["last_skipped"] = last_skipped if last_graded is not None: record["last_graded"] = last_graded if last_viewed is not None: record["last_viewed"] = last_viewed if furthest_viewed is not None: previous = record["furthest_viewed"] if isinstance(previous, int): record["furthest_viewed"] = max(previous, furthest_viewed) else: record["furthest_viewed"] = furthest_viewed stamp = _coerce_activity(at) record["last_activity"] = stamp if stamp is not None else pd.Timestamp.now(tz="UTC") rest = table[table["username"] != username] added = pd.DataFrame([record]) if rest.empty: return normalize_progress(added) updated = pd.concat([rest, added], ignore_index=True) return normalize_progress(updated) def rename_progress_user( df: pd.DataFrame, old_name: str, new_name: str ) -> pd.DataFrame: """Rename one navigation row. No-op when the names match.""" table = normalize_progress(df) if not old_name or old_name == new_name: return table table.loc[table["username"] == old_name, "username"] = new_name return normalize_progress(table) def drop_progress_user(df: pd.DataFrame, username: str) -> pd.DataFrame: """Remove one user's navigation row.""" table = normalize_progress(df) return table[table["username"] != username].reset_index(drop=True) def progress_tables_equal(left: pd.DataFrame, right: pd.DataFrame) -> bool: """True when normalized navigation tables serialize identically.""" return to_progress_csv_bytes(left) == to_progress_csv_bytes(right) def to_progress_csv_bytes(df: pd.DataFrame) -> bytes: """Serialize the navigation table as UTF-8 CSV (empty cells if never set).""" out = normalize_progress(df).copy() for col in PROGRESS_INDEX_COLUMNS: out[col] = out[col].map( lambda v: "" if v is None or pd.isna(v) else str(int(v)) ) out["last_activity"] = out["last_activity"].map(_activity_csv_cell) return out.to_csv(index=False).encode("utf-8") @dataclass(frozen=True) class ProgressSyncFile: """One jurisdiction navigation table that should be copied to the dataset.""" path: str merged: pd.DataFrame def plan_progress_sync( bucket_by_path: dict[str, pd.DataFrame], dataset_by_path: dict[str, pd.DataFrame], ) -> list[ProgressSyncFile]: """Return progress files whose bucket snapshot differs from the dataset.""" planned: list[ProgressSyncFile] = [] for path in sorted(bucket_by_path): bucket = normalize_progress(bucket_by_path[path]) dataset = normalize_progress(dataset_by_path.get(path)) if progress_tables_equal(bucket, dataset): continue planned.append(ProgressSyncFile(path=path, merged=bucket)) return planned def parse_card_number(raw: object, total_cards: int) -> tuple[int | None, str | None]: """Parse a 1-based card number, or return ``(None, error)``.""" if isinstance(raw, bool) or isinstance(raw, float): return None, "Card number must be digits only." if isinstance(raw, int): text = str(raw) elif raw is None: text = "" else: text = str(raw).strip() if not text: return None, "Enter a card number." if not re.fullmatch(r"[0-9]+", text): return None, "Card number must be digits only." number = int(text) if number <= 0: return None, "Card number must be at least 1." if total_cards <= 0: return None, "There are no cards to jump to." if number > total_cards: return None, f"Card number must be at most {total_cards}." return number, None def landing_card_index(last_viewed: object, total_cards: int) -> int: """Return the 0-based deck index to open after login.""" if total_cards <= 0: return 0 number = _coerce_card_index(last_viewed) if number is None: return 0 return min(number, total_cards) - 1 def is_graded_verdict(verdict: object) -> bool: """True for a Pass or fail reason. Empty/None means ungraded (skip/view).""" if verdict is None: return False if isinstance(verdict, float) and pd.isna(verdict): return False return bool(str(verdict).strip()) def count_graded_verdicts(verdicts: dict[object, object]) -> int: """Count Pass/fail entries. Skip and view do not belong in this map.""" return sum(1 for value in verdicts.values() if is_graded_verdict(value)) def other_fail_verdict(raw: object) -> str: """Return custom Other text, or ``Fail`` when the field is empty.""" if raw is None: return "Fail" if isinstance(raw, float) and pd.isna(raw): return "Fail" text = str(raw).strip() return text if text else "Fail" def other_reason_prefill( existing: object, *, pass_label: str, fail_reasons: tuple[str, ...], ) -> str: """Return stored Other-fail text, or ``""`` for Pass / named fail reasons.""" if existing is None: return "" if isinstance(existing, float) and pd.isna(existing): return "" label = str(existing).strip() if not label or label == pass_label or label in fail_reasons: return "" return label