grading-answers / src /grades.py
giuseppecuccunm's picture
Deploy grading app interface improvements
f01c104 verified
Raw History Blame Contribute Delete
17.5 kB
"""Sparse grading rows shared by the Space, migrate script, and hourly sync.
Live HF-mode grades are sparse CSVs keyed by ``(term, field_of_law, question_id)``.
When the same key appears twice, the row with the later ``modified_at`` wins.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
import pandas as pd
GRADE_COLUMNS: tuple[str, ...] = (
"term",
"field_of_law",
"question_id",
"verdict",
"created_at",
"modified_at",
)
GRADE_KEY: tuple[str, ...] = ("term", "field_of_law", "question_id")
_MIN_TS = pd.Timestamp.min.tz_localize("UTC")
def empty_grades() -> pd.DataFrame:
"""Return a zero-row frame with the sparse grade schema."""
return pd.DataFrame(columns=list(GRADE_COLUMNS))
def answers_csv_path(path: str) -> str:
"""Map ``*_answers.parquet`` (or already-csv) paths to the CSV filename."""
if path.endswith(".parquet"):
return path[: -len(".parquet")] + ".csv"
return path
def normalize_grades(df: pd.DataFrame | None) -> pd.DataFrame:
"""Keep graded rows only, with UTC timestamps and one row per key."""
if df is None or df.empty:
return empty_grades()
out = df.copy()
if "created_at" not in out.columns:
out["created_at"] = pd.NaT
if "modified_at" not in out.columns:
out["modified_at"] = pd.NaT
if "time_stamp" in out.columns:
legacy = pd.to_datetime(out["time_stamp"], errors="coerce", utc=True)
out["created_at"] = pd.to_datetime(
out["created_at"], errors="coerce", utc=True
).fillna(legacy)
out["modified_at"] = pd.to_datetime(
out["modified_at"], errors="coerce", utc=True
).fillna(legacy)
for col in GRADE_KEY + ("verdict",):
if col not in out.columns:
out[col] = None
out = out[out["verdict"].notna() & (out["verdict"].astype(str).str.strip() != "")]
if out.empty:
return empty_grades()
out = out.copy()
out["created_at"] = pd.to_datetime(out["created_at"], errors="coerce", utc=True)
out["modified_at"] = pd.to_datetime(out["modified_at"], errors="coerce", utc=True)
out["modified_at"] = out["modified_at"].fillna(out["created_at"])
return _dedupe_latest(out)[list(GRADE_COLUMNS)].reset_index(drop=True)
def merge_grades(left: pd.DataFrame, right: pd.DataFrame) -> pd.DataFrame:
"""Union grades by key; newer ``modified_at`` wins on conflict."""
frames = [normalize_grades(left), normalize_grades(right)]
frames = [frame for frame in frames if not frame.empty]
if not frames:
return empty_grades()
combined = pd.concat(frames, ignore_index=True)
return _dedupe_latest(combined).reset_index(drop=True)
def newer_or_missing_rows(bucket: pd.DataFrame, dataset: pd.DataFrame) -> pd.DataFrame:
"""Bucket rows whose key is absent in the dataset or strictly newer."""
bucket_n = normalize_grades(bucket)
dataset_n = normalize_grades(dataset)
if bucket_n.empty or dataset_n.empty:
return bucket_n
right = dataset_n[list(GRADE_KEY) + ["modified_at"]].rename(
columns={"modified_at": "dataset_modified_at"}
)
left = bucket_n.merge(right, on=list(GRADE_KEY), how="left")
missing = left["dataset_modified_at"].isna()
newer = left["modified_at"] > left["dataset_modified_at"]
picked = left.loc[missing | newer, list(GRADE_COLUMNS)]
if picked.empty:
return empty_grades()
return picked.reset_index(drop=True)
def apply_bucket_upsert(
dataset: pd.DataFrame, bucket: pd.DataFrame
) -> tuple[pd.DataFrame, int]:
"""Apply new or newer bucket rows onto a dataset snapshot.
Returns the merged grades and how many bucket rows were inserted or updated
(including Pass→Fail when ``modified_at`` is newer).
"""
dataset_n = normalize_grades(dataset)
updates = newer_or_missing_rows(bucket, dataset_n)
if updates.empty:
return dataset_n, 0
return merge_grades(dataset_n, updates), len(updates)
def apply_grades_to_template(
template: pd.DataFrame, grades: pd.DataFrame
) -> pd.DataFrame:
"""Left-join sparse grades onto the swipe-deck template."""
drop = [
col
for col in ("verdict", "time_stamp", "created_at", "modified_at")
if col in template.columns
]
base = template.drop(columns=drop)
sparse = normalize_grades(grades)
merged = base.merge(sparse, on=list(GRADE_KEY), how="left")
merged["time_stamp"] = merged["modified_at"]
return merged
def to_csv_bytes(grades: pd.DataFrame) -> bytes:
"""Serialize sparse grades as UTF-8 CSV (ISO-8601 timestamps)."""
to_write = normalize_grades(grades).copy()
for col in ("created_at", "modified_at"):
to_write[col] = to_write[col].map(_isoformat)
return to_write.to_csv(index=False).encode("utf-8")
@dataclass(frozen=True)
class GradeSyncFile:
"""One user answers file that should be committed to the dataset."""
path: str
merged: pd.DataFrame
n_updates: int
def plan_grade_sync(
bucket_by_path: dict[str, pd.DataFrame],
dataset_by_path: dict[str, pd.DataFrame],
) -> list[GradeSyncFile]:
"""Return only files where the bucket has new or newer rows."""
planned: list[GradeSyncFile] = []
for path in sorted(bucket_by_path):
dataset_df = dataset_by_path.get(path, empty_grades())
merged, n_updates = apply_bucket_upsert(dataset_df, bucket_by_path[path])
if n_updates:
planned.append(GradeSyncFile(path=path, merged=merged, n_updates=n_updates))
return planned
def _dedupe_latest(df: pd.DataFrame) -> pd.DataFrame:
ranked = df.copy()
ranked["_rank"] = ranked["modified_at"].fillna(_MIN_TS)
ranked = ranked.sort_values("_rank")
ranked = ranked.drop_duplicates(subset=list(GRADE_KEY), keep="last")
return ranked.drop(columns=["_rank"])
def _isoformat(value: object) -> str:
if pd.isna(value):
return ""
return pd.Timestamp(value).isoformat()
PROGRESS_COLUMNS: tuple[str, ...] = (
"username",
"last_skipped",
"last_graded",
"last_viewed",
"furthest_viewed",
"last_activity",
)
PROGRESS_INDEX_COLUMNS: tuple[str, ...] = (
"last_skipped",
"last_graded",
"last_viewed",
"furthest_viewed",
)
def user_progress_csv_path(jurisdiction: str) -> str:
"""Bucket/dataset path for the per-jurisdiction navigation table."""
return f"{jurisdiction}/users/user_progress.csv"
def empty_progress() -> pd.DataFrame:
"""Return a zero-row navigation table."""
return pd.DataFrame(columns=list(PROGRESS_COLUMNS))
def empty_progress_record(username: str) -> dict[str, object]:
"""Return an empty navigation row for ``username``."""
return {
"username": username,
"last_skipped": None,
"last_graded": None,
"last_viewed": None,
"furthest_viewed": None,
"last_activity": None,
}
def progress_pointers_from_graded_count(n_grade: int) -> dict[str, int | None]:
"""1-based demo pointers from how many cards a user has graded."""
if n_grade <= 0:
return {
"last_skipped": None,
"last_graded": None,
"last_viewed": None,
"furthest_viewed": None,
}
last = int(n_grade)
if n_grade >= 5:
skipped = max(1, n_grade // 4)
if skipped == last:
skipped = max(1, last - 1)
elif n_grade >= 2:
skipped = 1
else:
skipped = None
return {
"last_skipped": skipped,
"last_graded": last,
"last_viewed": last,
"furthest_viewed": last,
}
def progress_for_jumps(
stored: dict[str, object],
*,
freeze: dict[str, object] | None = None,
displayed_last_viewed: object = None,
use_displayed_last_viewed: bool = False,
) -> dict[str, object]:
"""Sidebar jump values. ``freeze`` wins (view-as snapshot); else live progress."""
source = freeze if freeze is not None else stored
last_viewed = source.get("last_viewed")
if freeze is None and use_displayed_last_viewed:
last_viewed = displayed_last_viewed
return {
"last_skipped": source.get("last_skipped"),
"last_graded": source.get("last_graded"),
"last_viewed": last_viewed,
"furthest_viewed": source.get("furthest_viewed"),
}
def _coerce_card_index(value: object) -> int | None:
if value is None:
return None
if isinstance(value, bool):
return None
if isinstance(value, int):
return value if value > 0 else None
if isinstance(value, float):
if pd.isna(value):
return None
number = int(value)
return number if number > 0 else None
text = str(value).strip()
if not text:
return None
try:
number = int(text)
except ValueError:
return None
return number if number > 0 else None
def _coerce_activity(value: object) -> pd.Timestamp | None:
if value is None or isinstance(value, bool):
return None
if isinstance(value, str) and not value.strip():
return None
stamp = pd.to_datetime(value, utc=True, errors="coerce")
if pd.isna(stamp):
return None
return pd.Timestamp(stamp)
def _activity_csv_cell(value: object) -> str:
stamp = _coerce_activity(value)
return "" if stamp is None else stamp.isoformat()
def format_last_active(value: object) -> str:
"""Admin-card label for a stored ``last_activity`` timestamp."""
stamp = _coerce_activity(value)
if stamp is None:
return "Last active: —"
month_names = (
"Jan",
"Feb",
"Mar",
"Apr",
"May",
"Jun",
"Jul",
"Aug",
"Sep",
"Oct",
"Nov",
"Dec",
)
return (
f"Last active: {stamp.hour:02d}:{stamp.minute:02d}, "
f"{month_names[stamp.month - 1]} {stamp.day:02d} {stamp.year:04d}"
)
def progress_activity_by_user(df: pd.DataFrame) -> dict[str, object]:
"""Map username to ``last_activity`` (or ``None``) for the admin console."""
table = normalize_progress(df)
activity: dict[str, object] = {}
for _, row in table.iterrows():
activity[str(row["username"])] = _coerce_activity(row["last_activity"])
return activity
def normalize_progress(df: pd.DataFrame | None) -> pd.DataFrame:
"""One row per username; invalid card indices become empty."""
if df is None or df.empty:
return empty_progress()
out = df.copy()
for col in PROGRESS_COLUMNS:
if col not in out.columns:
out[col] = None
out["username"] = out["username"].map(
lambda v: "" if pd.isna(v) else str(v).strip()
)
out = out[out["username"] != ""]
if out.empty:
return empty_progress()
for col in PROGRESS_INDEX_COLUMNS:
out[col] = out[col].map(_coerce_card_index)
out["last_activity"] = out["last_activity"].map(_coerce_activity)
out = out.drop_duplicates(subset=["username"], keep="last")
out = out.sort_values("username").reset_index(drop=True)
return out[list(PROGRESS_COLUMNS)]
def get_progress_record(df: pd.DataFrame, username: str) -> dict[str, object]:
"""Return the navigation row for ``username``, or an empty record."""
table = normalize_progress(df)
match = table[table["username"] == username]
if match.empty:
return empty_progress_record(username)
row = match.iloc[0]
record: dict[str, object] = {"username": username}
for col in PROGRESS_INDEX_COLUMNS:
value = row[col]
record[col] = None if pd.isna(value) else int(value)
record["last_activity"] = _coerce_activity(row["last_activity"])
return record
def apply_progress_event(
df: pd.DataFrame,
username: str,
*,
last_skipped: int | None = None,
last_graded: int | None = None,
last_viewed: int | None = None,
furthest_viewed: int | None = None,
at: object | None = None,
) -> pd.DataFrame:
"""Update one user's navigation row. ``furthest_viewed`` only grows."""
table = normalize_progress(df)
record = get_progress_record(table, username)
if last_skipped is not None:
record["last_skipped"] = last_skipped
if last_graded is not None:
record["last_graded"] = last_graded
if last_viewed is not None:
record["last_viewed"] = last_viewed
if furthest_viewed is not None:
previous = record["furthest_viewed"]
if isinstance(previous, int):
record["furthest_viewed"] = max(previous, furthest_viewed)
else:
record["furthest_viewed"] = furthest_viewed
stamp = _coerce_activity(at)
record["last_activity"] = stamp if stamp is not None else pd.Timestamp.now(tz="UTC")
rest = table[table["username"] != username]
added = pd.DataFrame([record])
if rest.empty:
return normalize_progress(added)
updated = pd.concat([rest, added], ignore_index=True)
return normalize_progress(updated)
def rename_progress_user(
df: pd.DataFrame, old_name: str, new_name: str
) -> pd.DataFrame:
"""Rename one navigation row. No-op when the names match."""
table = normalize_progress(df)
if not old_name or old_name == new_name:
return table
table.loc[table["username"] == old_name, "username"] = new_name
return normalize_progress(table)
def drop_progress_user(df: pd.DataFrame, username: str) -> pd.DataFrame:
"""Remove one user's navigation row."""
table = normalize_progress(df)
return table[table["username"] != username].reset_index(drop=True)
def progress_tables_equal(left: pd.DataFrame, right: pd.DataFrame) -> bool:
"""True when normalized navigation tables serialize identically."""
return to_progress_csv_bytes(left) == to_progress_csv_bytes(right)
def to_progress_csv_bytes(df: pd.DataFrame) -> bytes:
"""Serialize the navigation table as UTF-8 CSV (empty cells if never set)."""
out = normalize_progress(df).copy()
for col in PROGRESS_INDEX_COLUMNS:
out[col] = out[col].map(
lambda v: "" if v is None or pd.isna(v) else str(int(v))
)
out["last_activity"] = out["last_activity"].map(_activity_csv_cell)
return out.to_csv(index=False).encode("utf-8")
@dataclass(frozen=True)
class ProgressSyncFile:
"""One jurisdiction navigation table that should be copied to the dataset."""
path: str
merged: pd.DataFrame
def plan_progress_sync(
bucket_by_path: dict[str, pd.DataFrame],
dataset_by_path: dict[str, pd.DataFrame],
) -> list[ProgressSyncFile]:
"""Return progress files whose bucket snapshot differs from the dataset."""
planned: list[ProgressSyncFile] = []
for path in sorted(bucket_by_path):
bucket = normalize_progress(bucket_by_path[path])
dataset = normalize_progress(dataset_by_path.get(path))
if progress_tables_equal(bucket, dataset):
continue
planned.append(ProgressSyncFile(path=path, merged=bucket))
return planned
def parse_card_number(raw: object, total_cards: int) -> tuple[int | None, str | None]:
"""Parse a 1-based card number, or return ``(None, error)``."""
if isinstance(raw, bool) or isinstance(raw, float):
return None, "Card number must be digits only."
if isinstance(raw, int):
text = str(raw)
elif raw is None:
text = ""
else:
text = str(raw).strip()
if not text:
return None, "Enter a card number."
if not re.fullmatch(r"[0-9]+", text):
return None, "Card number must be digits only."
number = int(text)
if number <= 0:
return None, "Card number must be at least 1."
if total_cards <= 0:
return None, "There are no cards to jump to."
if number > total_cards:
return None, f"Card number must be at most {total_cards}."
return number, None
def landing_card_index(last_viewed: object, total_cards: int) -> int:
"""Return the 0-based deck index to open after login."""
if total_cards <= 0:
return 0
number = _coerce_card_index(last_viewed)
if number is None:
return 0
return min(number, total_cards) - 1
def is_graded_verdict(verdict: object) -> bool:
"""True for a Pass or fail reason. Empty/None means ungraded (skip/view)."""
if verdict is None:
return False
if isinstance(verdict, float) and pd.isna(verdict):
return False
return bool(str(verdict).strip())
def count_graded_verdicts(verdicts: dict[object, object]) -> int:
"""Count Pass/fail entries. Skip and view do not belong in this map."""
return sum(1 for value in verdicts.values() if is_graded_verdict(value))
def other_fail_verdict(raw: object) -> str:
"""Return custom Other text, or ``Fail`` when the field is empty."""
if raw is None:
return "Fail"
if isinstance(raw, float) and pd.isna(raw):
return "Fail"
text = str(raw).strip()
return text if text else "Fail"
def other_reason_prefill(
existing: object,
*,
pass_label: str,
fail_reasons: tuple[str, ...],
) -> str:
"""Return stored Other-fail text, or ``""`` for Pass / named fail reasons."""
if existing is None:
return ""
if isinstance(existing, float) and pd.isna(existing):
return ""
label = str(existing).strip()
if not label or label == pass_label or label in fail_reasons:
return ""
return label