grading-answers / src /stats.py
giuseppecuccunm's picture
Deploy grading app interface improvements
f01c104 verified
Raw History Blame Contribute Delete
26 kB
"""Pass / fail / ungraded counts and themed dashboard HTML for the Stats page."""
from __future__ import annotations
import difflib
import html
import re
from collections import Counter
from dataclasses import dataclass
import pandas as pd
from grades import is_graded_verdict, normalize_grades
DEFAULT_FAIL_REASONS: tuple[str, ...] = (
"Completeness",
"Relevance",
"Clarity",
"Correctness",
)
OTHER_REASON = "Other"
_TYPO_CUTOFF = 0.8
_MIN_EXTRA_FAIL_COUNT = 3 # more than twice
_ANSWERS_SUFFIXES = ("_answers.parquet", "_answers.csv")
SWIPE_TEMPLATE_COLUMNS: tuple[str, ...] = (
"term",
"field_of_law",
"concept",
"section",
"section_index",
"question_id",
"group",
"part",
"question",
"question_index",
"answer",
"source_fields",
"source_prompts",
)
def missing_template_columns(columns: object) -> tuple[str, ...]:
"""Return swipe-deck columns absent from a template frame."""
present = {str(name) for name in columns}
return tuple(name for name in SWIPE_TEMPLATE_COLUMNS if name not in present)
def is_test_user(record: object) -> bool:
"""True when the users.json record is flagged ``test_user``."""
return _true_flag(record, "test_user")
def is_admin(record: object) -> bool:
"""True when the users.json record is flagged ``admin``."""
return _true_flag(record, "admin")
def is_trashed(record: object) -> bool:
"""True when the users.json record is flagged ``trashed``."""
return _true_flag(record, "trashed")
def _true_flag(record: object, key: str) -> bool:
if not isinstance(record, dict):
return False
return bool(record.get(key))
def production_usernames(users: dict[str, object]) -> set[str]:
"""Registered graders: not test users and not trashed."""
return {
name
for name, record in users.items()
if not is_test_user(record) and not is_trashed(record)
}
_SPARSE_TRUE_FLAGS = ("test_user", "admin", "trashed")
def compact_user_record(record: object) -> dict[str, object]:
"""Copy a users.json row, keeping only ``true`` boolean flags."""
if not isinstance(record, dict):
return {"password": record}
row = dict(record)
for key in _SPARSE_TRUE_FLAGS:
if key not in row:
continue
if row[key]:
row[key] = True
else:
del row[key]
return row
def compact_users(users: dict[str, object]) -> dict[str, dict[str, object]]:
"""Compact every account record in a users.json mapping."""
return {name: compact_user_record(record) for name, record in users.items()}
def stamp_test_user_flags(
users: dict[str, object],
*,
production_names: frozenset[str],
) -> dict[str, dict[str, object]]:
"""Copy accounts; set ``test_user`` only for names outside the allowlist."""
stamped: dict[str, dict[str, object]] = {}
for name, record in users.items():
row = compact_user_record(record)
if name not in production_names:
row["test_user"] = True
else:
row.pop("test_user", None)
stamped[name] = compact_user_record(row)
return stamped
def apply_account_edit(
users: dict[str, object],
old_name: str,
*,
new_username: str = "",
new_password_hash: str | None = None,
) -> tuple[dict[str, dict[str, object]], str | None, str]:
"""Rename and/or rehash one account. Empty fields are left unchanged."""
if old_name not in users:
return compact_users(users), "Username not found", old_name
wanted = new_username.strip()
resulting = wanted or old_name
if wanted and wanted != old_name and wanted in users:
return compact_users(users), "Username already exists", old_name
updated = compact_users(users)
row = dict(updated.pop(old_name))
if new_password_hash:
row["password"] = new_password_hash
if not wanted and not new_password_hash:
return compact_users(users), "Nothing to save", old_name
updated[resulting] = compact_user_record(row)
return updated, None, resulting
def new_user_record(
password_hash: str, *, test_user: bool = False
) -> dict[str, object]:
"""Build a sparse users.json row. ``test_user`` is omitted when false."""
row: dict[str, object] = {"password": password_hash}
if test_user:
row["test_user"] = True
return compact_user_record(row)
def account_edit_ready(new_username: str, new_password: str) -> bool:
"""True when Save may run: at least one edit field has text."""
return bool(str(new_username or "").strip() or str(new_password or ""))
def set_account_trashed(
users: dict[str, object],
name: str,
*,
trashed: bool = True,
) -> tuple[dict[str, dict[str, object]], str | None]:
"""Toggle the sparse ``trashed`` flag on one account."""
if name not in users:
return compact_users(users), "Username not found"
updated = compact_users(users)
row = dict(updated[name])
if trashed:
row["trashed"] = True
else:
row.pop("trashed", None)
updated[name] = compact_user_record(row)
return updated, None
def delete_account(
users: dict[str, object], name: str
) -> tuple[dict[str, dict[str, object]], str | None]:
"""Remove one account from users.json."""
if name not in users:
return compact_users(users), "Username not found"
updated = compact_users(users)
updated.pop(name, None)
return updated, None
def partition_console_users(
users: dict[str, object],
*,
omit: str | None = None,
) -> tuple[list[str], list[str]]:
"""Split registered names into graders vs test users (trashed stay listed).
``omit`` hides one account (the signed-in admin) without hiding other admins.
"""
graders: list[str] = []
testers: list[str] = []
skipped = omit or ""
for name in sorted(users):
if name == skipped:
continue
if is_test_user(users[name]):
testers.append(name)
else:
graders.append(name)
return graders, testers
def can_persist_grades(view_as: str | None) -> bool:
"""False while an admin is in read-only View as."""
return not bool(view_as)
def display_username(session_username: str, view_as: str | None) -> str:
"""Name whose deck/stats to show. Login identity stays ``session_username``."""
return str(view_as) if view_as else session_username
def ensure_admin_account(
users: dict[str, object],
*,
password_hash: str,
) -> dict[str, dict[str, object]]:
"""Insert or refresh the ``admin`` account (admin + test_user, not trashed)."""
updated = compact_users(users)
row = dict(updated.get("admin") or {})
row["password"] = password_hash
row["admin"] = True
row["test_user"] = True
row.pop("trashed", None)
updated["admin"] = compact_user_record(row)
return updated
@dataclass(frozen=True)
class GradeCounts:
"""One grader's (or a rolled-up group's) Pass / fail / ungraded split."""
passed: int
failed: int
ungraded: int
fail_reasons: dict[str, int]
@property
def graded(self) -> int:
return self.passed + self.failed
@property
def total(self) -> int:
return self.passed + self.failed + self.ungraded
def add(self, other: GradeCounts) -> GradeCounts:
reasons = dict(self.fail_reasons)
for name, count in other.fail_reasons.items():
reasons[name] = reasons.get(name, 0) + count
return GradeCounts(
passed=self.passed + other.passed,
failed=self.failed + other.failed,
ungraded=self.ungraded + other.ungraded,
fail_reasons=reasons,
)
def empty_counts(fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS) -> GradeCounts:
"""Zeroed counts with a slot for each named fail reason plus Other."""
return GradeCounts(
passed=0,
failed=0,
ungraded=0,
fail_reasons=_empty_reasons(fail_reasons),
)
def sum_counts(*items: GradeCounts | None) -> GradeCounts:
"""Add counts; skip None. Empty input yields zeros with default reasons."""
acc: GradeCounts | None = None
for item in items:
if item is None:
continue
acc = item if acc is None else acc.add(item)
return acc if acc is not None else empty_counts()
def percent(part: int, whole: int) -> float:
"""Return ``part`` as a percentage of ``whole``, or 0 when whole is 0."""
if whole <= 0:
return 0.0
return 100.0 * part / whole
def segment_widths(parts: tuple[int, ...]) -> tuple[float, ...]:
"""Percent widths that sum to 100 when any part is positive."""
total = sum(parts)
if total <= 0:
return tuple(0.0 for _ in parts)
rounded = [round(100.0 * n / total, 4) for n in parts]
drift = round(100.0 - sum(rounded), 4)
for index in range(len(rounded) - 1, -1, -1):
if parts[index] > 0:
rounded[index] = round(rounded[index] + drift, 4)
break
return tuple(rounded)
def username_from_answers_filename(name: str) -> str | None:
"""Parse ``alice`` from ``alice_answers.parquet`` or ``…/alice_answers.csv``."""
base = name.rsplit("/", 1)[-1]
for suffix in _ANSWERS_SUFFIXES:
if base.endswith(suffix):
user = base[: -len(suffix)]
return user or None
return None
def grades_to_verdicts(
df: pd.DataFrame | None,
) -> dict[tuple[object, object, object], object]:
"""Map sparse or template-shaped grade rows to ``(term, fol, qid) → verdict``."""
graded = normalize_grades(df)
out: dict[tuple[object, object, object], object] = {}
if graded.empty:
return out
for term, field, question_id, verdict in zip(
graded["term"],
graded["field_of_law"],
graded["question_id"],
graded["verdict"],
strict=True,
):
out[(term, field, question_id)] = verdict
return out
def overlay_verdicts(
stored: dict[tuple[object, object, object], object],
session: dict[object, object],
) -> dict[tuple[object, object, object], object]:
"""Copy ``stored``, then apply in-session Pass/fail (and ignore empty skip)."""
merged = dict(stored)
for key, verdict in session.items():
if not is_graded_verdict(verdict):
continue
if not isinstance(key, tuple) or len(key) != 3:
continue
merged[(key[0], key[1], key[2])] = verdict
return merged
def normalize_fail_label(raw: object) -> str:
"""Strip, collapse whitespace, and casefold a fail verdict for clustering."""
text = re.sub(r"\s+", " ", str(raw).strip())
return text.casefold()
def match_canonical_fail(
label: str,
fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS,
*,
cutoff: float = _TYPO_CUTOFF,
) -> str | None:
"""Return a button fail reason when ``label`` matches exactly or as a typo."""
folded = normalize_fail_label(label)
if not folded:
return None
by_fold = {normalize_fail_label(name): name for name in fail_reasons}
if folded in by_fold:
return by_fold[folded]
matches = difflib.get_close_matches(
folded, list(by_fold.keys()), n=1, cutoff=cutoff
)
if not matches:
return None
return by_fold[matches[0]]
def cluster_fail_labels(
labels: list[str],
fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS,
*,
cutoff: float = _TYPO_CUTOFF,
) -> dict[str, int]:
"""Map display labels to counts, folding case and typos onto button reasons."""
counts: dict[str, int] = {name: 0 for name in fail_reasons}
free_raw: list[str] = []
for raw in labels:
text = re.sub(r"\s+", " ", str(raw).strip())
if not text:
continue
canonical = match_canonical_fail(text, fail_reasons, cutoff=cutoff)
if canonical is not None:
counts[canonical] = counts.get(canonical, 0) + 1
continue
free_raw.append(text)
# Cluster remaining free-text: casefold first, then typo-fold onto cluster keys.
fold_to_display: dict[str, str] = {}
fold_counts: Counter[str] = Counter()
display_votes: dict[str, Counter[str]] = {}
for text in free_raw:
folded = normalize_fail_label(text)
if folded in fold_to_display:
key = folded
else:
matches = difflib.get_close_matches(
folded, list(fold_to_display.keys()), n=1, cutoff=cutoff
)
key = matches[0] if matches else folded
if key not in fold_to_display:
fold_to_display[key] = text
fold_counts[key] += 1
display_votes.setdefault(key, Counter())[text] += 1
for key, n in fold_counts.items():
display = display_votes[key].most_common(1)[0][0]
counts[display] = counts.get(display, 0) + n
return counts
def fail_type_rows(
catalog_counts: dict[str, int],
fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS,
*,
min_extra_count: int = _MIN_EXTRA_FAIL_COUNT,
) -> tuple[str, ...]:
"""Button reasons, then free-text clusters with count > 2, then Other if needed."""
rows = list(fail_reasons)
extras = [
(name, n)
for name, n in catalog_counts.items()
if name not in fail_reasons and name != OTHER_REASON and n >= min_extra_count
]
extras.sort(key=lambda item: (-item[1], item[0].casefold()))
rows.extend(name for name, _ in extras)
leftover = sum(
n
for name, n in catalog_counts.items()
if name not in fail_reasons and name != OTHER_REASON and n < min_extra_count
)
leftover += catalog_counts.get(OTHER_REASON, 0)
if leftover > 0:
rows.append(OTHER_REASON)
return tuple(rows)
def project_fail_reasons(
counts: GradeCounts,
row_names: tuple[str, ...],
fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS,
) -> dict[str, int]:
"""Project a grader's fail_reasons onto the shared Fail by type row catalog."""
projected = {name: 0 for name in row_names}
named = set(row_names)
for name, n in counts.fail_reasons.items():
if name in named and name != OTHER_REASON:
projected[name] = projected.get(name, 0) + n
elif OTHER_REASON in named:
projected[OTHER_REASON] = projected.get(OTHER_REASON, 0) + n
# Ensure button reasons exist even when absent from the grader's dict.
for name in fail_reasons:
if name in projected:
projected.setdefault(name, 0)
return projected
def toggle_open_grader(open_names: list[str], name: str) -> list[str]:
"""Append ``name`` if absent; remove it if present. Preserve click order."""
if name in open_names:
return [item for item in open_names if item != name]
return [*open_names, name]
def prune_open_graders(
open_names: list[str], available: set[str] | frozenset[str]
) -> list[str]:
"""Drop open graders that are no longer in the Other graders list."""
return [name for name in open_names if name in available]
def counts_from_verdicts(
verdicts: dict[object, object],
total_cards: int,
*,
pass_label: str = "Pass",
fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS,
) -> GradeCounts:
"""Split verdicts into Pass, fail (by clustered reason), and ungraded cards."""
passed = 0
failed = 0
fail_labels: list[str] = []
for verdict in verdicts.values():
if not is_graded_verdict(verdict):
continue
label = str(verdict).strip()
if label == pass_label:
passed += 1
continue
failed += 1
fail_labels.append(label)
reasons = cluster_fail_labels(fail_labels, fail_reasons)
# Preserve Other slot for callers that expect it when nothing free-texted.
reasons.setdefault(OTHER_REASON, 0)
ungraded = max(0, int(total_cards) - passed - failed)
return GradeCounts(
passed=passed,
failed=failed,
ungraded=ungraded,
fail_reasons=reasons,
)
def dashboard_html(
*,
everyone: GradeCounts,
you: GradeCounts,
you_name: str,
others: list[tuple[str, GradeCounts]],
jurisdictions: list[tuple[str, str, GradeCounts]],
fail_reasons: tuple[str, ...] = DEFAULT_FAIL_REASONS,
you_is_test_user: bool = False,
open_others: list[tuple[str, GradeCounts]] | None = None,
) -> str:
"""Return the Stats page body (hero, aggregated, you, other graders)."""
open_others = open_others or []
catalog = dict(everyone.fail_reasons)
if you_is_test_user:
for name, n in you.fail_reasons.items():
catalog[name] = catalog.get(name, 0) + n
row_names = fail_type_rows(catalog, fail_reasons)
styles: list[str] = []
everyone_bar = _stacked_bar_html("stats-bar-everyone", everyone, styles)
reason_block = _fail_reasons_html(
"stats-reason-everyone",
everyone,
row_names,
styles,
)
juri_block = _jurisdiction_block_html(jurisdictions, styles)
open_names = {name for name, _ in open_others}
others_block = _others_grid_html(others, styles, open_names=open_names)
you_role = "Test user (excluded from aggregation)" if you_is_test_user else "Grader"
you_panel = _grader_panel_html(
title=you_name,
role=you_role,
counts=you,
team_graded=everyone.graded,
bar_id="stats-bar-you",
reason_prefix="stats-reason-you",
row_names=row_names,
styles=styles,
caption_self=True,
)
open_panels = []
for index, (name, counts) in enumerate(open_others):
open_panels.append(
_grader_panel_html(
title=name,
role="Grader",
counts=counts,
team_graded=everyone.graded,
bar_id=f"stats-bar-open-{_css_id(name)}-{index}",
reason_prefix=f"stats-reason-open-{_css_id(name)}-{index}",
row_names=row_names,
styles=styles,
caption_self=False,
)
)
style_tag = f"<style>{''.join(styles)}</style>" if styles else ""
return (
f"{style_tag}"
'<div class="stats-page">'
'<div class="stats-hero">'
'<h1 class="stats-hero-title">Grading stats</h1>'
'<p class="stats-hero-kicker">'
"Pass, fail, and ungraded cards across every jurisdiction with a template."
"</p>"
"</div>"
'<section class="stats-panel">'
'<div class="stats-panel-head">'
"<h2>Aggregated results</h2>"
'<p class="stats-muted">All graders · all jurisdictions</p>'
"</div>"
f"{_kpi_row_html(everyone)}"
f"{everyone_bar}"
f"{_legend_html()}"
f"{reason_block}"
f"{juri_block}"
"</section>"
f"{you_panel}"
'<section class="stats-panel">'
'<div class="stats-panel-head">'
"<h2>Other graders</h2>"
'<p class="stats-muted">Click on graders to see their stats</p>'
"</div>"
f"{others_block}"
"</section>"
f"{''.join(open_panels)}"
"</div>"
)
def _empty_reasons(fail_reasons: tuple[str, ...]) -> dict[str, int]:
return {name: 0 for name in (*fail_reasons, OTHER_REASON)}
def _fmt(value: int) -> str:
return f"{value:,}"
def _fmt_pct(value: float) -> str:
if value == 0:
return "0%"
if value < 0.1:
return f"{value:.2f}%"
if value < 10:
return f"{value:.1f}%"
return f"{value:.0f}%"
def _css_id(label: str) -> str:
slug = re.sub(r"[^a-zA-Z0-9_-]+", "-", label).strip("-")
return slug or "x"
def _kpi_row_html(counts: GradeCounts) -> str:
total = counts.total
tiles = (
("pass", "Pass", counts.passed),
("fail", "Fail", counts.failed),
("ungraded", "Ungraded", counts.ungraded),
)
parts = []
for kind, label, value in tiles:
parts.append(
f'<div class="stats-kpi stats-kpi-{kind}">'
f'<div class="stats-kpi-label">{label}</div>'
f'<div class="stats-kpi-value">{_fmt(value)}</div>'
f'<div class="stats-kpi-pct">{_fmt_pct(percent(value, total))} of deck</div>'
"</div>"
)
return f'<div class="stats-kpi-row">{"".join(parts)}</div>'
def _stacked_bar_html(bar_id: str, counts: GradeCounts, styles: list[str]) -> str:
widths = segment_widths((counts.passed, counts.failed, counts.ungraded))
styles.append(
f"#{bar_id} .seg-pass{{width:{widths[0]:.4f}%!important}}"
f"#{bar_id} .seg-fail{{width:{widths[1]:.4f}%!important}}"
f"#{bar_id} .seg-ungraded{{width:{widths[2]:.4f}%!important}}"
)
label = (
f"{_fmt(counts.passed)} pass, {_fmt(counts.failed)} fail, "
f"{_fmt(counts.ungraded)} ungraded"
)
return (
f'<div class="stats-bar" id="{bar_id}" role="img" '
f'aria-label="{html.escape(label)}">'
'<span class="seg seg-pass"></span>'
'<span class="seg seg-fail"></span>'
'<span class="seg seg-ungraded"></span>'
"</div>"
)
def _legend_html() -> str:
chips = (
("pass", "Pass"),
("fail", "Fail"),
("ungraded", "Ungraded"),
)
parts = [
f'<span class="stats-legend-chip stats-legend-{kind}">'
f'<span class="stats-legend-swatch"></span>{label}</span>'
for kind, label in chips
]
return f'<div class="stats-legend">{"".join(parts)}</div>'
def _grader_panel_html(
*,
title: str,
role: str,
counts: GradeCounts,
team_graded: int,
bar_id: str,
reason_prefix: str,
row_names: tuple[str, ...],
styles: list[str],
caption_self: bool,
) -> str:
"""You-style panel: KPIs, bar, legend, Fail by type, share caption."""
bar = _stacked_bar_html(bar_id, counts, styles)
reasons = _fail_reasons_html(reason_prefix, counts, row_names, styles)
share = percent(counts.graded, team_graded)
who = "You have" if caption_self else f"{html.escape(title)} has"
possessive = "your" if caption_self else "their"
return (
'<section class="stats-panel stats-panel-you">'
'<div class="stats-panel-head">'
f"<h2>{html.escape(title)}</h2>"
f'<p class="stats-muted">{html.escape(role)}</p>'
"</div>"
f"{_kpi_row_html(counts)}"
f"{bar}"
f"{_legend_html()}"
f"{reasons}"
'<p class="stats-caption">'
f"{who} graded {_fmt(counts.graded)} of {_fmt(counts.total)} cards "
f"({_fmt_pct(percent(counts.graded, counts.total))} of {possessive} deck). "
f"That is {_fmt_pct(share)} of every card graded by the team."
"</p>"
"</section>"
)
def _fail_reasons_html(
prefix: str,
counts: GradeCounts,
row_names: tuple[str, ...],
styles: list[str],
) -> str:
if counts.failed <= 0:
return '<p class="stats-caption">No fails recorded yet.</p>'
projected = project_fail_reasons(counts, row_names)
peak = max((projected.get(name, 0) for name in row_names), default=0) or 1
rows = []
for name in row_names:
n = projected.get(name, 0)
row_id = f"{prefix}-{_css_id(name)}"
width = round(100.0 * n / peak, 4)
styles.append(f"#{row_id} .stats-reason-fill{{width:{width:.4f}%!important}}")
rows.append(
f'<div class="stats-reason-row" id="{row_id}">'
f'<div class="stats-reason-name">{html.escape(name)}</div>'
'<div class="stats-reason-track"><div class="stats-reason-fill"></div></div>'
f'<div class="stats-reason-n">{_fmt(n)}</div>'
"</div>"
)
return (
'<div class="stats-reasons">'
'<div class="stats-subhead">Fail by type</div>'
f"{''.join(rows)}"
"</div>"
)
def _jurisdiction_block_html(
jurisdictions: list[tuple[str, str, GradeCounts]],
styles: list[str],
) -> str:
if len(jurisdictions) < 2:
return ""
rows = []
for code, label, counts in jurisdictions:
bar_id = f"stats-bar-j-{_css_id(code)}"
bar = _stacked_bar_html(bar_id, counts, styles)
rows.append(
'<div class="stats-juri-row">'
f'<div class="stats-juri-name">{html.escape(label)}</div>'
f"{bar}"
f'<div class="stats-juri-nums">{_fmt(counts.passed)} · '
f"{_fmt(counts.failed)} · {_fmt(counts.ungraded)}</div>"
"</div>"
)
return (
'<div class="stats-juris">'
'<div class="stats-subhead">By jurisdiction</div>'
f"{''.join(rows)}"
"</div>"
)
def _others_grid_html(
others: list[tuple[str, GradeCounts]],
styles: list[str],
*,
open_names: set[str] | frozenset[str] | None = None,
) -> str:
if not others:
return '<p class="stats-caption">No other graders yet.</p>'
open_names = open_names or set()
cards = []
for name, counts in others:
bar_id = f"stats-bar-user-{_css_id(name)}"
bar = _stacked_bar_html(bar_id, counts, styles)
open_cls = " stats-user-card-open" if name in open_names else ""
cards.append(
f'<div class="stats-user-card{open_cls}" role="button" tabindex="0" '
f'data-grader="{html.escape(name, quote=True)}">'
f'<div class="stats-user-name">{html.escape(name)}</div>'
f"{bar}"
f'<div class="stats-user-nums">'
f'<span class="stats-num-pass">{_fmt(counts.passed)} pass</span>'
f'<span class="stats-num-fail">{_fmt(counts.failed)} fail</span>'
f'<span class="stats-num-ungraded">{_fmt(counts.ungraded)} left</span>'
"</div>"
"</div>"
)
return f'<div class="stats-users">{"".join(cards)}</div>'