Spaces:
Sleeping
Sleeping
Download app/training/logo_eval.py from Rthur2003/crowncode-backend: direct link, hf CLI and curl.
- Browser
- Download file 7.57 kB
-
https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/training/logo_eval.py
- Command line
-
hf download hf://spaces/Rthur2003/crowncode-backend/app/training/logo_eval.py
-
curl -L -o logo_eval.py https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/training/logo_eval.py
7.57 kB
| """ | |
| Leave-One-Generator-Out (LOGO) evaluation for AURIS. | |
| The reviewer's #1 priority: standard 5-fold CV shuffles all samples together, | |
| so it never tests whether the model generalizes to an AI generator it has | |
| never seen during training. This script holds out one AI generator at a time, | |
| trains on everything else (all human sources + all remaining AI generators), | |
| and reports how the model performs on the unseen generator. | |
| Human samples are never held out — they remain in the training set for every | |
| fold, since the reviewer's concern is specifically about generator-level | |
| generalization on the AI side. | |
| Usage: | |
| python -m app.training.logo_eval | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import json | |
| import sys | |
| import warnings | |
| from pathlib import Path | |
| from typing import Any | |
| import numpy as np | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[2])) | |
| import lightgbm as lgb | |
| from sklearn.exceptions import ConvergenceWarning | |
| from sklearn.metrics import ( | |
| accuracy_score, | |
| average_precision_score, | |
| balanced_accuracy_score, | |
| f1_score, | |
| matthews_corrcoef, | |
| precision_score, | |
| recall_score, | |
| roc_auc_score, | |
| roc_curve, | |
| ) | |
| from sklearn.preprocessing import StandardScaler | |
| DATASET_DIR = Path("D:/CrownCode/DataSet") | |
| FEATURES_WITH_META = DATASET_DIR / "features_with_meta.csv" | |
| OUTPUT_CSV = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables/logo_results.csv" | |
| _EXCLUDED_COLUMNS = { | |
| "file_path", "label_int", "duration_sec", "sample_rate", | |
| "genre", "generator", "ai_model", "meta_sample_rate", "meta_duration_sec", | |
| "match_source", | |
| } | |
| # LightGBM config matching the best-performing candidate from train_classifier.py | |
| _LGBM_PARAMS = dict( | |
| n_estimators=300, | |
| max_depth=-1, | |
| learning_rate=0.05, | |
| num_leaves=31, | |
| subsample=0.8, | |
| colsample_bytree=0.8, | |
| min_child_samples=20, | |
| reg_alpha=0.1, | |
| reg_lambda=1.0, | |
| class_weight="balanced", | |
| random_state=42, | |
| verbose=-1, | |
| ) | |
| def _load() -> tuple[np.ndarray, np.ndarray, list[str], list[str]]: | |
| with open(FEATURES_WITH_META, "r", encoding="utf-8") as f: | |
| reader = csv.DictReader(f) | |
| fieldnames = reader.fieldnames or [] | |
| feature_cols = [c for c in fieldnames if c not in _EXCLUDED_COLUMNS] | |
| rows = list(reader) | |
| X = np.array([[float(r[c]) for c in feature_cols] for r in rows], dtype=np.float32) | |
| X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0) | |
| y = np.array([int(r["label_int"]) for r in rows], dtype=np.int32) | |
| generators = [r["generator"] for r in rows] | |
| return X, y, generators, feature_cols | |
| def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float: | |
| fpr, tpr, thresholds = roc_curve(y_true, y_prob) | |
| j_scores = tpr - fpr | |
| return float(thresholds[np.argmax(j_scores)]) | |
| def _metrics(y_true: np.ndarray, y_prob: np.ndarray, threshold: float) -> dict[str, float]: | |
| y_pred = (y_prob >= threshold).astype(int) | |
| out: dict[str, float] = { | |
| "n_test": int(len(y_true)), | |
| "n_ai_test": int(np.sum(y_true == 1)), | |
| "n_human_test": int(np.sum(y_true == 0)), | |
| "accuracy": round(float(accuracy_score(y_true, y_pred)), 4), | |
| "precision": round(float(precision_score(y_true, y_pred, zero_division=0)), 4), | |
| "recall": round(float(recall_score(y_true, y_pred, zero_division=0)), 4), | |
| "f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4), | |
| "balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4), | |
| "mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4) if len(set(y_pred)) > 1 else 0.0, | |
| } | |
| # ROC-AUC / PR-AUC require both classes present in the held-out generator's | |
| # test fold; a single-class generator fold cannot report them. | |
| if len(set(y_true.tolist())) > 1: | |
| out["roc_auc"] = round(float(roc_auc_score(y_true, y_prob)), 4) | |
| out["pr_auc"] = round(float(average_precision_score(y_true, y_prob)), 4) | |
| else: | |
| out["roc_auc"] = None | |
| out["pr_auc"] = None | |
| return out | |
| def run() -> dict[str, Any]: | |
| X, y, generators, feature_cols = _load() | |
| generators_arr = np.array(generators) | |
| ai_generators = sorted(set(g for g, label in zip(generators, y) if label == 1)) | |
| print(f"AI generators found: {ai_generators}") | |
| print(f"Total samples: {len(y)} (AI={int(np.sum(y == 1))}, Human={int(np.sum(y == 0))})") | |
| results: dict[str, dict] = {} | |
| for held_out in ai_generators: | |
| test_mask = generators_arr == held_out | |
| train_mask = ~test_mask | |
| X_train, y_train = X[train_mask], y[train_mask] | |
| X_test, y_test = X[test_mask], y[test_mask] | |
| if len(set(y_train.tolist())) < 2: | |
| print(f" Skipping {held_out}: training set has only one class after holdout") | |
| continue | |
| scaler = StandardScaler() | |
| X_train_scaled = scaler.fit_transform(X_train) | |
| X_test_scaled = scaler.transform(X_test) | |
| model = lgb.LGBMClassifier(**_LGBM_PARAMS) | |
| with warnings.catch_warnings(): | |
| warnings.simplefilter("ignore", category=ConvergenceWarning) | |
| model.fit(X_train_scaled, y_train) | |
| y_prob_train = model.predict_proba(X_train_scaled)[:, 1] | |
| threshold = _optimal_threshold(y_train, y_prob_train) | |
| y_prob_test = model.predict_proba(X_test_scaled)[:, 1] | |
| metrics = _metrics(y_test, y_prob_test, threshold) | |
| metrics["held_out_generator"] = held_out | |
| metrics["threshold_from_train"] = round(threshold, 4) | |
| metrics["n_train"] = int(len(y_train)) | |
| results[held_out] = metrics | |
| auc_str = f"{metrics['roc_auc']:.4f}" if metrics["roc_auc"] is not None else "N/A (single class)" | |
| print( | |
| f" Held out: {held_out:25s} n_test={metrics['n_test']:4d} " | |
| f"AUC={auc_str} F1={metrics['f1']:.4f} " | |
| f"BalAcc={metrics['balanced_accuracy']:.4f} MCC={metrics['mcc']:.4f} " | |
| f"Recall={metrics['recall']:.4f}" | |
| ) | |
| # ── Aggregate: mean/std across generators with a valid AUC ── | |
| valid_aucs = [r["roc_auc"] for r in results.values() if r["roc_auc"] is not None] | |
| summary = { | |
| "mean_roc_auc": round(float(np.mean(valid_aucs)), 4) if valid_aucs else None, | |
| "std_roc_auc": round(float(np.std(valid_aucs)), 4) if valid_aucs else None, | |
| "n_generators_evaluated": len(results), | |
| "reference_5fold_cv_auc": 0.9548, # from training_results.json, same-distribution CV | |
| } | |
| print("\n" + "=" * 70) | |
| print("LOGO SUMMARY") | |
| print("=" * 70) | |
| print(f" Mean ROC-AUC across held-out generators: {summary['mean_roc_auc']}") | |
| print(f" Std ROC-AUC across held-out generators: {summary['std_roc_auc']}") | |
| print(f" Reference (in-distribution 5-fold CV): {summary['reference_5fold_cv_auc']}") | |
| # ── Write CSV ── | |
| OUTPUT_CSV.parent.mkdir(parents=True, exist_ok=True) | |
| fieldnames = [ | |
| "held_out_generator", "n_train", "n_test", "n_ai_test", "n_human_test", | |
| "threshold_from_train", "accuracy", "precision", "recall", "f1", | |
| "balanced_accuracy", "mcc", "roc_auc", "pr_auc", | |
| ] | |
| with open(OUTPUT_CSV, "w", newline="", encoding="utf-8") as f: | |
| writer = csv.DictWriter(f, fieldnames=fieldnames) | |
| writer.writeheader() | |
| for gen in ai_generators: | |
| if gen in results: | |
| writer.writerow({k: results[gen].get(k) for k in fieldnames}) | |
| print(f"\nOutput: {OUTPUT_CSV}") | |
| return {"per_generator": results, "summary": summary} | |
| if __name__ == "__main__": | |
| run() | |