Instructions to use Zipeng365/WISP with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Scikit-learn
How to use Zipeng365/WISP with Scikit-learn:
from huggingface_hub import hf_hub_download import joblib model = joblib.load( hf_hub_download("Zipeng365/WISP", "sklearn_model.joblib") ) # only load pickle files from sources you trust # read more about it here https://skops.readthedocs.io/en/stable/persistence.html - Notebooks
- Google Colab
- Kaggle
Download validation/model_validation.json from Zipeng365/WISP: direct link, hf CLI and curl.
- Browser
- Download file 460 kB
-
https://huggingface.co/Zipeng365/WISP/resolve/main/validation/model_validation.json
- Command line
-
hf download hf://Zipeng365/WISP/validation/model_validation.json
-
curl -L -o model_validation.json https://huggingface.co/Zipeng365/WISP/resolve/main/validation/model_validation.json
460 kB
| { | |
| "generated_at_utc": "2026-10-02T12:57:11.649548+00:00", | |
| "slurm_job_id": "55937950", | |
| "environment": { | |
| "wisp-har": "0.1.0", | |
| "numpy": "2.3.5", | |
| "scipy": "1.17.1", | |
| "scikit-learn": "1.7.2", | |
| "joblib": "1.5.3", | |
| "threadpoolctl": "3.6.0", | |
| "python": "3.12.13" | |
| }, | |
| "code_provenance": { | |
| "release_wheel": "wisp_har-0.1.0-py3-none-any.whl", | |
| "release_wheel_sha256": "89c1cccabaf095b66fe8d4b82a007b868316302ca613d6bfe1a9a8b2995876b0", | |
| "installed_checkpoint_api_sha256": "261bf3c45c2ef7f5a21ff188533aa19a46a6774e261fb6ee1d0f25cfb794b47f", | |
| "installed_package_source_sha256": { | |
| "wisp/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", | |
| "wisp/ablations.py": "d92d2cc5a369ca5d6a3baac156355b0b21b7816778ba9e40eb0c898bafaf8354", | |
| "wisp/cis_algorithms.py": "c7b9b262d9bceed5f6a31821a26aeb044350b6af759d100876b3967a36a93be5", | |
| "wisp/core/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", | |
| "wisp/core/data.py": "a39ff2a5a8cf67c9e59e944bb065e8274ccd7d18b319efeba46044e6cc372306", | |
| "wisp/core/metrics.py": "113b8461fa7edd883cebd961b77e15dde9ffc8c3dbf538e313c246e350838c7c", | |
| "wisp/cpu/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", | |
| "wisp/cpu/algorithms.py": "b1c855877ef70ccc4abb0726c626933053120272fcfeb8d5129603c6f59500e4", | |
| "wisp/cpu/features.py": "aab2b76f8dbf7d7faf9b4b1785b2596a2596ba4a805a18e0c856b0ea581ef668", | |
| "wisp/cpu/hmm.py": "2c42f3c2076aed5143f4d7b850f88b0b6486ceab72ca291184a9c5859e0ddae0", | |
| "wisp/cpu/probability.py": "1232011f6eac18d45fd13580f96e79b9b8b088f8661a863130fe7084b1e67cc9", | |
| "wisp/registry.py": "a7a96dfa10ca77d95032e4e62e1a526fc53503ee4618fa6194a73a97ae523828", | |
| "wisp/search/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", | |
| "wisp/search/cascade.py": "2f2d19573a69afe12015657f13a89caf0d28466ffd703a31f2afd951c7d1e245", | |
| "wisp/search/controller.py": "895240d4ba5b79088766a972a76112d15fa40d4e6fa4f5440bf47b6f2a0ec451", | |
| "wisp/search/dataset_fingerprint.py": "8e976b35e37534c4fd795d529d63ac01cb0e95b87df4f090791d0d6611fb08ca", | |
| "wisp/search/grammar.py": "a038e582182aa1bed8b03a173f02b6977730dbcde07f44a3cd721f87e6d08971", | |
| "wisp/search/motifs.py": "5d0375afbb4cec325ccb0d476f27c797d0e335e6d4e292a16b9330f3565a7a44", | |
| "wisp/search/operator_program.py": "654975111d32566a712dd7ae8e386c577f9fc0196dad8582939ab018f32ff8ec", | |
| "wisp/search/posterior.py": "0fb27135a136f9fc77c0bdc6facb959a85587e629e4d99c40f95ff4262507e1f", | |
| "wisp/search/profiles.py": "0acb26c966b72721b42f9e90183186ddb1db713a42d451d886103e0ebf7a16a6", | |
| "wisp/utils/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", | |
| "wisp/utils/jsonl.py": "ab9819981cf6f56436edada19b01cfb11d78a3af40d05504c7b9740384441daa", | |
| "wisp_release/__init__.py": "c4c3a0ae71ab4eb449ac357d5cddaa40b897a6bdc683c5713e019f3fcc5027a2", | |
| "wisp_release/checkpoints.py": "261bf3c45c2ef7f5a21ff188533aa19a46a6774e261fb6ee1d0f25cfb794b47f", | |
| "wisp_release/cli.py": "4d4f7106bd5eca4c8496c55b0a03c967df589cf1ec2e7f365db0ec6798c980a5", | |
| "wisp_release/data.py": "5dab92de198c7bca28fb620f830168be3bf936bb3c8a8c9ca4dd5762c6cf0a89", | |
| "wisp_release/methods.py": "2f1eac83ccb9fe7c8d0e2a7520f746df7f522470af8dab65322a261acc17abcd", | |
| "wisp_release/scoring.py": "aa1e89dd83b7295e0dbe087b441c13bae3247a04129f4710006da9bc49386dc5", | |
| "wisp_release/selection.py": "7077a0bfef5c9ec3371c64b0bfdef1d285ee9f9902142e2f5676d0f0462d72b6" | |
| }, | |
| "installed_sources_exactly_match_release_wheel": true, | |
| "notice": "Wheel digest identifies the built release artifact; source hashes identify the actual installed implementation used by these model checks." | |
| }, | |
| "expected_available_checkpoints": 360, | |
| "baselines_installed_or_tested": false, | |
| "training_runs_performed": 0, | |
| "records": [ | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 56197147, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "passed", | |
| "n_test_windows": 861, | |
| "exact_frozen_predictions_equal": true, | |
| "original_input_exact_frozen_predictions_equal": true, | |
| "original_input_differing_prediction_count": 0, | |
| "differing_prediction_count": 0, | |
| "original_and_fresh_X_exactly_equal": true, | |
| "original_and_fresh_X_max_absolute_difference": 0.0, | |
| "time_coordinate_check": { | |
| "raw_time_values_exactly_equal": false, | |
| "per_subject_stable_chronological_permutation_identical": { | |
| "f3": true, | |
| "m1": true, | |
| "m3": true, | |
| "m7": true | |
| }, | |
| "original_first5": [ | |
| 0.0, | |
| 2.5, | |
| 5.0, | |
| 7.5, | |
| 0.0 | |
| ], | |
| "fresh_first5": [ | |
| 0.0, | |
| 1.0, | |
| 2.0, | |
| 3.0, | |
| 0.0 | |
| ], | |
| "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.46857990441875785, | |
| "accuracy": 0.70267131242741, | |
| "balanced_accuracy": 0.45147912566266135, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.46857990441875785, | |
| "accuracy": 0.70267131242741, | |
| "balanced_accuracy": 0.45147912566266135, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c8089b71a53b67ff1eee86b112362230e51a34a25664cf0652786ab7db00188f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 5.281506538391113 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 56614267, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "passed", | |
| "n_test_windows": 861, | |
| "exact_frozen_predictions_equal": true, | |
| "original_input_exact_frozen_predictions_equal": true, | |
| "original_input_differing_prediction_count": 0, | |
| "differing_prediction_count": 0, | |
| "original_and_fresh_X_exactly_equal": true, | |
| "original_and_fresh_X_max_absolute_difference": 0.0, | |
| "time_coordinate_check": { | |
| "raw_time_values_exactly_equal": false, | |
| "per_subject_stable_chronological_permutation_identical": { | |
| "f3": true, | |
| "m1": true, | |
| "m3": true, | |
| "m7": true | |
| }, | |
| "original_first5": [ | |
| 0.0, | |
| 2.5, | |
| 5.0, | |
| 7.5, | |
| 0.0 | |
| ], | |
| "fresh_first5": [ | |
| 0.0, | |
| 1.0, | |
| 2.0, | |
| 3.0, | |
| 0.0 | |
| ], | |
| "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5008386012173559, | |
| "accuracy": 0.7096399535423926, | |
| "balanced_accuracy": 0.4878142449093867, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5008386012173559, | |
| "accuracy": 0.7096399535423926, | |
| "balanced_accuracy": 0.4878142449093867, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "874f0b547ea16f5520641d01894fc4836451d749795dec2e805544925b56944b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 1.4547859011217952 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 213724, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "passed", | |
| "n_test_windows": 861, | |
| "exact_frozen_predictions_equal": true, | |
| "original_input_exact_frozen_predictions_equal": true, | |
| "original_input_differing_prediction_count": 0, | |
| "differing_prediction_count": 0, | |
| "original_and_fresh_X_exactly_equal": true, | |
| "original_and_fresh_X_max_absolute_difference": 0.0, | |
| "time_coordinate_check": { | |
| "raw_time_values_exactly_equal": false, | |
| "per_subject_stable_chronological_permutation_identical": { | |
| "f3": true, | |
| "m1": true, | |
| "m3": true, | |
| "m7": true | |
| }, | |
| "original_first5": [ | |
| 0.0, | |
| 2.5, | |
| 5.0, | |
| 7.5, | |
| 0.0 | |
| ], | |
| "fresh_first5": [ | |
| 0.0, | |
| 1.0, | |
| 2.0, | |
| 3.0, | |
| 0.0 | |
| ], | |
| "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5804510063302952, | |
| "accuracy": 0.7061556329849012, | |
| "balanced_accuracy": 0.5332788991510314, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5804510063302952, | |
| "accuracy": 0.7061556329849012, | |
| "balanced_accuracy": 0.5332788991510314, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0729c0358473d08aa0f07abd071f248aca98c576592b692872feaddc172090be", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 3.528837164863944 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 144715563, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "passed", | |
| "n_test_windows": 861, | |
| "exact_frozen_predictions_equal": true, | |
| "original_input_exact_frozen_predictions_equal": true, | |
| "original_input_differing_prediction_count": 0, | |
| "differing_prediction_count": 0, | |
| "original_and_fresh_X_exactly_equal": true, | |
| "original_and_fresh_X_max_absolute_difference": 0.0, | |
| "time_coordinate_check": { | |
| "raw_time_values_exactly_equal": false, | |
| "per_subject_stable_chronological_permutation_identical": { | |
| "f3": true, | |
| "m1": true, | |
| "m3": true, | |
| "m7": true | |
| }, | |
| "original_first5": [ | |
| 0.0, | |
| 2.5, | |
| 5.0, | |
| 7.5, | |
| 0.0 | |
| ], | |
| "fresh_first5": [ | |
| 0.0, | |
| 1.0, | |
| 2.0, | |
| 3.0, | |
| 0.0 | |
| ], | |
| "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4271967810219083, | |
| "accuracy": 0.6806039488966318, | |
| "balanced_accuracy": 0.45020425062766495, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4271967810219083, | |
| "accuracy": 0.6806039488966318, | |
| "balanced_accuracy": 0.45020425062766495, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "943a9f9c75edb5b5425533d5fbbc376e39e838942d32562c7b46f0d83a5c653d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 5.605986746959388 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 435753, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.45512076153931874, | |
| "accuracy": 0.5837837837837838, | |
| "balanced_accuracy": 0.5301783700824152, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e1bde8c4ad01f3152f254f754e19c9ed214d790ca66450317b462ca941e52afd", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 0.09336361195892096 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 239385, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4543957425721672, | |
| "accuracy": 0.5704504504504504, | |
| "balanced_accuracy": 0.5233597226947379, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4f4a0d1fc0ab03fd159a8dd028bb870360eb0e7c89f779b36e050e2bb56a9c8d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 0.0711149936541915 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 195613, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4444367338790675, | |
| "accuracy": 0.5585585585585585, | |
| "balanced_accuracy": 0.5129764381142231, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5ef9f923e8d30c0a7148c0fea06920e1addff9d89cbe6ce1c161db4ebf5b7998", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 7, | |
| 7, | |
| 7, | |
| 7 | |
| ], | |
| "verification_seconds": 0.09933141898363829 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 108071, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6315284716337148, | |
| "accuracy": 0.6824742268041237, | |
| "balanced_accuracy": 0.5954227178234918, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a3646800e53bd55b507cd0545383377d8b1fd0d4e5e4e21b94e79de64c717bdd", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.0682113254442811 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 212796, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7140298291222251, | |
| "accuracy": 0.7793814432989691, | |
| "balanced_accuracy": 0.6659095889792745, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5ca91cbb72343bbe53d408cd5267fe1bf4e70a06e02b0afd65f2b6b8d6ef8c1e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07314625475555658 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 108151, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6091680266775356, | |
| "accuracy": 0.5711340206185567, | |
| "balanced_accuracy": 0.5892073482822214, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d12df30d0ab256561d9b583bfe910dfedb13138949b410bf219bc2ec7d622533", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07153434678912163 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 215198, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.39874385233957527, | |
| "accuracy": 0.555992141453831, | |
| "balanced_accuracy": 0.45164475266682325, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "650947efac20189d66cd0c69348da623efd950daf0ad759b404b6e72b64505df", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.0834163036197424 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 213799, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.38419952735589424, | |
| "accuracy": 0.5520628683693517, | |
| "balanced_accuracy": 0.4058504505983019, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bfb5d044c9caa32d3b496619b49e2e60251318e4100e771fccb2b364885acc5f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07257030159235 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 109750, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.37838383738902187, | |
| "accuracy": 0.48919449901768175, | |
| "balanced_accuracy": 0.42086382580606824, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "10415ecf33b468d8584ef8e41d076d63fc4cafb231db82cd1d682737a43ca307", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.06706883013248444 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 215136, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7160946161678324, | |
| "accuracy": 0.803921568627451, | |
| "balanced_accuracy": 0.7463469409476607, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c47064840f9187a5cfa3fc181f829aab1ba77c8754f79db9fcff50e9c62a98c2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07794979959726334 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 267079, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6873594627210284, | |
| "accuracy": 0.8215686274509804, | |
| "balanced_accuracy": 0.7955491343583891, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "19a43e6143a7572be809bec98c7e517eb332cfd030fe87b86f78490a0c8463fa", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 0.06673328019678593 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 265353, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.64200883315541, | |
| "accuracy": 0.788235294117647, | |
| "balanced_accuracy": 0.7487627012767595, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9d3b9fb55b2ff1b4f05643df09594a84438e24bbdfd8beb6cd0b95b045607e35", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 0.0789025230333209 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed42/model.pkl", | |
| "bytes": 128855, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7773584180364441, | |
| "accuracy": 0.8507273259787171, | |
| "balanced_accuracy": 0.7698689175281773, | |
| "worst_class_f1": 0.5946502057613169 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "efa2e8604213ef4989368eedc7e077746724d0e45eba640c93ed8a6df76b2bad", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.0911676436662674 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed43/model.pkl", | |
| "bytes": 64391, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7707432326452675, | |
| "accuracy": 0.8394025187933223, | |
| "balanced_accuracy": 0.7700530056485361, | |
| "worst_class_f1": 0.6070409134157945 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "feac6c70266738be626a463ec0bd7deb778bdd63e4acaaa0027692959aec4598", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.1280159205198288 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed44/model.pkl", | |
| "bytes": 64845, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7583360117244344, | |
| "accuracy": 0.8415503270526213, | |
| "balanced_accuracy": 0.7514210173129554, | |
| "worst_class_f1": 0.5420944558521561 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fc5accae13589359550c006de9301f3c37970d3332cdb9edfa166cb28b8491fa", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.0684780403971672 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed42/model.pkl", | |
| "bytes": 347933090, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6832491969161993, | |
| "accuracy": 0.7742128337983261, | |
| "balanced_accuracy": 0.6653189192827791, | |
| "worst_class_f1": 0.44139650872817954 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3b55ea01da5f0a0d1832836c99428f6ae17ebfc6c18101aad6633fbc89d0a8c6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.8010620893910527 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed43/model.pkl", | |
| "bytes": 130511, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.748965216157762, | |
| "accuracy": 0.8414707054603427, | |
| "balanced_accuracy": 0.7496277163543152, | |
| "worst_class_f1": 0.48459958932238195 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a1f9521a70baff07ae0e2705279c268dae32f3bf01088791a01b740a2d0eee7b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09537813626229763 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed44/model.pkl", | |
| "bytes": 592029080, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7095731117734799, | |
| "accuracy": 0.7794938222399362, | |
| "balanced_accuracy": 0.7155070471172339, | |
| "worst_class_f1": 0.5171102661596958 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c77d246c5eec2ee3c943bbc43e366ebedfb234cfe3c19dbd2341daafec7efffa", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 1.529852494597435 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed42/model.pkl", | |
| "bytes": 130236, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7293373556148361, | |
| "accuracy": 0.8356660001904218, | |
| "balanced_accuracy": 0.7377749513518659, | |
| "worst_class_f1": 0.47665847665847666 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "48284dba20870ad2877b9189df3bec278653f04c8a3c99812f0840188a0d657e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.10741668753325939 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed43/model.pkl", | |
| "bytes": 64221, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7109350771181784, | |
| "accuracy": 0.8188136722841093, | |
| "balanced_accuracy": 0.7237143821038571, | |
| "worst_class_f1": 0.45794392523364486 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3c3152512aed2c8bbb1c16efd7942035151f9e89e139a58c8abfc2f9158470b2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "verification_seconds": 0.07644576393067837 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed44/model.pkl", | |
| "bytes": 128604, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7151153882450071, | |
| "accuracy": 0.8309054555841188, | |
| "balanced_accuracy": 0.7169997827310548, | |
| "worst_class_f1": 0.4375 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "66e2b8cd1c80f44d36e2ff62422ad6509515f99359a52b0fc62b543192605664", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "verification_seconds": 0.09162094537168741 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed42/model.pkl", | |
| "bytes": 127524, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7519431599340909, | |
| "accuracy": 0.84189453125, | |
| "balanced_accuracy": 0.7382519385358287, | |
| "worst_class_f1": 0.5137614678899083 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0e9ce884e630f754d915db2bf33fe41392c659e2fd04a569225695902c90db0f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "verification_seconds": 0.1081920899450779 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed43/model.pkl", | |
| "bytes": 164122, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7641903008260709, | |
| "accuracy": 0.84794921875, | |
| "balanced_accuracy": 0.7437808965170654, | |
| "worst_class_f1": 0.570273003033367 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "966eab1af20383e4ba9a7aa3d014fe22d7dde2c1b34eb75681cb90563f999662", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09193179570138454 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed44/model.pkl", | |
| "bytes": 175628, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7666007836147859, | |
| "accuracy": 0.85, | |
| "balanced_accuracy": 0.7458548085139014, | |
| "worst_class_f1": 0.5743174924165824 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "34fddd3eed7a6389bf16550b29df02b8646dd08bd5a8961363070b236323a6af", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09885944984853268 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed42/model.pkl", | |
| "bytes": 164896, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7478668985168254, | |
| "accuracy": 0.8256184407796102, | |
| "balanced_accuracy": 0.7282771903274528, | |
| "worst_class_f1": 0.5169927909371782 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7ded22cb68e6229b81b7d16ca5c0edaa46e66099a10fbc5c90d8af138a92c99c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09246528707444668 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed43/model.pkl", | |
| "bytes": 161763, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7447052020556495, | |
| "accuracy": 0.8229010494752623, | |
| "balanced_accuracy": 0.7243578069099286, | |
| "worst_class_f1": 0.5125 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a888536d0d5f7dc270067e8ea39665c7cd4e1e29a331fa57600076b157985bab", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09998060762882233 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed44/model.pkl", | |
| "bytes": 133859, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7477843382157683, | |
| "accuracy": 0.8235569715142429, | |
| "balanced_accuracy": 0.7262215740897333, | |
| "worst_class_f1": 0.5230125523012552 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9e3ad341c1e95d21e5bcd9cd71ae1eda615c27c49f8d67c2bd75c563da10a1ac", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.100236008875072 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 212600, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7664598552903535, | |
| "accuracy": 0.9142131979695431, | |
| "balanced_accuracy": 0.759350201817826, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "91782879414932f4fd1d83de2b64f157f80a4203e71dd9cebc63de3836655b2c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.111138129606843 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 109344, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7779119924157182, | |
| "accuracy": 0.9030456852791878, | |
| "balanced_accuracy": 0.771893748615854, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f987480f734d991975f9214d65fb667624b757b644f4187dd9eb17b44236ff80", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10117201320827007 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 215498, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7588925510712723, | |
| "accuracy": 0.9055837563451776, | |
| "balanced_accuracy": 0.7528336709974588, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "dc05c3132f80de17ce71294cf01693bac44d720b3374913fabdf232c8c762ba9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09738574642688036 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 539572, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7654332232404667, | |
| "accuracy": 0.8809418036428254, | |
| "balanced_accuracy": 0.7650026167595115, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cf2712de8548876ab47ead3c76641d2aff86285a081d09d3e3e1a1fee26bcd7b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12981910444796085 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 318789, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7475567873753366, | |
| "accuracy": 0.8431808085295425, | |
| "balanced_accuracy": 0.7409736397320528, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f207d6cb99c6be9e0229de253aeb5c554106e6808b8630e28c87738f9f7582b3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12939182203263044 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 216225, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7643324610484091, | |
| "accuracy": 0.8698356286095069, | |
| "balanced_accuracy": 0.7816075184988293, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2bbd335eeb771d9febd05165213d6c533578169d0a8d57b0127a817714d12005", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10225305519998074 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 542893, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6063961064701442, | |
| "accuracy": 0.8345864661654135, | |
| "balanced_accuracy": 0.6476178172323037, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c250dac4028d1978b59d5ccb9dfc9881c040b1623291d93d23e68351b1e10a6a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12861014250665903 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 540198, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6786222922383697, | |
| "accuracy": 0.8883747831116252, | |
| "balanced_accuracy": 0.7045844794521647, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c780ed09a5710ffeda2ecaf05f41dbdc62d24781faba84ac4d25c6c3f06098a2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12846562545746565 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 542016, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6059659135596054, | |
| "accuracy": 0.8340080971659919, | |
| "balanced_accuracy": 0.6479127025827084, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ad2ecac326b372c3dde4f10e2b6079b0eed91a7cd39fe073aec238131c0dc91f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13145661167800426 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 212454, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6328932422472617, | |
| "accuracy": 0.8124012638230648, | |
| "balanced_accuracy": 0.658274984974044, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e00af7926b2d0764a2ee380dab3e15d8608f58be432b0441448491310da1bff0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10056019108742476 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 215087, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5656274211024497, | |
| "accuracy": 0.7918641390205371, | |
| "balanced_accuracy": 0.5810575952113626, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e9d96df627dad5f8005a5dde54b70ae26ec7fcc47b7571fe2b429156d40a72bc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10281260497868061 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 213083, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5766493081312468, | |
| "accuracy": 0.7969984202211691, | |
| "balanced_accuracy": 0.5883944723651199, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cabbe2a146876bdf4d60ef3b7511d624112af70f087adaf6362efc6d24842f1f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09937153570353985 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 319198, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6476508962514093, | |
| "accuracy": 0.8443037974683544, | |
| "balanced_accuracy": 0.6502484226519598, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7f022f0cb93b1f3d1b3333bc9280f649705c197ede1e8c632a915c4648190fcc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12414687685668468 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 541362, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.618716820825527, | |
| "accuracy": 0.8510548523206751, | |
| "balanced_accuracy": 0.6259556576317066, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e6cb7ff7b7eabaeebceb94181be0b077fc467ab9693a583392ee8e5eb829a29b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.14895725715905428 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 210604, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6790563959748898, | |
| "accuracy": 0.8611814345991561, | |
| "balanced_accuracy": 0.6645928651150633, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "913f6893df2fc3e36e06434f73869862a53c966f78f3011f0906c81900623bf7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09429517947137356 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 701279, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7672489965511144, | |
| "accuracy": 0.780731384931708, | |
| "balanced_accuracy": 0.790075815022735, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b8dcc5c64a5343a93e548def0659ae9f69aa59f26b9efae8b18ef60c04999072", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13250253908336163 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 592198, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7835365865957504, | |
| "accuracy": 0.7968864737846967, | |
| "balanced_accuracy": 0.8124908062429019, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "86844eb8a9e175f68acfb8f2777e19dc90862f4423fa1cb06adf28f84b65fa1b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13126813992857933 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 589580, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7870352562954467, | |
| "accuracy": 0.804817153767073, | |
| "balanced_accuracy": 0.8210987640639349, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b37545d5ae857e640fab4e7933811f530e9ca1d9d6d5ddc7f26ee1e9e6f5cfd7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12988745048642159 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 175868950, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.745442544578232, | |
| "accuracy": 0.7687480869299052, | |
| "balanced_accuracy": 0.7353281827704078, | |
| "worst_class_f1": 0.4176533907427341 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f7a4e86ede990675448cc131f8c22ce073b831c3e82ae1722c8e9e24578d3842", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.49978523049503565 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 594940, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8188924951875662, | |
| "accuracy": 0.8221610039791858, | |
| "balanced_accuracy": 0.8239290176963354, | |
| "worst_class_f1": 0.39609483960948394 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "aa10b8a5133b9e53941adf6e1b24b5b5c50a0f2ed7f993b95ab3dad2f1499321", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13092663139104843 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 593446, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.204170427930421e-18 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.78724631788864, | |
| "accuracy": 0.7976737067646159, | |
| "balanced_accuracy": 0.8042867736850741, | |
| "worst_class_f1": 0.014492753623188406 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6d275ae2416840f97d5e25539b64c1cfd4c7550e8ecf4182f57a1a2401f5928c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12541959341615438 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 625820, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8096769252697558, | |
| "accuracy": 0.8215525402335121, | |
| "balanced_accuracy": 0.8207685956698325, | |
| "worst_class_f1": 0.45045045045045046 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "61832a626e710b6e585baa509e8143bc38aa82c5998d30a35721bdda870a6894", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12353577744215727 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 592488, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8028690592578791, | |
| "accuracy": 0.8247081098138214, | |
| "balanced_accuracy": 0.8126440443873308, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "23861efe32842b0b497613aa381f5e4552922c858b2b4394ed7de391860aabb9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12496596947312355 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 593567, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8063658973106798, | |
| "accuracy": 0.8302303565793626, | |
| "balanced_accuracy": 0.8136036002493117, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "19983a22327bbb26e1f068f170d5f667f86a83a1fe5e24ec5bb2571f55ae1319", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12128795497119427 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 677345, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.783907803140329, | |
| "accuracy": 0.8038985783379745, | |
| "balanced_accuracy": 0.7821143484738341, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "be9943a93e0b583388c6ce972347ee31a40a057a579f1cfd9b0b4aa8ecb99d00", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1648122714832425 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 556104, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 6.245004513516506e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7776997415582797, | |
| "accuracy": 0.7975963652352338, | |
| "balanced_accuracy": 0.7793900579312314, | |
| "worst_class_f1": 0.03298350824587706 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3fc594cc17fd2cc89b882953aaa152c3f481210354d4e629425eb45734e14ee5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12396656069904566 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 115192, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.697018026421252, | |
| "accuracy": 0.7243148175289462, | |
| "balanced_accuracy": 0.6970510428018644, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3ab1b44e1e59ea0d0756fb61416e2b05ce9b5704a8b1c373c28cd561cefa0f9c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08868241589516401 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 593151, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8641234565792023, | |
| "accuracy": 0.8536787908985218, | |
| "balanced_accuracy": 0.8719389189768827, | |
| "worst_class_f1": 0.6344238975817923 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1e177f034f0d5272a98ddc35fe6e14cfd7c30f4617902a21d2d51fe77d08ee2f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13063144125044346 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 628812, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8815252691975555, | |
| "accuracy": 0.8659691081215745, | |
| "balanced_accuracy": 0.8895848428301525, | |
| "worst_class_f1": 0.6130884041331802 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5f59b1c56c4e5143ce16e0d4e3e0298a77be839efe4cd4f788a05c9b2f633888", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12962188199162483 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 591703, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8515240355978488, | |
| "accuracy": 0.8400597907324364, | |
| "balanced_accuracy": 0.8586857351484299, | |
| "worst_class_f1": 0.5885797950219619 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "82ccc3b08284c3d6639155e8047459ee651a6610ebc6ca5d814873a92b058b53", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1400800608098507 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 107027779, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9535440866847888, | |
| "accuracy": 0.9505915100904663, | |
| "balanced_accuracy": 0.9547187190113678, | |
| "worst_class_f1": 0.917960088691796 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0ba2051deeb8953941a5ba1097cf67c02bae3e0400ac3ba552d5ba6d92e523fd", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.30445832666009665 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 29561530, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9664924083067158, | |
| "accuracy": 0.9659011830201809, | |
| "balanced_accuracy": 0.9675183372111787, | |
| "worst_class_f1": 0.9343065693430657 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cd92a861ddcc1fb2e319dc39b7255dabf7596ea0c792856d5377e044f89bacaa", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.15698592364788055 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 66429575, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9631621411148588, | |
| "accuracy": 0.9610299234516354, | |
| "balanced_accuracy": 0.9639777161822713, | |
| "worst_class_f1": 0.933920704845815 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2d11e5bb8191faa9ea6f33fc59a552fca8f9da0161b1ab257b15cc7e2caa5164", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.23916628491133451 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 108937791, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9488100219032743, | |
| "accuracy": 0.9364968597348221, | |
| "balanced_accuracy": 0.9585341464521031, | |
| "worst_class_f1": 0.8946236559139785 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e92e2c69e08c43c5b43a1ee7424453a0c6efcece3d351138a1ecd40cde17b434", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 9, | |
| 6, | |
| 5 | |
| ], | |
| "verification_seconds": 0.2992333984002471 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 69160983, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9688875446818674, | |
| "accuracy": 0.9630146545708305, | |
| "balanced_accuracy": 0.9736233967271118, | |
| "worst_class_f1": 0.9453781512605042 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9a0ec2e26f0c92f0a9bee79d4d16c82361fb07bad0e8ba3f6cf78873284fbbff", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.23830208834260702 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 69220315, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9739522200817022, | |
| "accuracy": 0.9685973482205164, | |
| "balanced_accuracy": 0.9786360981639619, | |
| "worst_class_f1": 0.9436325678496869 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "49db770dfcc481f405a102acfe78a27064e9424221313042063188f1a8a2ad48", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.2255033189430833 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 70056227, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9667081563909887, | |
| "accuracy": 0.9611436950146628, | |
| "balanced_accuracy": 0.9678508084578631, | |
| "worst_class_f1": 0.929384965831435 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6111562ba3832d7c834a167ef77c6f8639fe1d979d35017ebbaf8bdede9c839f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.23397216200828552 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 30878124, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.960645742786498, | |
| "accuracy": 0.9545454545454546, | |
| "balanced_accuracy": 0.9613495525574918, | |
| "worst_class_f1": 0.9248291571753986 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7741bfa8ce2e05112fac9f950e9075a2973e9a81f0ada3d9e676ebc487ae4d7d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.18209315743297338 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 30208604, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9646038665262866, | |
| "accuracy": 0.9582111436950147, | |
| "balanced_accuracy": 0.9668511854523635, | |
| "worst_class_f1": 0.9269406392694064 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1f3ae2ef84aa939b59d3b1ef0d07f5bc5e4b87b5fe487f86bd013733f9daf6ad", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.17018190491944551 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 29162012, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9666138221553023, | |
| "accuracy": 0.9584487534626038, | |
| "balanced_accuracy": 0.972777362601331, | |
| "worst_class_f1": 0.9322709163346613 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "22e3d999f78b0ea25cf621e6b0e2233717892e85c2fe821b120231da482b9dd6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.16035774070769548 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 67966647, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9643939005024231, | |
| "accuracy": 0.953601108033241, | |
| "balanced_accuracy": 0.9703785911125241, | |
| "worst_class_f1": 0.9221556886227545 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "edf212eef5bab87a08152d61291dc7ef0e01114f220d4e372ba4d42b06af6242", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.22190038301050663 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 68830105, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.959464542814932, | |
| "accuracy": 0.9473684210526315, | |
| "balanced_accuracy": 0.9656081680613511, | |
| "worst_class_f1": 0.9123434704830053 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6777cd4ce496465a3deb12a74d58cabfdc498145ce3485b9b09b3688ee09aea5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.23183409683406353 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 29370288, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": -1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9665822259579404, | |
| "accuracy": 0.9585714285714285, | |
| "balanced_accuracy": 0.9665539314970886, | |
| "worst_class_f1": 0.9095238095238095 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3772a56d9802e1201b220323acfab33e90e13c7c1e5fdd315fef49329842bac4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 9, | |
| 6, | |
| 5 | |
| ], | |
| "verification_seconds": 0.15483754873275757 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 29652538, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9612307601150556, | |
| "accuracy": 0.9535714285714286, | |
| "balanced_accuracy": 0.9622524467374151, | |
| "worst_class_f1": 0.9134615384615384 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "79b0bbd8d789f273915019109e1c79b4012e400ce4336b80a8b2de213b376acf", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.16102203354239464 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 109540579, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9508254122131812, | |
| "accuracy": 0.9378571428571428, | |
| "balanced_accuracy": 0.9511741971307087, | |
| "worst_class_f1": 0.88 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "448bed549863e5f10d2843e937a4c4c3fd282cd15171c5f20e4970facf451d81", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.3187501523643732 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 812416, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.387570196628973, | |
| "accuracy": 0.4189173789173789, | |
| "balanced_accuracy": 0.42617659454061835, | |
| "worst_class_f1": 0.13793103448275862 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "036a2ff2ca69e09e785fcf0b0ad2b7cd5b2c14996b915a4840c17eb4e98f44d7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13104734662920237 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 586110, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3844396970116084, | |
| "accuracy": 0.4151566951566952, | |
| "balanced_accuracy": 0.4250397919631618, | |
| "worst_class_f1": 0.16263736263736264 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "54c708e15be4215fa32d885d85257bd09be95e83ec0beb097a7eb4ff74944b40", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12016687728464603 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 702211, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3870428869869673, | |
| "accuracy": 0.4168660968660969, | |
| "balanced_accuracy": 0.42767719047090413, | |
| "worst_class_f1": 0.1568627450980392 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b6a53d74eabde5c5dfe0f9c1be47bfe0d641ecddde0739efda79fc92933aae38", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.14498190488666296 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 366487, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3281143240077257, | |
| "accuracy": 0.3518880208333333, | |
| "balanced_accuracy": 0.36986021552619264, | |
| "worst_class_f1": 0.1663286004056795 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0cbcab1b6bd4e29ba9d1d1532339c60ceda1eb623ff8b635de0d7a2058f9fd57", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09013545699417591 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 479767, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3368172298791864, | |
| "accuracy": 0.3588324652777778, | |
| "balanced_accuracy": 0.3818676836393975, | |
| "worst_class_f1": 0.1725417439703154 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5f30f5071527b5e6a91313eece5fd547570f5b23f7859d6feb5a9adb21bed997", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 8, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10633725114166737 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 589671, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.34027982278741586, | |
| "accuracy": 0.3645833333333333, | |
| "balanced_accuracy": 0.3811477225820268, | |
| "worst_class_f1": 0.18385650224215247 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3316aff7e86d626e671de4aeb0b58720856bfee06b6b5bebe482193ca7e62cd4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 11, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11465025693178177 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 856939, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4157316719433681, | |
| "accuracy": 0.4486997635933806, | |
| "balanced_accuracy": 0.4557413938998671, | |
| "worst_class_f1": 0.1743119266055046 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c33c1d7f183f138b4d34ac5927b9156a7fd20c731affa8c8165db2239189b3ed", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13862699456512928 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 591055, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.41011206610734857, | |
| "accuracy": 0.4459810874704492, | |
| "balanced_accuracy": 0.4506340420650674, | |
| "worst_class_f1": 0.1487603305785124 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "06f7274c1dafdffdd23e29988cef74286bacfd6ec13b28f4ed2f677f8361a04b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13259280938655138 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 678051, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3979308233637672, | |
| "accuracy": 0.4329787234042553, | |
| "balanced_accuracy": 0.4453328739886988, | |
| "worst_class_f1": 0.11907164480322906 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bde21d7f651fc6c81a941892af381009a94022df1980f1e9a01ed5b6018dae5a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1591173354536295 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 856835, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3957192396436544, | |
| "accuracy": 0.4207221350078493, | |
| "balanced_accuracy": 0.42766245503552947, | |
| "worst_class_f1": 0.2186046511627907 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b3faa0b76c9aba0701a6cdb52920174ae1521ea97aa2a3f64821dfe08d997ae6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.16633606050163507 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 555115, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3848142778821959, | |
| "accuracy": 0.4088360618972864, | |
| "balanced_accuracy": 0.4180010374181177, | |
| "worst_class_f1": 0.2053388090349076 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d435fe4444ee31e12f4aeff2d08da4da90b1581e55789f78c53b63e459cbda72", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.14687734376639128 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 678876, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3791269371977908, | |
| "accuracy": 0.40165956492487104, | |
| "balanced_accuracy": 0.4179393559999468, | |
| "worst_class_f1": 0.1830708661417323 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "63f4c3fc491c715125760737b25c16405481f68eaefd3945ce31338650723c6e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.1423840904608369 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 164963, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.38673689720064064, | |
| "accuracy": 0.4308221457894153, | |
| "balanced_accuracy": 0.3889641395556819, | |
| "worst_class_f1": 0.18230563002680966 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bea2ddaf698e130e949026e1d06658401971805e9a3192abd36778e73b885c37", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.08014317229390144 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 587574, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.40823881367737425, | |
| "accuracy": 0.44291578830578054, | |
| "balanced_accuracy": 0.45349802538725736, | |
| "worst_class_f1": 0.18556701030927836 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fa9923bc96f898e37075dc1fe779e76b14f44e27caf6deda4068d31d0c756f62", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11033798847347498 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 589309, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4140329275306399, | |
| "accuracy": 0.4480195273493842, | |
| "balanced_accuracy": 0.45934733635186287, | |
| "worst_class_f1": 0.1822849807445443 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3ed51ab0acde02f67a2049d119fc0a1d83ea9cdebf3833321f09c7ec12c487f1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 8, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12434007879346609 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 681220, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.47706378206167416, | |
| "accuracy": 0.5076923076923077, | |
| "balanced_accuracy": 0.5092512560398281, | |
| "worst_class_f1": 0.1962864721485411 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b45b73fd069a7912e03d1855f10a92784abbc3953e9651cee273a6532f692464", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.1404802268370986 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 554213, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.49312844038510845, | |
| "accuracy": 0.5299145299145299, | |
| "balanced_accuracy": 0.5207230526700427, | |
| "worst_class_f1": 0.1807909604519774 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0752056393d300870f405f1eaa6c1ea04582279181d1e136badfa8ce80222518", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 5, | |
| 11 | |
| ], | |
| "verification_seconds": 0.11337910685688257 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 344185, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4666524181153792, | |
| "accuracy": 0.5035897435897436, | |
| "balanced_accuracy": 0.49701966308229717, | |
| "worst_class_f1": 0.16022099447513813 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a58060dc11572fe3ccdce4a6d2f0150fcbdf558914403de04c4b8d2f11cfc7ae", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.10804144851863384 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 364507, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4370155710734781, | |
| "accuracy": 0.4697265625, | |
| "balanced_accuracy": 0.48011368668483795, | |
| "worst_class_f1": 0.1203585147247119 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "70840706528cb04f82ac77a1d4c92016dc45ee93a2f96c56a01f09966d31b925", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08191533200442791 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 697475, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4512327358194982, | |
| "accuracy": 0.4790581597222222, | |
| "balanced_accuracy": 0.49275046578712517, | |
| "worst_class_f1": 0.1793478260869565 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5bb2a4d812e31bc98fcc59fb93920ddf01a1167ee426e56150098a931c74aba4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 4, | |
| 5, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12659502308815718 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 451638, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4471513148059462, | |
| "accuracy": 0.4729817708333333, | |
| "balanced_accuracy": 0.48961303342143864, | |
| "worst_class_f1": 0.17073170731707318 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "89fd62d69fb8334e6474dd91d725669067166c5e62c3178ea6068d6478bfd641", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 11, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.1261815158650279 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 477487, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5345935850365053, | |
| "accuracy": 0.58274231678487, | |
| "balanced_accuracy": 0.5655624417274536, | |
| "worst_class_f1": 0.21656050955414013 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ba5cf262812bbe919af9e320dbc6d871a6ea1cd10c91c3aa70e16e6d79de8da5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 11, | |
| 4, | |
| 5 | |
| ], | |
| "verification_seconds": 0.10365157946944237 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 701127, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5360904467830878, | |
| "accuracy": 0.5872340425531914, | |
| "balanced_accuracy": 0.5652497977212918, | |
| "worst_class_f1": 0.18181818181818182 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c9b0240e9e9b1e24a6818aea545aca798ed834851723fe5cbc1cf6bee074c309", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 11, | |
| 5, | |
| 11 | |
| ], | |
| "verification_seconds": 0.12794150412082672 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 593415, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.540710085532877, | |
| "accuracy": 0.5888888888888889, | |
| "balanced_accuracy": 0.5713526491135847, | |
| "worst_class_f1": 0.2018348623853211 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "777bc6f479fcba72984e9e1aed22661d6ac8fc65036a2d58606ec75a7df3ae22", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 4, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.117134939879179 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 928585, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5292660108853414, | |
| "accuracy": 0.5708679076026015, | |
| "balanced_accuracy": 0.5517760462043491, | |
| "worst_class_f1": 0.24623115577889448 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f6cb6f46235235f76ebec4dcaab5d78ef7b237742ceb943f4c0c956f71c6fe40", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 5, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.1802399018779397 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 920365, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5224325352669028, | |
| "accuracy": 0.5642520744561561, | |
| "balanced_accuracy": 0.5433392429912146, | |
| "worst_class_f1": 0.224 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2df7e442bc4600f05e8c723f33877fb92f60d551306cd69ff7175049bf7c43e4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 11, | |
| 11, | |
| 8 | |
| ], | |
| "verification_seconds": 0.1654700394719839 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 678702, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4902724278530045, | |
| "accuracy": 0.5289302534200493, | |
| "balanced_accuracy": 0.5163140375736235, | |
| "worst_class_f1": 0.23756906077348067 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "935dd179ad44b3b02aa5e8545f12db3b4099b241eb890ce780d3c323c5ac9664", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.13488131761550903 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 116863, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.48107076640171, | |
| "accuracy": 0.5211361366914457, | |
| "balanced_accuracy": 0.47297209841401533, | |
| "worst_class_f1": 0.2077562326869806 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "66385fc30b94f06aa0449ed217198ccc132752ce4dc4535b14f709b3d2f412f0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.09273753501474857 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 587010, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5040897734143406, | |
| "accuracy": 0.5419948962609564, | |
| "balanced_accuracy": 0.5311132155244158, | |
| "worst_class_f1": 0.20030349013657056 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "500ca32f859e3fcc82c2e63641bcf90b276c39ef76b25897c20105bd34b33b76", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.11589611694216728 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 678966, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.49092845263416923, | |
| "accuracy": 0.5254632197936314, | |
| "balanced_accuracy": 0.520122834067514, | |
| "worst_class_f1": 0.19607843137254902 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "32de18c5708819ab41108eb5d133a31cb1df50743b541d3e44ec033066f0c252", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.15122493356466293 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 144553, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6001622909093439, | |
| "accuracy": 0.8245280257573541, | |
| "balanced_accuracy": 0.592896134674992, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "64fdaaef017498ecc7315aa50318895b526ece1f7ca5565d2852c8c94b6fdb7a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10048750694841146 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 194458, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6212723756520478, | |
| "accuracy": 0.8232108883360164, | |
| "balanced_accuracy": 0.6083823342657297, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c743297bc7b09082982f875e32a2efb54f4f9ab3417e51248ca11c64c51198fb", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.08981783781200647 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 341656, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5636640533035754, | |
| "accuracy": 0.8281867408166252, | |
| "balanced_accuracy": 0.5371753995032529, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4ab88443c7cb1bd1291caabd5f666cda2fc23b1f3a7f4f61b082131ba58e5b5f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.1375412354245782 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 336594, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5364918657912368, | |
| "accuracy": 0.8251804330392943, | |
| "balanced_accuracy": 0.5041208793661618, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e2f6ac7eea957553f6812a95af8140f0b47ea32fc36a0a021b72e7b65b2c5ad4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.12254564091563225 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 343025, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5134095420062846, | |
| "accuracy": 0.801523656776263, | |
| "balanced_accuracy": 0.48543136741220744, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "32e9553c9d5a4159a79f670139ae0ce81021f2ffcc2c7691b0af3ae030e54425", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.1299411579966545 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 341859, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.49811428812826414, | |
| "accuracy": 0.7797380379577653, | |
| "balanced_accuracy": 0.4614286887159727, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0e901ee03a0b25696fd04c6454816ac40d9fd6a9c620b0aaad388f502b3061ce", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.12822435796260834 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 73411, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5740855391481829, | |
| "accuracy": 0.8384826434450674, | |
| "balanced_accuracy": 0.6331106148564649, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "064ae588a90768eb6a818b751dad91f2a4bdc110ea0cc3b879d0655073fdfad1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08763671666383743 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 147505, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5887274453964119, | |
| "accuracy": 0.8454014076106405, | |
| "balanced_accuracy": 0.6058360488052064, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "31dc121ea36092e092175c7e0e5407da27c8a1a661baa1a52a283e93892565a3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10630783904343843 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 266734, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6268461641781113, | |
| "accuracy": 0.8302516998687821, | |
| "balanced_accuracy": 0.6435828090925986, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6b9344e70583d77530fe8a694440936b094483f06e6e582834b1a3e2e09d9230", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.10888167563825846 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 288208, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.596128960568406, | |
| "accuracy": 0.8317962835512732, | |
| "balanced_accuracy": 0.598804660316241, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6a22c61813ec149f540c737bd34e3e7488eb05fbfefe348c9d6f5a4f626bc0be", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12140228878706694 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 338086, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5905239929548161, | |
| "accuracy": 0.8220233998623537, | |
| "balanced_accuracy": 0.5860314266823297, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cc486db4a47978da0ec68a942e819276c79206dd4a18237dde884231bcd3ac9e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.1356620490550995 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 339803, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.612778808268588, | |
| "accuracy": 0.846662078458362, | |
| "balanced_accuracy": 0.5768670269769565, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "378d36025950f1bb47310bcca00bd87e5482be55e84ac0630cffb98b8b060883", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.1284400476142764 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 358831, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7350769324712363, | |
| "accuracy": 0.8778756035217268, | |
| "balanced_accuracy": 0.6931079259273305, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "8d02a047a1b63170714775755c37eb944a608a7ee30bdd074c911ff15f0efc95", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.13610275462269783 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 342490, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7479329190105534, | |
| "accuracy": 0.8950582220959955, | |
| "balanced_accuracy": 0.7162155645562306, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0d649153cc7bd5c8480e6281471419759e3781edb27a967ed22233fc16775f34", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.13554767239838839 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 337538, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7998097634781114, | |
| "accuracy": 0.9067026412950866, | |
| "balanced_accuracy": 0.7779843281151284, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d385cbf979f12f2e1fa3a129f70e1a4d7f3f117a83b9efd811e9ea8a0fe59b7c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.1348281092941761 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed42/model.pkl", | |
| "bytes": 385670, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7354886950705765, | |
| "accuracy": 0.7267326732673267, | |
| "balanced_accuracy": 0.7416392821031345, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "642323bd092f7af4bff81261d49a95f8d6bdd9c03bb991382807f96cd65a1b68", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.10621493961662054 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed43/model.pkl", | |
| "bytes": 389579, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8188387012671772, | |
| "accuracy": 0.8198019801980198, | |
| "balanced_accuracy": 0.8294384057971014, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c98af6a1ad5fa4a5acf72dbb151c2c39e8bbcff3547fb378b2655c3128fbcc6d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.10488037578761578 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed44/model.pkl", | |
| "bytes": 388588, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7493625750406817, | |
| "accuracy": 0.7485148514851485, | |
| "balanced_accuracy": 0.761945989214695, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1a1e5ee70767382efbcf8077bd8f6b6a408593fe493ffa43a86c593202d523a5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.10792993381619453 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed42/model.pkl", | |
| "bytes": 388809, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8135257169451995, | |
| "accuracy": 0.8317214700193424, | |
| "balanced_accuracy": 0.8389626741846908, | |
| "worst_class_f1": 0.5352112676056338 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b6b4cad4225d36f51a87164af7f5ae2cb9686f93958bb781e20b0cd92fe2b143", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.11736003495752811 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed43/model.pkl", | |
| "bytes": 389448, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8224755450020339, | |
| "accuracy": 0.8355899419729207, | |
| "balanced_accuracy": 0.8430465618254703, | |
| "worst_class_f1": 0.5526315789473685 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ea44f640f353240f0fe43c2f07b0671b8ef4d9b14da5b93e0c141cf6f051a4df", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.1024811640381813 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed44/model.pkl", | |
| "bytes": 99292, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7123372087360301, | |
| "accuracy": 0.7330754352030948, | |
| "balanced_accuracy": 0.74444591280854, | |
| "worst_class_f1": 0.11428571428571428 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e4fa923163e155f243de5e81ab01b0dfd68897c2d30b760261d78939901d607b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.06621815077960491 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed42/model.pkl", | |
| "bytes": 6680248, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8878789083727651, | |
| "accuracy": 0.884393063583815, | |
| "balanced_accuracy": 0.8913043478260869, | |
| "worst_class_f1": 0.5542168674698795 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ce8c0b79823ca0e9198b9705a7708e6417d0465501e611c72e131206ae690b15", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.13278221990913153 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed43/model.pkl", | |
| "bytes": 16192319, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9111111111111111, | |
| "accuracy": 0.9113680154142582, | |
| "balanced_accuracy": 0.9166666666666666, | |
| "worst_class_f1": 0.6666666666666666 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b03a1eba1b9ba1d5f14bf9454de555f8b0214b3e6020f363d0e32d2f7ca345b3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.13645241782069206 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed44/model.pkl", | |
| "bytes": 16911563, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.877856486947396, | |
| "accuracy": 0.8959537572254336, | |
| "balanced_accuracy": 0.8731884057971014, | |
| "worst_class_f1": 0.6571428571428571 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "619961db831ef97b705ff4251ab9de1c1e5f62c9d9a174974ca4372ce9b765d3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.14094430953264236 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed42/model.pkl", | |
| "bytes": 387408, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9296020533702118, | |
| "accuracy": 0.9295499021526419, | |
| "balanced_accuracy": 0.9320460673468762, | |
| "worst_class_f1": 0.676923076923077 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c12d0ad33340bdfaa17619d0480c221971f88ad98b2f9782328b8ec64c775bd9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.11122406274080276 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed43/model.pkl", | |
| "bytes": 100328, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.888480483344134, | |
| "accuracy": 0.8904109589041096, | |
| "balanced_accuracy": 0.8949656539393648, | |
| "worst_class_f1": 0.5423728813559322 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cd5e821662f61f83e72b853c5ec5c1bc544b8d90ff9eb27351f715dfc7ef1654", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.0694976132363081 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed44/model.pkl", | |
| "bytes": 291189, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8694335167724162, | |
| "accuracy": 0.8688845401174168, | |
| "balanced_accuracy": 0.8672814378072213, | |
| "worst_class_f1": 0.64 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a7d10e300109478314430a4eb69c1a26724258d09f1f95448fc6061bfa54862a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.091588887386024 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed42/model.pkl", | |
| "bytes": 30961327, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8318056467541005, | |
| "accuracy": 0.8230616302186878, | |
| "balanced_accuracy": 0.8346920289855072, | |
| "worst_class_f1": 0.5 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c3c4a01d438ce5ecd6b04091d0a8514f61ef8601fa4ed5227c473308b0af42d9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.16053823940455914 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed43/model.pkl", | |
| "bytes": 30830438, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8414983164983165, | |
| "accuracy": 0.8489065606361829, | |
| "balanced_accuracy": 0.8333333333333334, | |
| "worst_class_f1": 0.46464646464646464 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cb2eb0d0b2dfc1e6692fdeb57a8fbcae3132439f68cd7192c107bbc48a118b89", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.16630718670785427 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed44/model.pkl", | |
| "bytes": 30621176, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8392896169265338, | |
| "accuracy": 0.8469184890656064, | |
| "balanced_accuracy": 0.8315217391304349, | |
| "worst_class_f1": 0.46464646464646464 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e9b6153280a1a0515086dcb052677e4ef9525a52e4eaba7988f5974d590abef0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.17363407742232084 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 153957344, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7892275662433231, | |
| "accuracy": 0.7895833333333333, | |
| "balanced_accuracy": 0.7732288959196837, | |
| "worst_class_f1": 0.5088757396449705 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7e051b14cf200359bc93dc42d1c644a3d70195c75d53c83595ff78536fbbca83", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.4136338597163558 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 323691551, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7900666383115289, | |
| "accuracy": 0.8010416666666667, | |
| "balanced_accuracy": 0.7741262465535576, | |
| "worst_class_f1": 0.5810055865921788 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f7e788b278cc8e1725664d17aae66ab8d38966f5da8db575d689330f152375b4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7362617207691073 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 153261502, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7730199712967488, | |
| "accuracy": 0.7885416666666667, | |
| "balanced_accuracy": 0.7593348479587491, | |
| "worst_class_f1": 0.55 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "338412c40b053ad5446733dc73170b258270ec38714701590ab4bfd4a8f69e47", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.36440687999129295 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 142963268, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7225936836830401, | |
| "accuracy": 0.7149677057006459, | |
| "balanced_accuracy": 0.705294925589822, | |
| "worst_class_f1": 0.37209302325581395 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5430b3d4daa1933336b3353f154efd50193f86f80dcc357acecafc03c7a1d3bf", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.3679047701880336 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 140553320, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7164108253593273, | |
| "accuracy": 0.7197416456051671, | |
| "balanced_accuracy": 0.7004261415634176, | |
| "worst_class_f1": 0.4806201550387597 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c46d20bb9312101456100e660befe4efadf1cf076805c45dd995d23d9ccae00e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.39510233141481876 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 141989394, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6871823068269264, | |
| "accuracy": 0.7071047458579051, | |
| "balanced_accuracy": 0.6740489832241326, | |
| "worst_class_f1": 0.31683168316831684 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7a846466c5eebdb50b19c86a9569e1a8dfa50385fd1351113a32907dac90d898", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.3557204445824027 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 315612091, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7864181898198394, | |
| "accuracy": 0.7759418374091209, | |
| "balanced_accuracy": 0.766090744467668, | |
| "worst_class_f1": 0.45454545454545453 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "8741865403f39a1d2025c6d1b9d8b298731828a521f2fa6fef533931b1d0c060", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7392814699560404 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 153243266, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.80773716013251, | |
| "accuracy": 0.7974223397224058, | |
| "balanced_accuracy": 0.7904072939498322, | |
| "worst_class_f1": 0.5039370078740157 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c62309370b896062d65a2c3de815063290b1b90c15357ae94b902e47c1321565", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.41625876631587744 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 153626016, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8019267714455053, | |
| "accuracy": 0.7941176470588235, | |
| "balanced_accuracy": 0.7839061568806102, | |
| "worst_class_f1": 0.49624060150375937 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "23fadf9558497c26725749a608d94a3f3e9b6552e270d86357084b0663d62761", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.3832047041505575 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 152149574, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7777656452114684, | |
| "accuracy": 0.7796555086122847, | |
| "balanced_accuracy": 0.7665927790732726, | |
| "worst_class_f1": 0.3953488372093023 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3092788b844f02df039c5bd7471138d84d5ec2d4dfc0c32cf3632e5d4da2adbd", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.3757688459008932 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 152238152, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7607235262767121, | |
| "accuracy": 0.7731556711082223, | |
| "balanced_accuracy": 0.7535080157000026, | |
| "worst_class_f1": 0.3684210526315789 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "870373f958f71bf32099b6a24355c38355e07b6531bd6f4a3b1232df1e59b0f2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.39853435661643744 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 319477175, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7532136587464725, | |
| "accuracy": 0.7598310042248944, | |
| "balanced_accuracy": 0.7411083111974465, | |
| "worst_class_f1": 0.40229885057471265 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6c875bc9fe2f42aa55055a78d56124883050a090c9eb0843500b5d5c870a3d8b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7252659574151039 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 151505892, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7806518780446584, | |
| "accuracy": 0.7801668806161746, | |
| "balanced_accuracy": 0.7721625612556755, | |
| "worst_class_f1": 0.2857142857142857 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "264d7c463966746c13527a9b1f6e957909c6f22d3d65ef956de8b3a3af631cd2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.38599200546741486 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 318869947, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7932082732315537, | |
| "accuracy": 0.7910783055198973, | |
| "balanced_accuracy": 0.7835228970239744, | |
| "worst_class_f1": 0.3378995433789954 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3f8b66d97b8aaa91894b0675e2e2358ba7a0c91ff60c3f319cb383925951dcee", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7330380454659462 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 152256374, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7644344111861626, | |
| "accuracy": 0.7708600770218228, | |
| "balanced_accuracy": 0.7561859057947079, | |
| "worst_class_f1": 0.3404255319148936 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "af81927a19858af47b2fb5e51a8649cc34e27cde0b64614136a14bff375c5f0e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.3918878575786948 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed42/model.pkl", | |
| "bytes": 193291, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9574373809904692, | |
| "accuracy": 0.9519050593379138, | |
| "balanced_accuracy": 0.9570683091449005, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6b84666bf1cf734a5028a6ad927d030ab5001d3c948d5ce9a5adfd41fac2f467", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08527306746691465 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed43/model.pkl", | |
| "bytes": 196772, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9234438902270412, | |
| "accuracy": 0.915053091817614, | |
| "balanced_accuracy": 0.9239575066889878, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bcbe55be438c66f2ff1596569efcd1b0f119bf8930351c32e578edc73c50562e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12321723159402609 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed44/model.pkl", | |
| "bytes": 297193, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9406545864398269, | |
| "accuracy": 0.9344159900062461, | |
| "balanced_accuracy": 0.9396017944519662, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4b8ae35ce0b9243acac79c368aaf530801f25329219fb9a261379dd0e8491d85", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12391958478838205 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed42/model.pkl", | |
| "bytes": 100780, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9274071442306097, | |
| "accuracy": 0.9279907084785134, | |
| "balanced_accuracy": 0.9303482022468934, | |
| "worst_class_f1": 0.7922705314009661 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4cf0e5c4409bcd82ca7fdb4c22a3c8bca473f0f20a164320d5c944bff525fcdc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.0929145747795701 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed43/model.pkl", | |
| "bytes": 597009, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9573547671478485, | |
| "accuracy": 0.9506387921022067, | |
| "balanced_accuracy": 0.9599025452743013, | |
| "worst_class_f1": 0.8265895953757225 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fc6e3ad30248d370e9a115f9934da1a4a6bcf5f258e65cee5e75c1517be6ed2e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1140044592320919 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed44/model.pkl", | |
| "bytes": 100739, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9486688403645763, | |
| "accuracy": 0.9419279907084785, | |
| "balanced_accuracy": 0.9491802208534699, | |
| "worst_class_f1": 0.8306010928961749 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3fd5f1bda39e2b35f9f9c9a01447d6e6def59a6117463fb9839d208a4d8b74ad", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08163740299642086 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed42/model.pkl", | |
| "bytes": 62111308, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8517262119654153, | |
| "accuracy": 0.8604014598540146, | |
| "balanced_accuracy": 0.8410551716685949, | |
| "worst_class_f1": 0.5333333333333333 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b583f9f7e0ef26632e0c3a974d8c4ab3131c1151c650395164b7427e4c8c2dd8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.23181306663900614 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed43/model.pkl", | |
| "bytes": 62886878, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8209642589780709, | |
| "accuracy": 0.8394160583941606, | |
| "balanced_accuracy": 0.8033008632095653, | |
| "worst_class_f1": 0.5384615384615384 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a6d79a8bda87c56e6c8c7c67224f349e1e58e4360275e55869232059844bcfec", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.2601744942367077 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed44/model.pkl", | |
| "bytes": 63063120, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8023272337691744, | |
| "accuracy": 0.8147810218978102, | |
| "balanced_accuracy": 0.7802617209412323, | |
| "worst_class_f1": 0.5333333333333333 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "65be58d783f38b502915d98d4653c04896fa4f9c020ebf6fd3fa2c628e66188d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.24120864365249872 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed42/model.pkl", | |
| "bytes": 508988, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9753302851247804, | |
| "accuracy": 0.9764453961456103, | |
| "balanced_accuracy": 0.9768624909957174, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7c7b18f7c4b28dfa856b9d137d3b9b30dd31321d0f44a67b7e4a9c958bb79990", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08564670383930206 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed43/model.pkl", | |
| "bytes": 202528697, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9777781580379844, | |
| "accuracy": 0.9796573875802997, | |
| "balanced_accuracy": 0.9795458509744225, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3408f2b33a059bf8d413702989205a6cbe26e63950ca4c79d86a394e8c8a54b2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.8584732040762901 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed44/model.pkl", | |
| "bytes": 100381, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9702034685152359, | |
| "accuracy": 0.9721627408993576, | |
| "balanced_accuracy": 0.9695197659483373, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "aa815e1e89f51a870ce1317e6de1533067fc8b3c13fc95460ed97384d76c78a5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.07312445435672998 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed42/model.pkl", | |
| "bytes": 56120005, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7621147901273506, | |
| "accuracy": 0.7752192982456141, | |
| "balanced_accuracy": 0.7490324061766814, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2042c93b8c87d47a56542875b1c96ba42a779fa55d459d847becec8bb24ab696", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.22116222511976957 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed43/model.pkl", | |
| "bytes": 58285096, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7501886834700756, | |
| "accuracy": 0.7642543859649122, | |
| "balanced_accuracy": 0.7350324796363008, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "446c26982f12fa7809412901d7da73a936e567375ac9d4c20800e4ea5052fb23", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.25193033926188946 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed44/model.pkl", | |
| "bytes": 56854569, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.760305665820137, | |
| "accuracy": 0.7702850877192983, | |
| "balanced_accuracy": 0.7455118406391396, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6cf67eb93bfe4f2e570d63e3aa237bd864a21d2f10755012fb0ca68d3adaf488", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.22516581136733294 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 1441696758, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7257920521665745, | |
| "accuracy": 0.7153008962868118, | |
| "balanced_accuracy": 0.7167385507142605, | |
| "worst_class_f1": 0.24921728240450847 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c15e5c665d7296a50f5da08fc485c98b6587d4710d61fff886d8d0bd9d868353", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.7825141521170735 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 1440491744, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7311220616483857, | |
| "accuracy": 0.7195902688860435, | |
| "balanced_accuracy": 0.721265618929603, | |
| "worst_class_f1": 0.2734422262552934 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ae134fbb2388fff33f837da08fea166bc34896eb2145b9abcc51dded51dc3f26", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.7973577231168747 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 1454190198, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7372074685460596, | |
| "accuracy": 0.7245198463508322, | |
| "balanced_accuracy": 0.7259944037246389, | |
| "worst_class_f1": 0.35683629675045986 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9947c4ed8e471c90de672c7379927e557a3c43505ce6547e95e0332965987f10", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.7477301387116313 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 1415103478, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6909213215278447, | |
| "accuracy": 0.6976187476376464, | |
| "balanced_accuracy": 0.6950960789059522, | |
| "worst_class_f1": 0.3468507333908542 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "434ed4bdc14514ab3e125b0032326ccfff4b11dd900528031dddbb158080136d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.5760244950652122 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 1410096822, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6693677517933854, | |
| "accuracy": 0.6783419427995464, | |
| "balanced_accuracy": 0.6754648043483025, | |
| "worst_class_f1": 0.21124361158432708 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ae2013720b730a92e701ea61ec7c5cba041461b6730b88c893d715e2cb5fd3a1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.713862843811512 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 1412228576, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6790566568076546, | |
| "accuracy": 0.6857124858258788, | |
| "balanced_accuracy": 0.682652741791336, | |
| "worst_class_f1": 0.27692307692307694 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "72e5d2ab67810a31dc43d07fbcf1fb1e75e27ed539bdf73216270d54e5f1089c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.726566475816071 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 1006207, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8497583589701082, | |
| "accuracy": 0.8565003513703443, | |
| "balanced_accuracy": 0.857026570430238, | |
| "worst_class_f1": 0.33 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3ec5e9a17b1d91fcd980c62acbaad8d3c664ba57bc3a4647423f766794ee10f9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.1348606338724494 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 638844, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8221419805287655, | |
| "accuracy": 0.8338018271257905, | |
| "balanced_accuracy": 0.8360105567768358, | |
| "worst_class_f1": 0.15869311551925322 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "00e790c44211d2e9101f4b7989cebaccb4b24f1c8d472be774bdf4677f14fb10", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.1066829888150096 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 1346779355, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8333028215152704, | |
| "accuracy": 0.8427266338721012, | |
| "balanced_accuracy": 0.8448529483213124, | |
| "worst_class_f1": 0.33367037411526795 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3a6544700bc41317febaa9a6fccf239bea2eaa23679f880b1d874ff9c035da32", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.545041259378195 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 1403333366, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7105930492098491, | |
| "accuracy": 0.7057796620736954, | |
| "balanced_accuracy": 0.7063862046110593, | |
| "worst_class_f1": 0.40115025161754136 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f6eb8ecc81f2dbd82dc2198a7fce9b1a6e87335a4c4894ddaf9fa1b49f0e1b8d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.6965157566592097 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 1402171766, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7084641883414828, | |
| "accuracy": 0.7075877548475591, | |
| "balanced_accuracy": 0.7081053741669864, | |
| "worst_class_f1": 0.3187889581478183 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "dccc1936bb4b4df120ba985021080edaff742caed891e7d0a4bd65ab5d3b9e35", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.6935139382258058 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 1403424512, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6952479691226653, | |
| "accuracy": 0.6926865764698548, | |
| "balanced_accuracy": 0.6933706911358393, | |
| "worst_class_f1": 0.34602649006622516 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "57c44e13d0ba8d4c6c512726c084d375dbe48c6814b88286ed92c508e5835cf7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.691866286098957 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 641366, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7203206678406642, | |
| "accuracy": 0.7189364132504125, | |
| "balanced_accuracy": 0.7196280071775887, | |
| "worst_class_f1": 0.12974051896207583 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7a53a1691892c5ace8e41e723b388da68e5714799557855900d73ee03dfde3cd", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.0962999165058136 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 246094, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6694266295253583, | |
| "accuracy": 0.6790836400558447, | |
| "balanced_accuracy": 0.6768085236883788, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fbb2ab9c1cee267c109d60f3d6b771a9b4e6f1069056306d1d07c99bdbc5e45c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.0871832063421607 | |
| }, | |
| { | |
| "method": "wisp_evolution", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 642689, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7068318523799468, | |
| "accuracy": 0.7071963447137962, | |
| "balanced_accuracy": 0.7070325543416646, | |
| "worst_class_f1": 0.1347248576850095 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a389973ba7102f94da04fba77bae30fec1c3a3ba6e259d813aa62ab81060aa79", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.10736861545592546 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 213075, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "passed", | |
| "n_test_windows": 861, | |
| "exact_frozen_predictions_equal": true, | |
| "original_input_exact_frozen_predictions_equal": true, | |
| "original_input_differing_prediction_count": 0, | |
| "differing_prediction_count": 0, | |
| "original_and_fresh_X_exactly_equal": true, | |
| "original_and_fresh_X_max_absolute_difference": 0.0, | |
| "time_coordinate_check": { | |
| "raw_time_values_exactly_equal": false, | |
| "per_subject_stable_chronological_permutation_identical": { | |
| "f3": true, | |
| "m1": true, | |
| "m3": true, | |
| "m7": true | |
| }, | |
| "original_first5": [ | |
| 0.0, | |
| 2.5, | |
| 5.0, | |
| 7.5, | |
| 0.0 | |
| ], | |
| "fresh_first5": [ | |
| 0.0, | |
| 1.0, | |
| 2.0, | |
| 3.0, | |
| 0.0 | |
| ], | |
| "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5339423877092813, | |
| "accuracy": 0.7015098722415796, | |
| "balanced_accuracy": 0.5192755280407102, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5339423877092813, | |
| "accuracy": 0.7015098722415796, | |
| "balanced_accuracy": 0.5192755280407102, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d7e9ecb9baabeede43934e670f6e0efcdeb29e1110056d33362d5cad8cc0870f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 3.478103124536574 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 56473111, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "passed", | |
| "n_test_windows": 861, | |
| "exact_frozen_predictions_equal": true, | |
| "original_input_exact_frozen_predictions_equal": true, | |
| "original_input_differing_prediction_count": 0, | |
| "differing_prediction_count": 0, | |
| "original_and_fresh_X_exactly_equal": true, | |
| "original_and_fresh_X_max_absolute_difference": 0.0, | |
| "time_coordinate_check": { | |
| "raw_time_values_exactly_equal": false, | |
| "per_subject_stable_chronological_permutation_identical": { | |
| "f3": true, | |
| "m1": true, | |
| "m3": true, | |
| "m7": true | |
| }, | |
| "original_first5": [ | |
| 0.0, | |
| 2.5, | |
| 5.0, | |
| 7.5, | |
| 0.0 | |
| ], | |
| "fresh_first5": [ | |
| 0.0, | |
| 1.0, | |
| 2.0, | |
| 3.0, | |
| 0.0 | |
| ], | |
| "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.44804969372888687, | |
| "accuracy": 0.6991869918699187, | |
| "balanced_accuracy": 0.48086825598802657, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.44804969372888687, | |
| "accuracy": 0.6991869918699187, | |
| "balanced_accuracy": 0.48086825598802657, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "69b629122b9d9b7f4dc5cbf40b767b9391f12f8d6327ea817f385e736638f265", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 1.414209634065628 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 387105, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.42336890869963556, | |
| "accuracy": 0.5448648648648649, | |
| "balanced_accuracy": 0.5103515571350251, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0593f16f80eb5d1a010e55c9534cc105a671f91294b48a68990c46d0b27ddb1e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 7, | |
| 7, | |
| 7 | |
| ], | |
| "verification_seconds": 0.11762447189539671 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 388857, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.44265859848601236, | |
| "accuracy": 0.5553153153153153, | |
| "balanced_accuracy": 0.5262091686071412, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0133cd4b226973eebe7a77576171dfe2440d216a3c440f8652f4379d6e9cd136", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 7, | |
| 7, | |
| 7, | |
| 7 | |
| ], | |
| "verification_seconds": 0.10366932395845652 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 197938, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.45020993038951257, | |
| "accuracy": 0.556036036036036, | |
| "balanced_accuracy": 0.5254028034871413, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0ef0f6220e2db9c4c853ba6158dc3924effaefaca7f40fd87f196c3259fd3030", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 7, | |
| 7, | |
| 7, | |
| 7 | |
| ], | |
| "verification_seconds": 0.07649926003068686 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 213222, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6839360804687519, | |
| "accuracy": 0.6371134020618556, | |
| "balanced_accuracy": 0.6487853885528353, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "83d5ea4d49be6e2c8eac6cd6363cbecf629412f9ac345d791a04dd6444c24824", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07935636956244707 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 108796, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6914456270187225, | |
| "accuracy": 0.6412371134020619, | |
| "balanced_accuracy": 0.6577139599814067, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "beda21721a3b563425f0da94f37fb3e5e901fb85cecdb3874fa0c7649a3629be", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.06602407619357109 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 211297, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6265840120876497, | |
| "accuracy": 0.6164948453608248, | |
| "balanced_accuracy": 0.6150310331521384, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bdf4f5b49f0ee29aa30c2ca28ef1e400302d793f8bb08b94b53773c69d50c48c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.08080927841365337 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 213351, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3984637132349555, | |
| "accuracy": 0.5677799607072691, | |
| "balanced_accuracy": 0.4565550938263096, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "75ae6aa393878d072945e96f452b96e09e9cf40fb3b2ea8457e186d70b05e841", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07724830415099859 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 211953, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3947840773031935, | |
| "accuracy": 0.5540275049115914, | |
| "balanced_accuracy": 0.4383977872139755, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d82f5778b37d6f94a75882921e38848d8207348565af2d0d39c3db90326f7cb8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07652495242655277 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 542820, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.41233788381060416, | |
| "accuracy": 0.5717092337917485, | |
| "balanced_accuracy": 0.46000228267069465, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a809a108f835a4ee2214a0948773f4689220fcd9cfd5deb9f8698001ba8902f9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.11057593394070864 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 419921, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7684496006640077, | |
| "accuracy": 0.8529411764705882, | |
| "balanced_accuracy": 0.8162091423863549, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "51e1d01636b9a0cac92c6cc300ec4e48cce07d310e2e1f190017f13e488a5ee3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.09686489589512348 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 544269, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7017359507404785, | |
| "accuracy": 0.8137254901960784, | |
| "balanced_accuracy": 0.7581788689951661, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "251a88a762c7a5b4b2590567ef275d02cb441de8a0f7e4747da5d76fcc8311f1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 0, | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "verification_seconds": 0.08747569378465414 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 214370, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6492442092593446, | |
| "accuracy": 0.8176470588235294, | |
| "balanced_accuracy": 0.7580917818206289, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5af62dd1e9a42bc066cfd01cfdda16039fcba40a4061a905a997d948728dc8a6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 5 | |
| ], | |
| "verification_seconds": 0.07864306773990393 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed42/model.pkl", | |
| "bytes": 161894, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.766990439714226, | |
| "accuracy": 0.8378404764229229, | |
| "balanced_accuracy": 0.7554500861212902, | |
| "worst_class_f1": 0.6008610086100861 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "229c7dee72504b72d5605c5150c08e9d3f077a27a287382225e8aa2b179929d1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.0840184036642313 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed43/model.pkl", | |
| "bytes": 129303, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7330825808731976, | |
| "accuracy": 0.819974616811481, | |
| "balanced_accuracy": 0.7079910899772252, | |
| "worst_class_f1": 0.5666456096020215 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b65dfa5093e81d27982784c93a20a87a0bf1930e29ffcc01b8e26ece25504f54", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09258068632334471 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed44/model.pkl", | |
| "bytes": 161270, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7623005521588372, | |
| "accuracy": 0.8392072634970223, | |
| "balanced_accuracy": 0.7456711136811414, | |
| "worst_class_f1": 0.6124694376528117 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "07b9eec05901e515f03e25eb39a3c842dc98b7982e302188bcfda892d1bf1869", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.1038873614743352 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed42/model.pkl", | |
| "bytes": 138656, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7486620464593752, | |
| "accuracy": 0.8456556396970905, | |
| "balanced_accuracy": 0.7473717694082199, | |
| "worst_class_f1": 0.4661016949152542 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5c92495d9331f02daf8560445df4e83b9e98d59510b3c7375f378d40904e45b9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.07749845832586288 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed43/model.pkl", | |
| "bytes": 255990, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.735485533953722, | |
| "accuracy": 0.8366879234754883, | |
| "balanced_accuracy": 0.7346841985437063, | |
| "worst_class_f1": 0.43897216274089934 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "866e338432a7bb4b5f8ee276c77ce81e136694c2c3804ccc1574b56810a86e01", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "verification_seconds": 0.1168006956577301 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed44/model.pkl", | |
| "bytes": 130598, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7593223381430728, | |
| "accuracy": 0.8434635312873655, | |
| "balanced_accuracy": 0.7578615088048173, | |
| "worst_class_f1": 0.5243128964059197 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2685a44c752f4432440d3a31e6e8576d149ad136752d6e573c9cb0484f638eb5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "verification_seconds": 0.12646049074828625 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed42/model.pkl", | |
| "bytes": 161894, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7062470015513294, | |
| "accuracy": 0.8255736456250595, | |
| "balanced_accuracy": 0.6981685147812106, | |
| "worst_class_f1": 0.48322147651006714 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0207c109e8e11f6f7fbca6ae8e7624f00f59b83cd39b14b918bd7ac4913d9a1a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.08643207233399153 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed43/model.pkl", | |
| "bytes": 287812, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7108976698220147, | |
| "accuracy": 0.8287156050652195, | |
| "balanced_accuracy": 0.7053300591182281, | |
| "worst_class_f1": 0.48825065274151436 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "92f17f4830088544f470bcc13a91313fcd66ad20a772abb355604a78a1188344", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.10610962845385075 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed44/model.pkl", | |
| "bytes": 161581, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.724010717738153, | |
| "accuracy": 0.8322384080738836, | |
| "balanced_accuracy": 0.7199091773450284, | |
| "worst_class_f1": 0.5288831835686778 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a1f2eab36e31ca790baae3ac9cef3cf645c08d2e2fb9c6c6290d4de65d64b818", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.08403156418353319 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed42/model.pkl", | |
| "bytes": 115443, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7488290281373933, | |
| "accuracy": 0.839453125, | |
| "balanced_accuracy": 0.742297928519201, | |
| "worst_class_f1": 0.5042174320524836 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "619d5842e7c5ec71c173c938e126a128095d1bddcc300489956ed39bfccd6de4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.20588375721126795 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed43/model.pkl", | |
| "bytes": 118067, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7570308365882163, | |
| "accuracy": 0.84384765625, | |
| "balanced_accuracy": 0.7481699905884054, | |
| "worst_class_f1": 0.5209756097560976 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "afcb98b4c2dc1350bcb9e53d800a22a5cb5794874689933b475b88d2775ae41a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.24759656377136707 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed44/model.pkl", | |
| "bytes": 117459, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7596369938969753, | |
| "accuracy": 0.8490234375, | |
| "balanced_accuracy": 0.7538129256318069, | |
| "worst_class_f1": 0.5155393053016454 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a2211a1bb2fd15c90b224ac7ea5131aa6d59957b4e3460875c6eb26500ea7fc1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.2584470985457301 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed42/model.pkl", | |
| "bytes": 128065, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7551038036075215, | |
| "accuracy": 0.83273988005997, | |
| "balanced_accuracy": 0.7455861736760279, | |
| "worst_class_f1": 0.49612403100775193 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9f2d52d91c2bc5624fc0d918954b4aff119e5c78ad73416cbb83c1b72ff605ae", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.08599592372775078 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed43/model.pkl", | |
| "bytes": 226783, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.742279603672112, | |
| "accuracy": 0.8223388305847077, | |
| "balanced_accuracy": 0.7216110695134941, | |
| "worst_class_f1": 0.5005302226935313 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bfee24f11c6a76538adbfec619fa58865fdfa08020e9762ec6cf1bdc81812e0f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.10002798307687044 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "capture24_wearable_activity_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed44/model.pkl", | |
| "bytes": 161581, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7523348286903604, | |
| "accuracy": 0.825243628185907, | |
| "balanced_accuracy": 0.732207260115527, | |
| "worst_class_f1": 0.5368852459016393 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "87ef00aa9a441fa21d0cd24ab3db92c9829499a2634920694f8ab9ed45e59fae", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.09194480534642935 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 213222, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7551308251068606, | |
| "accuracy": 0.8979695431472081, | |
| "balanced_accuracy": 0.7583531872830543, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "52c0487222b97840207d833ce067c1ef209aa9dc6e76b3879cbd48791285e781", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10122111812233925 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 215616, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7773091987294367, | |
| "accuracy": 0.9116751269035533, | |
| "balanced_accuracy": 0.7694623232746799, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9be8c786b60ba44925ce330a3331a3791c2f192e8fb4f2af95c7b285d3a8a020", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13180649187415838 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 542386, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7518046343995171, | |
| "accuracy": 0.9121827411167512, | |
| "balanced_accuracy": 0.7516438722909802, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "562b1a9ca6109509524f71787309d109f44981b05a31982f6310f10bb2a3a2d1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13094902504235506 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 215890, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7811112855157368, | |
| "accuracy": 0.9018214127054642, | |
| "balanced_accuracy": 0.7979395775270787, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "04858553cf1d77303f80f81d2b6fdfcf8f32f846f53b127a0a3531369698cc29", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.10303804371505976 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 316666, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7801701329373244, | |
| "accuracy": 0.8716126166148378, | |
| "balanced_accuracy": 0.8032376592902427, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a50203df8da14032b31032f1609cbcbb997363c09bead7da66293a46d2b92e27", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12858885247260332 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 542875, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7646305706831946, | |
| "accuracy": 0.8796090626388272, | |
| "balanced_accuracy": 0.7734379168736917, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e11c60884f89f1a630a9c14a24705ad1d829e9edb77ab045cdc9142313f76c54", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12358776666224003 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 468799, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6182859411274892, | |
| "accuracy": 0.8502024291497976, | |
| "balanced_accuracy": 0.6535983881129549, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "03ee367d4f45f272040bc32e1f223c210c8ebbb1d9a14e5cf8de193227c52071", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11528230179101229 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 540105, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6311436177301671, | |
| "accuracy": 0.8704453441295547, | |
| "balanced_accuracy": 0.659241603110375, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ba772f5e17588452cbde29336e69acfd4a44c07e565347509a33c81dfc20d447", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1269427239894867 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 543332, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.578905837151691, | |
| "accuracy": 0.8536726431463274, | |
| "balanced_accuracy": 0.6249674973278894, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1e68c747889a8f2466fcf2805645a3014692e747feae48b6abbc19bd014381ac", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13287620432674885 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 215742, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7511955410337825, | |
| "accuracy": 0.8684834123222749, | |
| "balanced_accuracy": 0.7689192612204052, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "97b4a33e5eac4d4c97d0369dc142118fcab3db8345e5cd25563d334b77df938f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.0989870810881257 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 108796, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6954366439264663, | |
| "accuracy": 0.8203001579778831, | |
| "balanced_accuracy": 0.7168607714759039, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0f1c5108cf1dcb1eafcdd881b076eefd0b12eda59dc5e4da1d1ce463951b9e5c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.06485389173030853 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 214370, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.627228447883459, | |
| "accuracy": 0.8052922590837283, | |
| "balanced_accuracy": 0.6445631276390363, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "034d9a680e4312bff79652317774770d18ef487648d866333c627807748d3cc8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09241700731217861 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 541762, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6183927693961376, | |
| "accuracy": 0.8535864978902954, | |
| "balanced_accuracy": 0.6268799948210574, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "52e3563e30b4160d8a3ad7eef9b961d2486c1534ca26592d490a0a865589c313", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1271434649825096 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 540415, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.634073183238398, | |
| "accuracy": 0.8556962025316456, | |
| "balanced_accuracy": 0.639406051515471, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "501fee63a547a6aa191e6c86ca1eaa5b18a4ea9206b981b94e35a14cc5bb0d60", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12751690112054348 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "domino_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 216185, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6367216796558524, | |
| "accuracy": 0.8265822784810126, | |
| "balanced_accuracy": 0.6282511507353351, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f86c3d295afb1e24fce9d7afe9efce9f8a687ebe5f2bc10734dcf9bbab1e9e94", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09841653145849705 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 591456, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.784293009526535, | |
| "accuracy": 0.8009986782200029, | |
| "balanced_accuracy": 0.8151819951716208, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b9db54ada58d01aa9a88928b52628850e3057f6820f1f726b87c01a7ec6c01c6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12829209584742785 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 592918, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7594087239448236, | |
| "accuracy": 0.7810251138199442, | |
| "balanced_accuracy": 0.7894529186109617, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2996f25843ced7e146f6d85bdc6ed30a5857576e422a2520979e15a487384379", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1241801893338561 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 592080, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7708773764124208, | |
| "accuracy": 0.7893963871346747, | |
| "balanced_accuracy": 0.7973621988392794, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "68ea2185c96db68527916dc00959d4927af1085b20db8920929f0de67084fa8e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.15180608443915844 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 182063436, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.729251179192184, | |
| "accuracy": 0.756351392715029, | |
| "balanced_accuracy": 0.7251885891431922, | |
| "worst_class_f1": 0.46808510638297873 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1f41ba703dce73869d5aff7dc3d8de09327e83b4138dc1e06ec26ee24f54f13f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.4741174401715398 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 704450, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8159433539759252, | |
| "accuracy": 0.8123660850933578, | |
| "balanced_accuracy": 0.8184655317757509, | |
| "worst_class_f1": 0.4541832669322709 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f5ffa07e6b96255ed93b361e398740732f20f6e502ae788bb6c76b543f141e8d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1623753560706973 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 590016, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 6.591949208711867e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7652924699658761, | |
| "accuracy": 0.7745638200183654, | |
| "balanced_accuracy": 0.7834846740797345, | |
| "worst_class_f1": 0.018083182640144666 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2826758bc5ee17021e7b8c471ed11739e13983e8ab8736f866423e454d9cd29a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11671893298625946 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 588032, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7873681523995203, | |
| "accuracy": 0.8100347112653834, | |
| "balanced_accuracy": 0.798674374826554, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b911a5c1d76ec88f7fac386b71f1f69977910349f1fc35de1264c8bd172e8578", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12262417282909155 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 591573, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8000927882449437, | |
| "accuracy": 0.8117702745345535, | |
| "balanced_accuracy": 0.8150741858340492, | |
| "worst_class_f1": 0.1109350237717908 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "dc92c8b944619d5abf2b682e0d7188c78f6b63b88229d2bd6225e8b3e6d84d7a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12705540098249912 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 595035, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 6.938893903907228e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7726750498433423, | |
| "accuracy": 0.7914168507415589, | |
| "balanced_accuracy": 0.7876774055916419, | |
| "worst_class_f1": 0.12089810017271158 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1fef32d0456a2818005badf5de3c075659b4041b40d875d109a102ffef16fc25", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12606476619839668 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 591928, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 6.938893903907228e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7993431026428096, | |
| "accuracy": 0.8082954712003517, | |
| "balanced_accuracy": 0.8063771798704091, | |
| "worst_class_f1": 0.05865921787709497 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2596f8d542a9f3892df755c8747f15daa65971ef21cb8e0d5eabf040bdc997cf", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.14119288697838783 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 589011, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7867416975202401, | |
| "accuracy": 0.7984757438077092, | |
| "balanced_accuracy": 0.7980594826403815, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0729d3043a36b77566ef9d77526fe81526baa018aa9a8f44bc39038a5993d1e0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12311080563813448 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 230479, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 1.3877787807814457e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7190410117062849, | |
| "accuracy": 0.7367726806390151, | |
| "balanced_accuracy": 0.7161055800914486, | |
| "worst_class_f1": 0.046008119079837616 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b675a28ea876f097481cf0f186babf1d3cdca36f673c75141a0f273a7996fc7f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08946351986378431 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 592354, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8792767127830559, | |
| "accuracy": 0.8633117422355091, | |
| "balanced_accuracy": 0.8866686986182304, | |
| "worst_class_f1": 0.5959183673469388 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4a0a57580d103371b297935817eb138c41e2332f1a074b9aaf94587d2aa2137f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12175817228853703 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 592217, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8686373451288947, | |
| "accuracy": 0.8555057299451918, | |
| "balanced_accuracy": 0.8761138489486704, | |
| "worst_class_f1": 0.5566433566433566 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "48de32930bc6b94187f3ff3a8e354264ffbb3064b1cdd2944edd2a8fec7d730c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12206245306879282 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "gotov_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 592860, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9048190879441402, | |
| "accuracy": 0.8947018767646571, | |
| "balanced_accuracy": 0.9067162256666195, | |
| "worst_class_f1": 0.7234567901234568 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c113ffde305f5212b8ea0674e68c244d7987fc618b708aa4f2b207d14f2d0613", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.16558631416410208 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 66847331, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9603439013050938, | |
| "accuracy": 0.9589422407794015, | |
| "balanced_accuracy": 0.9617768102403584, | |
| "worst_class_f1": 0.9343065693430657 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0a1e0ed18c21d6fe8be9fadea73e11bdc65f47b66feca80a5bdf9528a2f7bfae", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.21702496614307165 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 106353855, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": -1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9494755982485993, | |
| "accuracy": 0.9443284620737648, | |
| "balanced_accuracy": 0.9507084789963877, | |
| "worst_class_f1": 0.9022222222222223 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6e11892bdd3343eb962fd1e690ede41e7567bdc368dad79b041a027683f8bb3c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.2986576007679105 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 106727683, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9536047751541364, | |
| "accuracy": 0.9491997216423104, | |
| "balanced_accuracy": 0.9554135283314116, | |
| "worst_class_f1": 0.911504424778761 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9f08a3ce13a054fb2a954aa97b14c1e7a045f9f7e5bad43d4e2fa904ba04dfc4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.32660636119544506 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 109082947, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9507493107548275, | |
| "accuracy": 0.9378925331472435, | |
| "balanced_accuracy": 0.9605487984667551, | |
| "worst_class_f1": 0.8957055214723927 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0b3c854a78d5dca17fa61bd74992831ab14ae6b17eb5144adb36420497308d1c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.3146402854472399 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 69488437, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9722661390125834, | |
| "accuracy": 0.9658060013956734, | |
| "balanced_accuracy": 0.9770293664024313, | |
| "worst_class_f1": 0.9409282700421941 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "393ce4ed2452236ddd27745fe07aaf9f40ca0a0f1590ed38ba28625c3f9d9fa9", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 9, | |
| 6, | |
| 5 | |
| ], | |
| "verification_seconds": 0.2198030510917306 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 109039171, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.94426634615951, | |
| "accuracy": 0.9288206559665039, | |
| "balanced_accuracy": 0.9550853661302577, | |
| "worst_class_f1": 0.8724279835390947 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ea8efece2d0848cc16ede574aae1180ee2908669c260f4bc42c5292611a8f360", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.3040814511477947 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 70548283, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": -1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9603615365507692, | |
| "accuracy": 0.9538123167155426, | |
| "balanced_accuracy": 0.9612722403991679, | |
| "worst_class_f1": 0.9227272727272727 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7b3956cbd291ede95168ff2a6d28b63d26a43e9684e037ba627c7cc0fe44ecdc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.21898382529616356 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 30411068, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9635137430915, | |
| "accuracy": 0.9567448680351907, | |
| "balanced_accuracy": 0.9634035317435918, | |
| "worst_class_f1": 0.9223744292237442 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fc0ebebd742e3e0c5b9199f5a3caec835db45198021d4c7de2272ac359577b3c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.15453871339559555 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 30402066, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9654073531705973, | |
| "accuracy": 0.9611436950146628, | |
| "balanced_accuracy": 0.9659125049875467, | |
| "worst_class_f1": 0.9363636363636364 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "78f540cf3087c0fc79bdfdce82373310fa77b3a76a188f6dcc1886f2b797d5e3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.15123513340950012 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 67924027, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9655985809409403, | |
| "accuracy": 0.9556786703601108, | |
| "balanced_accuracy": 0.9711245386810653, | |
| "worst_class_f1": 0.9302325581395349 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "78d8142ced6b45000763b28a2e8f88d73eb1e31be36a8bc9d585e0d18c2d71db", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 5, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.2322320556268096 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 67799611, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": -1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9596676620980877, | |
| "accuracy": 0.9494459833795014, | |
| "balanced_accuracy": 0.9635122651082073, | |
| "worst_class_f1": 0.9222614840989399 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7fd6a793318e44a366fbf31ab425824f2708bc411e1cb270ed3e0558955403cf", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.22738107945770025 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 68072341, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9637962821709289, | |
| "accuracy": 0.9529085872576177, | |
| "balanced_accuracy": 0.9693982750317192, | |
| "worst_class_f1": 0.9217081850533808 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "16e4dbdabe73170e232579c7865b2d7c8fb39c4b29d1c8914be212e7746161d7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 9, | |
| 6, | |
| 5, | |
| 9 | |
| ], | |
| "verification_seconds": 0.22421606816351414 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 109733535, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9501422250540463, | |
| "accuracy": 0.9364285714285714, | |
| "balanced_accuracy": 0.9499210582168228, | |
| "worst_class_f1": 0.8758782201405152 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5006890e670eab4c6f96ada58978af59f03154b2635f4eb5b83194181695436e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 9, | |
| 6, | |
| 5 | |
| ], | |
| "verification_seconds": 0.30270709563046694 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 109431135, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9462109554549561, | |
| "accuracy": 0.9328571428571428, | |
| "balanced_accuracy": 0.9473380715803916, | |
| "worst_class_f1": 0.8752941176470588 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "be1b71a6c9d79a676fd10b4333db9054eb6b4753ae89352609c5ced206de97db", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 9, | |
| 6, | |
| 5 | |
| ], | |
| "verification_seconds": 0.30106195248663425 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "handy_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 68605399, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9602680505925019, | |
| "accuracy": 0.9514285714285714, | |
| "balanced_accuracy": 0.959909060884063, | |
| "worst_class_f1": 0.9 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3d1e2a382d39244fbe4348d3104bf3a5d4e9f76d1e9b9550bae9c6018b66a18b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 9, | |
| 6, | |
| 5 | |
| ], | |
| "verification_seconds": 0.2278675800189376 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 2370891003, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3943996945388775, | |
| "accuracy": 0.4341880341880342, | |
| "balanced_accuracy": 0.39529889172768035, | |
| "worst_class_f1": 0.19811320754716982 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "8bb6acfedc4c11f2f59ab482a6c8feed7f8dc61f9443b19a94b8e045b8101d54", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.696244320832193 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 517285, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3797364195371058, | |
| "accuracy": 0.4098005698005698, | |
| "balanced_accuracy": 0.42310013825699694, | |
| "worst_class_f1": 0.16806722689075632 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c2d4ca643c7a6400fdade9b344ad2e6d13f19f479ddc6c9ebb331624bd7f8fe0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11076897941529751 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 2371692608, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.39416052775079996, | |
| "accuracy": 0.42883190883190886, | |
| "balanced_accuracy": 0.4205249580366873, | |
| "worst_class_f1": 0.20245398773006135 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "253153fe7336aebebac63efcae12e0aa58a4fee4f665559059d24832b45e17e7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.713018720969558 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 2266198779, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.3493869479288254, | |
| "accuracy": 0.3809678819444444, | |
| "balanced_accuracy": 0.35520574414892814, | |
| "worst_class_f1": 0.19378427787934185 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "90a68d971230a9617c8a99d8c006bce5c14cc5775b165be199052df2c4593584", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.515500565059483 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 587116, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.34019784103660566, | |
| "accuracy": 0.365234375, | |
| "balanced_accuracy": 0.3786343493232046, | |
| "worst_class_f1": 0.17903930131004367 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "66228228fad97bfe648e33ec5b21af6b237e923adcee81eef79a9417f7c8f4e0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 11, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.09785876236855984 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 2263995968, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.351469207855779, | |
| "accuracy": 0.376953125, | |
| "balanced_accuracy": 0.3768039684244342, | |
| "worst_class_f1": 0.20352781546811397 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3e448a2c29814480b6e97bd5671df91708d6137083295326421759d1f0574c04", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 8, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.504460323601961 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 2367772155, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.42957514443609257, | |
| "accuracy": 0.46879432624113476, | |
| "balanced_accuracy": 0.4255324702472358, | |
| "worst_class_f1": 0.2266857962697274 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a3798422bee96ade1b4974e2344a3c099c228c611e41dbd3c2fd6a2b18584709", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.693867314606905 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 590184, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.41079080082797637, | |
| "accuracy": 0.4462174940898345, | |
| "balanced_accuracy": 0.4527289147376499, | |
| "worst_class_f1": 0.17002237136465326 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d905f128c139edd4960352e27701d0df14c5f322cebd010aa0c1a9d8edfcc2a8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11217740643769503 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 2368916672, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4233880396109819, | |
| "accuracy": 0.46418439716312054, | |
| "balanced_accuracy": 0.4487659825863075, | |
| "worst_class_f1": 0.2023121387283237 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2dd211a680d732fbc2e3c2b442c5a170550f79ffe5b0d2e31e2027ce78902969", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.683686343953013 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 2296903419, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4045781945009205, | |
| "accuracy": 0.4273379681542947, | |
| "balanced_accuracy": 0.3983868486508851, | |
| "worst_class_f1": 0.1737142857142857 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0381189d16ad8a9551ee512689162cf20ae4f80bdbbc1d82677d4498f0057dec", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.528864155523479 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 585468, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.38515500821634396, | |
| "accuracy": 0.40805113254092845, | |
| "balanced_accuracy": 0.42030076334721334, | |
| "worst_class_f1": 0.21149425287356322 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "2d4f8d1e7b0e7d2852bb5ee91fec80d00e6136819f72561d193957072fe6ac37", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11319855600595474 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 2296580288, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.40008244076906796, | |
| "accuracy": 0.42419825072886297, | |
| "balanced_accuracy": 0.41809533022760476, | |
| "worst_class_f1": 0.17865429234338748 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "8c39d17a9ba8b720249d3fb8640328b464aca798b8f4fc04f4f34e86c94b1189", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 3.9741314267739654 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 2313538299, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.42894317767551826, | |
| "accuracy": 0.4700987462554089, | |
| "balanced_accuracy": 0.4321963492839216, | |
| "worst_class_f1": 0.15950920245398773 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6a51c532c58da2852a903abe948896fbdd819fdc8572f883a670f835c608a3f5", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.683324318379164 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 363896, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.39903742181842566, | |
| "accuracy": 0.43459447464773104, | |
| "balanced_accuracy": 0.4465491869813594, | |
| "worst_class_f1": 0.16531165311653118 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6d43b1687051d63a031e99ed8f2b7aca80f066fd8b0dc718ed16e8e649743f76", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13601224217563868 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_left_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 2313291200, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 5.551115123125783e-17, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.43042334665685333, | |
| "accuracy": 0.47043159880173085, | |
| "balanced_accuracy": 0.4597562844733023, | |
| "worst_class_f1": 0.1883656509695291 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "8b97fbb28bb726002a9024e9af4eae8a83272abaa7fb69fa019ed75ce9fc4ea8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.670859377831221 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 455682, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.47826892471136084, | |
| "accuracy": 0.5122507122507123, | |
| "balanced_accuracy": 0.5079849702546556, | |
| "worst_class_f1": 0.1615598885793872 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "9b12660beb807daa7f15ce68676b5af6f29c826ec651eafa4e67926a5f2d11cc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 8, | |
| 8, | |
| 11 | |
| ], | |
| "verification_seconds": 0.11805807705968618 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 455953, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4827274993532969, | |
| "accuracy": 0.5171509971509971, | |
| "balanced_accuracy": 0.5125902139658777, | |
| "worst_class_f1": 0.18289085545722714 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e8376233adffc905ad61b15b2c5366a3241faf126ff6e84a162a0b92d4dd1e82", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.11957382317632437 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 2274877760, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5013992889364879, | |
| "accuracy": 0.5433618233618234, | |
| "balanced_accuracy": 0.5148112236986051, | |
| "worst_class_f1": 0.1736111111111111 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0b8f8bcc9da51d65a7218dca78a0e35a5f69043839c3ddbd0c51ca532761b29b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 4, | |
| 11, | |
| 8 | |
| ], | |
| "verification_seconds": 4.500875387340784 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 2180587131, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 4.163336342344337e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4734412043443698, | |
| "accuracy": 0.4984809027777778, | |
| "balanced_accuracy": 0.47314577660732104, | |
| "worst_class_f1": 0.11463046757164404 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "00f2a7d26793bbda7faca4d0cab16e7e1c69649f3fa0593af78db3a8e8998c6e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 8 | |
| ], | |
| "verification_seconds": 4.1530782505869865 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 590496, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.45353736391094296, | |
| "accuracy": 0.4861111111111111, | |
| "balanced_accuracy": 0.49245402085307655, | |
| "worst_class_f1": 0.16145833333333334 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "94f311255d7cdb7256e5919d96345d998dea9ab5bb9f2b25c632272554a832c1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 13, | |
| 4, | |
| 5, | |
| 13 | |
| ], | |
| "verification_seconds": 0.11939528677612543 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 2177927744, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 5.551115123125783e-17, | |
| "worst_class_f1": 1.3877787807814457e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4605656968569808, | |
| "accuracy": 0.4949001736111111, | |
| "balanced_accuracy": 0.48605973746000264, | |
| "worst_class_f1": 0.12435233160621761 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d6fe8ad972163b8638f89e9b60f3a9a077145b665d872e3e01372408ea514ffa", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 4.361154975369573 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 588184, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5421584351505209, | |
| "accuracy": 0.5908983451536644, | |
| "balanced_accuracy": 0.5719084457993553, | |
| "worst_class_f1": 0.20958083832335328 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "1418f6e3f196aa4e7835f8ffb632c7eb9962ec8208ed298e9026a71163035fa2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 5, | |
| 11, | |
| 4, | |
| 5 | |
| ], | |
| "verification_seconds": 0.11561560537666082 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 517285, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5426752741368953, | |
| "accuracy": 0.5897163120567376, | |
| "balanced_accuracy": 0.5765607711259935, | |
| "worst_class_f1": 0.23384615384615384 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fe2a7d032ee656e4a399e1977b4754a7c5a2b1ee18e3f9f3039150718f61388e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 5, | |
| 11 | |
| ], | |
| "verification_seconds": 0.11062014661729336 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 476284, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5340476395279575, | |
| "accuracy": 0.5822695035460993, | |
| "balanced_accuracy": 0.563630793774831, | |
| "worst_class_f1": 0.20125786163522014 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "679a7bdceb4ee46b969c415ca020e61c58b6da683decb765ca52bad685d61d9e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 4, | |
| 11, | |
| 11 | |
| ], | |
| "verification_seconds": 0.09793207608163357 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 455682, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 5.551115123125783e-17, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.49901099119769765, | |
| "accuracy": 0.5366674142184347, | |
| "balanced_accuracy": 0.5248976016389348, | |
| "worst_class_f1": 0.2107843137254902 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fd31fa757b80f2f3e9293a0644895165773fc9cd67c265449ab5a7c1e736d6db", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 11 | |
| ], | |
| "verification_seconds": 0.10844338033348322 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 590055, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5129585454890183, | |
| "accuracy": 0.5566270464229648, | |
| "balanced_accuracy": 0.5378205618305084, | |
| "worst_class_f1": 0.2288135593220339 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "17c70707b59020594e415426a34af60b624aacd54716082cb0ce5277954bf4ff", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 13, | |
| 11, | |
| 8, | |
| 14 | |
| ], | |
| "verification_seconds": 0.11518295481801033 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 2209661888, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5301043138101286, | |
| "accuracy": 0.5714285714285714, | |
| "balanced_accuracy": 0.5399403495460648, | |
| "worst_class_f1": 0.2031063321385902 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3bb3447bf3d2446e4c4123cd2a4a5dd6dab50a51b38d01d4b3bb14a9dd1accd2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 4, | |
| 11, | |
| 8 | |
| ], | |
| "verification_seconds": 4.420359159819782 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 587878, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5020630813880627, | |
| "accuracy": 0.5415510928658605, | |
| "balanced_accuracy": 0.5286122763951864, | |
| "worst_class_f1": 0.19607843137254902 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "44d6aac704b945e7ff7ec9e5ba5b47d45db1db7633bcbd7de43740b7abbaa5f8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.11303388141095638 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 587116, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.4988862797334431, | |
| "accuracy": 0.5389992233440586, | |
| "balanced_accuracy": 0.5251543776171204, | |
| "worst_class_f1": 0.18507462686567164 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "dadf463570cbb51bb37ff17643f647d22434739e1fd95f28e907ddf52983100d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 11, | |
| 11, | |
| 8, | |
| 11 | |
| ], | |
| "verification_seconds": 0.1164759211242199 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "harmes_right_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 2209648832, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5176601134537039, | |
| "accuracy": 0.5586375235770553, | |
| "balanced_accuracy": 0.5295711176752078, | |
| "worst_class_f1": 0.19439868204283361 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d6582b4369b37f15f5b67c4c5553536fb5f11a55e15632a8119a77d2ee4e1314", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 4, | |
| 11, | |
| 8 | |
| ], | |
| "verification_seconds": 4.430933751165867 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 128945, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6096479050548741, | |
| "accuracy": 0.8319918044782673, | |
| "balanced_accuracy": 0.5923564159046437, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "39a72f089a60c8d2136b1cdb37d6ff4fce32c67244d8b4d6446fe295ae3cb8a0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.07963023521006107 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 156083, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.612918638732734, | |
| "accuracy": 0.845602224498756, | |
| "balanced_accuracy": 0.6040208282401819, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b884279bc5116301f86f9b9b047d6f39f83e2e70bfaa0cd5222de24f83fc07b2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.45693342853337526 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 128949, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6096479050548741, | |
| "accuracy": 0.8319918044782673, | |
| "balanced_accuracy": 0.5923564159046437, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "651faac607a3a3956c8a909c4870272b49bb201510102066fb552d6ec61ce156", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.08445706032216549 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 128949, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5928265629244448, | |
| "accuracy": 0.8043303929430633, | |
| "balanced_accuracy": 0.5810361579109827, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6c67e257799eb3499b65b1b828816e1037d87739dc13515f9f4ad8b63c1c6a8e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.0769574511796236 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 338789, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5121815033880968, | |
| "accuracy": 0.8032611601176156, | |
| "balanced_accuracy": 0.4962577766609905, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "7964b0b8c3e4c7ec761024654e6a14184907702f09386733bf025a7210105825", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.1298271780833602 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 340825, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5351559377826424, | |
| "accuracy": 0.8064688585939588, | |
| "balanced_accuracy": 0.5059498177634308, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4b42b63d5c748bba1e5c7ad609b3add2aa005086dbbf7df23ce0d3289d228211", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.13942116685211658 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 169676, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6116900772477086, | |
| "accuracy": 0.8550638196349756, | |
| "balanced_accuracy": 0.6305240862826924, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "516514293771fba294dc22e9f4c7efbccdaed62bd58283a201cf01591d03660f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 2 | |
| ], | |
| "verification_seconds": 0.07959313318133354 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 342837, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.587618207448066, | |
| "accuracy": 0.8846475008946678, | |
| "balanced_accuracy": 0.5780258345781247, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "978a3d968f4687a3eecb529adbe28f9a88c6a7c8409af7f49e2d4fab7f31ba27", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.13611499685794115 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 128949, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6320530481952188, | |
| "accuracy": 0.8665155672193725, | |
| "balanced_accuracy": 0.6538620348912504, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "81d368579274b11fadb3b8f92ba55513ecf36c70c2cf638fb8895a34a4419e1d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.07920224126428366 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 128945, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6098265825969379, | |
| "accuracy": 0.865244322092223, | |
| "balanced_accuracy": 0.6081597599365783, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b2b4e2c420fd84f538b1cd9e327354e336a78e2f6ad7c744cd8706e3007b61a2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.0839704629033804 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 338435, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.5497180265570948, | |
| "accuracy": 0.8531314521679284, | |
| "balanced_accuracy": 0.5354421050786498, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3ad8b51c6de3b613bff453f4991eedf86577f18c58db396b41430c409dbf9757", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.12315355986356735 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 128949, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6098265825969379, | |
| "accuracy": 0.865244322092223, | |
| "balanced_accuracy": 0.6081597599365783, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6df4173edc966f77135834c37ddde485c62dd8a7489cdf08d71302625756b309", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.08461639657616615 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 128945, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8173963795908751, | |
| "accuracy": 0.882277761999432, | |
| "balanced_accuracy": 0.8327524147291605, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4d739e8c0c327536274dd2edb686f25c1e1ff652b33d34572941cd1795429b7b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.07930546812713146 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 339455, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7775502482937635, | |
| "accuracy": 0.9034365237148537, | |
| "balanced_accuracy": 0.7482305404696603, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a92a42dba50c70fda2bcf31af9d5ea060d0e2cb14b427c970dd581f56464ee72", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.12170023191720247 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "iuwds_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 128949, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8173963795908751, | |
| "accuracy": 0.882277761999432, | |
| "balanced_accuracy": 0.8327524147291605, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a393b658599dd887f9a42ad8d4ec97a8306cd036643de52f930d8ab567be5f24", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 3, | |
| 3, | |
| 3, | |
| 3 | |
| ], | |
| "verification_seconds": 0.08189722429960966 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed42/model.pkl", | |
| "bytes": 195692, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7428036236211897, | |
| "accuracy": 0.7366336633663366, | |
| "balanced_accuracy": 0.7463115099427031, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "35cd0793db79f56c657858da9d950bcd00c29755da0cc46324942c3d41acbbd1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.10329711902886629 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed43/model.pkl", | |
| "bytes": 386577, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7530918965544832, | |
| "accuracy": 0.7485148514851485, | |
| "balanced_accuracy": 0.7627043309740479, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "86b1f9bd71d34a67a3985181fb212ae5dcc4a3582f3284a3e4e33eb71ef91128", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.10184361599385738 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed44/model.pkl", | |
| "bytes": 197143, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.777767255892256, | |
| "accuracy": 0.7762376237623763, | |
| "balanced_accuracy": 0.7874346983485001, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "8f894de960f765a8739605e6bf46bed77abd05ce0cc4bfaacdcdee07791209e1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.06998705118894577 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed42/model.pkl", | |
| "bytes": 194609, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7958838404465718, | |
| "accuracy": 0.816247582205029, | |
| "balanced_accuracy": 0.8247764415664509, | |
| "worst_class_f1": 0.5833333333333334 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6be3c89830b29436a93355544ab8c9e74d31a11c22c5488573952b93f79e38f6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.06705038249492645 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed43/model.pkl", | |
| "bytes": 195124, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7607287596852985, | |
| "accuracy": 0.7543520309477756, | |
| "balanced_accuracy": 0.7665576406325713, | |
| "worst_class_f1": 0.45569620253164556 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "edbf66cb6e12b62cb7adc6a093779254ba6d098e467ebbd80726961354e293c6", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.09524028841406107 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed44/model.pkl", | |
| "bytes": 197143, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.833983599459939, | |
| "accuracy": 0.8317214700193424, | |
| "balanced_accuracy": 0.8397702744372494, | |
| "worst_class_f1": 0.47191011235955055 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c8a1ff51de6357d56b65587d5ea4760668f86d6326649fecde826baa6bf43f89", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.07438390795141459 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed42/model.pkl", | |
| "bytes": 788723, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8583320296777582, | |
| "accuracy": 0.859344894026975, | |
| "balanced_accuracy": 0.8676731078904992, | |
| "worst_class_f1": 0.6268656716417911 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a124969a171b4ab7fea49bc4e000e578b62d9192bcf3e99a00e8262c985996ce", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 8, | |
| 8, | |
| 8, | |
| 8 | |
| ], | |
| "verification_seconds": 0.10094030201435089 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed43/model.pkl", | |
| "bytes": 6921228, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9111111111111111, | |
| "accuracy": 0.9113680154142582, | |
| "balanced_accuracy": 0.9166666666666666, | |
| "worst_class_f1": 0.6666666666666666 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d26957d48806f1c581b02fbb2ce1d9804b0c119afb4ff19f5e18074c04f3e094", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.11249503772705793 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed44/model.pkl", | |
| "bytes": 793838, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8565516714672254, | |
| "accuracy": 0.8574181117533719, | |
| "balanced_accuracy": 0.8658212560386475, | |
| "worst_class_f1": 0.6268656716417911 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "29ea043d1216b22c5db967c576795c72dfbaa855dd0a0daed56bd362d6208493", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 1, | |
| 1, | |
| 1, | |
| 1 | |
| ], | |
| "verification_seconds": 0.10749175865203142 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed42/model.pkl", | |
| "bytes": 194609, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": -1.1102230246251565e-16 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9702950466044578, | |
| "accuracy": 0.9686888454011742, | |
| "balanced_accuracy": 0.9701297607010448, | |
| "worst_class_f1": 0.9113924050632911 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "d8022f233350561fc606adbbb20334d0587ba8d1708c268ca2df0ea3d3d7112c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.08102338202297688 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed43/model.pkl", | |
| "bytes": 554981, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8595948062369086, | |
| "accuracy": 0.8532289628180039, | |
| "balanced_accuracy": 0.8617290192113246, | |
| "worst_class_f1": 0.5348837209302325 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b3f038f9a4d1e2cd1901c8d8f3acb225525868a12ec6fb2aeb82c3b3ef85297b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.1226800736039877 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed44/model.pkl", | |
| "bytes": 612279, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8849608669246121, | |
| "accuracy": 0.8806262230919765, | |
| "balanced_accuracy": 0.8867179121855563, | |
| "worst_class_f1": 0.6666666666666666 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "337f75cf5520357b464199ddeb5123a7679f98e14fe336f555e99cbe798c383b", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.2504494357854128 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed42/model.pkl", | |
| "bytes": 30675898, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7478307728307728, | |
| "accuracy": 0.7594433399602386, | |
| "balanced_accuracy": 0.7422123541887592, | |
| "worst_class_f1": 0.4444444444444444 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "bdccff3234086735c8efb97bf95b41ba974f47277cecb6fc82cee8f6740a1100", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.15433111507445574 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed43/model.pkl", | |
| "bytes": 30830438, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8414983164983165, | |
| "accuracy": 0.8489065606361829, | |
| "balanced_accuracy": 0.8333333333333334, | |
| "worst_class_f1": 0.46464646464646464 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "44b3cf13580087f18e4c668aacea2a0e039b3bcce22129cfc25dc37e0c768753", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.15820361580699682 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "mhealth_right_lower_arm_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed44/model.pkl", | |
| "bytes": 30621176, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8392896169265338, | |
| "accuracy": 0.8469184890656064, | |
| "balanced_accuracy": 0.8315217391304349, | |
| "worst_class_f1": 0.46464646464646464 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0b525eb01e391ac3c106189748e4774e7051f1072797c28a5ce394e6a4f09249", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 6, | |
| 6, | |
| 6, | |
| 6 | |
| ], | |
| "verification_seconds": 0.15374513156712055 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 321909179, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.782808883037621, | |
| "accuracy": 0.7777777777777778, | |
| "balanced_accuracy": 0.7664396920134046, | |
| "worst_class_f1": 0.47619047619047616 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "712daccfd21c9116e7b220a7e4d733fde83e22603313fa5c47013ae707f6600d", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.715706367045641 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 154309196, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7863210822620976, | |
| "accuracy": 0.7899305555555556, | |
| "balanced_accuracy": 0.7717697486014151, | |
| "worst_class_f1": 0.5317919075144508 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "4da96f750e45494cfac7b52690d7d888adfef142eade782702938801620d8c96", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.3765896176919341 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 322817461, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7406621280602123, | |
| "accuracy": 0.7579861111111111, | |
| "balanced_accuracy": 0.7268762175805494, | |
| "worst_class_f1": 0.47619047619047616 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "fd9bc35b63ca0e620a4d8eaedf5ccee58908a2d06fd92f1c6c8a5f8a0da2b8ff", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7351966826245189 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 297563171, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6989074572715867, | |
| "accuracy": 0.7000842459983151, | |
| "balanced_accuracy": 0.6760508882503385, | |
| "worst_class_f1": 0.37383177570093457 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "20467c0f35c90da5501b1c5e905192ca30b0011f6a0b806bbe80ba40c88a6d4a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.6626511849462986 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 295829813, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6940231797042055, | |
| "accuracy": 0.698680146026397, | |
| "balanced_accuracy": 0.6768125701153013, | |
| "worst_class_f1": 0.368 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6b0dd5effa866c8a1b29138c87f1a2055a54407a47522fd9e796403938c60fba", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.6669249190017581 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 141851858, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7247158739945666, | |
| "accuracy": 0.7245155855096883, | |
| "balanced_accuracy": 0.7101332033880609, | |
| "worst_class_f1": 0.4036697247706422 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "41b7e3278f9ac8c30e9788240174fc0fe373634d1bed67671b1d59f27b713c75", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.3645658940076828 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 153892642, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7831926412023957, | |
| "accuracy": 0.7792465300727033, | |
| "balanced_accuracy": 0.7655895754624013, | |
| "worst_class_f1": 0.46956521739130436 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "972b07f92063078910321412edc2b110d833ae0d5afbdd23f3664a72e9fd21e8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.420327321626246 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 153585740, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7977890388559085, | |
| "accuracy": 0.7868473231989425, | |
| "balanced_accuracy": 0.7789270668058506, | |
| "worst_class_f1": 0.4444444444444444 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b01c9d65130a785c746ee0d9705ee0a27eb52f58da411e93babb7384ff13a88f", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.41771274618804455 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 317604151, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7959494363633283, | |
| "accuracy": 0.7762723066754792, | |
| "balanced_accuracy": 0.7728997159742116, | |
| "worst_class_f1": 0.4915254237288136 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6f0c6b5613e76df8b1424c9301da581d80049278ae7cecd53d6b43268d034f54", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7287461068481207 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 152149574, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7777656452114684, | |
| "accuracy": 0.7796555086122847, | |
| "balanced_accuracy": 0.7665927790732726, | |
| "worst_class_f1": 0.3953488372093023 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5246675fb1a261b6a6dbeef965cc6adfed66bb83873766ab4202da68bc22f685", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.4086458645761013 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 152113740, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.76412050853765, | |
| "accuracy": 0.7796555086122847, | |
| "balanced_accuracy": 0.7541487816381051, | |
| "worst_class_f1": 0.43243243243243246 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b3cc4c609f8c65135908f512fd3f194b51dbabc49f780e18451de7911e041df0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.38193516433238983 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 152259654, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7785577846318382, | |
| "accuracy": 0.7890802729931752, | |
| "balanced_accuracy": 0.7690620670775831, | |
| "worst_class_f1": 0.44086021505376344 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "333a97f3219b97286c436e88ec4777d7825c319ab3f463c23b66d70f4998ea78", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 23, | |
| 23, | |
| 23, | |
| 23 | |
| ], | |
| "verification_seconds": 0.4068180527538061 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 319455669, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7699247674457195, | |
| "accuracy": 0.7750320924261874, | |
| "balanced_accuracy": 0.7602778385960877, | |
| "worst_class_f1": 0.3711340206185567 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "832929b8272271bc6b27792732d474ff816d1e7d6a357c558f43dbe33d3b4fb1", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.7088611256331205 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 149126524, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7920311703422366, | |
| "accuracy": 0.7907573812580231, | |
| "balanced_accuracy": 0.7814562061777864, | |
| "worst_class_f1": 0.42857142857142855 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "51e33652aa3f4d4d6f61e8aabe58672305870b0a45568de614155507f0203d8a", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.3706858614459634 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "paal_adl_wrist_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 151948870, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7804989539681273, | |
| "accuracy": 0.7894736842105263, | |
| "balanced_accuracy": 0.7705393743488225, | |
| "worst_class_f1": 0.3958333333333333 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b5dfdd5b52fc55364dcd0441c1fb7cd39b1663a1978d04b6b1d68a417738ff88", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 20, | |
| 20, | |
| 20, | |
| 20 | |
| ], | |
| "verification_seconds": 0.3796448949724436 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed42/model.pkl", | |
| "bytes": 194655, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.926126828018268, | |
| "accuracy": 0.9237976264834479, | |
| "balanced_accuracy": 0.9259866160332428, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "888da1e0d42747468ef5c32fd88620d2f20a1532cc44ef9a2e78001430d36ed2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08373156283050776 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed43/model.pkl", | |
| "bytes": 99194, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9373390528660345, | |
| "accuracy": 0.9400374765771393, | |
| "balanced_accuracy": 0.9314587981199641, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "082730a7b8c759e90da1ece504c471b8277f2fb9e14d9111b26fd7faf11401dc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.062161377631127834 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed44/model.pkl", | |
| "bytes": 196468, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9106480562075838, | |
| "accuracy": 0.905683947532792, | |
| "balanced_accuracy": 0.9113062332361342, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6a9ca05ac9ea12ce9141c63cd033d1d6dced14ac4ee24e2868f6dd69c125f657", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.08026024792343378 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed42/model.pkl", | |
| "bytes": 519536, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9607069422188861, | |
| "accuracy": 0.9529616724738676, | |
| "balanced_accuracy": 0.9629197423085207, | |
| "worst_class_f1": 0.8265895953757225 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "72256fd2b7de2b50a1665ec7cd03547aec9f69d6a7d2953fd578575ab7420509", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.12377436179667711 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed43/model.pkl", | |
| "bytes": 198347, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9424888593632138, | |
| "accuracy": 0.9419279907084785, | |
| "balanced_accuracy": 0.9454272899289901, | |
| "worst_class_f1": 0.78 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "5820902bf30bd558226d0c2b6e0195d1bfc841411c0932b6e335fc86b80a2828", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11050238087773323 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed44/model.pkl", | |
| "bytes": 694157, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 1.1102230246251565e-16, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9446475832944937, | |
| "accuracy": 0.9413472706155633, | |
| "balanced_accuracy": 0.9465516272892364, | |
| "worst_class_f1": 0.829971181556196 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "44fb853e3a01699c4db66ebb77e1a58d9d2c6669d67de80d9cd352f3e335bfe8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.1427327049896121 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed42/model.pkl", | |
| "bytes": 62795362, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8304680692494896, | |
| "accuracy": 0.8458029197080292, | |
| "balanced_accuracy": 0.8113854490423412, | |
| "worst_class_f1": 0.5358851674641149 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "62d7318cf47d9efb6f2dd3e6ca60a8c65d299f33df64875e2fb0b5bdae1acbf8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.24092491995543242 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed43/model.pkl", | |
| "bytes": 63752910, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8248201686199893, | |
| "accuracy": 0.8412408759124088, | |
| "balanced_accuracy": 0.8068507107128275, | |
| "worst_class_f1": 0.5384615384615384 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e1e5bca8da318977784e09f71c58727705efd5b86c185d8a831cf08c6cfe4add", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.2815354336053133 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed44/model.pkl", | |
| "bytes": 62588844, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8211673302288754, | |
| "accuracy": 0.8403284671532847, | |
| "balanced_accuracy": 0.8039473847047613, | |
| "worst_class_f1": 0.5373134328358209 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "508284c3f5768f28860e48ea9ea057daef15ab81d29ef6d8c688855020581ff2", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.23504982236772776 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed42/model.pkl", | |
| "bytes": 197040, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": -1.1102230246251565e-16, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": -1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9561457774540599, | |
| "accuracy": 0.9571734475374732, | |
| "balanced_accuracy": 0.9573821531111663, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c53f9bc0825d8cab7fa0ba36c553909cde9373814972c6e31681651047106a63", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.11584278754889965 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed43/model.pkl", | |
| "bytes": 99194, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 1.1102230246251565e-16, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9651552354318038, | |
| "accuracy": 0.9657387580299786, | |
| "balanced_accuracy": 0.9661267981171673, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cb26ba3b108f180b33bf984874f47e6270771b5d673ef51fb723691ea21f1f89", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.0970451133325696 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed44/model.pkl", | |
| "bytes": 855341, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": -1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.9767973966100094, | |
| "accuracy": 0.9775160599571735, | |
| "balanced_accuracy": 0.9768469624234878, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f745b4bde373b714c90c335b8c5cd65b3986e78241eb48a6959b1afe626fd144", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.13156522531062365 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed42/model.pkl", | |
| "bytes": 56033925, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7675404704390513, | |
| "accuracy": 0.7796052631578947, | |
| "balanced_accuracy": 0.7538230037404436, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "463314bab2079ab16263df8e7c5799c8269b32556b58448ce9c3f56c1b304585", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.2576223621144891 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed43/model.pkl", | |
| "bytes": 56014725, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7623319002781765, | |
| "accuracy": 0.7763157894736842, | |
| "balanced_accuracy": 0.7505501469112027, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "0dfb08aea1d728fce3389d61e02d54faef0a3f9a7ab613f2292ee037997060b7", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.24346212297677994 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "pamap2_hand_imu_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed44/model.pkl", | |
| "bytes": 56797677, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 1.1102230246251565e-16, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8976064609725526, | |
| "accuracy": 0.9133771929824561, | |
| "balanced_accuracy": 0.8840144162810198, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6070cd99340fe4ff4f92fd11653fee008f03b3648d0406b8b1a52dbb24069af3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 4, | |
| 4, | |
| 4, | |
| 4 | |
| ], | |
| "verification_seconds": 0.24854245502501726 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed42/model.pkl", | |
| "bytes": 247303, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.774132262918264, | |
| "accuracy": 0.7813700384122919, | |
| "balanced_accuracy": 0.7826521562757212, | |
| "worst_class_f1": 0.20817120622568094 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6697b406a31e23bea6b952561315e8473da41831f2c3d30d6b16fd958eb877c0", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.11846522241830826 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed43/model.pkl", | |
| "bytes": 1458792410, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7181558240095015, | |
| "accuracy": 0.7084507042253522, | |
| "balanced_accuracy": 0.7099612955750377, | |
| "worst_class_f1": 0.24347826086956523 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "a18797baf8a57ae9d67afe35877e16545ffc71a22a3e9db470615d891e277787", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.8971761520951986 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 0, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed44/model.pkl", | |
| "bytes": 1444945600, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.75091127432255, | |
| "accuracy": 0.742573623559539, | |
| "balanced_accuracy": 0.7438359518494067, | |
| "worst_class_f1": 0.29577464788732394 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "51fb1abc29411f60dedc5d31c452b05a8b6ab852c7267e5ce83b385ee30abc70", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.8266507973894477 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed42/model.pkl", | |
| "bytes": 1406824768, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.688070847675892, | |
| "accuracy": 0.6940909663600857, | |
| "balanced_accuracy": 0.6928882984232319, | |
| "worst_class_f1": 0.4200772200772201 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "cce00ee54265fd663eb667bf5dc0e95effaea966d64ac748b8ed1ff4d188ebd4", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.7140465062111616 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed43/model.pkl", | |
| "bytes": 1409985088, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6931772279311114, | |
| "accuracy": 0.6961068413758347, | |
| "balanced_accuracy": 0.6949493182373486, | |
| "worst_class_f1": 0.39651416122004357 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "3da4f861e14224e8cd3d13f3937b341320acfc02b616be0df87bcc63a5cdf558", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.7301150457933545 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 1, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed44/model.pkl", | |
| "bytes": 1411856544, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 8.326672684688674e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6725085349439842, | |
| "accuracy": 0.6828146654907395, | |
| "balanced_accuracy": 0.6791010276060436, | |
| "worst_class_f1": 0.17481203007518797 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ded420ffe0a3aa5e6334b29b25d289202890519cfa1b8ab889ff46e9ca079592", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.6930082300677896 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed42/model.pkl", | |
| "bytes": 1348189179, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8465069012503283, | |
| "accuracy": 0.8521433591004919, | |
| "balanced_accuracy": 0.8542029870340069, | |
| "worst_class_f1": 0.43606255749770007 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "6de1903ee0fc8da851336f261ae566cd1a945f62a4124899188d869387491f71", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.5633741300553083 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed43/model.pkl", | |
| "bytes": 1350693119, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8476215612403961, | |
| "accuracy": 0.8535488404778636, | |
| "balanced_accuracy": 0.8553756774503352, | |
| "worst_class_f1": 0.421875 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "e11c85e09a30475ece633f18dd44eb2f2279e55ac903e386b2448f30c6ecf6f3", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.5569702116772532 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 2, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed44/model.pkl", | |
| "bytes": 1342968119, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.8392289410102078, | |
| "accuracy": 0.8493323963457484, | |
| "balanced_accuracy": 0.8512439984972445, | |
| "worst_class_f1": 0.3232533889468196 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "baa4b60ce41b27b2cbf0eae874ed6b4aef19ce3c464167a3f641c8d10888c9d8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.5231892075389624 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed42/model.pkl", | |
| "bytes": 1402628768, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 5.551115123125783e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.70277937521498, | |
| "accuracy": 0.7029740008728724, | |
| "balanced_accuracy": 0.7036263928622836, | |
| "worst_class_f1": 0.41839080459770117 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "c1fc3a1fa5bb8666f48ab3a3c198c59c5a5da8a006d64341ed8b2488c4d3838c", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.695559806190431 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed43/model.pkl", | |
| "bytes": 1402410746, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6994422646652237, | |
| "accuracy": 0.695180497537253, | |
| "balanced_accuracy": 0.695636607852856, | |
| "worst_class_f1": 0.4831130690161527 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "aafccdb6f0cb69725ae4197d1db86d820a4db509aada4a9922c4b2dddfc16dcc", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.6815116200596094 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 3, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed44/model.pkl", | |
| "bytes": 1401175926, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.6921004642618573, | |
| "accuracy": 0.6919384001496353, | |
| "balanced_accuracy": 0.6924893673415535, | |
| "worst_class_f1": 0.2857142857142857 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "b9984a03b030ccf5a06961c0fb690909e20be3b0c4aea245c76a29995ef1320e", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 2.714278470724821 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 42, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed42/model.pkl", | |
| "bytes": 642964, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 2.7755575615628914e-17 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7263448253080516, | |
| "accuracy": 0.7255996953928163, | |
| "balanced_accuracy": 0.7267649113642366, | |
| "worst_class_f1": 0.13160518444666003 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "ecbb4e620aef4f605b9a0657077e8589d0c086d0886507bd4d344048bb336d59", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.12464579846709967 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 43, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed43/model.pkl", | |
| "bytes": 643937, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7221425984704912, | |
| "accuracy": 0.729343825358548, | |
| "balanced_accuracy": 0.7322211437109059, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "57c7e0ea59741b1dc82941501812441efc800bee4b335c72b05ae7bce4cd0c54", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.11194508150219917 | |
| }, | |
| { | |
| "method": "wisp_random", | |
| "dataset": "wisdm_watch_accel_v1", | |
| "fold": 4, | |
| "seed": 44, | |
| "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed44/model.pkl", | |
| "bytes": 643950, | |
| "load": "passed", | |
| "synthetic_prediction": "passed", | |
| "frozen_prediction": { | |
| "status": "not_run" | |
| }, | |
| "saved_prediction_metrics": { | |
| "status": "passed", | |
| "paper_metric_deltas": { | |
| "macro_f1": 0.0, | |
| "accuracy": 0.0, | |
| "balanced_accuracy": 0.0, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "metrics": { | |
| "macro_f1": 0.7196462587809807, | |
| "accuracy": 0.7240766594745526, | |
| "balanced_accuracy": 0.7256429459103105, | |
| "worst_class_f1": 0.0 | |
| }, | |
| "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." | |
| }, | |
| "sha256": "f7822a30c9af1ab31909c8eee97cd6685b0825d809e80ae86a3378ea161df9f8", | |
| "source_status": "load_and_smoke_verified", | |
| "synthetic_prediction_class_ids": [ | |
| 10, | |
| 10, | |
| 10, | |
| 10 | |
| ], | |
| "verification_seconds": 0.10468210000544786 | |
| } | |
| ], | |
| "completed": 360, | |
| "status_counts": { | |
| "load_and_smoke_verified": 360 | |
| }, | |
| "full_frozen_prediction_status_counts": { | |
| "passed": 6, | |
| "not_run": 354 | |
| }, | |
| "saved_prediction_metric_status_counts": { | |
| "passed": 360 | |
| }, | |
| "public_model_text_hygiene": { | |
| "files_checked": 363, | |
| "passed": true, | |
| "failed_relative_paths": [], | |
| "scope": "Public model JSON/CSV and validation report only; no credential file is read." | |
| } | |
| } | |