Scikit-learn
human-activity-recognition
wearable
wrist
time-series
cpu
scikit-learn
WISP / validation /model_validation.json
Zipeng365's picture
Add files using upload-large-folder tool
10ef792 verified
Raw History Blame Contribute Delete
460 kB
{
"generated_at_utc": "2026-10-02T12:57:11.649548+00:00",
"slurm_job_id": "55937950",
"environment": {
"wisp-har": "0.1.0",
"numpy": "2.3.5",
"scipy": "1.17.1",
"scikit-learn": "1.7.2",
"joblib": "1.5.3",
"threadpoolctl": "3.6.0",
"python": "3.12.13"
},
"code_provenance": {
"release_wheel": "wisp_har-0.1.0-py3-none-any.whl",
"release_wheel_sha256": "89c1cccabaf095b66fe8d4b82a007b868316302ca613d6bfe1a9a8b2995876b0",
"installed_checkpoint_api_sha256": "261bf3c45c2ef7f5a21ff188533aa19a46a6774e261fb6ee1d0f25cfb794b47f",
"installed_package_source_sha256": {
"wisp/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b",
"wisp/ablations.py": "d92d2cc5a369ca5d6a3baac156355b0b21b7816778ba9e40eb0c898bafaf8354",
"wisp/cis_algorithms.py": "c7b9b262d9bceed5f6a31821a26aeb044350b6af759d100876b3967a36a93be5",
"wisp/core/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b",
"wisp/core/data.py": "a39ff2a5a8cf67c9e59e944bb065e8274ccd7d18b319efeba46044e6cc372306",
"wisp/core/metrics.py": "113b8461fa7edd883cebd961b77e15dde9ffc8c3dbf538e313c246e350838c7c",
"wisp/cpu/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b",
"wisp/cpu/algorithms.py": "b1c855877ef70ccc4abb0726c626933053120272fcfeb8d5129603c6f59500e4",
"wisp/cpu/features.py": "aab2b76f8dbf7d7faf9b4b1785b2596a2596ba4a805a18e0c856b0ea581ef668",
"wisp/cpu/hmm.py": "2c42f3c2076aed5143f4d7b850f88b0b6486ceab72ca291184a9c5859e0ddae0",
"wisp/cpu/probability.py": "1232011f6eac18d45fd13580f96e79b9b8b088f8661a863130fe7084b1e67cc9",
"wisp/registry.py": "a7a96dfa10ca77d95032e4e62e1a526fc53503ee4618fa6194a73a97ae523828",
"wisp/search/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b",
"wisp/search/cascade.py": "2f2d19573a69afe12015657f13a89caf0d28466ffd703a31f2afd951c7d1e245",
"wisp/search/controller.py": "895240d4ba5b79088766a972a76112d15fa40d4e6fa4f5440bf47b6f2a0ec451",
"wisp/search/dataset_fingerprint.py": "8e976b35e37534c4fd795d529d63ac01cb0e95b87df4f090791d0d6611fb08ca",
"wisp/search/grammar.py": "a038e582182aa1bed8b03a173f02b6977730dbcde07f44a3cd721f87e6d08971",
"wisp/search/motifs.py": "5d0375afbb4cec325ccb0d476f27c797d0e335e6d4e292a16b9330f3565a7a44",
"wisp/search/operator_program.py": "654975111d32566a712dd7ae8e386c577f9fc0196dad8582939ab018f32ff8ec",
"wisp/search/posterior.py": "0fb27135a136f9fc77c0bdc6facb959a85587e629e4d99c40f95ff4262507e1f",
"wisp/search/profiles.py": "0acb26c966b72721b42f9e90183186ddb1db713a42d451d886103e0ebf7a16a6",
"wisp/utils/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b",
"wisp/utils/jsonl.py": "ab9819981cf6f56436edada19b01cfb11d78a3af40d05504c7b9740384441daa",
"wisp_release/__init__.py": "c4c3a0ae71ab4eb449ac357d5cddaa40b897a6bdc683c5713e019f3fcc5027a2",
"wisp_release/checkpoints.py": "261bf3c45c2ef7f5a21ff188533aa19a46a6774e261fb6ee1d0f25cfb794b47f",
"wisp_release/cli.py": "4d4f7106bd5eca4c8496c55b0a03c967df589cf1ec2e7f365db0ec6798c980a5",
"wisp_release/data.py": "5dab92de198c7bca28fb620f830168be3bf936bb3c8a8c9ca4dd5762c6cf0a89",
"wisp_release/methods.py": "2f1eac83ccb9fe7c8d0e2a7520f746df7f522470af8dab65322a261acc17abcd",
"wisp_release/scoring.py": "aa1e89dd83b7295e0dbe087b441c13bae3247a04129f4710006da9bc49386dc5",
"wisp_release/selection.py": "7077a0bfef5c9ec3371c64b0bfdef1d285ee9f9902142e2f5676d0f0462d72b6"
},
"installed_sources_exactly_match_release_wheel": true,
"notice": "Wheel digest identifies the built release artifact; source hashes identify the actual installed implementation used by these model checks."
},
"expected_available_checkpoints": 360,
"baselines_installed_or_tested": false,
"training_runs_performed": 0,
"records": [
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 56197147,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "passed",
"n_test_windows": 861,
"exact_frozen_predictions_equal": true,
"original_input_exact_frozen_predictions_equal": true,
"original_input_differing_prediction_count": 0,
"differing_prediction_count": 0,
"original_and_fresh_X_exactly_equal": true,
"original_and_fresh_X_max_absolute_difference": 0.0,
"time_coordinate_check": {
"raw_time_values_exactly_equal": false,
"per_subject_stable_chronological_permutation_identical": {
"f3": true,
"m1": true,
"m3": true,
"m7": true
},
"original_first5": [
0.0,
2.5,
5.0,
7.5,
0.0
],
"fresh_first5": [
0.0,
1.0,
2.0,
3.0,
0.0
],
"explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled."
},
"metrics": {
"macro_f1": 0.46857990441875785,
"accuracy": 0.70267131242741,
"balanced_accuracy": 0.45147912566266135,
"worst_class_f1": 0.0
},
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.46857990441875785,
"accuracy": 0.70267131242741,
"balanced_accuracy": 0.45147912566266135,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c8089b71a53b67ff1eee86b112362230e51a34a25664cf0652786ab7db00188f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 5.281506538391113
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 56614267,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "passed",
"n_test_windows": 861,
"exact_frozen_predictions_equal": true,
"original_input_exact_frozen_predictions_equal": true,
"original_input_differing_prediction_count": 0,
"differing_prediction_count": 0,
"original_and_fresh_X_exactly_equal": true,
"original_and_fresh_X_max_absolute_difference": 0.0,
"time_coordinate_check": {
"raw_time_values_exactly_equal": false,
"per_subject_stable_chronological_permutation_identical": {
"f3": true,
"m1": true,
"m3": true,
"m7": true
},
"original_first5": [
0.0,
2.5,
5.0,
7.5,
0.0
],
"fresh_first5": [
0.0,
1.0,
2.0,
3.0,
0.0
],
"explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled."
},
"metrics": {
"macro_f1": 0.5008386012173559,
"accuracy": 0.7096399535423926,
"balanced_accuracy": 0.4878142449093867,
"worst_class_f1": 0.0
},
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5008386012173559,
"accuracy": 0.7096399535423926,
"balanced_accuracy": 0.4878142449093867,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "874f0b547ea16f5520641d01894fc4836451d749795dec2e805544925b56944b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 1.4547859011217952
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 213724,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "passed",
"n_test_windows": 861,
"exact_frozen_predictions_equal": true,
"original_input_exact_frozen_predictions_equal": true,
"original_input_differing_prediction_count": 0,
"differing_prediction_count": 0,
"original_and_fresh_X_exactly_equal": true,
"original_and_fresh_X_max_absolute_difference": 0.0,
"time_coordinate_check": {
"raw_time_values_exactly_equal": false,
"per_subject_stable_chronological_permutation_identical": {
"f3": true,
"m1": true,
"m3": true,
"m7": true
},
"original_first5": [
0.0,
2.5,
5.0,
7.5,
0.0
],
"fresh_first5": [
0.0,
1.0,
2.0,
3.0,
0.0
],
"explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled."
},
"metrics": {
"macro_f1": 0.5804510063302952,
"accuracy": 0.7061556329849012,
"balanced_accuracy": 0.5332788991510314,
"worst_class_f1": 0.0
},
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5804510063302952,
"accuracy": 0.7061556329849012,
"balanced_accuracy": 0.5332788991510314,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0729c0358473d08aa0f07abd071f248aca98c576592b692872feaddc172090be",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 3.528837164863944
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 144715563,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "passed",
"n_test_windows": 861,
"exact_frozen_predictions_equal": true,
"original_input_exact_frozen_predictions_equal": true,
"original_input_differing_prediction_count": 0,
"differing_prediction_count": 0,
"original_and_fresh_X_exactly_equal": true,
"original_and_fresh_X_max_absolute_difference": 0.0,
"time_coordinate_check": {
"raw_time_values_exactly_equal": false,
"per_subject_stable_chronological_permutation_identical": {
"f3": true,
"m1": true,
"m3": true,
"m7": true
},
"original_first5": [
0.0,
2.5,
5.0,
7.5,
0.0
],
"fresh_first5": [
0.0,
1.0,
2.0,
3.0,
0.0
],
"explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled."
},
"metrics": {
"macro_f1": 0.4271967810219083,
"accuracy": 0.6806039488966318,
"balanced_accuracy": 0.45020425062766495,
"worst_class_f1": 0.0
},
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4271967810219083,
"accuracy": 0.6806039488966318,
"balanced_accuracy": 0.45020425062766495,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "943a9f9c75edb5b5425533d5fbbc376e39e838942d32562c7b46f0d83a5c653d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 5.605986746959388
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 435753,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.45512076153931874,
"accuracy": 0.5837837837837838,
"balanced_accuracy": 0.5301783700824152,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e1bde8c4ad01f3152f254f754e19c9ed214d790ca66450317b462ca941e52afd",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 0.09336361195892096
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 239385,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4543957425721672,
"accuracy": 0.5704504504504504,
"balanced_accuracy": 0.5233597226947379,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4f4a0d1fc0ab03fd159a8dd028bb870360eb0e7c89f779b36e050e2bb56a9c8d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 0.0711149936541915
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 195613,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4444367338790675,
"accuracy": 0.5585585585585585,
"balanced_accuracy": 0.5129764381142231,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5ef9f923e8d30c0a7148c0fea06920e1addff9d89cbe6ce1c161db4ebf5b7998",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
7,
7,
7,
7
],
"verification_seconds": 0.09933141898363829
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 108071,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6315284716337148,
"accuracy": 0.6824742268041237,
"balanced_accuracy": 0.5954227178234918,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a3646800e53bd55b507cd0545383377d8b1fd0d4e5e4e21b94e79de64c717bdd",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.0682113254442811
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 212796,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7140298291222251,
"accuracy": 0.7793814432989691,
"balanced_accuracy": 0.6659095889792745,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5ca91cbb72343bbe53d408cd5267fe1bf4e70a06e02b0afd65f2b6b8d6ef8c1e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07314625475555658
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 108151,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6091680266775356,
"accuracy": 0.5711340206185567,
"balanced_accuracy": 0.5892073482822214,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d12df30d0ab256561d9b583bfe910dfedb13138949b410bf219bc2ec7d622533",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07153434678912163
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 215198,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.39874385233957527,
"accuracy": 0.555992141453831,
"balanced_accuracy": 0.45164475266682325,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "650947efac20189d66cd0c69348da623efd950daf0ad759b404b6e72b64505df",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.0834163036197424
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 213799,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.38419952735589424,
"accuracy": 0.5520628683693517,
"balanced_accuracy": 0.4058504505983019,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bfb5d044c9caa32d3b496619b49e2e60251318e4100e771fccb2b364885acc5f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07257030159235
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 109750,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.37838383738902187,
"accuracy": 0.48919449901768175,
"balanced_accuracy": 0.42086382580606824,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "10415ecf33b468d8584ef8e41d076d63fc4cafb231db82cd1d682737a43ca307",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.06706883013248444
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 215136,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7160946161678324,
"accuracy": 0.803921568627451,
"balanced_accuracy": 0.7463469409476607,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c47064840f9187a5cfa3fc181f829aab1ba77c8754f79db9fcff50e9c62a98c2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07794979959726334
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 267079,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6873594627210284,
"accuracy": 0.8215686274509804,
"balanced_accuracy": 0.7955491343583891,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "19a43e6143a7572be809bec98c7e517eb332cfd030fe87b86f78490a0c8463fa",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 0.06673328019678593
},
{
"method": "wisp_evolution",
"dataset": "adl_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 265353,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.64200883315541,
"accuracy": 0.788235294117647,
"balanced_accuracy": 0.7487627012767595,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9d3b9fb55b2ff1b4f05643df09594a84438e24bbdfd8beb6cd0b95b045607e35",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 0.0789025230333209
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed42/model.pkl",
"bytes": 128855,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7773584180364441,
"accuracy": 0.8507273259787171,
"balanced_accuracy": 0.7698689175281773,
"worst_class_f1": 0.5946502057613169
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "efa2e8604213ef4989368eedc7e077746724d0e45eba640c93ed8a6df76b2bad",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.0911676436662674
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed43/model.pkl",
"bytes": 64391,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7707432326452675,
"accuracy": 0.8394025187933223,
"balanced_accuracy": 0.7700530056485361,
"worst_class_f1": 0.6070409134157945
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "feac6c70266738be626a463ec0bd7deb778bdd63e4acaaa0027692959aec4598",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.1280159205198288
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed44/model.pkl",
"bytes": 64845,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7583360117244344,
"accuracy": 0.8415503270526213,
"balanced_accuracy": 0.7514210173129554,
"worst_class_f1": 0.5420944558521561
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fc5accae13589359550c006de9301f3c37970d3332cdb9edfa166cb28b8491fa",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.0684780403971672
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed42/model.pkl",
"bytes": 347933090,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.6832491969161993,
"accuracy": 0.7742128337983261,
"balanced_accuracy": 0.6653189192827791,
"worst_class_f1": 0.44139650872817954
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3b55ea01da5f0a0d1832836c99428f6ae17ebfc6c18101aad6633fbc89d0a8c6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.8010620893910527
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed43/model.pkl",
"bytes": 130511,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.748965216157762,
"accuracy": 0.8414707054603427,
"balanced_accuracy": 0.7496277163543152,
"worst_class_f1": 0.48459958932238195
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a1f9521a70baff07ae0e2705279c268dae32f3bf01088791a01b740a2d0eee7b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09537813626229763
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed44/model.pkl",
"bytes": 592029080,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7095731117734799,
"accuracy": 0.7794938222399362,
"balanced_accuracy": 0.7155070471172339,
"worst_class_f1": 0.5171102661596958
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c77d246c5eec2ee3c943bbc43e366ebedfb234cfe3c19dbd2341daafec7efffa",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 1.529852494597435
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed42/model.pkl",
"bytes": 130236,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7293373556148361,
"accuracy": 0.8356660001904218,
"balanced_accuracy": 0.7377749513518659,
"worst_class_f1": 0.47665847665847666
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "48284dba20870ad2877b9189df3bec278653f04c8a3c99812f0840188a0d657e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.10741668753325939
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed43/model.pkl",
"bytes": 64221,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7109350771181784,
"accuracy": 0.8188136722841093,
"balanced_accuracy": 0.7237143821038571,
"worst_class_f1": 0.45794392523364486
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3c3152512aed2c8bbb1c16efd7942035151f9e89e139a58c8abfc2f9158470b2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
2,
2,
2,
2
],
"verification_seconds": 0.07644576393067837
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed44/model.pkl",
"bytes": 128604,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7151153882450071,
"accuracy": 0.8309054555841188,
"balanced_accuracy": 0.7169997827310548,
"worst_class_f1": 0.4375
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "66e2b8cd1c80f44d36e2ff62422ad6509515f99359a52b0fc62b543192605664",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
2,
2,
2,
2
],
"verification_seconds": 0.09162094537168741
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed42/model.pkl",
"bytes": 127524,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7519431599340909,
"accuracy": 0.84189453125,
"balanced_accuracy": 0.7382519385358287,
"worst_class_f1": 0.5137614678899083
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0e9ce884e630f754d915db2bf33fe41392c659e2fd04a569225695902c90db0f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
2,
2,
2,
2
],
"verification_seconds": 0.1081920899450779
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed43/model.pkl",
"bytes": 164122,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7641903008260709,
"accuracy": 0.84794921875,
"balanced_accuracy": 0.7437808965170654,
"worst_class_f1": 0.570273003033367
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "966eab1af20383e4ba9a7aa3d014fe22d7dde2c1b34eb75681cb90563f999662",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09193179570138454
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed44/model.pkl",
"bytes": 175628,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7666007836147859,
"accuracy": 0.85,
"balanced_accuracy": 0.7458548085139014,
"worst_class_f1": 0.5743174924165824
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "34fddd3eed7a6389bf16550b29df02b8646dd08bd5a8961363070b236323a6af",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09885944984853268
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed42/model.pkl",
"bytes": 164896,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7478668985168254,
"accuracy": 0.8256184407796102,
"balanced_accuracy": 0.7282771903274528,
"worst_class_f1": 0.5169927909371782
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7ded22cb68e6229b81b7d16ca5c0edaa46e66099a10fbc5c90d8af138a92c99c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09246528707444668
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed43/model.pkl",
"bytes": 161763,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7447052020556495,
"accuracy": 0.8229010494752623,
"balanced_accuracy": 0.7243578069099286,
"worst_class_f1": 0.5125
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a888536d0d5f7dc270067e8ea39665c7cd4e1e29a331fa57600076b157985bab",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09998060762882233
},
{
"method": "wisp_evolution",
"dataset": "capture24_wearable_activity_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed44/model.pkl",
"bytes": 133859,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7477843382157683,
"accuracy": 0.8235569715142429,
"balanced_accuracy": 0.7262215740897333,
"worst_class_f1": 0.5230125523012552
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9e3ad341c1e95d21e5bcd9cd71ae1eda615c27c49f8d67c2bd75c563da10a1ac",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.100236008875072
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed42/model.pkl",
"bytes": 212600,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7664598552903535,
"accuracy": 0.9142131979695431,
"balanced_accuracy": 0.759350201817826,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "91782879414932f4fd1d83de2b64f157f80a4203e71dd9cebc63de3836655b2c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.111138129606843
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed43/model.pkl",
"bytes": 109344,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7779119924157182,
"accuracy": 0.9030456852791878,
"balanced_accuracy": 0.771893748615854,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f987480f734d991975f9214d65fb667624b757b644f4187dd9eb17b44236ff80",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10117201320827007
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed44/model.pkl",
"bytes": 215498,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7588925510712723,
"accuracy": 0.9055837563451776,
"balanced_accuracy": 0.7528336709974588,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "dc05c3132f80de17ce71294cf01693bac44d720b3374913fabdf232c8c762ba9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.09738574642688036
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed42/model.pkl",
"bytes": 539572,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7654332232404667,
"accuracy": 0.8809418036428254,
"balanced_accuracy": 0.7650026167595115,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cf2712de8548876ab47ead3c76641d2aff86285a081d09d3e3e1a1fee26bcd7b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12981910444796085
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed43/model.pkl",
"bytes": 318789,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7475567873753366,
"accuracy": 0.8431808085295425,
"balanced_accuracy": 0.7409736397320528,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f207d6cb99c6be9e0229de253aeb5c554106e6808b8630e28c87738f9f7582b3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12939182203263044
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed44/model.pkl",
"bytes": 216225,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7643324610484091,
"accuracy": 0.8698356286095069,
"balanced_accuracy": 0.7816075184988293,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2bbd335eeb771d9febd05165213d6c533578169d0a8d57b0127a817714d12005",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10225305519998074
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed42/model.pkl",
"bytes": 542893,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6063961064701442,
"accuracy": 0.8345864661654135,
"balanced_accuracy": 0.6476178172323037,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c250dac4028d1978b59d5ccb9dfc9881c040b1623291d93d23e68351b1e10a6a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12861014250665903
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed43/model.pkl",
"bytes": 540198,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6786222922383697,
"accuracy": 0.8883747831116252,
"balanced_accuracy": 0.7045844794521647,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c780ed09a5710ffeda2ecaf05f41dbdc62d24781faba84ac4d25c6c3f06098a2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12846562545746565
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed44/model.pkl",
"bytes": 542016,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6059659135596054,
"accuracy": 0.8340080971659919,
"balanced_accuracy": 0.6479127025827084,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ad2ecac326b372c3dde4f10e2b6079b0eed91a7cd39fe073aec238131c0dc91f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13145661167800426
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed42/model.pkl",
"bytes": 212454,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6328932422472617,
"accuracy": 0.8124012638230648,
"balanced_accuracy": 0.658274984974044,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e00af7926b2d0764a2ee380dab3e15d8608f58be432b0441448491310da1bff0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10056019108742476
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed43/model.pkl",
"bytes": 215087,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5656274211024497,
"accuracy": 0.7918641390205371,
"balanced_accuracy": 0.5810575952113626,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e9d96df627dad5f8005a5dde54b70ae26ec7fcc47b7571fe2b429156d40a72bc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10281260497868061
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed44/model.pkl",
"bytes": 213083,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5766493081312468,
"accuracy": 0.7969984202211691,
"balanced_accuracy": 0.5883944723651199,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cabbe2a146876bdf4d60ef3b7511d624112af70f087adaf6362efc6d24842f1f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.09937153570353985
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed42/model.pkl",
"bytes": 319198,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6476508962514093,
"accuracy": 0.8443037974683544,
"balanced_accuracy": 0.6502484226519598,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7f022f0cb93b1f3d1b3333bc9280f649705c197ede1e8c632a915c4648190fcc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12414687685668468
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed43/model.pkl",
"bytes": 541362,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.618716820825527,
"accuracy": 0.8510548523206751,
"balanced_accuracy": 0.6259556576317066,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e6cb7ff7b7eabaeebceb94181be0b077fc467ab9693a583392ee8e5eb829a29b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.14895725715905428
},
{
"method": "wisp_evolution",
"dataset": "domino_watch_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed44/model.pkl",
"bytes": 210604,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6790563959748898,
"accuracy": 0.8611814345991561,
"balanced_accuracy": 0.6645928651150633,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "913f6893df2fc3e36e06434f73869862a53c966f78f3011f0906c81900623bf7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.09429517947137356
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 701279,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7672489965511144,
"accuracy": 0.780731384931708,
"balanced_accuracy": 0.790075815022735,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b8dcc5c64a5343a93e548def0659ae9f69aa59f26b9efae8b18ef60c04999072",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13250253908336163
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 592198,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7835365865957504,
"accuracy": 0.7968864737846967,
"balanced_accuracy": 0.8124908062429019,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "86844eb8a9e175f68acfb8f2777e19dc90862f4423fa1cb06adf28f84b65fa1b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13126813992857933
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 589580,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7870352562954467,
"accuracy": 0.804817153767073,
"balanced_accuracy": 0.8210987640639349,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b37545d5ae857e640fab4e7933811f530e9ca1d9d6d5ddc7f26ee1e9e6f5cfd7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12988745048642159
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 175868950,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.745442544578232,
"accuracy": 0.7687480869299052,
"balanced_accuracy": 0.7353281827704078,
"worst_class_f1": 0.4176533907427341
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f7a4e86ede990675448cc131f8c22ce073b831c3e82ae1722c8e9e24578d3842",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.49978523049503565
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 594940,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8188924951875662,
"accuracy": 0.8221610039791858,
"balanced_accuracy": 0.8239290176963354,
"worst_class_f1": 0.39609483960948394
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "aa10b8a5133b9e53941adf6e1b24b5b5c50a0f2ed7f993b95ab3dad2f1499321",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13092663139104843
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 593446,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.204170427930421e-18
},
"metrics": {
"macro_f1": 0.78724631788864,
"accuracy": 0.7976737067646159,
"balanced_accuracy": 0.8042867736850741,
"worst_class_f1": 0.014492753623188406
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6d275ae2416840f97d5e25539b64c1cfd4c7550e8ecf4182f57a1a2401f5928c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12541959341615438
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 625820,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8096769252697558,
"accuracy": 0.8215525402335121,
"balanced_accuracy": 0.8207685956698325,
"worst_class_f1": 0.45045045045045046
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "61832a626e710b6e585baa509e8143bc38aa82c5998d30a35721bdda870a6894",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12353577744215727
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 592488,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8028690592578791,
"accuracy": 0.8247081098138214,
"balanced_accuracy": 0.8126440443873308,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "23861efe32842b0b497613aa381f5e4552922c858b2b4394ed7de391860aabb9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12496596947312355
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 593567,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8063658973106798,
"accuracy": 0.8302303565793626,
"balanced_accuracy": 0.8136036002493117,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "19983a22327bbb26e1f068f170d5f667f86a83a1fe5e24ec5bb2571f55ae1319",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12128795497119427
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 677345,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.783907803140329,
"accuracy": 0.8038985783379745,
"balanced_accuracy": 0.7821143484738341,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "be9943a93e0b583388c6ce972347ee31a40a057a579f1cfd9b0b4aa8ecb99d00",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1648122714832425
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 556104,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 6.245004513516506e-17
},
"metrics": {
"macro_f1": 0.7776997415582797,
"accuracy": 0.7975963652352338,
"balanced_accuracy": 0.7793900579312314,
"worst_class_f1": 0.03298350824587706
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3fc594cc17fd2cc89b882953aaa152c3f481210354d4e629425eb45734e14ee5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12396656069904566
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 115192,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.697018026421252,
"accuracy": 0.7243148175289462,
"balanced_accuracy": 0.6970510428018644,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3ab1b44e1e59ea0d0756fb61416e2b05ce9b5704a8b1c373c28cd561cefa0f9c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08868241589516401
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 593151,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8641234565792023,
"accuracy": 0.8536787908985218,
"balanced_accuracy": 0.8719389189768827,
"worst_class_f1": 0.6344238975817923
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1e177f034f0d5272a98ddc35fe6e14cfd7c30f4617902a21d2d51fe77d08ee2f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13063144125044346
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 628812,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8815252691975555,
"accuracy": 0.8659691081215745,
"balanced_accuracy": 0.8895848428301525,
"worst_class_f1": 0.6130884041331802
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5f59b1c56c4e5143ce16e0d4e3e0298a77be839efe4cd4f788a05c9b2f633888",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12962188199162483
},
{
"method": "wisp_evolution",
"dataset": "gotov_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 591703,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8515240355978488,
"accuracy": 0.8400597907324364,
"balanced_accuracy": 0.8586857351484299,
"worst_class_f1": 0.5885797950219619
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "82ccc3b08284c3d6639155e8047459ee651a6610ebc6ca5d814873a92b058b53",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1400800608098507
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 107027779,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9535440866847888,
"accuracy": 0.9505915100904663,
"balanced_accuracy": 0.9547187190113678,
"worst_class_f1": 0.917960088691796
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0ba2051deeb8953941a5ba1097cf67c02bae3e0400ac3ba552d5ba6d92e523fd",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.30445832666009665
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 29561530,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9664924083067158,
"accuracy": 0.9659011830201809,
"balanced_accuracy": 0.9675183372111787,
"worst_class_f1": 0.9343065693430657
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cd92a861ddcc1fb2e319dc39b7255dabf7596ea0c792856d5377e044f89bacaa",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.15698592364788055
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 66429575,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9631621411148588,
"accuracy": 0.9610299234516354,
"balanced_accuracy": 0.9639777161822713,
"worst_class_f1": 0.933920704845815
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2d11e5bb8191faa9ea6f33fc59a552fca8f9da0161b1ab257b15cc7e2caa5164",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.23916628491133451
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 108937791,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9488100219032743,
"accuracy": 0.9364968597348221,
"balanced_accuracy": 0.9585341464521031,
"worst_class_f1": 0.8946236559139785
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e92e2c69e08c43c5b43a1ee7424453a0c6efcece3d351138a1ecd40cde17b434",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
9,
6,
5
],
"verification_seconds": 0.2992333984002471
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 69160983,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9688875446818674,
"accuracy": 0.9630146545708305,
"balanced_accuracy": 0.9736233967271118,
"worst_class_f1": 0.9453781512605042
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9a0ec2e26f0c92f0a9bee79d4d16c82361fb07bad0e8ba3f6cf78873284fbbff",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.23830208834260702
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 69220315,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9739522200817022,
"accuracy": 0.9685973482205164,
"balanced_accuracy": 0.9786360981639619,
"worst_class_f1": 0.9436325678496869
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "49db770dfcc481f405a102acfe78a27064e9424221313042063188f1a8a2ad48",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.2255033189430833
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 70056227,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9667081563909887,
"accuracy": 0.9611436950146628,
"balanced_accuracy": 0.9678508084578631,
"worst_class_f1": 0.929384965831435
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6111562ba3832d7c834a167ef77c6f8639fe1d979d35017ebbaf8bdede9c839f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.23397216200828552
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 30878124,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.960645742786498,
"accuracy": 0.9545454545454546,
"balanced_accuracy": 0.9613495525574918,
"worst_class_f1": 0.9248291571753986
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7741bfa8ce2e05112fac9f950e9075a2973e9a81f0ada3d9e676ebc487ae4d7d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.18209315743297338
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 30208604,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9646038665262866,
"accuracy": 0.9582111436950147,
"balanced_accuracy": 0.9668511854523635,
"worst_class_f1": 0.9269406392694064
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1f3ae2ef84aa939b59d3b1ef0d07f5bc5e4b87b5fe487f86bd013733f9daf6ad",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.17018190491944551
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 29162012,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9666138221553023,
"accuracy": 0.9584487534626038,
"balanced_accuracy": 0.972777362601331,
"worst_class_f1": 0.9322709163346613
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "22e3d999f78b0ea25cf621e6b0e2233717892e85c2fe821b120231da482b9dd6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.16035774070769548
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 67966647,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9643939005024231,
"accuracy": 0.953601108033241,
"balanced_accuracy": 0.9703785911125241,
"worst_class_f1": 0.9221556886227545
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "edf212eef5bab87a08152d61291dc7ef0e01114f220d4e372ba4d42b06af6242",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.22190038301050663
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 68830105,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.959464542814932,
"accuracy": 0.9473684210526315,
"balanced_accuracy": 0.9656081680613511,
"worst_class_f1": 0.9123434704830053
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6777cd4ce496465a3deb12a74d58cabfdc498145ce3485b9b09b3688ee09aea5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.23183409683406353
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 29370288,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": -1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9665822259579404,
"accuracy": 0.9585714285714285,
"balanced_accuracy": 0.9665539314970886,
"worst_class_f1": 0.9095238095238095
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3772a56d9802e1201b220323acfab33e90e13c7c1e5fdd315fef49329842bac4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
9,
6,
5
],
"verification_seconds": 0.15483754873275757
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 29652538,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9612307601150556,
"accuracy": 0.9535714285714286,
"balanced_accuracy": 0.9622524467374151,
"worst_class_f1": 0.9134615384615384
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "79b0bbd8d789f273915019109e1c79b4012e400ce4336b80a8b2de213b376acf",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.16102203354239464
},
{
"method": "wisp_evolution",
"dataset": "handy_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 109540579,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9508254122131812,
"accuracy": 0.9378571428571428,
"balanced_accuracy": 0.9511741971307087,
"worst_class_f1": 0.88
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "448bed549863e5f10d2843e937a4c4c3fd282cd15171c5f20e4970facf451d81",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.3187501523643732
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 812416,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.387570196628973,
"accuracy": 0.4189173789173789,
"balanced_accuracy": 0.42617659454061835,
"worst_class_f1": 0.13793103448275862
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "036a2ff2ca69e09e785fcf0b0ad2b7cd5b2c14996b915a4840c17eb4e98f44d7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13104734662920237
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 586110,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.3844396970116084,
"accuracy": 0.4151566951566952,
"balanced_accuracy": 0.4250397919631618,
"worst_class_f1": 0.16263736263736264
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "54c708e15be4215fa32d885d85257bd09be95e83ec0beb097a7eb4ff74944b40",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12016687728464603
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 702211,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3870428869869673,
"accuracy": 0.4168660968660969,
"balanced_accuracy": 0.42767719047090413,
"worst_class_f1": 0.1568627450980392
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b6a53d74eabde5c5dfe0f9c1be47bfe0d641ecddde0739efda79fc92933aae38",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.14498190488666296
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 366487,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3281143240077257,
"accuracy": 0.3518880208333333,
"balanced_accuracy": 0.36986021552619264,
"worst_class_f1": 0.1663286004056795
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0cbcab1b6bd4e29ba9d1d1532339c60ceda1eb623ff8b635de0d7a2058f9fd57",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.09013545699417591
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 479767,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3368172298791864,
"accuracy": 0.3588324652777778,
"balanced_accuracy": 0.3818676836393975,
"worst_class_f1": 0.1725417439703154
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5f30f5071527b5e6a91313eece5fd547570f5b23f7859d6feb5a9adb21bed997",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
8,
4,
4
],
"verification_seconds": 0.10633725114166737
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 589671,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.34027982278741586,
"accuracy": 0.3645833333333333,
"balanced_accuracy": 0.3811477225820268,
"worst_class_f1": 0.18385650224215247
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3316aff7e86d626e671de4aeb0b58720856bfee06b6b5bebe482193ca7e62cd4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
11,
4,
4
],
"verification_seconds": 0.11465025693178177
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 856939,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4157316719433681,
"accuracy": 0.4486997635933806,
"balanced_accuracy": 0.4557413938998671,
"worst_class_f1": 0.1743119266055046
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c33c1d7f183f138b4d34ac5927b9156a7fd20c731affa8c8165db2239189b3ed",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13862699456512928
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 591055,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.41011206610734857,
"accuracy": 0.4459810874704492,
"balanced_accuracy": 0.4506340420650674,
"worst_class_f1": 0.1487603305785124
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "06f7274c1dafdffdd23e29988cef74286bacfd6ec13b28f4ed2f677f8361a04b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13259280938655138
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 678051,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.3979308233637672,
"accuracy": 0.4329787234042553,
"balanced_accuracy": 0.4453328739886988,
"worst_class_f1": 0.11907164480322906
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bde21d7f651fc6c81a941892af381009a94022df1980f1e9a01ed5b6018dae5a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
4,
4,
4
],
"verification_seconds": 0.1591173354536295
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 856835,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3957192396436544,
"accuracy": 0.4207221350078493,
"balanced_accuracy": 0.42766245503552947,
"worst_class_f1": 0.2186046511627907
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b3faa0b76c9aba0701a6cdb52920174ae1521ea97aa2a3f64821dfe08d997ae6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.16633606050163507
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 555115,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3848142778821959,
"accuracy": 0.4088360618972864,
"balanced_accuracy": 0.4180010374181177,
"worst_class_f1": 0.2053388090349076
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d435fe4444ee31e12f4aeff2d08da4da90b1581e55789f78c53b63e459cbda72",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.14687734376639128
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 678876,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3791269371977908,
"accuracy": 0.40165956492487104,
"balanced_accuracy": 0.4179393559999468,
"worst_class_f1": 0.1830708661417323
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "63f4c3fc491c715125760737b25c16405481f68eaefd3945ce31338650723c6e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.1423840904608369
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 164963,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.38673689720064064,
"accuracy": 0.4308221457894153,
"balanced_accuracy": 0.3889641395556819,
"worst_class_f1": 0.18230563002680966
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bea2ddaf698e130e949026e1d06658401971805e9a3192abd36778e73b885c37",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
8
],
"verification_seconds": 0.08014317229390144
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 587574,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.40823881367737425,
"accuracy": 0.44291578830578054,
"balanced_accuracy": 0.45349802538725736,
"worst_class_f1": 0.18556701030927836
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fa9923bc96f898e37075dc1fe779e76b14f44e27caf6deda4068d31d0c756f62",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11033798847347498
},
{
"method": "wisp_evolution",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 589309,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4140329275306399,
"accuracy": 0.4480195273493842,
"balanced_accuracy": 0.45934733635186287,
"worst_class_f1": 0.1822849807445443
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3ed51ab0acde02f67a2049d119fc0a1d83ea9cdebf3833321f09c7ec12c487f1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
8,
4,
4
],
"verification_seconds": 0.12434007879346609
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 681220,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.47706378206167416,
"accuracy": 0.5076923076923077,
"balanced_accuracy": 0.5092512560398281,
"worst_class_f1": 0.1962864721485411
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b45b73fd069a7912e03d1855f10a92784abbc3953e9651cee273a6532f692464",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.1404802268370986
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 554213,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.49312844038510845,
"accuracy": 0.5299145299145299,
"balanced_accuracy": 0.5207230526700427,
"worst_class_f1": 0.1807909604519774
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0752056393d300870f405f1eaa6c1ea04582279181d1e136badfa8ce80222518",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
5,
11
],
"verification_seconds": 0.11337910685688257
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 344185,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.4666524181153792,
"accuracy": 0.5035897435897436,
"balanced_accuracy": 0.49701966308229717,
"worst_class_f1": 0.16022099447513813
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a58060dc11572fe3ccdce4a6d2f0150fcbdf558914403de04c4b8d2f11cfc7ae",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.10804144851863384
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 364507,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4370155710734781,
"accuracy": 0.4697265625,
"balanced_accuracy": 0.48011368668483795,
"worst_class_f1": 0.1203585147247119
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "70840706528cb04f82ac77a1d4c92016dc45ee93a2f96c56a01f09966d31b925",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08191533200442791
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 697475,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4512327358194982,
"accuracy": 0.4790581597222222,
"balanced_accuracy": 0.49275046578712517,
"worst_class_f1": 0.1793478260869565
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5bb2a4d812e31bc98fcc59fb93920ddf01a1167ee426e56150098a931c74aba4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
4,
5,
4
],
"verification_seconds": 0.12659502308815718
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 451638,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.4471513148059462,
"accuracy": 0.4729817708333333,
"balanced_accuracy": 0.48961303342143864,
"worst_class_f1": 0.17073170731707318
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "89fd62d69fb8334e6474dd91d725669067166c5e62c3178ea6068d6478bfd641",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
11,
8,
8
],
"verification_seconds": 0.1261815158650279
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 477487,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.5345935850365053,
"accuracy": 0.58274231678487,
"balanced_accuracy": 0.5655624417274536,
"worst_class_f1": 0.21656050955414013
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ba5cf262812bbe919af9e320dbc6d871a6ea1cd10c91c3aa70e16e6d79de8da5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
11,
4,
5
],
"verification_seconds": 0.10365157946944237
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 701127,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.5360904467830878,
"accuracy": 0.5872340425531914,
"balanced_accuracy": 0.5652497977212918,
"worst_class_f1": 0.18181818181818182
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c9b0240e9e9b1e24a6818aea545aca798ed834851723fe5cbc1cf6bee074c309",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
11,
5,
11
],
"verification_seconds": 0.12794150412082672
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 593415,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.540710085532877,
"accuracy": 0.5888888888888889,
"balanced_accuracy": 0.5713526491135847,
"worst_class_f1": 0.2018348623853211
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "777bc6f479fcba72984e9e1aed22661d6ac8fc65036a2d58606ec75a7df3ae22",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
4,
11,
11
],
"verification_seconds": 0.117134939879179
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 928585,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.5292660108853414,
"accuracy": 0.5708679076026015,
"balanced_accuracy": 0.5517760462043491,
"worst_class_f1": 0.24623115577889448
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f6cb6f46235235f76ebec4dcaab5d78ef7b237742ceb943f4c0c956f71c6fe40",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
5,
8,
8
],
"verification_seconds": 0.1802399018779397
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 920365,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5224325352669028,
"accuracy": 0.5642520744561561,
"balanced_accuracy": 0.5433392429912146,
"worst_class_f1": 0.224
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2df7e442bc4600f05e8c723f33877fb92f60d551306cd69ff7175049bf7c43e4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
11,
11,
8
],
"verification_seconds": 0.1654700394719839
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 678702,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.4902724278530045,
"accuracy": 0.5289302534200493,
"balanced_accuracy": 0.5163140375736235,
"worst_class_f1": 0.23756906077348067
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "935dd179ad44b3b02aa5e8545f12db3b4099b241eb890ce780d3c323c5ac9664",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.13488131761550903
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 116863,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.48107076640171,
"accuracy": 0.5211361366914457,
"balanced_accuracy": 0.47297209841401533,
"worst_class_f1": 0.2077562326869806
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "66385fc30b94f06aa0449ed217198ccc132752ce4dc4535b14f709b3d2f412f0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
8
],
"verification_seconds": 0.09273753501474857
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 587010,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.5040897734143406,
"accuracy": 0.5419948962609564,
"balanced_accuracy": 0.5311132155244158,
"worst_class_f1": 0.20030349013657056
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "500ca32f859e3fcc82c2e63641bcf90b276c39ef76b25897c20105bd34b33b76",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.11589611694216728
},
{
"method": "wisp_evolution",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 678966,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.49092845263416923,
"accuracy": 0.5254632197936314,
"balanced_accuracy": 0.520122834067514,
"worst_class_f1": 0.19607843137254902
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "32de18c5708819ab41108eb5d133a31cb1df50743b541d3e44ec033066f0c252",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.15122493356466293
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 144553,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6001622909093439,
"accuracy": 0.8245280257573541,
"balanced_accuracy": 0.592896134674992,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "64fdaaef017498ecc7315aa50318895b526ece1f7ca5565d2852c8c94b6fdb7a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10048750694841146
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 194458,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6212723756520478,
"accuracy": 0.8232108883360164,
"balanced_accuracy": 0.6083823342657297,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c743297bc7b09082982f875e32a2efb54f4f9ab3417e51248ca11c64c51198fb",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.08981783781200647
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 341656,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5636640533035754,
"accuracy": 0.8281867408166252,
"balanced_accuracy": 0.5371753995032529,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4ab88443c7cb1bd1291caabd5f666cda2fc23b1f3a7f4f61b082131ba58e5b5f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.1375412354245782
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 336594,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5364918657912368,
"accuracy": 0.8251804330392943,
"balanced_accuracy": 0.5041208793661618,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e2f6ac7eea957553f6812a95af8140f0b47ea32fc36a0a021b72e7b65b2c5ad4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.12254564091563225
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 343025,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5134095420062846,
"accuracy": 0.801523656776263,
"balanced_accuracy": 0.48543136741220744,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "32e9553c9d5a4159a79f670139ae0ce81021f2ffcc2c7691b0af3ae030e54425",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.1299411579966545
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 341859,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.49811428812826414,
"accuracy": 0.7797380379577653,
"balanced_accuracy": 0.4614286887159727,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0e901ee03a0b25696fd04c6454816ac40d9fd6a9c620b0aaad388f502b3061ce",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.12822435796260834
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 73411,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5740855391481829,
"accuracy": 0.8384826434450674,
"balanced_accuracy": 0.6331106148564649,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "064ae588a90768eb6a818b751dad91f2a4bdc110ea0cc3b879d0655073fdfad1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08763671666383743
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 147505,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5887274453964119,
"accuracy": 0.8454014076106405,
"balanced_accuracy": 0.6058360488052064,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "31dc121ea36092e092175c7e0e5407da27c8a1a661baa1a52a283e93892565a3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10630783904343843
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 266734,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6268461641781113,
"accuracy": 0.8302516998687821,
"balanced_accuracy": 0.6435828090925986,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6b9344e70583d77530fe8a694440936b094483f06e6e582834b1a3e2e09d9230",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.10888167563825846
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 288208,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.596128960568406,
"accuracy": 0.8317962835512732,
"balanced_accuracy": 0.598804660316241,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6a22c61813ec149f540c737bd34e3e7488eb05fbfefe348c9d6f5a4f626bc0be",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12140228878706694
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 338086,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5905239929548161,
"accuracy": 0.8220233998623537,
"balanced_accuracy": 0.5860314266823297,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cc486db4a47978da0ec68a942e819276c79206dd4a18237dde884231bcd3ac9e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.1356620490550995
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 339803,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.612778808268588,
"accuracy": 0.846662078458362,
"balanced_accuracy": 0.5768670269769565,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "378d36025950f1bb47310bcca00bd87e5482be55e84ac0630cffb98b8b060883",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.1284400476142764
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 358831,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7350769324712363,
"accuracy": 0.8778756035217268,
"balanced_accuracy": 0.6931079259273305,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "8d02a047a1b63170714775755c37eb944a608a7ee30bdd074c911ff15f0efc95",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.13610275462269783
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 342490,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7479329190105534,
"accuracy": 0.8950582220959955,
"balanced_accuracy": 0.7162155645562306,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0d649153cc7bd5c8480e6281471419759e3781edb27a967ed22233fc16775f34",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.13554767239838839
},
{
"method": "wisp_evolution",
"dataset": "iuwds_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 337538,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7998097634781114,
"accuracy": 0.9067026412950866,
"balanced_accuracy": 0.7779843281151284,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d385cbf979f12f2e1fa3a129f70e1a4d7f3f117a83b9efd811e9ea8a0fe59b7c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.1348281092941761
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed42/model.pkl",
"bytes": 385670,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7354886950705765,
"accuracy": 0.7267326732673267,
"balanced_accuracy": 0.7416392821031345,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "642323bd092f7af4bff81261d49a95f8d6bdd9c03bb991382807f96cd65a1b68",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
6,
6,
6
],
"verification_seconds": 0.10621493961662054
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed43/model.pkl",
"bytes": 389579,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8188387012671772,
"accuracy": 0.8198019801980198,
"balanced_accuracy": 0.8294384057971014,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c98af6a1ad5fa4a5acf72dbb151c2c39e8bbcff3547fb378b2655c3128fbcc6d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.10488037578761578
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed44/model.pkl",
"bytes": 388588,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7493625750406817,
"accuracy": 0.7485148514851485,
"balanced_accuracy": 0.761945989214695,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1a1e5ee70767382efbcf8077bd8f6b6a408593fe493ffa43a86c593202d523a5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.10792993381619453
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed42/model.pkl",
"bytes": 388809,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8135257169451995,
"accuracy": 0.8317214700193424,
"balanced_accuracy": 0.8389626741846908,
"worst_class_f1": 0.5352112676056338
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b6b4cad4225d36f51a87164af7f5ae2cb9686f93958bb781e20b0cd92fe2b143",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.11736003495752811
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed43/model.pkl",
"bytes": 389448,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8224755450020339,
"accuracy": 0.8355899419729207,
"balanced_accuracy": 0.8430465618254703,
"worst_class_f1": 0.5526315789473685
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ea44f640f353240f0fe43c2f07b0671b8ef4d9b14da5b93e0c141cf6f051a4df",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.1024811640381813
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed44/model.pkl",
"bytes": 99292,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.7123372087360301,
"accuracy": 0.7330754352030948,
"balanced_accuracy": 0.74444591280854,
"worst_class_f1": 0.11428571428571428
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e4fa923163e155f243de5e81ab01b0dfd68897c2d30b760261d78939901d607b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
8
],
"verification_seconds": 0.06621815077960491
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed42/model.pkl",
"bytes": 6680248,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8878789083727651,
"accuracy": 0.884393063583815,
"balanced_accuracy": 0.8913043478260869,
"worst_class_f1": 0.5542168674698795
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ce8c0b79823ca0e9198b9705a7708e6417d0465501e611c72e131206ae690b15",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.13278221990913153
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed43/model.pkl",
"bytes": 16192319,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9111111111111111,
"accuracy": 0.9113680154142582,
"balanced_accuracy": 0.9166666666666666,
"worst_class_f1": 0.6666666666666666
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b03a1eba1b9ba1d5f14bf9454de555f8b0214b3e6020f363d0e32d2f7ca345b3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.13645241782069206
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed44/model.pkl",
"bytes": 16911563,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.877856486947396,
"accuracy": 0.8959537572254336,
"balanced_accuracy": 0.8731884057971014,
"worst_class_f1": 0.6571428571428571
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "619961db831ef97b705ff4251ab9de1c1e5f62c9d9a174974ca4372ce9b765d3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.14094430953264236
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed42/model.pkl",
"bytes": 387408,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9296020533702118,
"accuracy": 0.9295499021526419,
"balanced_accuracy": 0.9320460673468762,
"worst_class_f1": 0.676923076923077
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c12d0ad33340bdfaa17619d0480c221971f88ad98b2f9782328b8ec64c775bd9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.11122406274080276
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed43/model.pkl",
"bytes": 100328,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.888480483344134,
"accuracy": 0.8904109589041096,
"balanced_accuracy": 0.8949656539393648,
"worst_class_f1": 0.5423728813559322
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cd5e821662f61f83e72b853c5ec5c1bc544b8d90ff9eb27351f715dfc7ef1654",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.0694976132363081
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed44/model.pkl",
"bytes": 291189,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8694335167724162,
"accuracy": 0.8688845401174168,
"balanced_accuracy": 0.8672814378072213,
"worst_class_f1": 0.64
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a7d10e300109478314430a4eb69c1a26724258d09f1f95448fc6061bfa54862a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.091588887386024
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed42/model.pkl",
"bytes": 30961327,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8318056467541005,
"accuracy": 0.8230616302186878,
"balanced_accuracy": 0.8346920289855072,
"worst_class_f1": 0.5
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c3c4a01d438ce5ecd6b04091d0a8514f61ef8601fa4ed5227c473308b0af42d9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.16053823940455914
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed43/model.pkl",
"bytes": 30830438,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8414983164983165,
"accuracy": 0.8489065606361829,
"balanced_accuracy": 0.8333333333333334,
"worst_class_f1": 0.46464646464646464
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cb2eb0d0b2dfc1e6692fdeb57a8fbcae3132439f68cd7192c107bbc48a118b89",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.16630718670785427
},
{
"method": "wisp_evolution",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed44/model.pkl",
"bytes": 30621176,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8392896169265338,
"accuracy": 0.8469184890656064,
"balanced_accuracy": 0.8315217391304349,
"worst_class_f1": 0.46464646464646464
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e9b6153280a1a0515086dcb052677e4ef9525a52e4eaba7988f5974d590abef0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.17363407742232084
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 153957344,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7892275662433231,
"accuracy": 0.7895833333333333,
"balanced_accuracy": 0.7732288959196837,
"worst_class_f1": 0.5088757396449705
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7e051b14cf200359bc93dc42d1c644a3d70195c75d53c83595ff78536fbbca83",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.4136338597163558
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 323691551,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7900666383115289,
"accuracy": 0.8010416666666667,
"balanced_accuracy": 0.7741262465535576,
"worst_class_f1": 0.5810055865921788
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f7e788b278cc8e1725664d17aae66ab8d38966f5da8db575d689330f152375b4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7362617207691073
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 153261502,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7730199712967488,
"accuracy": 0.7885416666666667,
"balanced_accuracy": 0.7593348479587491,
"worst_class_f1": 0.55
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "338412c40b053ad5446733dc73170b258270ec38714701590ab4bfd4a8f69e47",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.36440687999129295
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 142963268,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7225936836830401,
"accuracy": 0.7149677057006459,
"balanced_accuracy": 0.705294925589822,
"worst_class_f1": 0.37209302325581395
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5430b3d4daa1933336b3353f154efd50193f86f80dcc357acecafc03c7a1d3bf",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.3679047701880336
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 140553320,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7164108253593273,
"accuracy": 0.7197416456051671,
"balanced_accuracy": 0.7004261415634176,
"worst_class_f1": 0.4806201550387597
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c46d20bb9312101456100e660befe4efadf1cf076805c45dd995d23d9ccae00e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.39510233141481876
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 141989394,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.6871823068269264,
"accuracy": 0.7071047458579051,
"balanced_accuracy": 0.6740489832241326,
"worst_class_f1": 0.31683168316831684
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7a846466c5eebdb50b19c86a9569e1a8dfa50385fd1351113a32907dac90d898",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.3557204445824027
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 315612091,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7864181898198394,
"accuracy": 0.7759418374091209,
"balanced_accuracy": 0.766090744467668,
"worst_class_f1": 0.45454545454545453
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "8741865403f39a1d2025c6d1b9d8b298731828a521f2fa6fef533931b1d0c060",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7392814699560404
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 153243266,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.80773716013251,
"accuracy": 0.7974223397224058,
"balanced_accuracy": 0.7904072939498322,
"worst_class_f1": 0.5039370078740157
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c62309370b896062d65a2c3de815063290b1b90c15357ae94b902e47c1321565",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.41625876631587744
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 153626016,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8019267714455053,
"accuracy": 0.7941176470588235,
"balanced_accuracy": 0.7839061568806102,
"worst_class_f1": 0.49624060150375937
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "23fadf9558497c26725749a608d94a3f3e9b6552e270d86357084b0663d62761",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.3832047041505575
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 152149574,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7777656452114684,
"accuracy": 0.7796555086122847,
"balanced_accuracy": 0.7665927790732726,
"worst_class_f1": 0.3953488372093023
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3092788b844f02df039c5bd7471138d84d5ec2d4dfc0c32cf3632e5d4da2adbd",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.3757688459008932
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 152238152,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7607235262767121,
"accuracy": 0.7731556711082223,
"balanced_accuracy": 0.7535080157000026,
"worst_class_f1": 0.3684210526315789
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "870373f958f71bf32099b6a24355c38355e07b6531bd6f4a3b1232df1e59b0f2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.39853435661643744
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 319477175,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7532136587464725,
"accuracy": 0.7598310042248944,
"balanced_accuracy": 0.7411083111974465,
"worst_class_f1": 0.40229885057471265
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6c875bc9fe2f42aa55055a78d56124883050a090c9eb0843500b5d5c870a3d8b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7252659574151039
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 151505892,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7806518780446584,
"accuracy": 0.7801668806161746,
"balanced_accuracy": 0.7721625612556755,
"worst_class_f1": 0.2857142857142857
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "264d7c463966746c13527a9b1f6e957909c6f22d3d65ef956de8b3a3af631cd2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.38599200546741486
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 318869947,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7932082732315537,
"accuracy": 0.7910783055198973,
"balanced_accuracy": 0.7835228970239744,
"worst_class_f1": 0.3378995433789954
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3f8b66d97b8aaa91894b0675e2e2358ba7a0c91ff60c3f319cb383925951dcee",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7330380454659462
},
{
"method": "wisp_evolution",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 152256374,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7644344111861626,
"accuracy": 0.7708600770218228,
"balanced_accuracy": 0.7561859057947079,
"worst_class_f1": 0.3404255319148936
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "af81927a19858af47b2fb5e51a8649cc34e27cde0b64614136a14bff375c5f0e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.3918878575786948
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed42/model.pkl",
"bytes": 193291,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9574373809904692,
"accuracy": 0.9519050593379138,
"balanced_accuracy": 0.9570683091449005,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6b84666bf1cf734a5028a6ad927d030ab5001d3c948d5ce9a5adfd41fac2f467",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08527306746691465
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed43/model.pkl",
"bytes": 196772,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9234438902270412,
"accuracy": 0.915053091817614,
"balanced_accuracy": 0.9239575066889878,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bcbe55be438c66f2ff1596569efcd1b0f119bf8930351c32e578edc73c50562e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12321723159402609
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed44/model.pkl",
"bytes": 297193,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9406545864398269,
"accuracy": 0.9344159900062461,
"balanced_accuracy": 0.9396017944519662,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4b8ae35ce0b9243acac79c368aaf530801f25329219fb9a261379dd0e8491d85",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12391958478838205
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed42/model.pkl",
"bytes": 100780,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9274071442306097,
"accuracy": 0.9279907084785134,
"balanced_accuracy": 0.9303482022468934,
"worst_class_f1": 0.7922705314009661
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4cf0e5c4409bcd82ca7fdb4c22a3c8bca473f0f20a164320d5c944bff525fcdc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.0929145747795701
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed43/model.pkl",
"bytes": 597009,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9573547671478485,
"accuracy": 0.9506387921022067,
"balanced_accuracy": 0.9599025452743013,
"worst_class_f1": 0.8265895953757225
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fc6e3ad30248d370e9a115f9934da1a4a6bcf5f258e65cee5e75c1517be6ed2e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1140044592320919
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed44/model.pkl",
"bytes": 100739,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9486688403645763,
"accuracy": 0.9419279907084785,
"balanced_accuracy": 0.9491802208534699,
"worst_class_f1": 0.8306010928961749
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3fd5f1bda39e2b35f9f9c9a01447d6e6def59a6117463fb9839d208a4d8b74ad",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08163740299642086
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed42/model.pkl",
"bytes": 62111308,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8517262119654153,
"accuracy": 0.8604014598540146,
"balanced_accuracy": 0.8410551716685949,
"worst_class_f1": 0.5333333333333333
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b583f9f7e0ef26632e0c3a974d8c4ab3131c1151c650395164b7427e4c8c2dd8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.23181306663900614
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed43/model.pkl",
"bytes": 62886878,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8209642589780709,
"accuracy": 0.8394160583941606,
"balanced_accuracy": 0.8033008632095653,
"worst_class_f1": 0.5384615384615384
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a6d79a8bda87c56e6c8c7c67224f349e1e58e4360275e55869232059844bcfec",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.2601744942367077
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed44/model.pkl",
"bytes": 63063120,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8023272337691744,
"accuracy": 0.8147810218978102,
"balanced_accuracy": 0.7802617209412323,
"worst_class_f1": 0.5333333333333333
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "65be58d783f38b502915d98d4653c04896fa4f9c020ebf6fd3fa2c628e66188d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.24120864365249872
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed42/model.pkl",
"bytes": 508988,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9753302851247804,
"accuracy": 0.9764453961456103,
"balanced_accuracy": 0.9768624909957174,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7c7b18f7c4b28dfa856b9d137d3b9b30dd31321d0f44a67b7e4a9c958bb79990",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08564670383930206
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed43/model.pkl",
"bytes": 202528697,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9777781580379844,
"accuracy": 0.9796573875802997,
"balanced_accuracy": 0.9795458509744225,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3408f2b33a059bf8d413702989205a6cbe26e63950ca4c79d86a394e8c8a54b2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.8584732040762901
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed44/model.pkl",
"bytes": 100381,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9702034685152359,
"accuracy": 0.9721627408993576,
"balanced_accuracy": 0.9695197659483373,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "aa815e1e89f51a870ce1317e6de1533067fc8b3c13fc95460ed97384d76c78a5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.07312445435672998
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed42/model.pkl",
"bytes": 56120005,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7621147901273506,
"accuracy": 0.7752192982456141,
"balanced_accuracy": 0.7490324061766814,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2042c93b8c87d47a56542875b1c96ba42a779fa55d459d847becec8bb24ab696",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.22116222511976957
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed43/model.pkl",
"bytes": 58285096,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7501886834700756,
"accuracy": 0.7642543859649122,
"balanced_accuracy": 0.7350324796363008,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "446c26982f12fa7809412901d7da73a936e567375ac9d4c20800e4ea5052fb23",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.25193033926188946
},
{
"method": "wisp_evolution",
"dataset": "pamap2_hand_imu_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed44/model.pkl",
"bytes": 56854569,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.760305665820137,
"accuracy": 0.7702850877192983,
"balanced_accuracy": 0.7455118406391396,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6cf67eb93bfe4f2e570d63e3aa237bd864a21d2f10755012fb0ca68d3adaf488",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.22516581136733294
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed42/model.pkl",
"bytes": 1441696758,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7257920521665745,
"accuracy": 0.7153008962868118,
"balanced_accuracy": 0.7167385507142605,
"worst_class_f1": 0.24921728240450847
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c15e5c665d7296a50f5da08fc485c98b6587d4710d61fff886d8d0bd9d868353",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.7825141521170735
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed43/model.pkl",
"bytes": 1440491744,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7311220616483857,
"accuracy": 0.7195902688860435,
"balanced_accuracy": 0.721265618929603,
"worst_class_f1": 0.2734422262552934
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ae134fbb2388fff33f837da08fea166bc34896eb2145b9abcc51dded51dc3f26",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.7973577231168747
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed44/model.pkl",
"bytes": 1454190198,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7372074685460596,
"accuracy": 0.7245198463508322,
"balanced_accuracy": 0.7259944037246389,
"worst_class_f1": 0.35683629675045986
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9947c4ed8e471c90de672c7379927e557a3c43505ce6547e95e0332965987f10",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.7477301387116313
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed42/model.pkl",
"bytes": 1415103478,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6909213215278447,
"accuracy": 0.6976187476376464,
"balanced_accuracy": 0.6950960789059522,
"worst_class_f1": 0.3468507333908542
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "434ed4bdc14514ab3e125b0032326ccfff4b11dd900528031dddbb158080136d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.5760244950652122
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed43/model.pkl",
"bytes": 1410096822,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.6693677517933854,
"accuracy": 0.6783419427995464,
"balanced_accuracy": 0.6754648043483025,
"worst_class_f1": 0.21124361158432708
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ae2013720b730a92e701ea61ec7c5cba041461b6730b88c893d715e2cb5fd3a1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.713862843811512
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed44/model.pkl",
"bytes": 1412228576,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.6790566568076546,
"accuracy": 0.6857124858258788,
"balanced_accuracy": 0.682652741791336,
"worst_class_f1": 0.27692307692307694
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "72e5d2ab67810a31dc43d07fbcf1fb1e75e27ed539bdf73216270d54e5f1089c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.726566475816071
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed42/model.pkl",
"bytes": 1006207,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8497583589701082,
"accuracy": 0.8565003513703443,
"balanced_accuracy": 0.857026570430238,
"worst_class_f1": 0.33
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3ec5e9a17b1d91fcd980c62acbaad8d3c664ba57bc3a4647423f766794ee10f9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.1348606338724494
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed43/model.pkl",
"bytes": 638844,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.8221419805287655,
"accuracy": 0.8338018271257905,
"balanced_accuracy": 0.8360105567768358,
"worst_class_f1": 0.15869311551925322
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "00e790c44211d2e9101f4b7989cebaccb4b24f1c8d472be774bdf4677f14fb10",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.1066829888150096
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed44/model.pkl",
"bytes": 1346779355,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8333028215152704,
"accuracy": 0.8427266338721012,
"balanced_accuracy": 0.8448529483213124,
"worst_class_f1": 0.33367037411526795
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3a6544700bc41317febaa9a6fccf239bea2eaa23679f880b1d874ff9c035da32",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.545041259378195
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed42/model.pkl",
"bytes": 1403333366,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7105930492098491,
"accuracy": 0.7057796620736954,
"balanced_accuracy": 0.7063862046110593,
"worst_class_f1": 0.40115025161754136
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f6eb8ecc81f2dbd82dc2198a7fce9b1a6e87335a4c4894ddaf9fa1b49f0e1b8d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.6965157566592097
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed43/model.pkl",
"bytes": 1402171766,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7084641883414828,
"accuracy": 0.7075877548475591,
"balanced_accuracy": 0.7081053741669864,
"worst_class_f1": 0.3187889581478183
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "dccc1936bb4b4df120ba985021080edaff742caed891e7d0a4bd65ab5d3b9e35",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.6935139382258058
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed44/model.pkl",
"bytes": 1403424512,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.6952479691226653,
"accuracy": 0.6926865764698548,
"balanced_accuracy": 0.6933706911358393,
"worst_class_f1": 0.34602649006622516
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "57c44e13d0ba8d4c6c512726c084d375dbe48c6814b88286ed92c508e5835cf7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.691866286098957
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed42/model.pkl",
"bytes": 641366,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.7203206678406642,
"accuracy": 0.7189364132504125,
"balanced_accuracy": 0.7196280071775887,
"worst_class_f1": 0.12974051896207583
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7a53a1691892c5ace8e41e723b388da68e5714799557855900d73ee03dfde3cd",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.0962999165058136
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed43/model.pkl",
"bytes": 246094,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6694266295253583,
"accuracy": 0.6790836400558447,
"balanced_accuracy": 0.6768085236883788,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fbb2ab9c1cee267c109d60f3d6b771a9b4e6f1069056306d1d07c99bdbc5e45c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.0871832063421607
},
{
"method": "wisp_evolution",
"dataset": "wisdm_watch_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed44/model.pkl",
"bytes": 642689,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7068318523799468,
"accuracy": 0.7071963447137962,
"balanced_accuracy": 0.7070325543416646,
"worst_class_f1": 0.1347248576850095
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a389973ba7102f94da04fba77bae30fec1c3a3ba6e259d813aa62ab81060aa79",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.10736861545592546
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 213075,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "passed",
"n_test_windows": 861,
"exact_frozen_predictions_equal": true,
"original_input_exact_frozen_predictions_equal": true,
"original_input_differing_prediction_count": 0,
"differing_prediction_count": 0,
"original_and_fresh_X_exactly_equal": true,
"original_and_fresh_X_max_absolute_difference": 0.0,
"time_coordinate_check": {
"raw_time_values_exactly_equal": false,
"per_subject_stable_chronological_permutation_identical": {
"f3": true,
"m1": true,
"m3": true,
"m7": true
},
"original_first5": [
0.0,
2.5,
5.0,
7.5,
0.0
],
"fresh_first5": [
0.0,
1.0,
2.0,
3.0,
0.0
],
"explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled."
},
"metrics": {
"macro_f1": 0.5339423877092813,
"accuracy": 0.7015098722415796,
"balanced_accuracy": 0.5192755280407102,
"worst_class_f1": 0.0
},
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5339423877092813,
"accuracy": 0.7015098722415796,
"balanced_accuracy": 0.5192755280407102,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d7e9ecb9baabeede43934e670f6e0efcdeb29e1110056d33362d5cad8cc0870f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 3.478103124536574
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 56473111,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "passed",
"n_test_windows": 861,
"exact_frozen_predictions_equal": true,
"original_input_exact_frozen_predictions_equal": true,
"original_input_differing_prediction_count": 0,
"differing_prediction_count": 0,
"original_and_fresh_X_exactly_equal": true,
"original_and_fresh_X_max_absolute_difference": 0.0,
"time_coordinate_check": {
"raw_time_values_exactly_equal": false,
"per_subject_stable_chronological_permutation_identical": {
"f3": true,
"m1": true,
"m3": true,
"m7": true
},
"original_first5": [
0.0,
2.5,
5.0,
7.5,
0.0
],
"fresh_first5": [
0.0,
1.0,
2.0,
3.0,
0.0
],
"explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled."
},
"metrics": {
"macro_f1": 0.44804969372888687,
"accuracy": 0.6991869918699187,
"balanced_accuracy": 0.48086825598802657,
"worst_class_f1": 0.0
},
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.44804969372888687,
"accuracy": 0.6991869918699187,
"balanced_accuracy": 0.48086825598802657,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "69b629122b9d9b7f4dc5cbf40b767b9391f12f8d6327ea817f385e736638f265",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 1.414209634065628
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 387105,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.42336890869963556,
"accuracy": 0.5448648648648649,
"balanced_accuracy": 0.5103515571350251,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0593f16f80eb5d1a010e55c9534cc105a671f91294b48a68990c46d0b27ddb1e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
7,
7,
7
],
"verification_seconds": 0.11762447189539671
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 388857,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.44265859848601236,
"accuracy": 0.5553153153153153,
"balanced_accuracy": 0.5262091686071412,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0133cd4b226973eebe7a77576171dfe2440d216a3c440f8652f4379d6e9cd136",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
7,
7,
7,
7
],
"verification_seconds": 0.10366932395845652
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 197938,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.45020993038951257,
"accuracy": 0.556036036036036,
"balanced_accuracy": 0.5254028034871413,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0ef0f6220e2db9c4c853ba6158dc3924effaefaca7f40fd87f196c3259fd3030",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
7,
7,
7,
7
],
"verification_seconds": 0.07649926003068686
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 213222,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6839360804687519,
"accuracy": 0.6371134020618556,
"balanced_accuracy": 0.6487853885528353,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "83d5ea4d49be6e2c8eac6cd6363cbecf629412f9ac345d791a04dd6444c24824",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07935636956244707
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 108796,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6914456270187225,
"accuracy": 0.6412371134020619,
"balanced_accuracy": 0.6577139599814067,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "beda21721a3b563425f0da94f37fb3e5e901fb85cecdb3874fa0c7649a3629be",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.06602407619357109
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 211297,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6265840120876497,
"accuracy": 0.6164948453608248,
"balanced_accuracy": 0.6150310331521384,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bdf4f5b49f0ee29aa30c2ca28ef1e400302d793f8bb08b94b53773c69d50c48c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.08080927841365337
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 213351,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3984637132349555,
"accuracy": 0.5677799607072691,
"balanced_accuracy": 0.4565550938263096,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "75ae6aa393878d072945e96f452b96e09e9cf40fb3b2ea8457e186d70b05e841",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07724830415099859
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 211953,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.3947840773031935,
"accuracy": 0.5540275049115914,
"balanced_accuracy": 0.4383977872139755,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d82f5778b37d6f94a75882921e38848d8207348565af2d0d39c3db90326f7cb8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07652495242655277
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 542820,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.41233788381060416,
"accuracy": 0.5717092337917485,
"balanced_accuracy": 0.46000228267069465,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a809a108f835a4ee2214a0948773f4689220fcd9cfd5deb9f8698001ba8902f9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.11057593394070864
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 419921,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7684496006640077,
"accuracy": 0.8529411764705882,
"balanced_accuracy": 0.8162091423863549,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "51e1d01636b9a0cac92c6cc300ec4e48cce07d310e2e1f190017f13e488a5ee3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.09686489589512348
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 544269,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7017359507404785,
"accuracy": 0.8137254901960784,
"balanced_accuracy": 0.7581788689951661,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "251a88a762c7a5b4b2590567ef275d02cb441de8a0f7e4747da5d76fcc8311f1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
0,
0,
0,
0
],
"verification_seconds": 0.08747569378465414
},
{
"method": "wisp_random",
"dataset": "adl_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 214370,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6492442092593446,
"accuracy": 0.8176470588235294,
"balanced_accuracy": 0.7580917818206289,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5af62dd1e9a42bc066cfd01cfdda16039fcba40a4061a905a997d948728dc8a6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
5
],
"verification_seconds": 0.07864306773990393
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed42/model.pkl",
"bytes": 161894,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.766990439714226,
"accuracy": 0.8378404764229229,
"balanced_accuracy": 0.7554500861212902,
"worst_class_f1": 0.6008610086100861
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "229c7dee72504b72d5605c5150c08e9d3f077a27a287382225e8aa2b179929d1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.0840184036642313
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed43/model.pkl",
"bytes": 129303,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7330825808731976,
"accuracy": 0.819974616811481,
"balanced_accuracy": 0.7079910899772252,
"worst_class_f1": 0.5666456096020215
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b65dfa5093e81d27982784c93a20a87a0bf1930e29ffcc01b8e26ece25504f54",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09258068632334471
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed44/model.pkl",
"bytes": 161270,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7623005521588372,
"accuracy": 0.8392072634970223,
"balanced_accuracy": 0.7456711136811414,
"worst_class_f1": 0.6124694376528117
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "07b9eec05901e515f03e25eb39a3c842dc98b7982e302188bcfda892d1bf1869",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.1038873614743352
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed42/model.pkl",
"bytes": 138656,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7486620464593752,
"accuracy": 0.8456556396970905,
"balanced_accuracy": 0.7473717694082199,
"worst_class_f1": 0.4661016949152542
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5c92495d9331f02daf8560445df4e83b9e98d59510b3c7375f378d40904e45b9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.07749845832586288
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed43/model.pkl",
"bytes": 255990,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.735485533953722,
"accuracy": 0.8366879234754883,
"balanced_accuracy": 0.7346841985437063,
"worst_class_f1": 0.43897216274089934
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "866e338432a7bb4b5f8ee276c77ce81e136694c2c3804ccc1574b56810a86e01",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
2,
2,
2,
2
],
"verification_seconds": 0.1168006956577301
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed44/model.pkl",
"bytes": 130598,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7593223381430728,
"accuracy": 0.8434635312873655,
"balanced_accuracy": 0.7578615088048173,
"worst_class_f1": 0.5243128964059197
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2685a44c752f4432440d3a31e6e8576d149ad136752d6e573c9cb0484f638eb5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
2,
2,
2,
2
],
"verification_seconds": 0.12646049074828625
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed42/model.pkl",
"bytes": 161894,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7062470015513294,
"accuracy": 0.8255736456250595,
"balanced_accuracy": 0.6981685147812106,
"worst_class_f1": 0.48322147651006714
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0207c109e8e11f6f7fbca6ae8e7624f00f59b83cd39b14b918bd7ac4913d9a1a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.08643207233399153
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed43/model.pkl",
"bytes": 287812,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7108976698220147,
"accuracy": 0.8287156050652195,
"balanced_accuracy": 0.7053300591182281,
"worst_class_f1": 0.48825065274151436
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "92f17f4830088544f470bcc13a91313fcd66ad20a772abb355604a78a1188344",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.10610962845385075
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed44/model.pkl",
"bytes": 161581,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.724010717738153,
"accuracy": 0.8322384080738836,
"balanced_accuracy": 0.7199091773450284,
"worst_class_f1": 0.5288831835686778
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a1f2eab36e31ca790baae3ac9cef3cf645c08d2e2fb9c6c6290d4de65d64b818",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.08403156418353319
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed42/model.pkl",
"bytes": 115443,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7488290281373933,
"accuracy": 0.839453125,
"balanced_accuracy": 0.742297928519201,
"worst_class_f1": 0.5042174320524836
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "619d5842e7c5ec71c173c938e126a128095d1bddcc300489956ed39bfccd6de4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.20588375721126795
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed43/model.pkl",
"bytes": 118067,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7570308365882163,
"accuracy": 0.84384765625,
"balanced_accuracy": 0.7481699905884054,
"worst_class_f1": 0.5209756097560976
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "afcb98b4c2dc1350bcb9e53d800a22a5cb5794874689933b475b88d2775ae41a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.24759656377136707
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed44/model.pkl",
"bytes": 117459,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7596369938969753,
"accuracy": 0.8490234375,
"balanced_accuracy": 0.7538129256318069,
"worst_class_f1": 0.5155393053016454
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a2211a1bb2fd15c90b224ac7ea5131aa6d59957b4e3460875c6eb26500ea7fc1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.2584470985457301
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed42/model.pkl",
"bytes": 128065,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7551038036075215,
"accuracy": 0.83273988005997,
"balanced_accuracy": 0.7455861736760279,
"worst_class_f1": 0.49612403100775193
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9f2d52d91c2bc5624fc0d918954b4aff119e5c78ad73416cbb83c1b72ff605ae",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.08599592372775078
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed43/model.pkl",
"bytes": 226783,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.742279603672112,
"accuracy": 0.8223388305847077,
"balanced_accuracy": 0.7216110695134941,
"worst_class_f1": 0.5005302226935313
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bfee24f11c6a76538adbfec619fa58865fdfa08020e9762ec6cf1bdc81812e0f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.10002798307687044
},
{
"method": "wisp_random",
"dataset": "capture24_wearable_activity_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed44/model.pkl",
"bytes": 161581,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7523348286903604,
"accuracy": 0.825243628185907,
"balanced_accuracy": 0.732207260115527,
"worst_class_f1": 0.5368852459016393
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "87ef00aa9a441fa21d0cd24ab3db92c9829499a2634920694f8ab9ed45e59fae",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.09194480534642935
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed42/model.pkl",
"bytes": 213222,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7551308251068606,
"accuracy": 0.8979695431472081,
"balanced_accuracy": 0.7583531872830543,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "52c0487222b97840207d833ce067c1ef209aa9dc6e76b3879cbd48791285e781",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10122111812233925
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed43/model.pkl",
"bytes": 215616,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7773091987294367,
"accuracy": 0.9116751269035533,
"balanced_accuracy": 0.7694623232746799,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9be8c786b60ba44925ce330a3331a3791c2f192e8fb4f2af95c7b285d3a8a020",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13180649187415838
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed44/model.pkl",
"bytes": 542386,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7518046343995171,
"accuracy": 0.9121827411167512,
"balanced_accuracy": 0.7516438722909802,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "562b1a9ca6109509524f71787309d109f44981b05a31982f6310f10bb2a3a2d1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13094902504235506
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed42/model.pkl",
"bytes": 215890,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7811112855157368,
"accuracy": 0.9018214127054642,
"balanced_accuracy": 0.7979395775270787,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "04858553cf1d77303f80f81d2b6fdfcf8f32f846f53b127a0a3531369698cc29",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.10303804371505976
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed43/model.pkl",
"bytes": 316666,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7801701329373244,
"accuracy": 0.8716126166148378,
"balanced_accuracy": 0.8032376592902427,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a50203df8da14032b31032f1609cbcbb997363c09bead7da66293a46d2b92e27",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12858885247260332
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed44/model.pkl",
"bytes": 542875,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7646305706831946,
"accuracy": 0.8796090626388272,
"balanced_accuracy": 0.7734379168736917,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e11c60884f89f1a630a9c14a24705ad1d829e9edb77ab045cdc9142313f76c54",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12358776666224003
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed42/model.pkl",
"bytes": 468799,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6182859411274892,
"accuracy": 0.8502024291497976,
"balanced_accuracy": 0.6535983881129549,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "03ee367d4f45f272040bc32e1f223c210c8ebbb1d9a14e5cf8de193227c52071",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11528230179101229
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed43/model.pkl",
"bytes": 540105,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6311436177301671,
"accuracy": 0.8704453441295547,
"balanced_accuracy": 0.659241603110375,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ba772f5e17588452cbde29336e69acfd4a44c07e565347509a33c81dfc20d447",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1269427239894867
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed44/model.pkl",
"bytes": 543332,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.578905837151691,
"accuracy": 0.8536726431463274,
"balanced_accuracy": 0.6249674973278894,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1e68c747889a8f2466fcf2805645a3014692e747feae48b6abbc19bd014381ac",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13287620432674885
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed42/model.pkl",
"bytes": 215742,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7511955410337825,
"accuracy": 0.8684834123222749,
"balanced_accuracy": 0.7689192612204052,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "97b4a33e5eac4d4c97d0369dc142118fcab3db8345e5cd25563d334b77df938f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.0989870810881257
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed43/model.pkl",
"bytes": 108796,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6954366439264663,
"accuracy": 0.8203001579778831,
"balanced_accuracy": 0.7168607714759039,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0f1c5108cf1dcb1eafcdd881b076eefd0b12eda59dc5e4da1d1ce463951b9e5c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.06485389173030853
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed44/model.pkl",
"bytes": 214370,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.627228447883459,
"accuracy": 0.8052922590837283,
"balanced_accuracy": 0.6445631276390363,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "034d9a680e4312bff79652317774770d18ef487648d866333c627807748d3cc8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.09241700731217861
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed42/model.pkl",
"bytes": 541762,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6183927693961376,
"accuracy": 0.8535864978902954,
"balanced_accuracy": 0.6268799948210574,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "52e3563e30b4160d8a3ad7eef9b961d2486c1534ca26592d490a0a865589c313",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1271434649825096
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed43/model.pkl",
"bytes": 540415,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.634073183238398,
"accuracy": 0.8556962025316456,
"balanced_accuracy": 0.639406051515471,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "501fee63a547a6aa191e6c86ca1eaa5b18a4ea9206b981b94e35a14cc5bb0d60",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12751690112054348
},
{
"method": "wisp_random",
"dataset": "domino_watch_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed44/model.pkl",
"bytes": 216185,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6367216796558524,
"accuracy": 0.8265822784810126,
"balanced_accuracy": 0.6282511507353351,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f86c3d295afb1e24fce9d7afe9efce9f8a687ebe5f2bc10734dcf9bbab1e9e94",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.09841653145849705
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 591456,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.784293009526535,
"accuracy": 0.8009986782200029,
"balanced_accuracy": 0.8151819951716208,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b9db54ada58d01aa9a88928b52628850e3057f6820f1f726b87c01a7ec6c01c6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12829209584742785
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 592918,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7594087239448236,
"accuracy": 0.7810251138199442,
"balanced_accuracy": 0.7894529186109617,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2996f25843ced7e146f6d85bdc6ed30a5857576e422a2520979e15a487384379",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1241801893338561
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 592080,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7708773764124208,
"accuracy": 0.7893963871346747,
"balanced_accuracy": 0.7973621988392794,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "68ea2185c96db68527916dc00959d4927af1085b20db8920929f0de67084fa8e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.15180608443915844
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 182063436,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.729251179192184,
"accuracy": 0.756351392715029,
"balanced_accuracy": 0.7251885891431922,
"worst_class_f1": 0.46808510638297873
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1f41ba703dce73869d5aff7dc3d8de09327e83b4138dc1e06ec26ee24f54f13f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.4741174401715398
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 704450,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8159433539759252,
"accuracy": 0.8123660850933578,
"balanced_accuracy": 0.8184655317757509,
"worst_class_f1": 0.4541832669322709
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f5ffa07e6b96255ed93b361e398740732f20f6e502ae788bb6c76b543f141e8d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1623753560706973
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 590016,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 6.591949208711867e-17
},
"metrics": {
"macro_f1": 0.7652924699658761,
"accuracy": 0.7745638200183654,
"balanced_accuracy": 0.7834846740797345,
"worst_class_f1": 0.018083182640144666
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2826758bc5ee17021e7b8c471ed11739e13983e8ab8736f866423e454d9cd29a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11671893298625946
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 588032,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7873681523995203,
"accuracy": 0.8100347112653834,
"balanced_accuracy": 0.798674374826554,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b911a5c1d76ec88f7fac386b71f1f69977910349f1fc35de1264c8bd172e8578",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12262417282909155
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 591573,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8000927882449437,
"accuracy": 0.8117702745345535,
"balanced_accuracy": 0.8150741858340492,
"worst_class_f1": 0.1109350237717908
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "dc92c8b944619d5abf2b682e0d7188c78f6b63b88229d2bd6225e8b3e6d84d7a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12705540098249912
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 595035,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 6.938893903907228e-17
},
"metrics": {
"macro_f1": 0.7726750498433423,
"accuracy": 0.7914168507415589,
"balanced_accuracy": 0.7876774055916419,
"worst_class_f1": 0.12089810017271158
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1fef32d0456a2818005badf5de3c075659b4041b40d875d109a102ffef16fc25",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12606476619839668
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 591928,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 6.938893903907228e-17
},
"metrics": {
"macro_f1": 0.7993431026428096,
"accuracy": 0.8082954712003517,
"balanced_accuracy": 0.8063771798704091,
"worst_class_f1": 0.05865921787709497
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2596f8d542a9f3892df755c8747f15daa65971ef21cb8e0d5eabf040bdc997cf",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.14119288697838783
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 589011,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7867416975202401,
"accuracy": 0.7984757438077092,
"balanced_accuracy": 0.7980594826403815,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0729d3043a36b77566ef9d77526fe81526baa018aa9a8f44bc39038a5993d1e0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12311080563813448
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 230479,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 1.3877787807814457e-17
},
"metrics": {
"macro_f1": 0.7190410117062849,
"accuracy": 0.7367726806390151,
"balanced_accuracy": 0.7161055800914486,
"worst_class_f1": 0.046008119079837616
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b675a28ea876f097481cf0f186babf1d3cdca36f673c75141a0f273a7996fc7f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08946351986378431
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 592354,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8792767127830559,
"accuracy": 0.8633117422355091,
"balanced_accuracy": 0.8866686986182304,
"worst_class_f1": 0.5959183673469388
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4a0a57580d103371b297935817eb138c41e2332f1a074b9aaf94587d2aa2137f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12175817228853703
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 592217,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8686373451288947,
"accuracy": 0.8555057299451918,
"balanced_accuracy": 0.8761138489486704,
"worst_class_f1": 0.5566433566433566
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "48de32930bc6b94187f3ff3a8e354264ffbb3064b1cdd2944edd2a8fec7d730c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12206245306879282
},
{
"method": "wisp_random",
"dataset": "gotov_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 592860,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9048190879441402,
"accuracy": 0.8947018767646571,
"balanced_accuracy": 0.9067162256666195,
"worst_class_f1": 0.7234567901234568
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c113ffde305f5212b8ea0674e68c244d7987fc618b708aa4f2b207d14f2d0613",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.16558631416410208
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 66847331,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9603439013050938,
"accuracy": 0.9589422407794015,
"balanced_accuracy": 0.9617768102403584,
"worst_class_f1": 0.9343065693430657
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0a1e0ed18c21d6fe8be9fadea73e11bdc65f47b66feca80a5bdf9528a2f7bfae",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.21702496614307165
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 106353855,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": -1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9494755982485993,
"accuracy": 0.9443284620737648,
"balanced_accuracy": 0.9507084789963877,
"worst_class_f1": 0.9022222222222223
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6e11892bdd3343eb962fd1e690ede41e7567bdc368dad79b041a027683f8bb3c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.2986576007679105
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 106727683,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9536047751541364,
"accuracy": 0.9491997216423104,
"balanced_accuracy": 0.9554135283314116,
"worst_class_f1": 0.911504424778761
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9f08a3ce13a054fb2a954aa97b14c1e7a045f9f7e5bad43d4e2fa904ba04dfc4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.32660636119544506
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 109082947,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9507493107548275,
"accuracy": 0.9378925331472435,
"balanced_accuracy": 0.9605487984667551,
"worst_class_f1": 0.8957055214723927
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0b3c854a78d5dca17fa61bd74992831ab14ae6b17eb5144adb36420497308d1c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.3146402854472399
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 69488437,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9722661390125834,
"accuracy": 0.9658060013956734,
"balanced_accuracy": 0.9770293664024313,
"worst_class_f1": 0.9409282700421941
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "393ce4ed2452236ddd27745fe07aaf9f40ca0a0f1590ed38ba28625c3f9d9fa9",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
9,
6,
5
],
"verification_seconds": 0.2198030510917306
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 109039171,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.94426634615951,
"accuracy": 0.9288206559665039,
"balanced_accuracy": 0.9550853661302577,
"worst_class_f1": 0.8724279835390947
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ea8efece2d0848cc16ede574aae1180ee2908669c260f4bc42c5292611a8f360",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.3040814511477947
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 70548283,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": -1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9603615365507692,
"accuracy": 0.9538123167155426,
"balanced_accuracy": 0.9612722403991679,
"worst_class_f1": 0.9227272727272727
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7b3956cbd291ede95168ff2a6d28b63d26a43e9684e037ba627c7cc0fe44ecdc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.21898382529616356
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 30411068,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9635137430915,
"accuracy": 0.9567448680351907,
"balanced_accuracy": 0.9634035317435918,
"worst_class_f1": 0.9223744292237442
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fc0ebebd742e3e0c5b9199f5a3caec835db45198021d4c7de2272ac359577b3c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.15453871339559555
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 30402066,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9654073531705973,
"accuracy": 0.9611436950146628,
"balanced_accuracy": 0.9659125049875467,
"worst_class_f1": 0.9363636363636364
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "78f540cf3087c0fc79bdfdce82373310fa77b3a76a188f6dcc1886f2b797d5e3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.15123513340950012
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 67924027,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9655985809409403,
"accuracy": 0.9556786703601108,
"balanced_accuracy": 0.9711245386810653,
"worst_class_f1": 0.9302325581395349
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "78d8142ced6b45000763b28a2e8f88d73eb1e31be36a8bc9d585e0d18c2d71db",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
5,
5,
9
],
"verification_seconds": 0.2322320556268096
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 67799611,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": -1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9596676620980877,
"accuracy": 0.9494459833795014,
"balanced_accuracy": 0.9635122651082073,
"worst_class_f1": 0.9222614840989399
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7fd6a793318e44a366fbf31ab425824f2708bc411e1cb270ed3e0558955403cf",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.22738107945770025
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 68072341,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9637962821709289,
"accuracy": 0.9529085872576177,
"balanced_accuracy": 0.9693982750317192,
"worst_class_f1": 0.9217081850533808
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "16e4dbdabe73170e232579c7865b2d7c8fb39c4b29d1c8914be212e7746161d7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
9,
6,
5,
9
],
"verification_seconds": 0.22421606816351414
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 109733535,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9501422250540463,
"accuracy": 0.9364285714285714,
"balanced_accuracy": 0.9499210582168228,
"worst_class_f1": 0.8758782201405152
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5006890e670eab4c6f96ada58978af59f03154b2635f4eb5b83194181695436e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
9,
6,
5
],
"verification_seconds": 0.30270709563046694
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 109431135,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9462109554549561,
"accuracy": 0.9328571428571428,
"balanced_accuracy": 0.9473380715803916,
"worst_class_f1": 0.8752941176470588
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "be1b71a6c9d79a676fd10b4333db9054eb6b4753ae89352609c5ced206de97db",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
9,
6,
5
],
"verification_seconds": 0.30106195248663425
},
{
"method": "wisp_random",
"dataset": "handy_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 68605399,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9602680505925019,
"accuracy": 0.9514285714285714,
"balanced_accuracy": 0.959909060884063,
"worst_class_f1": 0.9
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3d1e2a382d39244fbe4348d3104bf3a5d4e9f76d1e9b9550bae9c6018b66a18b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
9,
6,
5
],
"verification_seconds": 0.2278675800189376
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 2370891003,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.3943996945388775,
"accuracy": 0.4341880341880342,
"balanced_accuracy": 0.39529889172768035,
"worst_class_f1": 0.19811320754716982
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "8bb6acfedc4c11f2f59ab482a6c8feed7f8dc61f9443b19a94b8e045b8101d54",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.696244320832193
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 517285,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.3797364195371058,
"accuracy": 0.4098005698005698,
"balanced_accuracy": 0.42310013825699694,
"worst_class_f1": 0.16806722689075632
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c2d4ca643c7a6400fdade9b344ad2e6d13f19f479ddc6c9ebb331624bd7f8fe0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11076897941529751
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 2371692608,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.39416052775079996,
"accuracy": 0.42883190883190886,
"balanced_accuracy": 0.4205249580366873,
"worst_class_f1": 0.20245398773006135
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "253153fe7336aebebac63efcae12e0aa58a4fee4f665559059d24832b45e17e7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.713018720969558
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 2266198779,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.3493869479288254,
"accuracy": 0.3809678819444444,
"balanced_accuracy": 0.35520574414892814,
"worst_class_f1": 0.19378427787934185
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "90a68d971230a9617c8a99d8c006bce5c14cc5775b165be199052df2c4593584",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.515500565059483
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 587116,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.34019784103660566,
"accuracy": 0.365234375,
"balanced_accuracy": 0.3786343493232046,
"worst_class_f1": 0.17903930131004367
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "66228228fad97bfe648e33ec5b21af6b237e923adcee81eef79a9417f7c8f4e0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
11,
4,
4
],
"verification_seconds": 0.09785876236855984
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 2263995968,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.351469207855779,
"accuracy": 0.376953125,
"balanced_accuracy": 0.3768039684244342,
"worst_class_f1": 0.20352781546811397
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3e448a2c29814480b6e97bd5671df91708d6137083295326421759d1f0574c04",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
8,
4,
4
],
"verification_seconds": 4.504460323601961
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 2367772155,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.42957514443609257,
"accuracy": 0.46879432624113476,
"balanced_accuracy": 0.4255324702472358,
"worst_class_f1": 0.2266857962697274
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a3798422bee96ade1b4974e2344a3c099c228c611e41dbd3c2fd6a2b18584709",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.693867314606905
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 590184,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.41079080082797637,
"accuracy": 0.4462174940898345,
"balanced_accuracy": 0.4527289147376499,
"worst_class_f1": 0.17002237136465326
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d905f128c139edd4960352e27701d0df14c5f322cebd010aa0c1a9d8edfcc2a8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11217740643769503
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 2368916672,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4233880396109819,
"accuracy": 0.46418439716312054,
"balanced_accuracy": 0.4487659825863075,
"worst_class_f1": 0.2023121387283237
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2dd211a680d732fbc2e3c2b442c5a170550f79ffe5b0d2e31e2027ce78902969",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.683686343953013
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 2296903419,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.4045781945009205,
"accuracy": 0.4273379681542947,
"balanced_accuracy": 0.3983868486508851,
"worst_class_f1": 0.1737142857142857
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0381189d16ad8a9551ee512689162cf20ae4f80bdbbc1d82677d4498f0057dec",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.528864155523479
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 585468,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.38515500821634396,
"accuracy": 0.40805113254092845,
"balanced_accuracy": 0.42030076334721334,
"worst_class_f1": 0.21149425287356322
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "2d4f8d1e7b0e7d2852bb5ee91fec80d00e6136819f72561d193957072fe6ac37",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11319855600595474
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 2296580288,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.40008244076906796,
"accuracy": 0.42419825072886297,
"balanced_accuracy": 0.41809533022760476,
"worst_class_f1": 0.17865429234338748
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "8c39d17a9ba8b720249d3fb8640328b464aca798b8f4fc04f4f34e86c94b1189",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 3.9741314267739654
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 2313538299,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.42894317767551826,
"accuracy": 0.4700987462554089,
"balanced_accuracy": 0.4321963492839216,
"worst_class_f1": 0.15950920245398773
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6a51c532c58da2852a903abe948896fbdd819fdc8572f883a670f835c608a3f5",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.683324318379164
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 363896,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.39903742181842566,
"accuracy": 0.43459447464773104,
"balanced_accuracy": 0.4465491869813594,
"worst_class_f1": 0.16531165311653118
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6d43b1687051d63a031e99ed8f2b7aca80f066fd8b0dc718ed16e8e649743f76",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13601224217563868
},
{
"method": "wisp_random",
"dataset": "harmes_left_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 2313291200,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 5.551115123125783e-17,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.43042334665685333,
"accuracy": 0.47043159880173085,
"balanced_accuracy": 0.4597562844733023,
"worst_class_f1": 0.1883656509695291
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "8b97fbb28bb726002a9024e9af4eae8a83272abaa7fb69fa019ed75ce9fc4ea8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.670859377831221
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 455682,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.47826892471136084,
"accuracy": 0.5122507122507123,
"balanced_accuracy": 0.5079849702546556,
"worst_class_f1": 0.1615598885793872
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "9b12660beb807daa7f15ce68676b5af6f29c826ec651eafa4e67926a5f2d11cc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
8,
8,
11
],
"verification_seconds": 0.11805807705968618
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 455953,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.4827274993532969,
"accuracy": 0.5171509971509971,
"balanced_accuracy": 0.5125902139658777,
"worst_class_f1": 0.18289085545722714
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e8376233adffc905ad61b15b2c5366a3241faf126ff6e84a162a0b92d4dd1e82",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
11,
11
],
"verification_seconds": 0.11957382317632437
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 2274877760,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5013992889364879,
"accuracy": 0.5433618233618234,
"balanced_accuracy": 0.5148112236986051,
"worst_class_f1": 0.1736111111111111
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0b8f8bcc9da51d65a7218dca78a0e35a5f69043839c3ddbd0c51ca532761b29b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
4,
11,
8
],
"verification_seconds": 4.500875387340784
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 2180587131,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 4.163336342344337e-17
},
"metrics": {
"macro_f1": 0.4734412043443698,
"accuracy": 0.4984809027777778,
"balanced_accuracy": 0.47314577660732104,
"worst_class_f1": 0.11463046757164404
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "00f2a7d26793bbda7faca4d0cab16e7e1c69649f3fa0593af78db3a8e8998c6e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
8
],
"verification_seconds": 4.1530782505869865
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 590496,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.45353736391094296,
"accuracy": 0.4861111111111111,
"balanced_accuracy": 0.49245402085307655,
"worst_class_f1": 0.16145833333333334
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "94f311255d7cdb7256e5919d96345d998dea9ab5bb9f2b25c632272554a832c1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
13,
4,
5,
13
],
"verification_seconds": 0.11939528677612543
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 2177927744,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 5.551115123125783e-17,
"worst_class_f1": 1.3877787807814457e-17
},
"metrics": {
"macro_f1": 0.4605656968569808,
"accuracy": 0.4949001736111111,
"balanced_accuracy": 0.48605973746000264,
"worst_class_f1": 0.12435233160621761
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d6fe8ad972163b8638f89e9b60f3a9a077145b665d872e3e01372408ea514ffa",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 4.361154975369573
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 588184,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.5421584351505209,
"accuracy": 0.5908983451536644,
"balanced_accuracy": 0.5719084457993553,
"worst_class_f1": 0.20958083832335328
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "1418f6e3f196aa4e7835f8ffb632c7eb9962ec8208ed298e9026a71163035fa2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
5,
11,
4,
5
],
"verification_seconds": 0.11561560537666082
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 517285,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.5426752741368953,
"accuracy": 0.5897163120567376,
"balanced_accuracy": 0.5765607711259935,
"worst_class_f1": 0.23384615384615384
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fe2a7d032ee656e4a399e1977b4754a7c5a2b1ee18e3f9f3039150718f61388e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
5,
11
],
"verification_seconds": 0.11062014661729336
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 476284,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.5340476395279575,
"accuracy": 0.5822695035460993,
"balanced_accuracy": 0.563630793774831,
"worst_class_f1": 0.20125786163522014
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "679a7bdceb4ee46b969c415ca020e61c58b6da683decb765ca52bad685d61d9e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
4,
11,
11
],
"verification_seconds": 0.09793207608163357
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 455682,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 5.551115123125783e-17,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.49901099119769765,
"accuracy": 0.5366674142184347,
"balanced_accuracy": 0.5248976016389348,
"worst_class_f1": 0.2107843137254902
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fd31fa757b80f2f3e9293a0644895165773fc9cd67c265449ab5a7c1e736d6db",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
11
],
"verification_seconds": 0.10844338033348322
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 590055,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5129585454890183,
"accuracy": 0.5566270464229648,
"balanced_accuracy": 0.5378205618305084,
"worst_class_f1": 0.2288135593220339
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "17c70707b59020594e415426a34af60b624aacd54716082cb0ce5277954bf4ff",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
13,
11,
8,
14
],
"verification_seconds": 0.11518295481801033
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 2209661888,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5301043138101286,
"accuracy": 0.5714285714285714,
"balanced_accuracy": 0.5399403495460648,
"worst_class_f1": 0.2031063321385902
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3bb3447bf3d2446e4c4123cd2a4a5dd6dab50a51b38d01d4b3bb14a9dd1accd2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
4,
11,
8
],
"verification_seconds": 4.420359159819782
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 587878,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.5020630813880627,
"accuracy": 0.5415510928658605,
"balanced_accuracy": 0.5286122763951864,
"worst_class_f1": 0.19607843137254902
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "44d6aac704b945e7ff7ec9e5ba5b47d45db1db7633bcbd7de43740b7abbaa5f8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
8,
8
],
"verification_seconds": 0.11303388141095638
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 587116,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.4988862797334431,
"accuracy": 0.5389992233440586,
"balanced_accuracy": 0.5251543776171204,
"worst_class_f1": 0.18507462686567164
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "dadf463570cbb51bb37ff17643f647d22434739e1fd95f28e907ddf52983100d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
11,
11,
8,
11
],
"verification_seconds": 0.1164759211242199
},
{
"method": "wisp_random",
"dataset": "harmes_right_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 2209648832,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.5176601134537039,
"accuracy": 0.5586375235770553,
"balanced_accuracy": 0.5295711176752078,
"worst_class_f1": 0.19439868204283361
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d6582b4369b37f15f5b67c4c5553536fb5f11a55e15632a8119a77d2ee4e1314",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
4,
11,
8
],
"verification_seconds": 4.430933751165867
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 128945,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6096479050548741,
"accuracy": 0.8319918044782673,
"balanced_accuracy": 0.5923564159046437,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "39a72f089a60c8d2136b1cdb37d6ff4fce32c67244d8b4d6446fe295ae3cb8a0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.07963023521006107
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 156083,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.612918638732734,
"accuracy": 0.845602224498756,
"balanced_accuracy": 0.6040208282401819,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b884279bc5116301f86f9b9b047d6f39f83e2e70bfaa0cd5222de24f83fc07b2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.45693342853337526
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 128949,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6096479050548741,
"accuracy": 0.8319918044782673,
"balanced_accuracy": 0.5923564159046437,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "651faac607a3a3956c8a909c4870272b49bb201510102066fb552d6ec61ce156",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.08445706032216549
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 128949,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5928265629244448,
"accuracy": 0.8043303929430633,
"balanced_accuracy": 0.5810361579109827,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6c67e257799eb3499b65b1b828816e1037d87739dc13515f9f4ad8b63c1c6a8e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.0769574511796236
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 338789,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5121815033880968,
"accuracy": 0.8032611601176156,
"balanced_accuracy": 0.4962577766609905,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "7964b0b8c3e4c7ec761024654e6a14184907702f09386733bf025a7210105825",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.1298271780833602
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 340825,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5351559377826424,
"accuracy": 0.8064688585939588,
"balanced_accuracy": 0.5059498177634308,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4b42b63d5c748bba1e5c7ad609b3add2aa005086dbbf7df23ce0d3289d228211",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.13942116685211658
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 169676,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6116900772477086,
"accuracy": 0.8550638196349756,
"balanced_accuracy": 0.6305240862826924,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "516514293771fba294dc22e9f4c7efbccdaed62bd58283a201cf01591d03660f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
2
],
"verification_seconds": 0.07959313318133354
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 342837,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.587618207448066,
"accuracy": 0.8846475008946678,
"balanced_accuracy": 0.5780258345781247,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "978a3d968f4687a3eecb529adbe28f9a88c6a7c8409af7f49e2d4fab7f31ba27",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.13611499685794115
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 128949,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6320530481952188,
"accuracy": 0.8665155672193725,
"balanced_accuracy": 0.6538620348912504,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "81d368579274b11fadb3b8f92ba55513ecf36c70c2cf638fb8895a34a4419e1d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.07920224126428366
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 128945,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6098265825969379,
"accuracy": 0.865244322092223,
"balanced_accuracy": 0.6081597599365783,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b2b4e2c420fd84f538b1cd9e327354e336a78e2f6ad7c744cd8706e3007b61a2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.0839704629033804
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 338435,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.5497180265570948,
"accuracy": 0.8531314521679284,
"balanced_accuracy": 0.5354421050786498,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3ad8b51c6de3b613bff453f4991eedf86577f18c58db396b41430c409dbf9757",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.12315355986356735
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 128949,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6098265825969379,
"accuracy": 0.865244322092223,
"balanced_accuracy": 0.6081597599365783,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6df4173edc966f77135834c37ddde485c62dd8a7489cdf08d71302625756b309",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.08461639657616615
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 128945,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8173963795908751,
"accuracy": 0.882277761999432,
"balanced_accuracy": 0.8327524147291605,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4d739e8c0c327536274dd2edb686f25c1e1ff652b33d34572941cd1795429b7b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.07930546812713146
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 339455,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7775502482937635,
"accuracy": 0.9034365237148537,
"balanced_accuracy": 0.7482305404696603,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a92a42dba50c70fda2bcf31af9d5ea060d0e2cb14b427c970dd581f56464ee72",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.12170023191720247
},
{
"method": "wisp_random",
"dataset": "iuwds_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 128949,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8173963795908751,
"accuracy": 0.882277761999432,
"balanced_accuracy": 0.8327524147291605,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a393b658599dd887f9a42ad8d4ec97a8306cd036643de52f930d8ab567be5f24",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
3,
3,
3,
3
],
"verification_seconds": 0.08189722429960966
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed42/model.pkl",
"bytes": 195692,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7428036236211897,
"accuracy": 0.7366336633663366,
"balanced_accuracy": 0.7463115099427031,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "35cd0793db79f56c657858da9d950bcd00c29755da0cc46324942c3d41acbbd1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.10329711902886629
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed43/model.pkl",
"bytes": 386577,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7530918965544832,
"accuracy": 0.7485148514851485,
"balanced_accuracy": 0.7627043309740479,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "86b1f9bd71d34a67a3985181fb212ae5dcc4a3582f3284a3e4e33eb71ef91128",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.10184361599385738
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed44/model.pkl",
"bytes": 197143,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.777767255892256,
"accuracy": 0.7762376237623763,
"balanced_accuracy": 0.7874346983485001,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "8f894de960f765a8739605e6bf46bed77abd05ce0cc4bfaacdcdee07791209e1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
8
],
"verification_seconds": 0.06998705118894577
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed42/model.pkl",
"bytes": 194609,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7958838404465718,
"accuracy": 0.816247582205029,
"balanced_accuracy": 0.8247764415664509,
"worst_class_f1": 0.5833333333333334
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6be3c89830b29436a93355544ab8c9e74d31a11c22c5488573952b93f79e38f6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.06705038249492645
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed43/model.pkl",
"bytes": 195124,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7607287596852985,
"accuracy": 0.7543520309477756,
"balanced_accuracy": 0.7665576406325713,
"worst_class_f1": 0.45569620253164556
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "edbf66cb6e12b62cb7adc6a093779254ba6d098e467ebbd80726961354e293c6",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.09524028841406107
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed44/model.pkl",
"bytes": 197143,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.833983599459939,
"accuracy": 0.8317214700193424,
"balanced_accuracy": 0.8397702744372494,
"worst_class_f1": 0.47191011235955055
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c8a1ff51de6357d56b65587d5ea4760668f86d6326649fecde826baa6bf43f89",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
8
],
"verification_seconds": 0.07438390795141459
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed42/model.pkl",
"bytes": 788723,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8583320296777582,
"accuracy": 0.859344894026975,
"balanced_accuracy": 0.8676731078904992,
"worst_class_f1": 0.6268656716417911
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a124969a171b4ab7fea49bc4e000e578b62d9192bcf3e99a00e8262c985996ce",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
8,
8,
8,
8
],
"verification_seconds": 0.10094030201435089
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed43/model.pkl",
"bytes": 6921228,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9111111111111111,
"accuracy": 0.9113680154142582,
"balanced_accuracy": 0.9166666666666666,
"worst_class_f1": 0.6666666666666666
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d26957d48806f1c581b02fbb2ce1d9804b0c119afb4ff19f5e18074c04f3e094",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.11249503772705793
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed44/model.pkl",
"bytes": 793838,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8565516714672254,
"accuracy": 0.8574181117533719,
"balanced_accuracy": 0.8658212560386475,
"worst_class_f1": 0.6268656716417911
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "29ea043d1216b22c5db967c576795c72dfbaa855dd0a0daed56bd362d6208493",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
1,
1,
1,
1
],
"verification_seconds": 0.10749175865203142
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed42/model.pkl",
"bytes": 194609,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": -1.1102230246251565e-16
},
"metrics": {
"macro_f1": 0.9702950466044578,
"accuracy": 0.9686888454011742,
"balanced_accuracy": 0.9701297607010448,
"worst_class_f1": 0.9113924050632911
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "d8022f233350561fc606adbbb20334d0587ba8d1708c268ca2df0ea3d3d7112c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.08102338202297688
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed43/model.pkl",
"bytes": 554981,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8595948062369086,
"accuracy": 0.8532289628180039,
"balanced_accuracy": 0.8617290192113246,
"worst_class_f1": 0.5348837209302325
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b3f038f9a4d1e2cd1901c8d8f3acb225525868a12ec6fb2aeb82c3b3ef85297b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.1226800736039877
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed44/model.pkl",
"bytes": 612279,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8849608669246121,
"accuracy": 0.8806262230919765,
"balanced_accuracy": 0.8867179121855563,
"worst_class_f1": 0.6666666666666666
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "337f75cf5520357b464199ddeb5123a7679f98e14fe336f555e99cbe798c383b",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.2504494357854128
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed42/model.pkl",
"bytes": 30675898,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7478307728307728,
"accuracy": 0.7594433399602386,
"balanced_accuracy": 0.7422123541887592,
"worst_class_f1": 0.4444444444444444
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "bdccff3234086735c8efb97bf95b41ba974f47277cecb6fc82cee8f6740a1100",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.15433111507445574
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed43/model.pkl",
"bytes": 30830438,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8414983164983165,
"accuracy": 0.8489065606361829,
"balanced_accuracy": 0.8333333333333334,
"worst_class_f1": 0.46464646464646464
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "44b3cf13580087f18e4c668aacea2a0e039b3bcce22129cfc25dc37e0c768753",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.15820361580699682
},
{
"method": "wisp_random",
"dataset": "mhealth_right_lower_arm_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed44/model.pkl",
"bytes": 30621176,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8392896169265338,
"accuracy": 0.8469184890656064,
"balanced_accuracy": 0.8315217391304349,
"worst_class_f1": 0.46464646464646464
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0b525eb01e391ac3c106189748e4774e7051f1072797c28a5ce394e6a4f09249",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
6,
6,
6,
6
],
"verification_seconds": 0.15374513156712055
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed42/model.pkl",
"bytes": 321909179,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.782808883037621,
"accuracy": 0.7777777777777778,
"balanced_accuracy": 0.7664396920134046,
"worst_class_f1": 0.47619047619047616
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "712daccfd21c9116e7b220a7e4d733fde83e22603313fa5c47013ae707f6600d",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.715706367045641
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed43/model.pkl",
"bytes": 154309196,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7863210822620976,
"accuracy": 0.7899305555555556,
"balanced_accuracy": 0.7717697486014151,
"worst_class_f1": 0.5317919075144508
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "4da96f750e45494cfac7b52690d7d888adfef142eade782702938801620d8c96",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.3765896176919341
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed44/model.pkl",
"bytes": 322817461,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7406621280602123,
"accuracy": 0.7579861111111111,
"balanced_accuracy": 0.7268762175805494,
"worst_class_f1": 0.47619047619047616
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "fd9bc35b63ca0e620a4d8eaedf5ccee58908a2d06fd92f1c6c8a5f8a0da2b8ff",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7351966826245189
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed42/model.pkl",
"bytes": 297563171,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.6989074572715867,
"accuracy": 0.7000842459983151,
"balanced_accuracy": 0.6760508882503385,
"worst_class_f1": 0.37383177570093457
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "20467c0f35c90da5501b1c5e905192ca30b0011f6a0b806bbe80ba40c88a6d4a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.6626511849462986
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed43/model.pkl",
"bytes": 295829813,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6940231797042055,
"accuracy": 0.698680146026397,
"balanced_accuracy": 0.6768125701153013,
"worst_class_f1": 0.368
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6b0dd5effa866c8a1b29138c87f1a2055a54407a47522fd9e796403938c60fba",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.6669249190017581
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed44/model.pkl",
"bytes": 141851858,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7247158739945666,
"accuracy": 0.7245155855096883,
"balanced_accuracy": 0.7101332033880609,
"worst_class_f1": 0.4036697247706422
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "41b7e3278f9ac8c30e9788240174fc0fe373634d1bed67671b1d59f27b713c75",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.3645658940076828
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed42/model.pkl",
"bytes": 153892642,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7831926412023957,
"accuracy": 0.7792465300727033,
"balanced_accuracy": 0.7655895754624013,
"worst_class_f1": 0.46956521739130436
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "972b07f92063078910321412edc2b110d833ae0d5afbdd23f3664a72e9fd21e8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.420327321626246
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed43/model.pkl",
"bytes": 153585740,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7977890388559085,
"accuracy": 0.7868473231989425,
"balanced_accuracy": 0.7789270668058506,
"worst_class_f1": 0.4444444444444444
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b01c9d65130a785c746ee0d9705ee0a27eb52f58da411e93babb7384ff13a88f",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.41771274618804455
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed44/model.pkl",
"bytes": 317604151,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7959494363633283,
"accuracy": 0.7762723066754792,
"balanced_accuracy": 0.7728997159742116,
"worst_class_f1": 0.4915254237288136
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6f0c6b5613e76df8b1424c9301da581d80049278ae7cecd53d6b43268d034f54",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7287461068481207
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed42/model.pkl",
"bytes": 152149574,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7777656452114684,
"accuracy": 0.7796555086122847,
"balanced_accuracy": 0.7665927790732726,
"worst_class_f1": 0.3953488372093023
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5246675fb1a261b6a6dbeef965cc6adfed66bb83873766ab4202da68bc22f685",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.4086458645761013
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed43/model.pkl",
"bytes": 152113740,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.76412050853765,
"accuracy": 0.7796555086122847,
"balanced_accuracy": 0.7541487816381051,
"worst_class_f1": 0.43243243243243246
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b3cc4c609f8c65135908f512fd3f194b51dbabc49f780e18451de7911e041df0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.38193516433238983
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed44/model.pkl",
"bytes": 152259654,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7785577846318382,
"accuracy": 0.7890802729931752,
"balanced_accuracy": 0.7690620670775831,
"worst_class_f1": 0.44086021505376344
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "333a97f3219b97286c436e88ec4777d7825c319ab3f463c23b66d70f4998ea78",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
23,
23,
23,
23
],
"verification_seconds": 0.4068180527538061
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed42/model.pkl",
"bytes": 319455669,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7699247674457195,
"accuracy": 0.7750320924261874,
"balanced_accuracy": 0.7602778385960877,
"worst_class_f1": 0.3711340206185567
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "832929b8272271bc6b27792732d474ff816d1e7d6a357c558f43dbe33d3b4fb1",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.7088611256331205
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed43/model.pkl",
"bytes": 149126524,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.7920311703422366,
"accuracy": 0.7907573812580231,
"balanced_accuracy": 0.7814562061777864,
"worst_class_f1": 0.42857142857142855
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "51e33652aa3f4d4d6f61e8aabe58672305870b0a45568de614155507f0203d8a",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.3706858614459634
},
{
"method": "wisp_random",
"dataset": "paal_adl_wrist_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed44/model.pkl",
"bytes": 151948870,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7804989539681273,
"accuracy": 0.7894736842105263,
"balanced_accuracy": 0.7705393743488225,
"worst_class_f1": 0.3958333333333333
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b5dfdd5b52fc55364dcd0441c1fb7cd39b1663a1978d04b6b1d68a417738ff88",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
20,
20,
20,
20
],
"verification_seconds": 0.3796448949724436
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed42/model.pkl",
"bytes": 194655,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.926126828018268,
"accuracy": 0.9237976264834479,
"balanced_accuracy": 0.9259866160332428,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "888da1e0d42747468ef5c32fd88620d2f20a1532cc44ef9a2e78001430d36ed2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08373156283050776
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed43/model.pkl",
"bytes": 99194,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9373390528660345,
"accuracy": 0.9400374765771393,
"balanced_accuracy": 0.9314587981199641,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "082730a7b8c759e90da1ece504c471b8277f2fb9e14d9111b26fd7faf11401dc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.062161377631127834
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed44/model.pkl",
"bytes": 196468,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9106480562075838,
"accuracy": 0.905683947532792,
"balanced_accuracy": 0.9113062332361342,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6a9ca05ac9ea12ce9141c63cd033d1d6dced14ac4ee24e2868f6dd69c125f657",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.08026024792343378
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed42/model.pkl",
"bytes": 519536,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9607069422188861,
"accuracy": 0.9529616724738676,
"balanced_accuracy": 0.9629197423085207,
"worst_class_f1": 0.8265895953757225
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "72256fd2b7de2b50a1665ec7cd03547aec9f69d6a7d2953fd578575ab7420509",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.12377436179667711
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed43/model.pkl",
"bytes": 198347,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9424888593632138,
"accuracy": 0.9419279907084785,
"balanced_accuracy": 0.9454272899289901,
"worst_class_f1": 0.78
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "5820902bf30bd558226d0c2b6e0195d1bfc841411c0932b6e335fc86b80a2828",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11050238087773323
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed44/model.pkl",
"bytes": 694157,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 1.1102230246251565e-16,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9446475832944937,
"accuracy": 0.9413472706155633,
"balanced_accuracy": 0.9465516272892364,
"worst_class_f1": 0.829971181556196
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "44fb853e3a01699c4db66ebb77e1a58d9d2c6669d67de80d9cd352f3e335bfe8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.1427327049896121
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed42/model.pkl",
"bytes": 62795362,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8304680692494896,
"accuracy": 0.8458029197080292,
"balanced_accuracy": 0.8113854490423412,
"worst_class_f1": 0.5358851674641149
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "62d7318cf47d9efb6f2dd3e6ca60a8c65d299f33df64875e2fb0b5bdae1acbf8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.24092491995543242
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed43/model.pkl",
"bytes": 63752910,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8248201686199893,
"accuracy": 0.8412408759124088,
"balanced_accuracy": 0.8068507107128275,
"worst_class_f1": 0.5384615384615384
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e1e5bca8da318977784e09f71c58727705efd5b86c185d8a831cf08c6cfe4add",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.2815354336053133
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed44/model.pkl",
"bytes": 62588844,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8211673302288754,
"accuracy": 0.8403284671532847,
"balanced_accuracy": 0.8039473847047613,
"worst_class_f1": 0.5373134328358209
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "508284c3f5768f28860e48ea9ea057daef15ab81d29ef6d8c688855020581ff2",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.23504982236772776
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed42/model.pkl",
"bytes": 197040,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": -1.1102230246251565e-16,
"accuracy": 0.0,
"balanced_accuracy": -1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9561457774540599,
"accuracy": 0.9571734475374732,
"balanced_accuracy": 0.9573821531111663,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c53f9bc0825d8cab7fa0ba36c553909cde9373814972c6e31681651047106a63",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.11584278754889965
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed43/model.pkl",
"bytes": 99194,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 1.1102230246251565e-16,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9651552354318038,
"accuracy": 0.9657387580299786,
"balanced_accuracy": 0.9661267981171673,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cb26ba3b108f180b33bf984874f47e6270771b5d673ef51fb723691ea21f1f89",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.0970451133325696
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed44/model.pkl",
"bytes": 855341,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": -1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.9767973966100094,
"accuracy": 0.9775160599571735,
"balanced_accuracy": 0.9768469624234878,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f745b4bde373b714c90c335b8c5cd65b3986e78241eb48a6959b1afe626fd144",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.13156522531062365
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed42/model.pkl",
"bytes": 56033925,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7675404704390513,
"accuracy": 0.7796052631578947,
"balanced_accuracy": 0.7538230037404436,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "463314bab2079ab16263df8e7c5799c8269b32556b58448ce9c3f56c1b304585",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.2576223621144891
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed43/model.pkl",
"bytes": 56014725,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7623319002781765,
"accuracy": 0.7763157894736842,
"balanced_accuracy": 0.7505501469112027,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "0dfb08aea1d728fce3389d61e02d54faef0a3f9a7ab613f2292ee037997060b7",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.24346212297677994
},
{
"method": "wisp_random",
"dataset": "pamap2_hand_imu_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed44/model.pkl",
"bytes": 56797677,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 1.1102230246251565e-16,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8976064609725526,
"accuracy": 0.9133771929824561,
"balanced_accuracy": 0.8840144162810198,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6070cd99340fe4ff4f92fd11653fee008f03b3648d0406b8b1a52dbb24069af3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
4,
4,
4,
4
],
"verification_seconds": 0.24854245502501726
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 0,
"seed": 42,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed42/model.pkl",
"bytes": 247303,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.774132262918264,
"accuracy": 0.7813700384122919,
"balanced_accuracy": 0.7826521562757212,
"worst_class_f1": 0.20817120622568094
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6697b406a31e23bea6b952561315e8473da41831f2c3d30d6b16fd958eb877c0",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.11846522241830826
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 0,
"seed": 43,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed43/model.pkl",
"bytes": 1458792410,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.7181558240095015,
"accuracy": 0.7084507042253522,
"balanced_accuracy": 0.7099612955750377,
"worst_class_f1": 0.24347826086956523
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "a18797baf8a57ae9d67afe35877e16545ffc71a22a3e9db470615d891e277787",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.8971761520951986
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 0,
"seed": 44,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed44/model.pkl",
"bytes": 1444945600,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.75091127432255,
"accuracy": 0.742573623559539,
"balanced_accuracy": 0.7438359518494067,
"worst_class_f1": 0.29577464788732394
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "51fb1abc29411f60dedc5d31c452b05a8b6ab852c7267e5ce83b385ee30abc70",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.8266507973894477
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 1,
"seed": 42,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed42/model.pkl",
"bytes": 1406824768,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.688070847675892,
"accuracy": 0.6940909663600857,
"balanced_accuracy": 0.6928882984232319,
"worst_class_f1": 0.4200772200772201
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "cce00ee54265fd663eb667bf5dc0e95effaea966d64ac748b8ed1ff4d188ebd4",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.7140465062111616
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 1,
"seed": 43,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed43/model.pkl",
"bytes": 1409985088,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.6931772279311114,
"accuracy": 0.6961068413758347,
"balanced_accuracy": 0.6949493182373486,
"worst_class_f1": 0.39651416122004357
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "3da4f861e14224e8cd3d13f3937b341320acfc02b616be0df87bcc63a5cdf558",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.7301150457933545
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 1,
"seed": 44,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed44/model.pkl",
"bytes": 1411856544,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 8.326672684688674e-17
},
"metrics": {
"macro_f1": 0.6725085349439842,
"accuracy": 0.6828146654907395,
"balanced_accuracy": 0.6791010276060436,
"worst_class_f1": 0.17481203007518797
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ded420ffe0a3aa5e6334b29b25d289202890519cfa1b8ab889ff46e9ca079592",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.6930082300677896
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 2,
"seed": 42,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed42/model.pkl",
"bytes": 1348189179,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.8465069012503283,
"accuracy": 0.8521433591004919,
"balanced_accuracy": 0.8542029870340069,
"worst_class_f1": 0.43606255749770007
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "6de1903ee0fc8da851336f261ae566cd1a945f62a4124899188d869387491f71",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.5633741300553083
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 2,
"seed": 43,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed43/model.pkl",
"bytes": 1350693119,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8476215612403961,
"accuracy": 0.8535488404778636,
"balanced_accuracy": 0.8553756774503352,
"worst_class_f1": 0.421875
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "e11c85e09a30475ece633f18dd44eb2f2279e55ac903e386b2448f30c6ecf6f3",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.5569702116772532
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 2,
"seed": 44,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed44/model.pkl",
"bytes": 1342968119,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.8392289410102078,
"accuracy": 0.8493323963457484,
"balanced_accuracy": 0.8512439984972445,
"worst_class_f1": 0.3232533889468196
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "baa4b60ce41b27b2cbf0eae874ed6b4aef19ce3c464167a3f641c8d10888c9d8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.5231892075389624
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 3,
"seed": 42,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed42/model.pkl",
"bytes": 1402628768,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 5.551115123125783e-17
},
"metrics": {
"macro_f1": 0.70277937521498,
"accuracy": 0.7029740008728724,
"balanced_accuracy": 0.7036263928622836,
"worst_class_f1": 0.41839080459770117
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "c1fc3a1fa5bb8666f48ab3a3c198c59c5a5da8a006d64341ed8b2488c4d3838c",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.695559806190431
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 3,
"seed": 43,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed43/model.pkl",
"bytes": 1402410746,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6994422646652237,
"accuracy": 0.695180497537253,
"balanced_accuracy": 0.695636607852856,
"worst_class_f1": 0.4831130690161527
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "aafccdb6f0cb69725ae4197d1db86d820a4db509aada4a9922c4b2dddfc16dcc",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.6815116200596094
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 3,
"seed": 44,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed44/model.pkl",
"bytes": 1401175926,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.6921004642618573,
"accuracy": 0.6919384001496353,
"balanced_accuracy": 0.6924893673415535,
"worst_class_f1": 0.2857142857142857
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "b9984a03b030ccf5a06961c0fb690909e20be3b0c4aea245c76a29995ef1320e",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 2.714278470724821
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 4,
"seed": 42,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed42/model.pkl",
"bytes": 642964,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 2.7755575615628914e-17
},
"metrics": {
"macro_f1": 0.7263448253080516,
"accuracy": 0.7255996953928163,
"balanced_accuracy": 0.7267649113642366,
"worst_class_f1": 0.13160518444666003
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "ecbb4e620aef4f605b9a0657077e8589d0c086d0886507bd4d344048bb336d59",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.12464579846709967
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 4,
"seed": 43,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed43/model.pkl",
"bytes": 643937,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7221425984704912,
"accuracy": 0.729343825358548,
"balanced_accuracy": 0.7322211437109059,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "57c7e0ea59741b1dc82941501812441efc800bee4b335c72b05ae7bce4cd0c54",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.11194508150219917
},
{
"method": "wisp_random",
"dataset": "wisdm_watch_accel_v1",
"fold": 4,
"seed": 44,
"model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed44/model.pkl",
"bytes": 643950,
"load": "passed",
"synthetic_prediction": "passed",
"frozen_prediction": {
"status": "not_run"
},
"saved_prediction_metrics": {
"status": "passed",
"paper_metric_deltas": {
"macro_f1": 0.0,
"accuracy": 0.0,
"balanced_accuracy": 0.0,
"worst_class_f1": 0.0
},
"metrics": {
"macro_f1": 0.7196462587809807,
"accuracy": 0.7240766594745526,
"balanced_accuracy": 0.7256429459103105,
"worst_class_f1": 0.0
},
"claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split."
},
"sha256": "f7822a30c9af1ab31909c8eee97cd6685b0825d809e80ae86a3378ea161df9f8",
"source_status": "load_and_smoke_verified",
"synthetic_prediction_class_ids": [
10,
10,
10,
10
],
"verification_seconds": 0.10468210000544786
}
],
"completed": 360,
"status_counts": {
"load_and_smoke_verified": 360
},
"full_frozen_prediction_status_counts": {
"passed": 6,
"not_run": 354
},
"saved_prediction_metric_status_counts": {
"passed": 360
},
"public_model_text_hygiene": {
"files_checked": 363,
"passed": true,
"failed_relative_paths": [],
"scope": "Public model JSON/CSV and validation report only; no credential file is read."
}
}