{ "generated_at_utc": "2026-10-02T12:57:11.649548+00:00", "slurm_job_id": "55937950", "environment": { "wisp-har": "0.1.0", "numpy": "2.3.5", "scipy": "1.17.1", "scikit-learn": "1.7.2", "joblib": "1.5.3", "threadpoolctl": "3.6.0", "python": "3.12.13" }, "code_provenance": { "release_wheel": "wisp_har-0.1.0-py3-none-any.whl", "release_wheel_sha256": "89c1cccabaf095b66fe8d4b82a007b868316302ca613d6bfe1a9a8b2995876b0", "installed_checkpoint_api_sha256": "261bf3c45c2ef7f5a21ff188533aa19a46a6774e261fb6ee1d0f25cfb794b47f", "installed_package_source_sha256": { "wisp/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", "wisp/ablations.py": "d92d2cc5a369ca5d6a3baac156355b0b21b7816778ba9e40eb0c898bafaf8354", "wisp/cis_algorithms.py": "c7b9b262d9bceed5f6a31821a26aeb044350b6af759d100876b3967a36a93be5", "wisp/core/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", "wisp/core/data.py": "a39ff2a5a8cf67c9e59e944bb065e8274ccd7d18b319efeba46044e6cc372306", "wisp/core/metrics.py": "113b8461fa7edd883cebd961b77e15dde9ffc8c3dbf538e313c246e350838c7c", "wisp/cpu/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", "wisp/cpu/algorithms.py": "b1c855877ef70ccc4abb0726c626933053120272fcfeb8d5129603c6f59500e4", "wisp/cpu/features.py": "aab2b76f8dbf7d7faf9b4b1785b2596a2596ba4a805a18e0c856b0ea581ef668", "wisp/cpu/hmm.py": "2c42f3c2076aed5143f4d7b850f88b0b6486ceab72ca291184a9c5859e0ddae0", "wisp/cpu/probability.py": "1232011f6eac18d45fd13580f96e79b9b8b088f8661a863130fe7084b1e67cc9", "wisp/registry.py": "a7a96dfa10ca77d95032e4e62e1a526fc53503ee4618fa6194a73a97ae523828", "wisp/search/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", "wisp/search/cascade.py": "2f2d19573a69afe12015657f13a89caf0d28466ffd703a31f2afd951c7d1e245", "wisp/search/controller.py": "895240d4ba5b79088766a972a76112d15fa40d4e6fa4f5440bf47b6f2a0ec451", "wisp/search/dataset_fingerprint.py": "8e976b35e37534c4fd795d529d63ac01cb0e95b87df4f090791d0d6611fb08ca", "wisp/search/grammar.py": "a038e582182aa1bed8b03a173f02b6977730dbcde07f44a3cd721f87e6d08971", "wisp/search/motifs.py": "5d0375afbb4cec325ccb0d476f27c797d0e335e6d4e292a16b9330f3565a7a44", "wisp/search/operator_program.py": "654975111d32566a712dd7ae8e386c577f9fc0196dad8582939ab018f32ff8ec", "wisp/search/posterior.py": "0fb27135a136f9fc77c0bdc6facb959a85587e629e4d99c40f95ff4262507e1f", "wisp/search/profiles.py": "0acb26c966b72721b42f9e90183186ddb1db713a42d451d886103e0ebf7a16a6", "wisp/utils/__init__.py": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b", "wisp/utils/jsonl.py": "ab9819981cf6f56436edada19b01cfb11d78a3af40d05504c7b9740384441daa", "wisp_release/__init__.py": "c4c3a0ae71ab4eb449ac357d5cddaa40b897a6bdc683c5713e019f3fcc5027a2", "wisp_release/checkpoints.py": "261bf3c45c2ef7f5a21ff188533aa19a46a6774e261fb6ee1d0f25cfb794b47f", "wisp_release/cli.py": "4d4f7106bd5eca4c8496c55b0a03c967df589cf1ec2e7f365db0ec6798c980a5", "wisp_release/data.py": "5dab92de198c7bca28fb620f830168be3bf936bb3c8a8c9ca4dd5762c6cf0a89", "wisp_release/methods.py": "2f1eac83ccb9fe7c8d0e2a7520f746df7f522470af8dab65322a261acc17abcd", "wisp_release/scoring.py": "aa1e89dd83b7295e0dbe087b441c13bae3247a04129f4710006da9bc49386dc5", "wisp_release/selection.py": "7077a0bfef5c9ec3371c64b0bfdef1d285ee9f9902142e2f5676d0f0462d72b6" }, "installed_sources_exactly_match_release_wheel": true, "notice": "Wheel digest identifies the built release artifact; source hashes identify the actual installed implementation used by these model checks." }, "expected_available_checkpoints": 360, "baselines_installed_or_tested": false, "training_runs_performed": 0, "records": [ { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 56197147, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "passed", "n_test_windows": 861, "exact_frozen_predictions_equal": true, "original_input_exact_frozen_predictions_equal": true, "original_input_differing_prediction_count": 0, "differing_prediction_count": 0, "original_and_fresh_X_exactly_equal": true, "original_and_fresh_X_max_absolute_difference": 0.0, "time_coordinate_check": { "raw_time_values_exactly_equal": false, "per_subject_stable_chronological_permutation_identical": { "f3": true, "m1": true, "m3": true, "m7": true }, "original_first5": [ 0.0, 2.5, 5.0, 7.5, 0.0 ], "fresh_first5": [ 0.0, 1.0, 2.0, 3.0, 0.0 ], "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." }, "metrics": { "macro_f1": 0.46857990441875785, "accuracy": 0.70267131242741, "balanced_accuracy": 0.45147912566266135, "worst_class_f1": 0.0 }, "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.46857990441875785, "accuracy": 0.70267131242741, "balanced_accuracy": 0.45147912566266135, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c8089b71a53b67ff1eee86b112362230e51a34a25664cf0652786ab7db00188f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 5.281506538391113 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 56614267, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "passed", "n_test_windows": 861, "exact_frozen_predictions_equal": true, "original_input_exact_frozen_predictions_equal": true, "original_input_differing_prediction_count": 0, "differing_prediction_count": 0, "original_and_fresh_X_exactly_equal": true, "original_and_fresh_X_max_absolute_difference": 0.0, "time_coordinate_check": { "raw_time_values_exactly_equal": false, "per_subject_stable_chronological_permutation_identical": { "f3": true, "m1": true, "m3": true, "m7": true }, "original_first5": [ 0.0, 2.5, 5.0, 7.5, 0.0 ], "fresh_first5": [ 0.0, 1.0, 2.0, 3.0, 0.0 ], "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." }, "metrics": { "macro_f1": 0.5008386012173559, "accuracy": 0.7096399535423926, "balanced_accuracy": 0.4878142449093867, "worst_class_f1": 0.0 }, "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5008386012173559, "accuracy": 0.7096399535423926, "balanced_accuracy": 0.4878142449093867, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "874f0b547ea16f5520641d01894fc4836451d749795dec2e805544925b56944b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 1.4547859011217952 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 213724, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "passed", "n_test_windows": 861, "exact_frozen_predictions_equal": true, "original_input_exact_frozen_predictions_equal": true, "original_input_differing_prediction_count": 0, "differing_prediction_count": 0, "original_and_fresh_X_exactly_equal": true, "original_and_fresh_X_max_absolute_difference": 0.0, "time_coordinate_check": { "raw_time_values_exactly_equal": false, "per_subject_stable_chronological_permutation_identical": { "f3": true, "m1": true, "m3": true, "m7": true }, "original_first5": [ 0.0, 2.5, 5.0, 7.5, 0.0 ], "fresh_first5": [ 0.0, 1.0, 2.0, 3.0, 0.0 ], "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." }, "metrics": { "macro_f1": 0.5804510063302952, "accuracy": 0.7061556329849012, "balanced_accuracy": 0.5332788991510314, "worst_class_f1": 0.0 }, "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5804510063302952, "accuracy": 0.7061556329849012, "balanced_accuracy": 0.5332788991510314, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0729c0358473d08aa0f07abd071f248aca98c576592b692872feaddc172090be", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 3.528837164863944 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 144715563, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "passed", "n_test_windows": 861, "exact_frozen_predictions_equal": true, "original_input_exact_frozen_predictions_equal": true, "original_input_differing_prediction_count": 0, "differing_prediction_count": 0, "original_and_fresh_X_exactly_equal": true, "original_and_fresh_X_max_absolute_difference": 0.0, "time_coordinate_check": { "raw_time_values_exactly_equal": false, "per_subject_stable_chronological_permutation_identical": { "f3": true, "m1": true, "m3": true, "m7": true }, "original_first5": [ 0.0, 2.5, 5.0, 7.5, 0.0 ], "fresh_first5": [ 0.0, 1.0, 2.0, 3.0, 0.0 ], "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." }, "metrics": { "macro_f1": 0.4271967810219083, "accuracy": 0.6806039488966318, "balanced_accuracy": 0.45020425062766495, "worst_class_f1": 0.0 }, "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4271967810219083, "accuracy": 0.6806039488966318, "balanced_accuracy": 0.45020425062766495, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "943a9f9c75edb5b5425533d5fbbc376e39e838942d32562c7b46f0d83a5c653d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 5.605986746959388 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 435753, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.45512076153931874, "accuracy": 0.5837837837837838, "balanced_accuracy": 0.5301783700824152, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e1bde8c4ad01f3152f254f754e19c9ed214d790ca66450317b462ca941e52afd", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 0.09336361195892096 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 239385, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4543957425721672, "accuracy": 0.5704504504504504, "balanced_accuracy": 0.5233597226947379, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4f4a0d1fc0ab03fd159a8dd028bb870360eb0e7c89f779b36e050e2bb56a9c8d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 0.0711149936541915 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 195613, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4444367338790675, "accuracy": 0.5585585585585585, "balanced_accuracy": 0.5129764381142231, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5ef9f923e8d30c0a7148c0fea06920e1addff9d89cbe6ce1c161db4ebf5b7998", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 7, 7, 7, 7 ], "verification_seconds": 0.09933141898363829 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 108071, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6315284716337148, "accuracy": 0.6824742268041237, "balanced_accuracy": 0.5954227178234918, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a3646800e53bd55b507cd0545383377d8b1fd0d4e5e4e21b94e79de64c717bdd", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.0682113254442811 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 212796, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7140298291222251, "accuracy": 0.7793814432989691, "balanced_accuracy": 0.6659095889792745, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5ca91cbb72343bbe53d408cd5267fe1bf4e70a06e02b0afd65f2b6b8d6ef8c1e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07314625475555658 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 108151, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6091680266775356, "accuracy": 0.5711340206185567, "balanced_accuracy": 0.5892073482822214, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d12df30d0ab256561d9b583bfe910dfedb13138949b410bf219bc2ec7d622533", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07153434678912163 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 215198, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.39874385233957527, "accuracy": 0.555992141453831, "balanced_accuracy": 0.45164475266682325, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "650947efac20189d66cd0c69348da623efd950daf0ad759b404b6e72b64505df", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.0834163036197424 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 213799, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.38419952735589424, "accuracy": 0.5520628683693517, "balanced_accuracy": 0.4058504505983019, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bfb5d044c9caa32d3b496619b49e2e60251318e4100e771fccb2b364885acc5f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07257030159235 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 109750, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.37838383738902187, "accuracy": 0.48919449901768175, "balanced_accuracy": 0.42086382580606824, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "10415ecf33b468d8584ef8e41d076d63fc4cafb231db82cd1d682737a43ca307", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.06706883013248444 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 215136, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7160946161678324, "accuracy": 0.803921568627451, "balanced_accuracy": 0.7463469409476607, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c47064840f9187a5cfa3fc181f829aab1ba77c8754f79db9fcff50e9c62a98c2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07794979959726334 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 267079, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6873594627210284, "accuracy": 0.8215686274509804, "balanced_accuracy": 0.7955491343583891, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "19a43e6143a7572be809bec98c7e517eb332cfd030fe87b86f78490a0c8463fa", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 0.06673328019678593 }, { "method": "wisp_evolution", "dataset": "adl_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/adl_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 265353, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.64200883315541, "accuracy": 0.788235294117647, "balanced_accuracy": 0.7487627012767595, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9d3b9fb55b2ff1b4f05643df09594a84438e24bbdfd8beb6cd0b95b045607e35", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 0.0789025230333209 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed42/model.pkl", "bytes": 128855, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7773584180364441, "accuracy": 0.8507273259787171, "balanced_accuracy": 0.7698689175281773, "worst_class_f1": 0.5946502057613169 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "efa2e8604213ef4989368eedc7e077746724d0e45eba640c93ed8a6df76b2bad", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.0911676436662674 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed43/model.pkl", "bytes": 64391, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7707432326452675, "accuracy": 0.8394025187933223, "balanced_accuracy": 0.7700530056485361, "worst_class_f1": 0.6070409134157945 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "feac6c70266738be626a463ec0bd7deb778bdd63e4acaaa0027692959aec4598", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.1280159205198288 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold0/seed44/model.pkl", "bytes": 64845, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7583360117244344, "accuracy": 0.8415503270526213, "balanced_accuracy": 0.7514210173129554, "worst_class_f1": 0.5420944558521561 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fc5accae13589359550c006de9301f3c37970d3332cdb9edfa166cb28b8491fa", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.0684780403971672 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed42/model.pkl", "bytes": 347933090, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.6832491969161993, "accuracy": 0.7742128337983261, "balanced_accuracy": 0.6653189192827791, "worst_class_f1": 0.44139650872817954 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3b55ea01da5f0a0d1832836c99428f6ae17ebfc6c18101aad6633fbc89d0a8c6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.8010620893910527 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed43/model.pkl", "bytes": 130511, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.748965216157762, "accuracy": 0.8414707054603427, "balanced_accuracy": 0.7496277163543152, "worst_class_f1": 0.48459958932238195 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a1f9521a70baff07ae0e2705279c268dae32f3bf01088791a01b740a2d0eee7b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09537813626229763 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold1/seed44/model.pkl", "bytes": 592029080, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7095731117734799, "accuracy": 0.7794938222399362, "balanced_accuracy": 0.7155070471172339, "worst_class_f1": 0.5171102661596958 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c77d246c5eec2ee3c943bbc43e366ebedfb234cfe3c19dbd2341daafec7efffa", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 1.529852494597435 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed42/model.pkl", "bytes": 130236, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7293373556148361, "accuracy": 0.8356660001904218, "balanced_accuracy": 0.7377749513518659, "worst_class_f1": 0.47665847665847666 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "48284dba20870ad2877b9189df3bec278653f04c8a3c99812f0840188a0d657e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.10741668753325939 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed43/model.pkl", "bytes": 64221, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7109350771181784, "accuracy": 0.8188136722841093, "balanced_accuracy": 0.7237143821038571, "worst_class_f1": 0.45794392523364486 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3c3152512aed2c8bbb1c16efd7942035151f9e89e139a58c8abfc2f9158470b2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 2, 2, 2, 2 ], "verification_seconds": 0.07644576393067837 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold2/seed44/model.pkl", "bytes": 128604, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7151153882450071, "accuracy": 0.8309054555841188, "balanced_accuracy": 0.7169997827310548, "worst_class_f1": 0.4375 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "66e2b8cd1c80f44d36e2ff62422ad6509515f99359a52b0fc62b543192605664", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 2, 2, 2, 2 ], "verification_seconds": 0.09162094537168741 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed42/model.pkl", "bytes": 127524, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7519431599340909, "accuracy": 0.84189453125, "balanced_accuracy": 0.7382519385358287, "worst_class_f1": 0.5137614678899083 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0e9ce884e630f754d915db2bf33fe41392c659e2fd04a569225695902c90db0f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 2, 2, 2, 2 ], "verification_seconds": 0.1081920899450779 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed43/model.pkl", "bytes": 164122, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7641903008260709, "accuracy": 0.84794921875, "balanced_accuracy": 0.7437808965170654, "worst_class_f1": 0.570273003033367 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "966eab1af20383e4ba9a7aa3d014fe22d7dde2c1b34eb75681cb90563f999662", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09193179570138454 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold3/seed44/model.pkl", "bytes": 175628, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7666007836147859, "accuracy": 0.85, "balanced_accuracy": 0.7458548085139014, "worst_class_f1": 0.5743174924165824 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "34fddd3eed7a6389bf16550b29df02b8646dd08bd5a8961363070b236323a6af", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09885944984853268 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed42/model.pkl", "bytes": 164896, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7478668985168254, "accuracy": 0.8256184407796102, "balanced_accuracy": 0.7282771903274528, "worst_class_f1": 0.5169927909371782 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7ded22cb68e6229b81b7d16ca5c0edaa46e66099a10fbc5c90d8af138a92c99c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09246528707444668 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed43/model.pkl", "bytes": 161763, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7447052020556495, "accuracy": 0.8229010494752623, "balanced_accuracy": 0.7243578069099286, "worst_class_f1": 0.5125 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a888536d0d5f7dc270067e8ea39665c7cd4e1e29a331fa57600076b157985bab", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09998060762882233 }, { "method": "wisp_evolution", "dataset": "capture24_wearable_activity_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/capture24_wearable_activity_v1/fold4/seed44/model.pkl", "bytes": 133859, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7477843382157683, "accuracy": 0.8235569715142429, "balanced_accuracy": 0.7262215740897333, "worst_class_f1": 0.5230125523012552 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9e3ad341c1e95d21e5bcd9cd71ae1eda615c27c49f8d67c2bd75c563da10a1ac", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.100236008875072 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed42/model.pkl", "bytes": 212600, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7664598552903535, "accuracy": 0.9142131979695431, "balanced_accuracy": 0.759350201817826, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "91782879414932f4fd1d83de2b64f157f80a4203e71dd9cebc63de3836655b2c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.111138129606843 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed43/model.pkl", "bytes": 109344, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7779119924157182, "accuracy": 0.9030456852791878, "balanced_accuracy": 0.771893748615854, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f987480f734d991975f9214d65fb667624b757b644f4187dd9eb17b44236ff80", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10117201320827007 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold0/seed44/model.pkl", "bytes": 215498, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7588925510712723, "accuracy": 0.9055837563451776, "balanced_accuracy": 0.7528336709974588, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "dc05c3132f80de17ce71294cf01693bac44d720b3374913fabdf232c8c762ba9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.09738574642688036 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed42/model.pkl", "bytes": 539572, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7654332232404667, "accuracy": 0.8809418036428254, "balanced_accuracy": 0.7650026167595115, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cf2712de8548876ab47ead3c76641d2aff86285a081d09d3e3e1a1fee26bcd7b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12981910444796085 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed43/model.pkl", "bytes": 318789, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7475567873753366, "accuracy": 0.8431808085295425, "balanced_accuracy": 0.7409736397320528, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f207d6cb99c6be9e0229de253aeb5c554106e6808b8630e28c87738f9f7582b3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12939182203263044 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold1/seed44/model.pkl", "bytes": 216225, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7643324610484091, "accuracy": 0.8698356286095069, "balanced_accuracy": 0.7816075184988293, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2bbd335eeb771d9febd05165213d6c533578169d0a8d57b0127a817714d12005", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10225305519998074 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed42/model.pkl", "bytes": 542893, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6063961064701442, "accuracy": 0.8345864661654135, "balanced_accuracy": 0.6476178172323037, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c250dac4028d1978b59d5ccb9dfc9881c040b1623291d93d23e68351b1e10a6a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12861014250665903 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed43/model.pkl", "bytes": 540198, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6786222922383697, "accuracy": 0.8883747831116252, "balanced_accuracy": 0.7045844794521647, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c780ed09a5710ffeda2ecaf05f41dbdc62d24781faba84ac4d25c6c3f06098a2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12846562545746565 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold2/seed44/model.pkl", "bytes": 542016, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6059659135596054, "accuracy": 0.8340080971659919, "balanced_accuracy": 0.6479127025827084, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ad2ecac326b372c3dde4f10e2b6079b0eed91a7cd39fe073aec238131c0dc91f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13145661167800426 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed42/model.pkl", "bytes": 212454, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6328932422472617, "accuracy": 0.8124012638230648, "balanced_accuracy": 0.658274984974044, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e00af7926b2d0764a2ee380dab3e15d8608f58be432b0441448491310da1bff0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10056019108742476 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed43/model.pkl", "bytes": 215087, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5656274211024497, "accuracy": 0.7918641390205371, "balanced_accuracy": 0.5810575952113626, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e9d96df627dad5f8005a5dde54b70ae26ec7fcc47b7571fe2b429156d40a72bc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10281260497868061 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold3/seed44/model.pkl", "bytes": 213083, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5766493081312468, "accuracy": 0.7969984202211691, "balanced_accuracy": 0.5883944723651199, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cabbe2a146876bdf4d60ef3b7511d624112af70f087adaf6362efc6d24842f1f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.09937153570353985 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed42/model.pkl", "bytes": 319198, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6476508962514093, "accuracy": 0.8443037974683544, "balanced_accuracy": 0.6502484226519598, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7f022f0cb93b1f3d1b3333bc9280f649705c197ede1e8c632a915c4648190fcc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12414687685668468 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed43/model.pkl", "bytes": 541362, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.618716820825527, "accuracy": 0.8510548523206751, "balanced_accuracy": 0.6259556576317066, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e6cb7ff7b7eabaeebceb94181be0b077fc467ab9693a583392ee8e5eb829a29b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.14895725715905428 }, { "method": "wisp_evolution", "dataset": "domino_watch_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/domino_watch_accel_v1/fold4/seed44/model.pkl", "bytes": 210604, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6790563959748898, "accuracy": 0.8611814345991561, "balanced_accuracy": 0.6645928651150633, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "913f6893df2fc3e36e06434f73869862a53c966f78f3011f0906c81900623bf7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.09429517947137356 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 701279, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7672489965511144, "accuracy": 0.780731384931708, "balanced_accuracy": 0.790075815022735, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b8dcc5c64a5343a93e548def0659ae9f69aa59f26b9efae8b18ef60c04999072", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13250253908336163 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 592198, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7835365865957504, "accuracy": 0.7968864737846967, "balanced_accuracy": 0.8124908062429019, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "86844eb8a9e175f68acfb8f2777e19dc90862f4423fa1cb06adf28f84b65fa1b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13126813992857933 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 589580, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7870352562954467, "accuracy": 0.804817153767073, "balanced_accuracy": 0.8210987640639349, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b37545d5ae857e640fab4e7933811f530e9ca1d9d6d5ddc7f26ee1e9e6f5cfd7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12988745048642159 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 175868950, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.745442544578232, "accuracy": 0.7687480869299052, "balanced_accuracy": 0.7353281827704078, "worst_class_f1": 0.4176533907427341 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f7a4e86ede990675448cc131f8c22ce073b831c3e82ae1722c8e9e24578d3842", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.49978523049503565 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 594940, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8188924951875662, "accuracy": 0.8221610039791858, "balanced_accuracy": 0.8239290176963354, "worst_class_f1": 0.39609483960948394 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "aa10b8a5133b9e53941adf6e1b24b5b5c50a0f2ed7f993b95ab3dad2f1499321", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13092663139104843 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 593446, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.204170427930421e-18 }, "metrics": { "macro_f1": 0.78724631788864, "accuracy": 0.7976737067646159, "balanced_accuracy": 0.8042867736850741, "worst_class_f1": 0.014492753623188406 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6d275ae2416840f97d5e25539b64c1cfd4c7550e8ecf4182f57a1a2401f5928c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12541959341615438 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 625820, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8096769252697558, "accuracy": 0.8215525402335121, "balanced_accuracy": 0.8207685956698325, "worst_class_f1": 0.45045045045045046 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "61832a626e710b6e585baa509e8143bc38aa82c5998d30a35721bdda870a6894", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12353577744215727 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 592488, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8028690592578791, "accuracy": 0.8247081098138214, "balanced_accuracy": 0.8126440443873308, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "23861efe32842b0b497613aa381f5e4552922c858b2b4394ed7de391860aabb9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12496596947312355 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 593567, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8063658973106798, "accuracy": 0.8302303565793626, "balanced_accuracy": 0.8136036002493117, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "19983a22327bbb26e1f068f170d5f667f86a83a1fe5e24ec5bb2571f55ae1319", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12128795497119427 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 677345, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.783907803140329, "accuracy": 0.8038985783379745, "balanced_accuracy": 0.7821143484738341, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "be9943a93e0b583388c6ce972347ee31a40a057a579f1cfd9b0b4aa8ecb99d00", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1648122714832425 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 556104, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 6.245004513516506e-17 }, "metrics": { "macro_f1": 0.7776997415582797, "accuracy": 0.7975963652352338, "balanced_accuracy": 0.7793900579312314, "worst_class_f1": 0.03298350824587706 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3fc594cc17fd2cc89b882953aaa152c3f481210354d4e629425eb45734e14ee5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12396656069904566 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 115192, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.697018026421252, "accuracy": 0.7243148175289462, "balanced_accuracy": 0.6970510428018644, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3ab1b44e1e59ea0d0756fb61416e2b05ce9b5704a8b1c373c28cd561cefa0f9c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08868241589516401 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 593151, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8641234565792023, "accuracy": 0.8536787908985218, "balanced_accuracy": 0.8719389189768827, "worst_class_f1": 0.6344238975817923 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1e177f034f0d5272a98ddc35fe6e14cfd7c30f4617902a21d2d51fe77d08ee2f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13063144125044346 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 628812, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8815252691975555, "accuracy": 0.8659691081215745, "balanced_accuracy": 0.8895848428301525, "worst_class_f1": 0.6130884041331802 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5f59b1c56c4e5143ce16e0d4e3e0298a77be839efe4cd4f788a05c9b2f633888", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12962188199162483 }, { "method": "wisp_evolution", "dataset": "gotov_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/gotov_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 591703, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8515240355978488, "accuracy": 0.8400597907324364, "balanced_accuracy": 0.8586857351484299, "worst_class_f1": 0.5885797950219619 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "82ccc3b08284c3d6639155e8047459ee651a6610ebc6ca5d814873a92b058b53", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1400800608098507 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 107027779, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9535440866847888, "accuracy": 0.9505915100904663, "balanced_accuracy": 0.9547187190113678, "worst_class_f1": 0.917960088691796 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0ba2051deeb8953941a5ba1097cf67c02bae3e0400ac3ba552d5ba6d92e523fd", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.30445832666009665 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 29561530, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9664924083067158, "accuracy": 0.9659011830201809, "balanced_accuracy": 0.9675183372111787, "worst_class_f1": 0.9343065693430657 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cd92a861ddcc1fb2e319dc39b7255dabf7596ea0c792856d5377e044f89bacaa", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.15698592364788055 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 66429575, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9631621411148588, "accuracy": 0.9610299234516354, "balanced_accuracy": 0.9639777161822713, "worst_class_f1": 0.933920704845815 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2d11e5bb8191faa9ea6f33fc59a552fca8f9da0161b1ab257b15cc7e2caa5164", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.23916628491133451 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 108937791, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9488100219032743, "accuracy": 0.9364968597348221, "balanced_accuracy": 0.9585341464521031, "worst_class_f1": 0.8946236559139785 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e92e2c69e08c43c5b43a1ee7424453a0c6efcece3d351138a1ecd40cde17b434", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 9, 6, 5 ], "verification_seconds": 0.2992333984002471 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 69160983, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9688875446818674, "accuracy": 0.9630146545708305, "balanced_accuracy": 0.9736233967271118, "worst_class_f1": 0.9453781512605042 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9a0ec2e26f0c92f0a9bee79d4d16c82361fb07bad0e8ba3f6cf78873284fbbff", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.23830208834260702 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 69220315, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9739522200817022, "accuracy": 0.9685973482205164, "balanced_accuracy": 0.9786360981639619, "worst_class_f1": 0.9436325678496869 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "49db770dfcc481f405a102acfe78a27064e9424221313042063188f1a8a2ad48", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.2255033189430833 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 70056227, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9667081563909887, "accuracy": 0.9611436950146628, "balanced_accuracy": 0.9678508084578631, "worst_class_f1": 0.929384965831435 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6111562ba3832d7c834a167ef77c6f8639fe1d979d35017ebbaf8bdede9c839f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.23397216200828552 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 30878124, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.960645742786498, "accuracy": 0.9545454545454546, "balanced_accuracy": 0.9613495525574918, "worst_class_f1": 0.9248291571753986 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7741bfa8ce2e05112fac9f950e9075a2973e9a81f0ada3d9e676ebc487ae4d7d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.18209315743297338 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 30208604, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9646038665262866, "accuracy": 0.9582111436950147, "balanced_accuracy": 0.9668511854523635, "worst_class_f1": 0.9269406392694064 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1f3ae2ef84aa939b59d3b1ef0d07f5bc5e4b87b5fe487f86bd013733f9daf6ad", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.17018190491944551 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 29162012, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9666138221553023, "accuracy": 0.9584487534626038, "balanced_accuracy": 0.972777362601331, "worst_class_f1": 0.9322709163346613 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "22e3d999f78b0ea25cf621e6b0e2233717892e85c2fe821b120231da482b9dd6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.16035774070769548 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 67966647, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9643939005024231, "accuracy": 0.953601108033241, "balanced_accuracy": 0.9703785911125241, "worst_class_f1": 0.9221556886227545 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "edf212eef5bab87a08152d61291dc7ef0e01114f220d4e372ba4d42b06af6242", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.22190038301050663 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 68830105, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.959464542814932, "accuracy": 0.9473684210526315, "balanced_accuracy": 0.9656081680613511, "worst_class_f1": 0.9123434704830053 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6777cd4ce496465a3deb12a74d58cabfdc498145ce3485b9b09b3688ee09aea5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.23183409683406353 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 29370288, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": -1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9665822259579404, "accuracy": 0.9585714285714285, "balanced_accuracy": 0.9665539314970886, "worst_class_f1": 0.9095238095238095 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3772a56d9802e1201b220323acfab33e90e13c7c1e5fdd315fef49329842bac4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 9, 6, 5 ], "verification_seconds": 0.15483754873275757 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 29652538, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9612307601150556, "accuracy": 0.9535714285714286, "balanced_accuracy": 0.9622524467374151, "worst_class_f1": 0.9134615384615384 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "79b0bbd8d789f273915019109e1c79b4012e400ce4336b80a8b2de213b376acf", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.16102203354239464 }, { "method": "wisp_evolution", "dataset": "handy_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/handy_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 109540579, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9508254122131812, "accuracy": 0.9378571428571428, "balanced_accuracy": 0.9511741971307087, "worst_class_f1": 0.88 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "448bed549863e5f10d2843e937a4c4c3fd282cd15171c5f20e4970facf451d81", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.3187501523643732 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 812416, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.387570196628973, "accuracy": 0.4189173789173789, "balanced_accuracy": 0.42617659454061835, "worst_class_f1": 0.13793103448275862 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "036a2ff2ca69e09e785fcf0b0ad2b7cd5b2c14996b915a4840c17eb4e98f44d7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13104734662920237 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 586110, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.3844396970116084, "accuracy": 0.4151566951566952, "balanced_accuracy": 0.4250397919631618, "worst_class_f1": 0.16263736263736264 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "54c708e15be4215fa32d885d85257bd09be95e83ec0beb097a7eb4ff74944b40", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12016687728464603 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 702211, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3870428869869673, "accuracy": 0.4168660968660969, "balanced_accuracy": 0.42767719047090413, "worst_class_f1": 0.1568627450980392 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b6a53d74eabde5c5dfe0f9c1be47bfe0d641ecddde0739efda79fc92933aae38", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.14498190488666296 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 366487, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3281143240077257, "accuracy": 0.3518880208333333, "balanced_accuracy": 0.36986021552619264, "worst_class_f1": 0.1663286004056795 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0cbcab1b6bd4e29ba9d1d1532339c60ceda1eb623ff8b635de0d7a2058f9fd57", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.09013545699417591 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 479767, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3368172298791864, "accuracy": 0.3588324652777778, "balanced_accuracy": 0.3818676836393975, "worst_class_f1": 0.1725417439703154 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5f30f5071527b5e6a91313eece5fd547570f5b23f7859d6feb5a9adb21bed997", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 8, 4, 4 ], "verification_seconds": 0.10633725114166737 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 589671, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.34027982278741586, "accuracy": 0.3645833333333333, "balanced_accuracy": 0.3811477225820268, "worst_class_f1": 0.18385650224215247 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3316aff7e86d626e671de4aeb0b58720856bfee06b6b5bebe482193ca7e62cd4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 11, 4, 4 ], "verification_seconds": 0.11465025693178177 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 856939, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4157316719433681, "accuracy": 0.4486997635933806, "balanced_accuracy": 0.4557413938998671, "worst_class_f1": 0.1743119266055046 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c33c1d7f183f138b4d34ac5927b9156a7fd20c731affa8c8165db2239189b3ed", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13862699456512928 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 591055, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.41011206610734857, "accuracy": 0.4459810874704492, "balanced_accuracy": 0.4506340420650674, "worst_class_f1": 0.1487603305785124 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "06f7274c1dafdffdd23e29988cef74286bacfd6ec13b28f4ed2f677f8361a04b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13259280938655138 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 678051, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.3979308233637672, "accuracy": 0.4329787234042553, "balanced_accuracy": 0.4453328739886988, "worst_class_f1": 0.11907164480322906 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bde21d7f651fc6c81a941892af381009a94022df1980f1e9a01ed5b6018dae5a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 4, 4, 4 ], "verification_seconds": 0.1591173354536295 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 856835, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3957192396436544, "accuracy": 0.4207221350078493, "balanced_accuracy": 0.42766245503552947, "worst_class_f1": 0.2186046511627907 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b3faa0b76c9aba0701a6cdb52920174ae1521ea97aa2a3f64821dfe08d997ae6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.16633606050163507 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 555115, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3848142778821959, "accuracy": 0.4088360618972864, "balanced_accuracy": 0.4180010374181177, "worst_class_f1": 0.2053388090349076 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d435fe4444ee31e12f4aeff2d08da4da90b1581e55789f78c53b63e459cbda72", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.14687734376639128 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 678876, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3791269371977908, "accuracy": 0.40165956492487104, "balanced_accuracy": 0.4179393559999468, "worst_class_f1": 0.1830708661417323 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "63f4c3fc491c715125760737b25c16405481f68eaefd3945ce31338650723c6e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.1423840904608369 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 164963, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.38673689720064064, "accuracy": 0.4308221457894153, "balanced_accuracy": 0.3889641395556819, "worst_class_f1": 0.18230563002680966 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bea2ddaf698e130e949026e1d06658401971805e9a3192abd36778e73b885c37", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 8 ], "verification_seconds": 0.08014317229390144 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 587574, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.40823881367737425, "accuracy": 0.44291578830578054, "balanced_accuracy": 0.45349802538725736, "worst_class_f1": 0.18556701030927836 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fa9923bc96f898e37075dc1fe779e76b14f44e27caf6deda4068d31d0c756f62", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11033798847347498 }, { "method": "wisp_evolution", "dataset": "harmes_left_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/harmes_left_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 589309, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4140329275306399, "accuracy": 0.4480195273493842, "balanced_accuracy": 0.45934733635186287, "worst_class_f1": 0.1822849807445443 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3ed51ab0acde02f67a2049d119fc0a1d83ea9cdebf3833321f09c7ec12c487f1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 8, 4, 4 ], "verification_seconds": 0.12434007879346609 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 681220, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.47706378206167416, "accuracy": 0.5076923076923077, "balanced_accuracy": 0.5092512560398281, "worst_class_f1": 0.1962864721485411 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b45b73fd069a7912e03d1855f10a92784abbc3953e9651cee273a6532f692464", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.1404802268370986 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 554213, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.49312844038510845, "accuracy": 0.5299145299145299, "balanced_accuracy": 0.5207230526700427, "worst_class_f1": 0.1807909604519774 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0752056393d300870f405f1eaa6c1ea04582279181d1e136badfa8ce80222518", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 5, 11 ], "verification_seconds": 0.11337910685688257 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 344185, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.4666524181153792, "accuracy": 0.5035897435897436, "balanced_accuracy": 0.49701966308229717, "worst_class_f1": 0.16022099447513813 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a58060dc11572fe3ccdce4a6d2f0150fcbdf558914403de04c4b8d2f11cfc7ae", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.10804144851863384 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 364507, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4370155710734781, "accuracy": 0.4697265625, "balanced_accuracy": 0.48011368668483795, "worst_class_f1": 0.1203585147247119 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "70840706528cb04f82ac77a1d4c92016dc45ee93a2f96c56a01f09966d31b925", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08191533200442791 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 697475, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4512327358194982, "accuracy": 0.4790581597222222, "balanced_accuracy": 0.49275046578712517, "worst_class_f1": 0.1793478260869565 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5bb2a4d812e31bc98fcc59fb93920ddf01a1167ee426e56150098a931c74aba4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 4, 5, 4 ], "verification_seconds": 0.12659502308815718 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 451638, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.4471513148059462, "accuracy": 0.4729817708333333, "balanced_accuracy": 0.48961303342143864, "worst_class_f1": 0.17073170731707318 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "89fd62d69fb8334e6474dd91d725669067166c5e62c3178ea6068d6478bfd641", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 11, 8, 8 ], "verification_seconds": 0.1261815158650279 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 477487, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.5345935850365053, "accuracy": 0.58274231678487, "balanced_accuracy": 0.5655624417274536, "worst_class_f1": 0.21656050955414013 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ba5cf262812bbe919af9e320dbc6d871a6ea1cd10c91c3aa70e16e6d79de8da5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 11, 4, 5 ], "verification_seconds": 0.10365157946944237 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 701127, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.5360904467830878, "accuracy": 0.5872340425531914, "balanced_accuracy": 0.5652497977212918, "worst_class_f1": 0.18181818181818182 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c9b0240e9e9b1e24a6818aea545aca798ed834851723fe5cbc1cf6bee074c309", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 11, 5, 11 ], "verification_seconds": 0.12794150412082672 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 593415, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.540710085532877, "accuracy": 0.5888888888888889, "balanced_accuracy": 0.5713526491135847, "worst_class_f1": 0.2018348623853211 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "777bc6f479fcba72984e9e1aed22661d6ac8fc65036a2d58606ec75a7df3ae22", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 4, 11, 11 ], "verification_seconds": 0.117134939879179 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 928585, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.5292660108853414, "accuracy": 0.5708679076026015, "balanced_accuracy": 0.5517760462043491, "worst_class_f1": 0.24623115577889448 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f6cb6f46235235f76ebec4dcaab5d78ef7b237742ceb943f4c0c956f71c6fe40", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 5, 8, 8 ], "verification_seconds": 0.1802399018779397 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 920365, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5224325352669028, "accuracy": 0.5642520744561561, "balanced_accuracy": 0.5433392429912146, "worst_class_f1": 0.224 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2df7e442bc4600f05e8c723f33877fb92f60d551306cd69ff7175049bf7c43e4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 11, 11, 8 ], "verification_seconds": 0.1654700394719839 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 678702, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.4902724278530045, "accuracy": 0.5289302534200493, "balanced_accuracy": 0.5163140375736235, "worst_class_f1": 0.23756906077348067 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "935dd179ad44b3b02aa5e8545f12db3b4099b241eb890ce780d3c323c5ac9664", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.13488131761550903 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 116863, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.48107076640171, "accuracy": 0.5211361366914457, "balanced_accuracy": 0.47297209841401533, "worst_class_f1": 0.2077562326869806 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "66385fc30b94f06aa0449ed217198ccc132752ce4dc4535b14f709b3d2f412f0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 8 ], "verification_seconds": 0.09273753501474857 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 587010, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.5040897734143406, "accuracy": 0.5419948962609564, "balanced_accuracy": 0.5311132155244158, "worst_class_f1": 0.20030349013657056 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "500ca32f859e3fcc82c2e63641bcf90b276c39ef76b25897c20105bd34b33b76", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.11589611694216728 }, { "method": "wisp_evolution", "dataset": "harmes_right_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/harmes_right_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 678966, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.49092845263416923, "accuracy": 0.5254632197936314, "balanced_accuracy": 0.520122834067514, "worst_class_f1": 0.19607843137254902 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "32de18c5708819ab41108eb5d133a31cb1df50743b541d3e44ec033066f0c252", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.15122493356466293 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 144553, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6001622909093439, "accuracy": 0.8245280257573541, "balanced_accuracy": 0.592896134674992, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "64fdaaef017498ecc7315aa50318895b526ece1f7ca5565d2852c8c94b6fdb7a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10048750694841146 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 194458, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6212723756520478, "accuracy": 0.8232108883360164, "balanced_accuracy": 0.6083823342657297, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c743297bc7b09082982f875e32a2efb54f4f9ab3417e51248ca11c64c51198fb", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.08981783781200647 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 341656, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5636640533035754, "accuracy": 0.8281867408166252, "balanced_accuracy": 0.5371753995032529, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4ab88443c7cb1bd1291caabd5f666cda2fc23b1f3a7f4f61b082131ba58e5b5f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.1375412354245782 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 336594, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5364918657912368, "accuracy": 0.8251804330392943, "balanced_accuracy": 0.5041208793661618, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e2f6ac7eea957553f6812a95af8140f0b47ea32fc36a0a021b72e7b65b2c5ad4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.12254564091563225 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 343025, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5134095420062846, "accuracy": 0.801523656776263, "balanced_accuracy": 0.48543136741220744, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "32e9553c9d5a4159a79f670139ae0ce81021f2ffcc2c7691b0af3ae030e54425", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.1299411579966545 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 341859, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.49811428812826414, "accuracy": 0.7797380379577653, "balanced_accuracy": 0.4614286887159727, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0e901ee03a0b25696fd04c6454816ac40d9fd6a9c620b0aaad388f502b3061ce", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.12822435796260834 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 73411, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5740855391481829, "accuracy": 0.8384826434450674, "balanced_accuracy": 0.6331106148564649, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "064ae588a90768eb6a818b751dad91f2a4bdc110ea0cc3b879d0655073fdfad1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08763671666383743 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 147505, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5887274453964119, "accuracy": 0.8454014076106405, "balanced_accuracy": 0.6058360488052064, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "31dc121ea36092e092175c7e0e5407da27c8a1a661baa1a52a283e93892565a3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10630783904343843 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 266734, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6268461641781113, "accuracy": 0.8302516998687821, "balanced_accuracy": 0.6435828090925986, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6b9344e70583d77530fe8a694440936b094483f06e6e582834b1a3e2e09d9230", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.10888167563825846 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 288208, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.596128960568406, "accuracy": 0.8317962835512732, "balanced_accuracy": 0.598804660316241, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6a22c61813ec149f540c737bd34e3e7488eb05fbfefe348c9d6f5a4f626bc0be", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12140228878706694 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 338086, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5905239929548161, "accuracy": 0.8220233998623537, "balanced_accuracy": 0.5860314266823297, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cc486db4a47978da0ec68a942e819276c79206dd4a18237dde884231bcd3ac9e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.1356620490550995 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 339803, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.612778808268588, "accuracy": 0.846662078458362, "balanced_accuracy": 0.5768670269769565, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "378d36025950f1bb47310bcca00bd87e5482be55e84ac0630cffb98b8b060883", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.1284400476142764 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 358831, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7350769324712363, "accuracy": 0.8778756035217268, "balanced_accuracy": 0.6931079259273305, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "8d02a047a1b63170714775755c37eb944a608a7ee30bdd074c911ff15f0efc95", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.13610275462269783 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 342490, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7479329190105534, "accuracy": 0.8950582220959955, "balanced_accuracy": 0.7162155645562306, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0d649153cc7bd5c8480e6281471419759e3781edb27a967ed22233fc16775f34", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.13554767239838839 }, { "method": "wisp_evolution", "dataset": "iuwds_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/iuwds_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 337538, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7998097634781114, "accuracy": 0.9067026412950866, "balanced_accuracy": 0.7779843281151284, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d385cbf979f12f2e1fa3a129f70e1a4d7f3f117a83b9efd811e9ea8a0fe59b7c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.1348281092941761 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed42/model.pkl", "bytes": 385670, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7354886950705765, "accuracy": 0.7267326732673267, "balanced_accuracy": 0.7416392821031345, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "642323bd092f7af4bff81261d49a95f8d6bdd9c03bb991382807f96cd65a1b68", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 6, 6, 6 ], "verification_seconds": 0.10621493961662054 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed43/model.pkl", "bytes": 389579, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8188387012671772, "accuracy": 0.8198019801980198, "balanced_accuracy": 0.8294384057971014, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c98af6a1ad5fa4a5acf72dbb151c2c39e8bbcff3547fb378b2655c3128fbcc6d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.10488037578761578 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold0/seed44/model.pkl", "bytes": 388588, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7493625750406817, "accuracy": 0.7485148514851485, "balanced_accuracy": 0.761945989214695, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1a1e5ee70767382efbcf8077bd8f6b6a408593fe493ffa43a86c593202d523a5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.10792993381619453 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed42/model.pkl", "bytes": 388809, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8135257169451995, "accuracy": 0.8317214700193424, "balanced_accuracy": 0.8389626741846908, "worst_class_f1": 0.5352112676056338 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b6b4cad4225d36f51a87164af7f5ae2cb9686f93958bb781e20b0cd92fe2b143", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.11736003495752811 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed43/model.pkl", "bytes": 389448, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8224755450020339, "accuracy": 0.8355899419729207, "balanced_accuracy": 0.8430465618254703, "worst_class_f1": 0.5526315789473685 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ea44f640f353240f0fe43c2f07b0671b8ef4d9b14da5b93e0c141cf6f051a4df", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.1024811640381813 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold1/seed44/model.pkl", "bytes": 99292, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.7123372087360301, "accuracy": 0.7330754352030948, "balanced_accuracy": 0.74444591280854, "worst_class_f1": 0.11428571428571428 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e4fa923163e155f243de5e81ab01b0dfd68897c2d30b760261d78939901d607b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 8 ], "verification_seconds": 0.06621815077960491 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed42/model.pkl", "bytes": 6680248, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8878789083727651, "accuracy": 0.884393063583815, "balanced_accuracy": 0.8913043478260869, "worst_class_f1": 0.5542168674698795 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ce8c0b79823ca0e9198b9705a7708e6417d0465501e611c72e131206ae690b15", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.13278221990913153 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed43/model.pkl", "bytes": 16192319, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9111111111111111, "accuracy": 0.9113680154142582, "balanced_accuracy": 0.9166666666666666, "worst_class_f1": 0.6666666666666666 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b03a1eba1b9ba1d5f14bf9454de555f8b0214b3e6020f363d0e32d2f7ca345b3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.13645241782069206 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold2/seed44/model.pkl", "bytes": 16911563, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.877856486947396, "accuracy": 0.8959537572254336, "balanced_accuracy": 0.8731884057971014, "worst_class_f1": 0.6571428571428571 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "619961db831ef97b705ff4251ab9de1c1e5f62c9d9a174974ca4372ce9b765d3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.14094430953264236 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed42/model.pkl", "bytes": 387408, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9296020533702118, "accuracy": 0.9295499021526419, "balanced_accuracy": 0.9320460673468762, "worst_class_f1": 0.676923076923077 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c12d0ad33340bdfaa17619d0480c221971f88ad98b2f9782328b8ec64c775bd9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.11122406274080276 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed43/model.pkl", "bytes": 100328, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.888480483344134, "accuracy": 0.8904109589041096, "balanced_accuracy": 0.8949656539393648, "worst_class_f1": 0.5423728813559322 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cd5e821662f61f83e72b853c5ec5c1bc544b8d90ff9eb27351f715dfc7ef1654", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.0694976132363081 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold3/seed44/model.pkl", "bytes": 291189, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8694335167724162, "accuracy": 0.8688845401174168, "balanced_accuracy": 0.8672814378072213, "worst_class_f1": 0.64 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a7d10e300109478314430a4eb69c1a26724258d09f1f95448fc6061bfa54862a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.091588887386024 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed42/model.pkl", "bytes": 30961327, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8318056467541005, "accuracy": 0.8230616302186878, "balanced_accuracy": 0.8346920289855072, "worst_class_f1": 0.5 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c3c4a01d438ce5ecd6b04091d0a8514f61ef8601fa4ed5227c473308b0af42d9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.16053823940455914 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed43/model.pkl", "bytes": 30830438, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8414983164983165, "accuracy": 0.8489065606361829, "balanced_accuracy": 0.8333333333333334, "worst_class_f1": 0.46464646464646464 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cb2eb0d0b2dfc1e6692fdeb57a8fbcae3132439f68cd7192c107bbc48a118b89", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.16630718670785427 }, { "method": "wisp_evolution", "dataset": "mhealth_right_lower_arm_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/mhealth_right_lower_arm_v1/fold4/seed44/model.pkl", "bytes": 30621176, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8392896169265338, "accuracy": 0.8469184890656064, "balanced_accuracy": 0.8315217391304349, "worst_class_f1": 0.46464646464646464 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e9b6153280a1a0515086dcb052677e4ef9525a52e4eaba7988f5974d590abef0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.17363407742232084 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 153957344, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7892275662433231, "accuracy": 0.7895833333333333, "balanced_accuracy": 0.7732288959196837, "worst_class_f1": 0.5088757396449705 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7e051b14cf200359bc93dc42d1c644a3d70195c75d53c83595ff78536fbbca83", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.4136338597163558 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 323691551, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7900666383115289, "accuracy": 0.8010416666666667, "balanced_accuracy": 0.7741262465535576, "worst_class_f1": 0.5810055865921788 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f7e788b278cc8e1725664d17aae66ab8d38966f5da8db575d689330f152375b4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7362617207691073 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 153261502, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7730199712967488, "accuracy": 0.7885416666666667, "balanced_accuracy": 0.7593348479587491, "worst_class_f1": 0.55 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "338412c40b053ad5446733dc73170b258270ec38714701590ab4bfd4a8f69e47", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.36440687999129295 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 142963268, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7225936836830401, "accuracy": 0.7149677057006459, "balanced_accuracy": 0.705294925589822, "worst_class_f1": 0.37209302325581395 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5430b3d4daa1933336b3353f154efd50193f86f80dcc357acecafc03c7a1d3bf", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.3679047701880336 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 140553320, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7164108253593273, "accuracy": 0.7197416456051671, "balanced_accuracy": 0.7004261415634176, "worst_class_f1": 0.4806201550387597 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c46d20bb9312101456100e660befe4efadf1cf076805c45dd995d23d9ccae00e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.39510233141481876 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 141989394, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.6871823068269264, "accuracy": 0.7071047458579051, "balanced_accuracy": 0.6740489832241326, "worst_class_f1": 0.31683168316831684 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7a846466c5eebdb50b19c86a9569e1a8dfa50385fd1351113a32907dac90d898", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.3557204445824027 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 315612091, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7864181898198394, "accuracy": 0.7759418374091209, "balanced_accuracy": 0.766090744467668, "worst_class_f1": 0.45454545454545453 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "8741865403f39a1d2025c6d1b9d8b298731828a521f2fa6fef533931b1d0c060", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7392814699560404 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 153243266, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.80773716013251, "accuracy": 0.7974223397224058, "balanced_accuracy": 0.7904072939498322, "worst_class_f1": 0.5039370078740157 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c62309370b896062d65a2c3de815063290b1b90c15357ae94b902e47c1321565", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.41625876631587744 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 153626016, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8019267714455053, "accuracy": 0.7941176470588235, "balanced_accuracy": 0.7839061568806102, "worst_class_f1": 0.49624060150375937 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "23fadf9558497c26725749a608d94a3f3e9b6552e270d86357084b0663d62761", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.3832047041505575 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 152149574, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7777656452114684, "accuracy": 0.7796555086122847, "balanced_accuracy": 0.7665927790732726, "worst_class_f1": 0.3953488372093023 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3092788b844f02df039c5bd7471138d84d5ec2d4dfc0c32cf3632e5d4da2adbd", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.3757688459008932 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 152238152, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7607235262767121, "accuracy": 0.7731556711082223, "balanced_accuracy": 0.7535080157000026, "worst_class_f1": 0.3684210526315789 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "870373f958f71bf32099b6a24355c38355e07b6531bd6f4a3b1232df1e59b0f2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.39853435661643744 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 319477175, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7532136587464725, "accuracy": 0.7598310042248944, "balanced_accuracy": 0.7411083111974465, "worst_class_f1": 0.40229885057471265 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6c875bc9fe2f42aa55055a78d56124883050a090c9eb0843500b5d5c870a3d8b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7252659574151039 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 151505892, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7806518780446584, "accuracy": 0.7801668806161746, "balanced_accuracy": 0.7721625612556755, "worst_class_f1": 0.2857142857142857 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "264d7c463966746c13527a9b1f6e957909c6f22d3d65ef956de8b3a3af631cd2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.38599200546741486 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 318869947, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7932082732315537, "accuracy": 0.7910783055198973, "balanced_accuracy": 0.7835228970239744, "worst_class_f1": 0.3378995433789954 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3f8b66d97b8aaa91894b0675e2e2358ba7a0c91ff60c3f319cb383925951dcee", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7330380454659462 }, { "method": "wisp_evolution", "dataset": "paal_adl_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/paal_adl_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 152256374, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7644344111861626, "accuracy": 0.7708600770218228, "balanced_accuracy": 0.7561859057947079, "worst_class_f1": 0.3404255319148936 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "af81927a19858af47b2fb5e51a8649cc34e27cde0b64614136a14bff375c5f0e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.3918878575786948 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed42/model.pkl", "bytes": 193291, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9574373809904692, "accuracy": 0.9519050593379138, "balanced_accuracy": 0.9570683091449005, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6b84666bf1cf734a5028a6ad927d030ab5001d3c948d5ce9a5adfd41fac2f467", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08527306746691465 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed43/model.pkl", "bytes": 196772, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9234438902270412, "accuracy": 0.915053091817614, "balanced_accuracy": 0.9239575066889878, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bcbe55be438c66f2ff1596569efcd1b0f119bf8930351c32e578edc73c50562e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12321723159402609 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold0/seed44/model.pkl", "bytes": 297193, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9406545864398269, "accuracy": 0.9344159900062461, "balanced_accuracy": 0.9396017944519662, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4b8ae35ce0b9243acac79c368aaf530801f25329219fb9a261379dd0e8491d85", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12391958478838205 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed42/model.pkl", "bytes": 100780, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9274071442306097, "accuracy": 0.9279907084785134, "balanced_accuracy": 0.9303482022468934, "worst_class_f1": 0.7922705314009661 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4cf0e5c4409bcd82ca7fdb4c22a3c8bca473f0f20a164320d5c944bff525fcdc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.0929145747795701 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed43/model.pkl", "bytes": 597009, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9573547671478485, "accuracy": 0.9506387921022067, "balanced_accuracy": 0.9599025452743013, "worst_class_f1": 0.8265895953757225 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fc6e3ad30248d370e9a115f9934da1a4a6bcf5f258e65cee5e75c1517be6ed2e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1140044592320919 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold1/seed44/model.pkl", "bytes": 100739, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9486688403645763, "accuracy": 0.9419279907084785, "balanced_accuracy": 0.9491802208534699, "worst_class_f1": 0.8306010928961749 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3fd5f1bda39e2b35f9f9c9a01447d6e6def59a6117463fb9839d208a4d8b74ad", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08163740299642086 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed42/model.pkl", "bytes": 62111308, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8517262119654153, "accuracy": 0.8604014598540146, "balanced_accuracy": 0.8410551716685949, "worst_class_f1": 0.5333333333333333 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b583f9f7e0ef26632e0c3a974d8c4ab3131c1151c650395164b7427e4c8c2dd8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.23181306663900614 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed43/model.pkl", "bytes": 62886878, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8209642589780709, "accuracy": 0.8394160583941606, "balanced_accuracy": 0.8033008632095653, "worst_class_f1": 0.5384615384615384 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a6d79a8bda87c56e6c8c7c67224f349e1e58e4360275e55869232059844bcfec", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.2601744942367077 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold2/seed44/model.pkl", "bytes": 63063120, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8023272337691744, "accuracy": 0.8147810218978102, "balanced_accuracy": 0.7802617209412323, "worst_class_f1": 0.5333333333333333 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "65be58d783f38b502915d98d4653c04896fa4f9c020ebf6fd3fa2c628e66188d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.24120864365249872 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed42/model.pkl", "bytes": 508988, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9753302851247804, "accuracy": 0.9764453961456103, "balanced_accuracy": 0.9768624909957174, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7c7b18f7c4b28dfa856b9d137d3b9b30dd31321d0f44a67b7e4a9c958bb79990", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08564670383930206 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed43/model.pkl", "bytes": 202528697, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9777781580379844, "accuracy": 0.9796573875802997, "balanced_accuracy": 0.9795458509744225, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3408f2b33a059bf8d413702989205a6cbe26e63950ca4c79d86a394e8c8a54b2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.8584732040762901 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold3/seed44/model.pkl", "bytes": 100381, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9702034685152359, "accuracy": 0.9721627408993576, "balanced_accuracy": 0.9695197659483373, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "aa815e1e89f51a870ce1317e6de1533067fc8b3c13fc95460ed97384d76c78a5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.07312445435672998 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed42/model.pkl", "bytes": 56120005, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7621147901273506, "accuracy": 0.7752192982456141, "balanced_accuracy": 0.7490324061766814, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2042c93b8c87d47a56542875b1c96ba42a779fa55d459d847becec8bb24ab696", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.22116222511976957 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed43/model.pkl", "bytes": 58285096, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7501886834700756, "accuracy": 0.7642543859649122, "balanced_accuracy": 0.7350324796363008, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "446c26982f12fa7809412901d7da73a936e567375ac9d4c20800e4ea5052fb23", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.25193033926188946 }, { "method": "wisp_evolution", "dataset": "pamap2_hand_imu_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/pamap2_hand_imu_v1/fold4/seed44/model.pkl", "bytes": 56854569, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.760305665820137, "accuracy": 0.7702850877192983, "balanced_accuracy": 0.7455118406391396, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6cf67eb93bfe4f2e570d63e3aa237bd864a21d2f10755012fb0ca68d3adaf488", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.22516581136733294 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed42/model.pkl", "bytes": 1441696758, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7257920521665745, "accuracy": 0.7153008962868118, "balanced_accuracy": 0.7167385507142605, "worst_class_f1": 0.24921728240450847 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c15e5c665d7296a50f5da08fc485c98b6587d4710d61fff886d8d0bd9d868353", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.7825141521170735 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed43/model.pkl", "bytes": 1440491744, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7311220616483857, "accuracy": 0.7195902688860435, "balanced_accuracy": 0.721265618929603, "worst_class_f1": 0.2734422262552934 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ae134fbb2388fff33f837da08fea166bc34896eb2145b9abcc51dded51dc3f26", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.7973577231168747 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold0/seed44/model.pkl", "bytes": 1454190198, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7372074685460596, "accuracy": 0.7245198463508322, "balanced_accuracy": 0.7259944037246389, "worst_class_f1": 0.35683629675045986 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9947c4ed8e471c90de672c7379927e557a3c43505ce6547e95e0332965987f10", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.7477301387116313 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed42/model.pkl", "bytes": 1415103478, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6909213215278447, "accuracy": 0.6976187476376464, "balanced_accuracy": 0.6950960789059522, "worst_class_f1": 0.3468507333908542 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "434ed4bdc14514ab3e125b0032326ccfff4b11dd900528031dddbb158080136d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.5760244950652122 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed43/model.pkl", "bytes": 1410096822, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.6693677517933854, "accuracy": 0.6783419427995464, "balanced_accuracy": 0.6754648043483025, "worst_class_f1": 0.21124361158432708 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ae2013720b730a92e701ea61ec7c5cba041461b6730b88c893d715e2cb5fd3a1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.713862843811512 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold1/seed44/model.pkl", "bytes": 1412228576, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.6790566568076546, "accuracy": 0.6857124858258788, "balanced_accuracy": 0.682652741791336, "worst_class_f1": 0.27692307692307694 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "72e5d2ab67810a31dc43d07fbcf1fb1e75e27ed539bdf73216270d54e5f1089c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.726566475816071 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed42/model.pkl", "bytes": 1006207, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8497583589701082, "accuracy": 0.8565003513703443, "balanced_accuracy": 0.857026570430238, "worst_class_f1": 0.33 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3ec5e9a17b1d91fcd980c62acbaad8d3c664ba57bc3a4647423f766794ee10f9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.1348606338724494 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed43/model.pkl", "bytes": 638844, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.8221419805287655, "accuracy": 0.8338018271257905, "balanced_accuracy": 0.8360105567768358, "worst_class_f1": 0.15869311551925322 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "00e790c44211d2e9101f4b7989cebaccb4b24f1c8d472be774bdf4677f14fb10", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.1066829888150096 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold2/seed44/model.pkl", "bytes": 1346779355, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8333028215152704, "accuracy": 0.8427266338721012, "balanced_accuracy": 0.8448529483213124, "worst_class_f1": 0.33367037411526795 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3a6544700bc41317febaa9a6fccf239bea2eaa23679f880b1d874ff9c035da32", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.545041259378195 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed42/model.pkl", "bytes": 1403333366, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7105930492098491, "accuracy": 0.7057796620736954, "balanced_accuracy": 0.7063862046110593, "worst_class_f1": 0.40115025161754136 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f6eb8ecc81f2dbd82dc2198a7fce9b1a6e87335a4c4894ddaf9fa1b49f0e1b8d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.6965157566592097 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed43/model.pkl", "bytes": 1402171766, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7084641883414828, "accuracy": 0.7075877548475591, "balanced_accuracy": 0.7081053741669864, "worst_class_f1": 0.3187889581478183 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "dccc1936bb4b4df120ba985021080edaff742caed891e7d0a4bd65ab5d3b9e35", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.6935139382258058 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold3/seed44/model.pkl", "bytes": 1403424512, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.6952479691226653, "accuracy": 0.6926865764698548, "balanced_accuracy": 0.6933706911358393, "worst_class_f1": 0.34602649006622516 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "57c44e13d0ba8d4c6c512726c084d375dbe48c6814b88286ed92c508e5835cf7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.691866286098957 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed42/model.pkl", "bytes": 641366, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.7203206678406642, "accuracy": 0.7189364132504125, "balanced_accuracy": 0.7196280071775887, "worst_class_f1": 0.12974051896207583 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7a53a1691892c5ace8e41e723b388da68e5714799557855900d73ee03dfde3cd", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.0962999165058136 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed43/model.pkl", "bytes": 246094, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6694266295253583, "accuracy": 0.6790836400558447, "balanced_accuracy": 0.6768085236883788, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fbb2ab9c1cee267c109d60f3d6b771a9b4e6f1069056306d1d07c99bdbc5e45c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.0871832063421607 }, { "method": "wisp_evolution", "dataset": "wisdm_watch_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_evolution/wisdm_watch_accel_v1/fold4/seed44/model.pkl", "bytes": 642689, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7068318523799468, "accuracy": 0.7071963447137962, "balanced_accuracy": 0.7070325543416646, "worst_class_f1": 0.1347248576850095 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a389973ba7102f94da04fba77bae30fec1c3a3ba6e259d813aa62ab81060aa79", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.10736861545592546 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 213075, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "passed", "n_test_windows": 861, "exact_frozen_predictions_equal": true, "original_input_exact_frozen_predictions_equal": true, "original_input_differing_prediction_count": 0, "differing_prediction_count": 0, "original_and_fresh_X_exactly_equal": true, "original_and_fresh_X_max_absolute_difference": 0.0, "time_coordinate_check": { "raw_time_values_exactly_equal": false, "per_subject_stable_chronological_permutation_identical": { "f3": true, "m1": true, "m3": true, "m7": true }, "original_first5": [ 0.0, 2.5, 5.0, 7.5, 0.0 ], "fresh_first5": [ 0.0, 1.0, 2.0, 3.0, 0.0 ], "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." }, "metrics": { "macro_f1": 0.5339423877092813, "accuracy": 0.7015098722415796, "balanced_accuracy": 0.5192755280407102, "worst_class_f1": 0.0 }, "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5339423877092813, "accuracy": 0.7015098722415796, "balanced_accuracy": 0.5192755280407102, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d7e9ecb9baabeede43934e670f6e0efcdeb29e1110056d33362d5cad8cc0870f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 3.478103124536574 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 56473111, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "passed", "n_test_windows": 861, "exact_frozen_predictions_equal": true, "original_input_exact_frozen_predictions_equal": true, "original_input_differing_prediction_count": 0, "differing_prediction_count": 0, "original_and_fresh_X_exactly_equal": true, "original_and_fresh_X_max_absolute_difference": 0.0, "time_coordinate_check": { "raw_time_values_exactly_equal": false, "per_subject_stable_chronological_permutation_identical": { "f3": true, "m1": true, "m3": true, "m7": true }, "original_first5": [ 0.0, 2.5, 5.0, 7.5, 0.0 ], "fresh_first5": [ 0.0, 1.0, 2.0, 3.0, 0.0 ], "explanation": "Original inputs use seconds; the public HF coordinate is a window ordinal. The decoder uses only within-participant chronological order, not time gaps. Numeric-coordinate inequality is reported, never concealed or silently shuffled." }, "metrics": { "macro_f1": 0.44804969372888687, "accuracy": 0.6991869918699187, "balanced_accuracy": 0.48086825598802657, "worst_class_f1": 0.0 }, "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "data_provenance": "Dual inference on original trusted frozen NPZ and a fresh export from pinned local HF benchmark staging; y/subject rows and chronological permutations checked before inference" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.44804969372888687, "accuracy": 0.6991869918699187, "balanced_accuracy": 0.48086825598802657, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "69b629122b9d9b7f4dc5cbf40b767b9391f12f8d6327ea817f385e736638f265", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 1.414209634065628 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 387105, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.42336890869963556, "accuracy": 0.5448648648648649, "balanced_accuracy": 0.5103515571350251, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0593f16f80eb5d1a010e55c9534cc105a671f91294b48a68990c46d0b27ddb1e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 7, 7, 7 ], "verification_seconds": 0.11762447189539671 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 388857, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.44265859848601236, "accuracy": 0.5553153153153153, "balanced_accuracy": 0.5262091686071412, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0133cd4b226973eebe7a77576171dfe2440d216a3c440f8652f4379d6e9cd136", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 7, 7, 7, 7 ], "verification_seconds": 0.10366932395845652 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 197938, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.45020993038951257, "accuracy": 0.556036036036036, "balanced_accuracy": 0.5254028034871413, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0ef0f6220e2db9c4c853ba6158dc3924effaefaca7f40fd87f196c3259fd3030", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 7, 7, 7, 7 ], "verification_seconds": 0.07649926003068686 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 213222, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6839360804687519, "accuracy": 0.6371134020618556, "balanced_accuracy": 0.6487853885528353, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "83d5ea4d49be6e2c8eac6cd6363cbecf629412f9ac345d791a04dd6444c24824", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07935636956244707 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 108796, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6914456270187225, "accuracy": 0.6412371134020619, "balanced_accuracy": 0.6577139599814067, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "beda21721a3b563425f0da94f37fb3e5e901fb85cecdb3874fa0c7649a3629be", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.06602407619357109 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 211297, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6265840120876497, "accuracy": 0.6164948453608248, "balanced_accuracy": 0.6150310331521384, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bdf4f5b49f0ee29aa30c2ca28ef1e400302d793f8bb08b94b53773c69d50c48c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.08080927841365337 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 213351, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3984637132349555, "accuracy": 0.5677799607072691, "balanced_accuracy": 0.4565550938263096, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "75ae6aa393878d072945e96f452b96e09e9cf40fb3b2ea8457e186d70b05e841", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07724830415099859 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 211953, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.3947840773031935, "accuracy": 0.5540275049115914, "balanced_accuracy": 0.4383977872139755, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d82f5778b37d6f94a75882921e38848d8207348565af2d0d39c3db90326f7cb8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07652495242655277 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 542820, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.41233788381060416, "accuracy": 0.5717092337917485, "balanced_accuracy": 0.46000228267069465, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a809a108f835a4ee2214a0948773f4689220fcd9cfd5deb9f8698001ba8902f9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.11057593394070864 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 419921, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7684496006640077, "accuracy": 0.8529411764705882, "balanced_accuracy": 0.8162091423863549, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "51e1d01636b9a0cac92c6cc300ec4e48cce07d310e2e1f190017f13e488a5ee3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.09686489589512348 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 544269, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7017359507404785, "accuracy": 0.8137254901960784, "balanced_accuracy": 0.7581788689951661, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "251a88a762c7a5b4b2590567ef275d02cb441de8a0f7e4747da5d76fcc8311f1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 0, 0, 0, 0 ], "verification_seconds": 0.08747569378465414 }, { "method": "wisp_random", "dataset": "adl_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/adl_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 214370, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6492442092593446, "accuracy": 0.8176470588235294, "balanced_accuracy": 0.7580917818206289, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5af62dd1e9a42bc066cfd01cfdda16039fcba40a4061a905a997d948728dc8a6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 5 ], "verification_seconds": 0.07864306773990393 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed42/model.pkl", "bytes": 161894, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.766990439714226, "accuracy": 0.8378404764229229, "balanced_accuracy": 0.7554500861212902, "worst_class_f1": 0.6008610086100861 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "229c7dee72504b72d5605c5150c08e9d3f077a27a287382225e8aa2b179929d1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.0840184036642313 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed43/model.pkl", "bytes": 129303, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7330825808731976, "accuracy": 0.819974616811481, "balanced_accuracy": 0.7079910899772252, "worst_class_f1": 0.5666456096020215 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b65dfa5093e81d27982784c93a20a87a0bf1930e29ffcc01b8e26ece25504f54", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09258068632334471 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold0/seed44/model.pkl", "bytes": 161270, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7623005521588372, "accuracy": 0.8392072634970223, "balanced_accuracy": 0.7456711136811414, "worst_class_f1": 0.6124694376528117 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "07b9eec05901e515f03e25eb39a3c842dc98b7982e302188bcfda892d1bf1869", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.1038873614743352 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed42/model.pkl", "bytes": 138656, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7486620464593752, "accuracy": 0.8456556396970905, "balanced_accuracy": 0.7473717694082199, "worst_class_f1": 0.4661016949152542 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5c92495d9331f02daf8560445df4e83b9e98d59510b3c7375f378d40904e45b9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.07749845832586288 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed43/model.pkl", "bytes": 255990, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.735485533953722, "accuracy": 0.8366879234754883, "balanced_accuracy": 0.7346841985437063, "worst_class_f1": 0.43897216274089934 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "866e338432a7bb4b5f8ee276c77ce81e136694c2c3804ccc1574b56810a86e01", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 2, 2, 2, 2 ], "verification_seconds": 0.1168006956577301 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold1/seed44/model.pkl", "bytes": 130598, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7593223381430728, "accuracy": 0.8434635312873655, "balanced_accuracy": 0.7578615088048173, "worst_class_f1": 0.5243128964059197 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2685a44c752f4432440d3a31e6e8576d149ad136752d6e573c9cb0484f638eb5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 2, 2, 2, 2 ], "verification_seconds": 0.12646049074828625 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed42/model.pkl", "bytes": 161894, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7062470015513294, "accuracy": 0.8255736456250595, "balanced_accuracy": 0.6981685147812106, "worst_class_f1": 0.48322147651006714 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0207c109e8e11f6f7fbca6ae8e7624f00f59b83cd39b14b918bd7ac4913d9a1a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.08643207233399153 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed43/model.pkl", "bytes": 287812, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7108976698220147, "accuracy": 0.8287156050652195, "balanced_accuracy": 0.7053300591182281, "worst_class_f1": 0.48825065274151436 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "92f17f4830088544f470bcc13a91313fcd66ad20a772abb355604a78a1188344", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.10610962845385075 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold2/seed44/model.pkl", "bytes": 161581, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.724010717738153, "accuracy": 0.8322384080738836, "balanced_accuracy": 0.7199091773450284, "worst_class_f1": 0.5288831835686778 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a1f2eab36e31ca790baae3ac9cef3cf645c08d2e2fb9c6c6290d4de65d64b818", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.08403156418353319 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed42/model.pkl", "bytes": 115443, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7488290281373933, "accuracy": 0.839453125, "balanced_accuracy": 0.742297928519201, "worst_class_f1": 0.5042174320524836 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "619d5842e7c5ec71c173c938e126a128095d1bddcc300489956ed39bfccd6de4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.20588375721126795 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed43/model.pkl", "bytes": 118067, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7570308365882163, "accuracy": 0.84384765625, "balanced_accuracy": 0.7481699905884054, "worst_class_f1": 0.5209756097560976 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "afcb98b4c2dc1350bcb9e53d800a22a5cb5794874689933b475b88d2775ae41a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.24759656377136707 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold3/seed44/model.pkl", "bytes": 117459, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7596369938969753, "accuracy": 0.8490234375, "balanced_accuracy": 0.7538129256318069, "worst_class_f1": 0.5155393053016454 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a2211a1bb2fd15c90b224ac7ea5131aa6d59957b4e3460875c6eb26500ea7fc1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.2584470985457301 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed42/model.pkl", "bytes": 128065, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7551038036075215, "accuracy": 0.83273988005997, "balanced_accuracy": 0.7455861736760279, "worst_class_f1": 0.49612403100775193 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9f2d52d91c2bc5624fc0d918954b4aff119e5c78ad73416cbb83c1b72ff605ae", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.08599592372775078 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed43/model.pkl", "bytes": 226783, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.742279603672112, "accuracy": 0.8223388305847077, "balanced_accuracy": 0.7216110695134941, "worst_class_f1": 0.5005302226935313 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bfee24f11c6a76538adbfec619fa58865fdfa08020e9762ec6cf1bdc81812e0f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.10002798307687044 }, { "method": "wisp_random", "dataset": "capture24_wearable_activity_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/capture24_wearable_activity_v1/fold4/seed44/model.pkl", "bytes": 161581, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7523348286903604, "accuracy": 0.825243628185907, "balanced_accuracy": 0.732207260115527, "worst_class_f1": 0.5368852459016393 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "87ef00aa9a441fa21d0cd24ab3db92c9829499a2634920694f8ab9ed45e59fae", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.09194480534642935 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed42/model.pkl", "bytes": 213222, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7551308251068606, "accuracy": 0.8979695431472081, "balanced_accuracy": 0.7583531872830543, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "52c0487222b97840207d833ce067c1ef209aa9dc6e76b3879cbd48791285e781", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10122111812233925 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed43/model.pkl", "bytes": 215616, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7773091987294367, "accuracy": 0.9116751269035533, "balanced_accuracy": 0.7694623232746799, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9be8c786b60ba44925ce330a3331a3791c2f192e8fb4f2af95c7b285d3a8a020", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13180649187415838 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/domino_watch_accel_v1/fold0/seed44/model.pkl", "bytes": 542386, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7518046343995171, "accuracy": 0.9121827411167512, "balanced_accuracy": 0.7516438722909802, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "562b1a9ca6109509524f71787309d109f44981b05a31982f6310f10bb2a3a2d1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13094902504235506 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed42/model.pkl", "bytes": 215890, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7811112855157368, "accuracy": 0.9018214127054642, "balanced_accuracy": 0.7979395775270787, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "04858553cf1d77303f80f81d2b6fdfcf8f32f846f53b127a0a3531369698cc29", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.10303804371505976 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed43/model.pkl", "bytes": 316666, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7801701329373244, "accuracy": 0.8716126166148378, "balanced_accuracy": 0.8032376592902427, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a50203df8da14032b31032f1609cbcbb997363c09bead7da66293a46d2b92e27", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12858885247260332 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/domino_watch_accel_v1/fold1/seed44/model.pkl", "bytes": 542875, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7646305706831946, "accuracy": 0.8796090626388272, "balanced_accuracy": 0.7734379168736917, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e11c60884f89f1a630a9c14a24705ad1d829e9edb77ab045cdc9142313f76c54", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12358776666224003 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed42/model.pkl", "bytes": 468799, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6182859411274892, "accuracy": 0.8502024291497976, "balanced_accuracy": 0.6535983881129549, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "03ee367d4f45f272040bc32e1f223c210c8ebbb1d9a14e5cf8de193227c52071", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11528230179101229 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed43/model.pkl", "bytes": 540105, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6311436177301671, "accuracy": 0.8704453441295547, "balanced_accuracy": 0.659241603110375, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ba772f5e17588452cbde29336e69acfd4a44c07e565347509a33c81dfc20d447", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1269427239894867 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/domino_watch_accel_v1/fold2/seed44/model.pkl", "bytes": 543332, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.578905837151691, "accuracy": 0.8536726431463274, "balanced_accuracy": 0.6249674973278894, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1e68c747889a8f2466fcf2805645a3014692e747feae48b6abbc19bd014381ac", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13287620432674885 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed42/model.pkl", "bytes": 215742, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7511955410337825, "accuracy": 0.8684834123222749, "balanced_accuracy": 0.7689192612204052, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "97b4a33e5eac4d4c97d0369dc142118fcab3db8345e5cd25563d334b77df938f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.0989870810881257 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed43/model.pkl", "bytes": 108796, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6954366439264663, "accuracy": 0.8203001579778831, "balanced_accuracy": 0.7168607714759039, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0f1c5108cf1dcb1eafcdd881b076eefd0b12eda59dc5e4da1d1ce463951b9e5c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.06485389173030853 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/domino_watch_accel_v1/fold3/seed44/model.pkl", "bytes": 214370, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.627228447883459, "accuracy": 0.8052922590837283, "balanced_accuracy": 0.6445631276390363, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "034d9a680e4312bff79652317774770d18ef487648d866333c627807748d3cc8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.09241700731217861 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed42/model.pkl", "bytes": 541762, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6183927693961376, "accuracy": 0.8535864978902954, "balanced_accuracy": 0.6268799948210574, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "52e3563e30b4160d8a3ad7eef9b961d2486c1534ca26592d490a0a865589c313", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1271434649825096 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed43/model.pkl", "bytes": 540415, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.634073183238398, "accuracy": 0.8556962025316456, "balanced_accuracy": 0.639406051515471, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "501fee63a547a6aa191e6c86ca1eaa5b18a4ea9206b981b94e35a14cc5bb0d60", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12751690112054348 }, { "method": "wisp_random", "dataset": "domino_watch_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/domino_watch_accel_v1/fold4/seed44/model.pkl", "bytes": 216185, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6367216796558524, "accuracy": 0.8265822784810126, "balanced_accuracy": 0.6282511507353351, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f86c3d295afb1e24fce9d7afe9efce9f8a687ebe5f2bc10734dcf9bbab1e9e94", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.09841653145849705 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 591456, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.784293009526535, "accuracy": 0.8009986782200029, "balanced_accuracy": 0.8151819951716208, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b9db54ada58d01aa9a88928b52628850e3057f6820f1f726b87c01a7ec6c01c6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12829209584742785 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 592918, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7594087239448236, "accuracy": 0.7810251138199442, "balanced_accuracy": 0.7894529186109617, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2996f25843ced7e146f6d85bdc6ed30a5857576e422a2520979e15a487384379", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1241801893338561 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 592080, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7708773764124208, "accuracy": 0.7893963871346747, "balanced_accuracy": 0.7973621988392794, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "68ea2185c96db68527916dc00959d4927af1085b20db8920929f0de67084fa8e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.15180608443915844 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 182063436, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.729251179192184, "accuracy": 0.756351392715029, "balanced_accuracy": 0.7251885891431922, "worst_class_f1": 0.46808510638297873 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1f41ba703dce73869d5aff7dc3d8de09327e83b4138dc1e06ec26ee24f54f13f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.4741174401715398 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 704450, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8159433539759252, "accuracy": 0.8123660850933578, "balanced_accuracy": 0.8184655317757509, "worst_class_f1": 0.4541832669322709 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f5ffa07e6b96255ed93b361e398740732f20f6e502ae788bb6c76b543f141e8d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1623753560706973 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 590016, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 6.591949208711867e-17 }, "metrics": { "macro_f1": 0.7652924699658761, "accuracy": 0.7745638200183654, "balanced_accuracy": 0.7834846740797345, "worst_class_f1": 0.018083182640144666 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2826758bc5ee17021e7b8c471ed11739e13983e8ab8736f866423e454d9cd29a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11671893298625946 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 588032, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7873681523995203, "accuracy": 0.8100347112653834, "balanced_accuracy": 0.798674374826554, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b911a5c1d76ec88f7fac386b71f1f69977910349f1fc35de1264c8bd172e8578", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12262417282909155 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 591573, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8000927882449437, "accuracy": 0.8117702745345535, "balanced_accuracy": 0.8150741858340492, "worst_class_f1": 0.1109350237717908 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "dc92c8b944619d5abf2b682e0d7188c78f6b63b88229d2bd6225e8b3e6d84d7a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12705540098249912 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 595035, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 6.938893903907228e-17 }, "metrics": { "macro_f1": 0.7726750498433423, "accuracy": 0.7914168507415589, "balanced_accuracy": 0.7876774055916419, "worst_class_f1": 0.12089810017271158 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1fef32d0456a2818005badf5de3c075659b4041b40d875d109a102ffef16fc25", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12606476619839668 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 591928, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 6.938893903907228e-17 }, "metrics": { "macro_f1": 0.7993431026428096, "accuracy": 0.8082954712003517, "balanced_accuracy": 0.8063771798704091, "worst_class_f1": 0.05865921787709497 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2596f8d542a9f3892df755c8747f15daa65971ef21cb8e0d5eabf040bdc997cf", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.14119288697838783 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 589011, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7867416975202401, "accuracy": 0.7984757438077092, "balanced_accuracy": 0.7980594826403815, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0729d3043a36b77566ef9d77526fe81526baa018aa9a8f44bc39038a5993d1e0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12311080563813448 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 230479, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 1.3877787807814457e-17 }, "metrics": { "macro_f1": 0.7190410117062849, "accuracy": 0.7367726806390151, "balanced_accuracy": 0.7161055800914486, "worst_class_f1": 0.046008119079837616 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b675a28ea876f097481cf0f186babf1d3cdca36f673c75141a0f273a7996fc7f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08946351986378431 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 592354, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8792767127830559, "accuracy": 0.8633117422355091, "balanced_accuracy": 0.8866686986182304, "worst_class_f1": 0.5959183673469388 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4a0a57580d103371b297935817eb138c41e2332f1a074b9aaf94587d2aa2137f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12175817228853703 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 592217, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8686373451288947, "accuracy": 0.8555057299451918, "balanced_accuracy": 0.8761138489486704, "worst_class_f1": 0.5566433566433566 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "48de32930bc6b94187f3ff3a8e354264ffbb3064b1cdd2944edd2a8fec7d730c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12206245306879282 }, { "method": "wisp_random", "dataset": "gotov_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/gotov_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 592860, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9048190879441402, "accuracy": 0.8947018767646571, "balanced_accuracy": 0.9067162256666195, "worst_class_f1": 0.7234567901234568 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c113ffde305f5212b8ea0674e68c244d7987fc618b708aa4f2b207d14f2d0613", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.16558631416410208 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 66847331, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9603439013050938, "accuracy": 0.9589422407794015, "balanced_accuracy": 0.9617768102403584, "worst_class_f1": 0.9343065693430657 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0a1e0ed18c21d6fe8be9fadea73e11bdc65f47b66feca80a5bdf9528a2f7bfae", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.21702496614307165 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 106353855, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": -1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9494755982485993, "accuracy": 0.9443284620737648, "balanced_accuracy": 0.9507084789963877, "worst_class_f1": 0.9022222222222223 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6e11892bdd3343eb962fd1e690ede41e7567bdc368dad79b041a027683f8bb3c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.2986576007679105 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 106727683, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9536047751541364, "accuracy": 0.9491997216423104, "balanced_accuracy": 0.9554135283314116, "worst_class_f1": 0.911504424778761 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9f08a3ce13a054fb2a954aa97b14c1e7a045f9f7e5bad43d4e2fa904ba04dfc4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.32660636119544506 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 109082947, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9507493107548275, "accuracy": 0.9378925331472435, "balanced_accuracy": 0.9605487984667551, "worst_class_f1": 0.8957055214723927 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0b3c854a78d5dca17fa61bd74992831ab14ae6b17eb5144adb36420497308d1c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.3146402854472399 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 69488437, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9722661390125834, "accuracy": 0.9658060013956734, "balanced_accuracy": 0.9770293664024313, "worst_class_f1": 0.9409282700421941 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "393ce4ed2452236ddd27745fe07aaf9f40ca0a0f1590ed38ba28625c3f9d9fa9", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 9, 6, 5 ], "verification_seconds": 0.2198030510917306 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 109039171, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.94426634615951, "accuracy": 0.9288206559665039, "balanced_accuracy": 0.9550853661302577, "worst_class_f1": 0.8724279835390947 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ea8efece2d0848cc16ede574aae1180ee2908669c260f4bc42c5292611a8f360", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.3040814511477947 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 70548283, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": -1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9603615365507692, "accuracy": 0.9538123167155426, "balanced_accuracy": 0.9612722403991679, "worst_class_f1": 0.9227272727272727 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7b3956cbd291ede95168ff2a6d28b63d26a43e9684e037ba627c7cc0fe44ecdc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.21898382529616356 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 30411068, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9635137430915, "accuracy": 0.9567448680351907, "balanced_accuracy": 0.9634035317435918, "worst_class_f1": 0.9223744292237442 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fc0ebebd742e3e0c5b9199f5a3caec835db45198021d4c7de2272ac359577b3c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.15453871339559555 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 30402066, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9654073531705973, "accuracy": 0.9611436950146628, "balanced_accuracy": 0.9659125049875467, "worst_class_f1": 0.9363636363636364 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "78f540cf3087c0fc79bdfdce82373310fa77b3a76a188f6dcc1886f2b797d5e3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.15123513340950012 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 67924027, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9655985809409403, "accuracy": 0.9556786703601108, "balanced_accuracy": 0.9711245386810653, "worst_class_f1": 0.9302325581395349 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "78d8142ced6b45000763b28a2e8f88d73eb1e31be36a8bc9d585e0d18c2d71db", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 5, 5, 9 ], "verification_seconds": 0.2322320556268096 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 67799611, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": -1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9596676620980877, "accuracy": 0.9494459833795014, "balanced_accuracy": 0.9635122651082073, "worst_class_f1": 0.9222614840989399 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7fd6a793318e44a366fbf31ab425824f2708bc411e1cb270ed3e0558955403cf", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.22738107945770025 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 68072341, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9637962821709289, "accuracy": 0.9529085872576177, "balanced_accuracy": 0.9693982750317192, "worst_class_f1": 0.9217081850533808 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "16e4dbdabe73170e232579c7865b2d7c8fb39c4b29d1c8914be212e7746161d7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 9, 6, 5, 9 ], "verification_seconds": 0.22421606816351414 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 109733535, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9501422250540463, "accuracy": 0.9364285714285714, "balanced_accuracy": 0.9499210582168228, "worst_class_f1": 0.8758782201405152 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5006890e670eab4c6f96ada58978af59f03154b2635f4eb5b83194181695436e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 9, 6, 5 ], "verification_seconds": 0.30270709563046694 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 109431135, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9462109554549561, "accuracy": 0.9328571428571428, "balanced_accuracy": 0.9473380715803916, "worst_class_f1": 0.8752941176470588 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "be1b71a6c9d79a676fd10b4333db9054eb6b4753ae89352609c5ced206de97db", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 9, 6, 5 ], "verification_seconds": 0.30106195248663425 }, { "method": "wisp_random", "dataset": "handy_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/handy_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 68605399, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9602680505925019, "accuracy": 0.9514285714285714, "balanced_accuracy": 0.959909060884063, "worst_class_f1": 0.9 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3d1e2a382d39244fbe4348d3104bf3a5d4e9f76d1e9b9550bae9c6018b66a18b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 9, 6, 5 ], "verification_seconds": 0.2278675800189376 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 2370891003, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.3943996945388775, "accuracy": 0.4341880341880342, "balanced_accuracy": 0.39529889172768035, "worst_class_f1": 0.19811320754716982 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "8bb6acfedc4c11f2f59ab482a6c8feed7f8dc61f9443b19a94b8e045b8101d54", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.696244320832193 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 517285, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.3797364195371058, "accuracy": 0.4098005698005698, "balanced_accuracy": 0.42310013825699694, "worst_class_f1": 0.16806722689075632 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c2d4ca643c7a6400fdade9b344ad2e6d13f19f479ddc6c9ebb331624bd7f8fe0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11076897941529751 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 2371692608, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.39416052775079996, "accuracy": 0.42883190883190886, "balanced_accuracy": 0.4205249580366873, "worst_class_f1": 0.20245398773006135 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "253153fe7336aebebac63efcae12e0aa58a4fee4f665559059d24832b45e17e7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.713018720969558 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 2266198779, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.3493869479288254, "accuracy": 0.3809678819444444, "balanced_accuracy": 0.35520574414892814, "worst_class_f1": 0.19378427787934185 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "90a68d971230a9617c8a99d8c006bce5c14cc5775b165be199052df2c4593584", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.515500565059483 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 587116, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.34019784103660566, "accuracy": 0.365234375, "balanced_accuracy": 0.3786343493232046, "worst_class_f1": 0.17903930131004367 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "66228228fad97bfe648e33ec5b21af6b237e923adcee81eef79a9417f7c8f4e0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 11, 4, 4 ], "verification_seconds": 0.09785876236855984 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 2263995968, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.351469207855779, "accuracy": 0.376953125, "balanced_accuracy": 0.3768039684244342, "worst_class_f1": 0.20352781546811397 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3e448a2c29814480b6e97bd5671df91708d6137083295326421759d1f0574c04", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 8, 4, 4 ], "verification_seconds": 4.504460323601961 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 2367772155, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.42957514443609257, "accuracy": 0.46879432624113476, "balanced_accuracy": 0.4255324702472358, "worst_class_f1": 0.2266857962697274 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a3798422bee96ade1b4974e2344a3c099c228c611e41dbd3c2fd6a2b18584709", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.693867314606905 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 590184, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.41079080082797637, "accuracy": 0.4462174940898345, "balanced_accuracy": 0.4527289147376499, "worst_class_f1": 0.17002237136465326 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d905f128c139edd4960352e27701d0df14c5f322cebd010aa0c1a9d8edfcc2a8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11217740643769503 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 2368916672, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4233880396109819, "accuracy": 0.46418439716312054, "balanced_accuracy": 0.4487659825863075, "worst_class_f1": 0.2023121387283237 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2dd211a680d732fbc2e3c2b442c5a170550f79ffe5b0d2e31e2027ce78902969", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.683686343953013 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 2296903419, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.4045781945009205, "accuracy": 0.4273379681542947, "balanced_accuracy": 0.3983868486508851, "worst_class_f1": 0.1737142857142857 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0381189d16ad8a9551ee512689162cf20ae4f80bdbbc1d82677d4498f0057dec", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.528864155523479 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 585468, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.38515500821634396, "accuracy": 0.40805113254092845, "balanced_accuracy": 0.42030076334721334, "worst_class_f1": 0.21149425287356322 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "2d4f8d1e7b0e7d2852bb5ee91fec80d00e6136819f72561d193957072fe6ac37", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11319855600595474 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 2296580288, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.40008244076906796, "accuracy": 0.42419825072886297, "balanced_accuracy": 0.41809533022760476, "worst_class_f1": 0.17865429234338748 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "8c39d17a9ba8b720249d3fb8640328b464aca798b8f4fc04f4f34e86c94b1189", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 3.9741314267739654 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 2313538299, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.42894317767551826, "accuracy": 0.4700987462554089, "balanced_accuracy": 0.4321963492839216, "worst_class_f1": 0.15950920245398773 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6a51c532c58da2852a903abe948896fbdd819fdc8572f883a670f835c608a3f5", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.683324318379164 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 363896, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.39903742181842566, "accuracy": 0.43459447464773104, "balanced_accuracy": 0.4465491869813594, "worst_class_f1": 0.16531165311653118 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6d43b1687051d63a031e99ed8f2b7aca80f066fd8b0dc718ed16e8e649743f76", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13601224217563868 }, { "method": "wisp_random", "dataset": "harmes_left_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/harmes_left_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 2313291200, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 5.551115123125783e-17, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.43042334665685333, "accuracy": 0.47043159880173085, "balanced_accuracy": 0.4597562844733023, "worst_class_f1": 0.1883656509695291 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "8b97fbb28bb726002a9024e9af4eae8a83272abaa7fb69fa019ed75ce9fc4ea8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.670859377831221 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 455682, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.47826892471136084, "accuracy": 0.5122507122507123, "balanced_accuracy": 0.5079849702546556, "worst_class_f1": 0.1615598885793872 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "9b12660beb807daa7f15ce68676b5af6f29c826ec651eafa4e67926a5f2d11cc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 8, 8, 11 ], "verification_seconds": 0.11805807705968618 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 455953, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.4827274993532969, "accuracy": 0.5171509971509971, "balanced_accuracy": 0.5125902139658777, "worst_class_f1": 0.18289085545722714 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e8376233adffc905ad61b15b2c5366a3241faf126ff6e84a162a0b92d4dd1e82", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 11, 11 ], "verification_seconds": 0.11957382317632437 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 2274877760, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5013992889364879, "accuracy": 0.5433618233618234, "balanced_accuracy": 0.5148112236986051, "worst_class_f1": 0.1736111111111111 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0b8f8bcc9da51d65a7218dca78a0e35a5f69043839c3ddbd0c51ca532761b29b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 4, 11, 8 ], "verification_seconds": 4.500875387340784 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 2180587131, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 4.163336342344337e-17 }, "metrics": { "macro_f1": 0.4734412043443698, "accuracy": 0.4984809027777778, "balanced_accuracy": 0.47314577660732104, "worst_class_f1": 0.11463046757164404 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "00f2a7d26793bbda7faca4d0cab16e7e1c69649f3fa0593af78db3a8e8998c6e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 8 ], "verification_seconds": 4.1530782505869865 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 590496, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.45353736391094296, "accuracy": 0.4861111111111111, "balanced_accuracy": 0.49245402085307655, "worst_class_f1": 0.16145833333333334 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "94f311255d7cdb7256e5919d96345d998dea9ab5bb9f2b25c632272554a832c1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 13, 4, 5, 13 ], "verification_seconds": 0.11939528677612543 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 2177927744, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 5.551115123125783e-17, "worst_class_f1": 1.3877787807814457e-17 }, "metrics": { "macro_f1": 0.4605656968569808, "accuracy": 0.4949001736111111, "balanced_accuracy": 0.48605973746000264, "worst_class_f1": 0.12435233160621761 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d6fe8ad972163b8638f89e9b60f3a9a077145b665d872e3e01372408ea514ffa", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 4.361154975369573 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 588184, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.5421584351505209, "accuracy": 0.5908983451536644, "balanced_accuracy": 0.5719084457993553, "worst_class_f1": 0.20958083832335328 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "1418f6e3f196aa4e7835f8ffb632c7eb9962ec8208ed298e9026a71163035fa2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 5, 11, 4, 5 ], "verification_seconds": 0.11561560537666082 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 517285, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.5426752741368953, "accuracy": 0.5897163120567376, "balanced_accuracy": 0.5765607711259935, "worst_class_f1": 0.23384615384615384 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fe2a7d032ee656e4a399e1977b4754a7c5a2b1ee18e3f9f3039150718f61388e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 5, 11 ], "verification_seconds": 0.11062014661729336 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 476284, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.5340476395279575, "accuracy": 0.5822695035460993, "balanced_accuracy": 0.563630793774831, "worst_class_f1": 0.20125786163522014 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "679a7bdceb4ee46b969c415ca020e61c58b6da683decb765ca52bad685d61d9e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 4, 11, 11 ], "verification_seconds": 0.09793207608163357 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 455682, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 5.551115123125783e-17, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.49901099119769765, "accuracy": 0.5366674142184347, "balanced_accuracy": 0.5248976016389348, "worst_class_f1": 0.2107843137254902 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fd31fa757b80f2f3e9293a0644895165773fc9cd67c265449ab5a7c1e736d6db", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 11 ], "verification_seconds": 0.10844338033348322 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 590055, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5129585454890183, "accuracy": 0.5566270464229648, "balanced_accuracy": 0.5378205618305084, "worst_class_f1": 0.2288135593220339 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "17c70707b59020594e415426a34af60b624aacd54716082cb0ce5277954bf4ff", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 13, 11, 8, 14 ], "verification_seconds": 0.11518295481801033 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 2209661888, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5301043138101286, "accuracy": 0.5714285714285714, "balanced_accuracy": 0.5399403495460648, "worst_class_f1": 0.2031063321385902 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3bb3447bf3d2446e4c4123cd2a4a5dd6dab50a51b38d01d4b3bb14a9dd1accd2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 4, 11, 8 ], "verification_seconds": 4.420359159819782 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 587878, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.5020630813880627, "accuracy": 0.5415510928658605, "balanced_accuracy": 0.5286122763951864, "worst_class_f1": 0.19607843137254902 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "44d6aac704b945e7ff7ec9e5ba5b47d45db1db7633bcbd7de43740b7abbaa5f8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 8, 8 ], "verification_seconds": 0.11303388141095638 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 587116, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.4988862797334431, "accuracy": 0.5389992233440586, "balanced_accuracy": 0.5251543776171204, "worst_class_f1": 0.18507462686567164 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "dadf463570cbb51bb37ff17643f647d22434739e1fd95f28e907ddf52983100d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 11, 11, 8, 11 ], "verification_seconds": 0.1164759211242199 }, { "method": "wisp_random", "dataset": "harmes_right_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/harmes_right_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 2209648832, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.5176601134537039, "accuracy": 0.5586375235770553, "balanced_accuracy": 0.5295711176752078, "worst_class_f1": 0.19439868204283361 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d6582b4369b37f15f5b67c4c5553536fb5f11a55e15632a8119a77d2ee4e1314", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 4, 11, 8 ], "verification_seconds": 4.430933751165867 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 128945, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6096479050548741, "accuracy": 0.8319918044782673, "balanced_accuracy": 0.5923564159046437, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "39a72f089a60c8d2136b1cdb37d6ff4fce32c67244d8b4d6446fe295ae3cb8a0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.07963023521006107 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 156083, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.612918638732734, "accuracy": 0.845602224498756, "balanced_accuracy": 0.6040208282401819, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b884279bc5116301f86f9b9b047d6f39f83e2e70bfaa0cd5222de24f83fc07b2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.45693342853337526 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 128949, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6096479050548741, "accuracy": 0.8319918044782673, "balanced_accuracy": 0.5923564159046437, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "651faac607a3a3956c8a909c4870272b49bb201510102066fb552d6ec61ce156", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.08445706032216549 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 128949, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5928265629244448, "accuracy": 0.8043303929430633, "balanced_accuracy": 0.5810361579109827, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6c67e257799eb3499b65b1b828816e1037d87739dc13515f9f4ad8b63c1c6a8e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.0769574511796236 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 338789, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5121815033880968, "accuracy": 0.8032611601176156, "balanced_accuracy": 0.4962577766609905, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "7964b0b8c3e4c7ec761024654e6a14184907702f09386733bf025a7210105825", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.1298271780833602 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 340825, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5351559377826424, "accuracy": 0.8064688585939588, "balanced_accuracy": 0.5059498177634308, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4b42b63d5c748bba1e5c7ad609b3add2aa005086dbbf7df23ce0d3289d228211", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.13942116685211658 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 169676, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6116900772477086, "accuracy": 0.8550638196349756, "balanced_accuracy": 0.6305240862826924, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "516514293771fba294dc22e9f4c7efbccdaed62bd58283a201cf01591d03660f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 2 ], "verification_seconds": 0.07959313318133354 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 342837, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.587618207448066, "accuracy": 0.8846475008946678, "balanced_accuracy": 0.5780258345781247, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "978a3d968f4687a3eecb529adbe28f9a88c6a7c8409af7f49e2d4fab7f31ba27", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.13611499685794115 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 128949, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6320530481952188, "accuracy": 0.8665155672193725, "balanced_accuracy": 0.6538620348912504, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "81d368579274b11fadb3b8f92ba55513ecf36c70c2cf638fb8895a34a4419e1d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.07920224126428366 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 128945, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6098265825969379, "accuracy": 0.865244322092223, "balanced_accuracy": 0.6081597599365783, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b2b4e2c420fd84f538b1cd9e327354e336a78e2f6ad7c744cd8706e3007b61a2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.0839704629033804 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 338435, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.5497180265570948, "accuracy": 0.8531314521679284, "balanced_accuracy": 0.5354421050786498, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3ad8b51c6de3b613bff453f4991eedf86577f18c58db396b41430c409dbf9757", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.12315355986356735 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 128949, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6098265825969379, "accuracy": 0.865244322092223, "balanced_accuracy": 0.6081597599365783, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6df4173edc966f77135834c37ddde485c62dd8a7489cdf08d71302625756b309", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.08461639657616615 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 128945, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8173963795908751, "accuracy": 0.882277761999432, "balanced_accuracy": 0.8327524147291605, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4d739e8c0c327536274dd2edb686f25c1e1ff652b33d34572941cd1795429b7b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.07930546812713146 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 339455, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7775502482937635, "accuracy": 0.9034365237148537, "balanced_accuracy": 0.7482305404696603, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a92a42dba50c70fda2bcf31af9d5ea060d0e2cb14b427c970dd581f56464ee72", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.12170023191720247 }, { "method": "wisp_random", "dataset": "iuwds_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/iuwds_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 128949, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8173963795908751, "accuracy": 0.882277761999432, "balanced_accuracy": 0.8327524147291605, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a393b658599dd887f9a42ad8d4ec97a8306cd036643de52f930d8ab567be5f24", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 3, 3, 3, 3 ], "verification_seconds": 0.08189722429960966 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed42/model.pkl", "bytes": 195692, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7428036236211897, "accuracy": 0.7366336633663366, "balanced_accuracy": 0.7463115099427031, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "35cd0793db79f56c657858da9d950bcd00c29755da0cc46324942c3d41acbbd1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.10329711902886629 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed43/model.pkl", "bytes": 386577, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7530918965544832, "accuracy": 0.7485148514851485, "balanced_accuracy": 0.7627043309740479, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "86b1f9bd71d34a67a3985181fb212ae5dcc4a3582f3284a3e4e33eb71ef91128", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.10184361599385738 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold0/seed44/model.pkl", "bytes": 197143, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.777767255892256, "accuracy": 0.7762376237623763, "balanced_accuracy": 0.7874346983485001, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "8f894de960f765a8739605e6bf46bed77abd05ce0cc4bfaacdcdee07791209e1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 8 ], "verification_seconds": 0.06998705118894577 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed42/model.pkl", "bytes": 194609, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7958838404465718, "accuracy": 0.816247582205029, "balanced_accuracy": 0.8247764415664509, "worst_class_f1": 0.5833333333333334 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6be3c89830b29436a93355544ab8c9e74d31a11c22c5488573952b93f79e38f6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.06705038249492645 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed43/model.pkl", "bytes": 195124, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7607287596852985, "accuracy": 0.7543520309477756, "balanced_accuracy": 0.7665576406325713, "worst_class_f1": 0.45569620253164556 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "edbf66cb6e12b62cb7adc6a093779254ba6d098e467ebbd80726961354e293c6", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.09524028841406107 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold1/seed44/model.pkl", "bytes": 197143, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.833983599459939, "accuracy": 0.8317214700193424, "balanced_accuracy": 0.8397702744372494, "worst_class_f1": 0.47191011235955055 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c8a1ff51de6357d56b65587d5ea4760668f86d6326649fecde826baa6bf43f89", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 8 ], "verification_seconds": 0.07438390795141459 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed42/model.pkl", "bytes": 788723, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8583320296777582, "accuracy": 0.859344894026975, "balanced_accuracy": 0.8676731078904992, "worst_class_f1": 0.6268656716417911 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a124969a171b4ab7fea49bc4e000e578b62d9192bcf3e99a00e8262c985996ce", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 8, 8, 8, 8 ], "verification_seconds": 0.10094030201435089 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed43/model.pkl", "bytes": 6921228, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9111111111111111, "accuracy": 0.9113680154142582, "balanced_accuracy": 0.9166666666666666, "worst_class_f1": 0.6666666666666666 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d26957d48806f1c581b02fbb2ce1d9804b0c119afb4ff19f5e18074c04f3e094", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.11249503772705793 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold2/seed44/model.pkl", "bytes": 793838, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8565516714672254, "accuracy": 0.8574181117533719, "balanced_accuracy": 0.8658212560386475, "worst_class_f1": 0.6268656716417911 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "29ea043d1216b22c5db967c576795c72dfbaa855dd0a0daed56bd362d6208493", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 1, 1, 1, 1 ], "verification_seconds": 0.10749175865203142 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed42/model.pkl", "bytes": 194609, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": -1.1102230246251565e-16 }, "metrics": { "macro_f1": 0.9702950466044578, "accuracy": 0.9686888454011742, "balanced_accuracy": 0.9701297607010448, "worst_class_f1": 0.9113924050632911 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "d8022f233350561fc606adbbb20334d0587ba8d1708c268ca2df0ea3d3d7112c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.08102338202297688 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed43/model.pkl", "bytes": 554981, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8595948062369086, "accuracy": 0.8532289628180039, "balanced_accuracy": 0.8617290192113246, "worst_class_f1": 0.5348837209302325 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b3f038f9a4d1e2cd1901c8d8f3acb225525868a12ec6fb2aeb82c3b3ef85297b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.1226800736039877 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold3/seed44/model.pkl", "bytes": 612279, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8849608669246121, "accuracy": 0.8806262230919765, "balanced_accuracy": 0.8867179121855563, "worst_class_f1": 0.6666666666666666 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "337f75cf5520357b464199ddeb5123a7679f98e14fe336f555e99cbe798c383b", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.2504494357854128 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed42/model.pkl", "bytes": 30675898, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7478307728307728, "accuracy": 0.7594433399602386, "balanced_accuracy": 0.7422123541887592, "worst_class_f1": 0.4444444444444444 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "bdccff3234086735c8efb97bf95b41ba974f47277cecb6fc82cee8f6740a1100", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.15433111507445574 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed43/model.pkl", "bytes": 30830438, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8414983164983165, "accuracy": 0.8489065606361829, "balanced_accuracy": 0.8333333333333334, "worst_class_f1": 0.46464646464646464 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "44b3cf13580087f18e4c668aacea2a0e039b3bcce22129cfc25dc37e0c768753", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.15820361580699682 }, { "method": "wisp_random", "dataset": "mhealth_right_lower_arm_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/mhealth_right_lower_arm_v1/fold4/seed44/model.pkl", "bytes": 30621176, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8392896169265338, "accuracy": 0.8469184890656064, "balanced_accuracy": 0.8315217391304349, "worst_class_f1": 0.46464646464646464 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0b525eb01e391ac3c106189748e4774e7051f1072797c28a5ce394e6a4f09249", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 6, 6, 6, 6 ], "verification_seconds": 0.15374513156712055 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed42/model.pkl", "bytes": 321909179, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.782808883037621, "accuracy": 0.7777777777777778, "balanced_accuracy": 0.7664396920134046, "worst_class_f1": 0.47619047619047616 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "712daccfd21c9116e7b220a7e4d733fde83e22603313fa5c47013ae707f6600d", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.715706367045641 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed43/model.pkl", "bytes": 154309196, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7863210822620976, "accuracy": 0.7899305555555556, "balanced_accuracy": 0.7717697486014151, "worst_class_f1": 0.5317919075144508 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "4da96f750e45494cfac7b52690d7d888adfef142eade782702938801620d8c96", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.3765896176919341 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold0/seed44/model.pkl", "bytes": 322817461, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7406621280602123, "accuracy": 0.7579861111111111, "balanced_accuracy": 0.7268762175805494, "worst_class_f1": 0.47619047619047616 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "fd9bc35b63ca0e620a4d8eaedf5ccee58908a2d06fd92f1c6c8a5f8a0da2b8ff", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7351966826245189 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed42/model.pkl", "bytes": 297563171, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.6989074572715867, "accuracy": 0.7000842459983151, "balanced_accuracy": 0.6760508882503385, "worst_class_f1": 0.37383177570093457 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "20467c0f35c90da5501b1c5e905192ca30b0011f6a0b806bbe80ba40c88a6d4a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.6626511849462986 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed43/model.pkl", "bytes": 295829813, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6940231797042055, "accuracy": 0.698680146026397, "balanced_accuracy": 0.6768125701153013, "worst_class_f1": 0.368 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6b0dd5effa866c8a1b29138c87f1a2055a54407a47522fd9e796403938c60fba", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.6669249190017581 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold1/seed44/model.pkl", "bytes": 141851858, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7247158739945666, "accuracy": 0.7245155855096883, "balanced_accuracy": 0.7101332033880609, "worst_class_f1": 0.4036697247706422 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "41b7e3278f9ac8c30e9788240174fc0fe373634d1bed67671b1d59f27b713c75", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.3645658940076828 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed42/model.pkl", "bytes": 153892642, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7831926412023957, "accuracy": 0.7792465300727033, "balanced_accuracy": 0.7655895754624013, "worst_class_f1": 0.46956521739130436 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "972b07f92063078910321412edc2b110d833ae0d5afbdd23f3664a72e9fd21e8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.420327321626246 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed43/model.pkl", "bytes": 153585740, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7977890388559085, "accuracy": 0.7868473231989425, "balanced_accuracy": 0.7789270668058506, "worst_class_f1": 0.4444444444444444 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b01c9d65130a785c746ee0d9705ee0a27eb52f58da411e93babb7384ff13a88f", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.41771274618804455 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold2/seed44/model.pkl", "bytes": 317604151, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7959494363633283, "accuracy": 0.7762723066754792, "balanced_accuracy": 0.7728997159742116, "worst_class_f1": 0.4915254237288136 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6f0c6b5613e76df8b1424c9301da581d80049278ae7cecd53d6b43268d034f54", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7287461068481207 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed42/model.pkl", "bytes": 152149574, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7777656452114684, "accuracy": 0.7796555086122847, "balanced_accuracy": 0.7665927790732726, "worst_class_f1": 0.3953488372093023 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5246675fb1a261b6a6dbeef965cc6adfed66bb83873766ab4202da68bc22f685", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.4086458645761013 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed43/model.pkl", "bytes": 152113740, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.76412050853765, "accuracy": 0.7796555086122847, "balanced_accuracy": 0.7541487816381051, "worst_class_f1": 0.43243243243243246 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b3cc4c609f8c65135908f512fd3f194b51dbabc49f780e18451de7911e041df0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.38193516433238983 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold3/seed44/model.pkl", "bytes": 152259654, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7785577846318382, "accuracy": 0.7890802729931752, "balanced_accuracy": 0.7690620670775831, "worst_class_f1": 0.44086021505376344 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "333a97f3219b97286c436e88ec4777d7825c319ab3f463c23b66d70f4998ea78", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 23, 23, 23, 23 ], "verification_seconds": 0.4068180527538061 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed42/model.pkl", "bytes": 319455669, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7699247674457195, "accuracy": 0.7750320924261874, "balanced_accuracy": 0.7602778385960877, "worst_class_f1": 0.3711340206185567 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "832929b8272271bc6b27792732d474ff816d1e7d6a357c558f43dbe33d3b4fb1", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.7088611256331205 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed43/model.pkl", "bytes": 149126524, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.7920311703422366, "accuracy": 0.7907573812580231, "balanced_accuracy": 0.7814562061777864, "worst_class_f1": 0.42857142857142855 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "51e33652aa3f4d4d6f61e8aabe58672305870b0a45568de614155507f0203d8a", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.3706858614459634 }, { "method": "wisp_random", "dataset": "paal_adl_wrist_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/paal_adl_wrist_accel_v1/fold4/seed44/model.pkl", "bytes": 151948870, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7804989539681273, "accuracy": 0.7894736842105263, "balanced_accuracy": 0.7705393743488225, "worst_class_f1": 0.3958333333333333 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b5dfdd5b52fc55364dcd0441c1fb7cd39b1663a1978d04b6b1d68a417738ff88", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 20, 20, 20, 20 ], "verification_seconds": 0.3796448949724436 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed42/model.pkl", "bytes": 194655, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.926126828018268, "accuracy": 0.9237976264834479, "balanced_accuracy": 0.9259866160332428, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "888da1e0d42747468ef5c32fd88620d2f20a1532cc44ef9a2e78001430d36ed2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08373156283050776 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed43/model.pkl", "bytes": 99194, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9373390528660345, "accuracy": 0.9400374765771393, "balanced_accuracy": 0.9314587981199641, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "082730a7b8c759e90da1ece504c471b8277f2fb9e14d9111b26fd7faf11401dc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.062161377631127834 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold0/seed44/model.pkl", "bytes": 196468, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9106480562075838, "accuracy": 0.905683947532792, "balanced_accuracy": 0.9113062332361342, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6a9ca05ac9ea12ce9141c63cd033d1d6dced14ac4ee24e2868f6dd69c125f657", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.08026024792343378 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed42/model.pkl", "bytes": 519536, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9607069422188861, "accuracy": 0.9529616724738676, "balanced_accuracy": 0.9629197423085207, "worst_class_f1": 0.8265895953757225 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "72256fd2b7de2b50a1665ec7cd03547aec9f69d6a7d2953fd578575ab7420509", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.12377436179667711 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed43/model.pkl", "bytes": 198347, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9424888593632138, "accuracy": 0.9419279907084785, "balanced_accuracy": 0.9454272899289901, "worst_class_f1": 0.78 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "5820902bf30bd558226d0c2b6e0195d1bfc841411c0932b6e335fc86b80a2828", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11050238087773323 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold1/seed44/model.pkl", "bytes": 694157, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 1.1102230246251565e-16, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9446475832944937, "accuracy": 0.9413472706155633, "balanced_accuracy": 0.9465516272892364, "worst_class_f1": 0.829971181556196 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "44fb853e3a01699c4db66ebb77e1a58d9d2c6669d67de80d9cd352f3e335bfe8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.1427327049896121 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed42/model.pkl", "bytes": 62795362, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8304680692494896, "accuracy": 0.8458029197080292, "balanced_accuracy": 0.8113854490423412, "worst_class_f1": 0.5358851674641149 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "62d7318cf47d9efb6f2dd3e6ca60a8c65d299f33df64875e2fb0b5bdae1acbf8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.24092491995543242 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed43/model.pkl", "bytes": 63752910, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8248201686199893, "accuracy": 0.8412408759124088, "balanced_accuracy": 0.8068507107128275, "worst_class_f1": 0.5384615384615384 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e1e5bca8da318977784e09f71c58727705efd5b86c185d8a831cf08c6cfe4add", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.2815354336053133 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold2/seed44/model.pkl", "bytes": 62588844, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8211673302288754, "accuracy": 0.8403284671532847, "balanced_accuracy": 0.8039473847047613, "worst_class_f1": 0.5373134328358209 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "508284c3f5768f28860e48ea9ea057daef15ab81d29ef6d8c688855020581ff2", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.23504982236772776 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed42/model.pkl", "bytes": 197040, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": -1.1102230246251565e-16, "accuracy": 0.0, "balanced_accuracy": -1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9561457774540599, "accuracy": 0.9571734475374732, "balanced_accuracy": 0.9573821531111663, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c53f9bc0825d8cab7fa0ba36c553909cde9373814972c6e31681651047106a63", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.11584278754889965 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed43/model.pkl", "bytes": 99194, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 1.1102230246251565e-16, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9651552354318038, "accuracy": 0.9657387580299786, "balanced_accuracy": 0.9661267981171673, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cb26ba3b108f180b33bf984874f47e6270771b5d673ef51fb723691ea21f1f89", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.0970451133325696 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold3/seed44/model.pkl", "bytes": 855341, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": -1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.9767973966100094, "accuracy": 0.9775160599571735, "balanced_accuracy": 0.9768469624234878, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f745b4bde373b714c90c335b8c5cd65b3986e78241eb48a6959b1afe626fd144", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.13156522531062365 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed42/model.pkl", "bytes": 56033925, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7675404704390513, "accuracy": 0.7796052631578947, "balanced_accuracy": 0.7538230037404436, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "463314bab2079ab16263df8e7c5799c8269b32556b58448ce9c3f56c1b304585", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.2576223621144891 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed43/model.pkl", "bytes": 56014725, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7623319002781765, "accuracy": 0.7763157894736842, "balanced_accuracy": 0.7505501469112027, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "0dfb08aea1d728fce3389d61e02d54faef0a3f9a7ab613f2292ee037997060b7", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.24346212297677994 }, { "method": "wisp_random", "dataset": "pamap2_hand_imu_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/pamap2_hand_imu_v1/fold4/seed44/model.pkl", "bytes": 56797677, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 1.1102230246251565e-16, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8976064609725526, "accuracy": 0.9133771929824561, "balanced_accuracy": 0.8840144162810198, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6070cd99340fe4ff4f92fd11653fee008f03b3648d0406b8b1a52dbb24069af3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 4, 4, 4, 4 ], "verification_seconds": 0.24854245502501726 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 0, "seed": 42, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed42/model.pkl", "bytes": 247303, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.774132262918264, "accuracy": 0.7813700384122919, "balanced_accuracy": 0.7826521562757212, "worst_class_f1": 0.20817120622568094 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6697b406a31e23bea6b952561315e8473da41831f2c3d30d6b16fd958eb877c0", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.11846522241830826 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 0, "seed": 43, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed43/model.pkl", "bytes": 1458792410, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.7181558240095015, "accuracy": 0.7084507042253522, "balanced_accuracy": 0.7099612955750377, "worst_class_f1": 0.24347826086956523 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "a18797baf8a57ae9d67afe35877e16545ffc71a22a3e9db470615d891e277787", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.8971761520951986 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 0, "seed": 44, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold0/seed44/model.pkl", "bytes": 1444945600, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.75091127432255, "accuracy": 0.742573623559539, "balanced_accuracy": 0.7438359518494067, "worst_class_f1": 0.29577464788732394 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "51fb1abc29411f60dedc5d31c452b05a8b6ab852c7267e5ce83b385ee30abc70", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.8266507973894477 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 1, "seed": 42, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed42/model.pkl", "bytes": 1406824768, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.688070847675892, "accuracy": 0.6940909663600857, "balanced_accuracy": 0.6928882984232319, "worst_class_f1": 0.4200772200772201 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "cce00ee54265fd663eb667bf5dc0e95effaea966d64ac748b8ed1ff4d188ebd4", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.7140465062111616 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 1, "seed": 43, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed43/model.pkl", "bytes": 1409985088, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.6931772279311114, "accuracy": 0.6961068413758347, "balanced_accuracy": 0.6949493182373486, "worst_class_f1": 0.39651416122004357 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "3da4f861e14224e8cd3d13f3937b341320acfc02b616be0df87bcc63a5cdf558", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.7301150457933545 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 1, "seed": 44, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold1/seed44/model.pkl", "bytes": 1411856544, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 8.326672684688674e-17 }, "metrics": { "macro_f1": 0.6725085349439842, "accuracy": 0.6828146654907395, "balanced_accuracy": 0.6791010276060436, "worst_class_f1": 0.17481203007518797 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ded420ffe0a3aa5e6334b29b25d289202890519cfa1b8ab889ff46e9ca079592", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.6930082300677896 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 2, "seed": 42, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed42/model.pkl", "bytes": 1348189179, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.8465069012503283, "accuracy": 0.8521433591004919, "balanced_accuracy": 0.8542029870340069, "worst_class_f1": 0.43606255749770007 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "6de1903ee0fc8da851336f261ae566cd1a945f62a4124899188d869387491f71", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.5633741300553083 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 2, "seed": 43, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed43/model.pkl", "bytes": 1350693119, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8476215612403961, "accuracy": 0.8535488404778636, "balanced_accuracy": 0.8553756774503352, "worst_class_f1": 0.421875 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "e11c85e09a30475ece633f18dd44eb2f2279e55ac903e386b2448f30c6ecf6f3", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.5569702116772532 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 2, "seed": 44, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold2/seed44/model.pkl", "bytes": 1342968119, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.8392289410102078, "accuracy": 0.8493323963457484, "balanced_accuracy": 0.8512439984972445, "worst_class_f1": 0.3232533889468196 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "baa4b60ce41b27b2cbf0eae874ed6b4aef19ce3c464167a3f641c8d10888c9d8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.5231892075389624 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 3, "seed": 42, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed42/model.pkl", "bytes": 1402628768, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 5.551115123125783e-17 }, "metrics": { "macro_f1": 0.70277937521498, "accuracy": 0.7029740008728724, "balanced_accuracy": 0.7036263928622836, "worst_class_f1": 0.41839080459770117 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "c1fc3a1fa5bb8666f48ab3a3c198c59c5a5da8a006d64341ed8b2488c4d3838c", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.695559806190431 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 3, "seed": 43, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed43/model.pkl", "bytes": 1402410746, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6994422646652237, "accuracy": 0.695180497537253, "balanced_accuracy": 0.695636607852856, "worst_class_f1": 0.4831130690161527 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "aafccdb6f0cb69725ae4197d1db86d820a4db509aada4a9922c4b2dddfc16dcc", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.6815116200596094 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 3, "seed": 44, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold3/seed44/model.pkl", "bytes": 1401175926, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.6921004642618573, "accuracy": 0.6919384001496353, "balanced_accuracy": 0.6924893673415535, "worst_class_f1": 0.2857142857142857 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "b9984a03b030ccf5a06961c0fb690909e20be3b0c4aea245c76a29995ef1320e", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 2.714278470724821 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 4, "seed": 42, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed42/model.pkl", "bytes": 642964, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 2.7755575615628914e-17 }, "metrics": { "macro_f1": 0.7263448253080516, "accuracy": 0.7255996953928163, "balanced_accuracy": 0.7267649113642366, "worst_class_f1": 0.13160518444666003 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "ecbb4e620aef4f605b9a0657077e8589d0c086d0886507bd4d344048bb336d59", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.12464579846709967 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 4, "seed": 43, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed43/model.pkl", "bytes": 643937, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7221425984704912, "accuracy": 0.729343825358548, "balanced_accuracy": 0.7322211437109059, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "57c7e0ea59741b1dc82941501812441efc800bee4b335c72b05ae7bce4cd0c54", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.11194508150219917 }, { "method": "wisp_random", "dataset": "wisdm_watch_accel_v1", "fold": 4, "seed": 44, "model_path": "models/wisp_random/wisdm_watch_accel_v1/fold4/seed44/model.pkl", "bytes": 643950, "load": "passed", "synthetic_prediction": "passed", "frozen_prediction": { "status": "not_run" }, "saved_prediction_metrics": { "status": "passed", "paper_metric_deltas": { "macro_f1": 0.0, "accuracy": 0.0, "balanced_accuracy": 0.0, "worst_class_f1": 0.0 }, "metrics": { "macro_f1": 0.7196462587809807, "accuracy": 0.7240766594745526, "balanced_accuracy": 0.7256429459103105, "worst_class_f1": 0.0 }, "claim_boundary": "Rescoring immutable saved predictions is not rerunning checkpoint inference on every frozen test split." }, "sha256": "f7822a30c9af1ab31909c8eee97cd6685b0825d809e80ae86a3378ea161df9f8", "source_status": "load_and_smoke_verified", "synthetic_prediction_class_ids": [ 10, 10, 10, 10 ], "verification_seconds": 0.10468210000544786 } ], "completed": 360, "status_counts": { "load_and_smoke_verified": 360 }, "full_frozen_prediction_status_counts": { "passed": 6, "not_run": 354 }, "saved_prediction_metric_status_counts": { "passed": 360 }, "public_model_text_hygiene": { "files_checked": 363, "passed": true, "failed_relative_paths": [], "scope": "Public model JSON/CSV and validation report only; no credential file is read." } }