diff --git a/README.md b/README.md index fbfd4282d355291192ed86d5a5c762ebb1dc81dc..17d12561b48a73a081252c081cfc030e97d34eba 100644 --- a/README.md +++ b/README.md @@ -1,74 +1,26 @@ --- -license: other +library_name: pytorch +tags: +- robotics +- vision-language-action +- latency-sensitive-bench +- multiple-checkpoints --- -# Latency-Sensitive Bench models -Inference-ready teachers and VLA policies for the supported benchmark tasks. +# LAGEN benchmark-models -## Layout +Inference checkpoints, original configurations and model provenance. -- `zero-latency//small-policy/` and `zero-latency//vla/`: models trained without latency. -- `latency-aware//small-policy/` and `latency-aware//vla/`: models trained for latency. Profile and fixed-2 training conditions are identified by run ID and bundle provenance. +| Experiment | Entry | +|---|---| +| Visual history | [Visual history](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history) | +| Latency in prompt | [Latency in prompt](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-in-prompt) | +| Latency transfer | [Latency transfer](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer) | +| Mean vs. profile training | [Mean vs. profile training](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/mean-vs-profile) | -MIKASA H8 conditioned inference bundles use `latency-aware/mikasa-intercept-grab-fast/vla/starvla--h8//`, where `` is `qwenoft`, `qwenpi_v3`, or `qwengr00t`. Each model has an SFT run `h8-conditioned-seed-reset-20260922-{profile|fixed-2}` and a DAgger run `h8-conditioned-dagger-20260923-{profile|fixed-2}`. +| Resource | Repository | +|---|---| +| benchmark-datasets | [latency-sensitive-bench/benchmark-datasets](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets) | +| profiles | [latency-sensitive-bench/profiles](https://huggingface.co/datasets/latency-sensitive-bench/profiles) | -## Hist8 VLA release: 18 models, 27 formal evaluations - -Flappy, Demon Attack and Deadly Corridor × QwenOFT, QwenGR00T and QwenPI v3 × zero/profile training. Each model is the evaluated step-5000 checkpoint, initialized from Qwen3-VL-4B revision `ebb281ec70b05090aa6165b016eac8ec08e71b17` with seed 42. - -Input is one current RGB image plus eight causal decision histories, passed as raw 0/1 transport state. Action horizon is **1**; the model directory suffix `h1` refers to this action horizon, while `hist8` in each run ID refers to input history. - -### Formal scores - -Each value is the mean of 100 episodes, seeds 1,000,000–1,000,099; parallel 32 and capacity 1. Flappy/Demon/Deadly caps are 3,600/7,200/3,600 raw frames at FPS/decision Hz 10/10, 60/15, 35/8.75. Deadly uses the corrected sf-render-v4 view (160×120 RGB with HUD). - -| Game / model | zero→zero | zero→profile | profile→profile | -|---|---:|---:|---:| -| flappy / qwenoft | 439.13 | 19.19 | 373.78 | -| flappy / qwengr00t | 428.48 | 6.62 | 414.52 | -| flappy / qwenpi_v3 | 408.72 | 5.58 | 344.48 | -| demon_attack / qwenoft | 2362.15 | 584.35 | 1452.80 | -| demon_attack / qwengr00t | 2370.90 | 231.90 | 1203.10 | -| demon_attack / qwenpi_v3 | 2350.95 | 112.30 | 907.30 | -| deadly_corridor / qwenoft | 2098.67 | 742.35 | 2091.23 | -| deadly_corridor / qwengr00t | 2100.46 | 341.76 | 2090.82 | -| deadly_corridor / qwenpi_v3 | 2104.09 | 140.94 | 1450.83 | - -### Downloadable models - -| Training | Game / model | Run | -|---|---|---| -| profile | flappy / qwenoft | [flappy-qwenoft-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-hist8-scratch-s42-v2) | -| profile | demon_attack / qwenoft | [demon_attack-qwenoft-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack-qwenoft-hist8-scratch-s42-v2) | -| profile | flappy / qwengr00t | [flappy-qwengr00t-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/flappy/vla/starvla-qwengr00t-h1/flappy-qwengr00t-hist8-scratch-s42-v2) | -| profile | demon_attack / qwengr00t | [demon_attack-qwengr00t-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/demon-attack/vla/starvla-qwengr00t-h1/demon_attack-qwengr00t-hist8-scratch-s42-v2) | -| profile | demon_attack / qwenpi_v3 | [demon_attack-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/demon-attack/vla/starvla-qwenpi_v3-h1/demon_attack-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6) | -| zero | flappy / qwenoft | [flappy-qwenoft-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-hist8-zero-scratch-s42-v2) | -| profile | flappy / qwenpi_v3 | [flappy-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/flappy/vla/starvla-qwenpi_v3-h1/flappy-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6) | -| zero | demon_attack / qwenoft | [demon_attack-qwenoft-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack-qwenoft-hist8-zero-scratch-s42-v2) | -| zero | flappy / qwengr00t | [flappy-qwengr00t-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/flappy/vla/starvla-qwengr00t-h1/flappy-qwengr00t-hist8-zero-scratch-s42-v2) | -| zero | flappy / qwenpi_v3 | [flappy-qwenpi_v3-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/flappy/vla/starvla-qwenpi_v3-h1/flappy-qwenpi_v3-hist8-zero-scratch-s42-v2) | -| zero | deadly_corridor / qwenoft | [deadly_corridor-qwenoft-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-hist8-zero-scratch-s42-v2) | -| zero | demon_attack / qwengr00t | [demon_attack-qwengr00t-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/demon-attack/vla/starvla-qwengr00t-h1/demon_attack-qwengr00t-hist8-zero-scratch-s42-v2) | -| profile | deadly_corridor / qwenpi_v3 | [deadly_corridor-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/deadly-corridor/vla/starvla-qwenpi_v3-h1/deadly_corridor-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6) | -| profile | deadly_corridor / qwenoft | [deadly_corridor-qwenoft-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-hist8-scratch-s42-v2) | -| profile | deadly_corridor / qwengr00t | [deadly_corridor-qwengr00t-hist8-scratch-s42-v2-workers16](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/deadly-corridor/vla/starvla-qwengr00t-h1/deadly_corridor-qwengr00t-hist8-scratch-s42-v2-workers16) | -| zero | demon_attack / qwenpi_v3 | [demon_attack-qwenpi_v3-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/demon-attack/vla/starvla-qwenpi_v3-h1/demon_attack-qwenpi_v3-hist8-zero-scratch-s42-v2) | -| zero | deadly_corridor / qwengr00t | [deadly_corridor-qwengr00t-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/deadly-corridor/vla/starvla-qwengr00t-h1/deadly_corridor-qwengr00t-hist8-zero-scratch-s42-v2) | -| zero | deadly_corridor / qwenpi_v3 | [deadly_corridor-qwenpi_v3-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/deadly-corridor/vla/starvla-qwenpi_v3-h1/deadly_corridor-qwenpi_v3-hist8-zero-scratch-s42-v2) | - -Every model contains its README, checkpoint, inference/training configuration, statistics, task contract, collection/training provenance, exact latency profile and episode-level evaluation evidence. The original detailed JSON index remains in the benchmark publication receipt. - -### Interpretation and reproduction - -The same zero-trained checkpoint is used in both evaluation environments. Each architecture has its own latency profile, so this is not a same-latency cross-architecture ranking. Zero and profile training use different teachers and collected data; differences do not isolate learning rate or a single training factor. Results use one training seed. - -Common training: two GPUs, bf16 ZeRO-2, batch 64/rank, accumulation 1, global batch 128, 16 workers/rank, prefetch 4, backbone LR 3e-6, action-head LR 1e-4, 100-step warmup, cosine min_lr_rate 1/30, 5,000 steps. Architecture-specific losses and other parameter groups are retained. - -Hist8 corrects the earlier -1/1 history input mismatch. Deadly also corrects an evaluation rendering mismatch; invalid older results are not included in this release. Profiles are preserved byte-for-byte with their sampling sidecars. Use each model task contract and pinned base-model revision when replaying. - -## Standard-Pipeline import: seven Gymnasium tasks - -AirRaid, Ant, HalfCheetah, Hopper, Humanoid, InvertedPendulum and Walker2d: 37 H1 VLA bundles (QwenOFT, QwenGR00T, QwenPI v3) and 28 Sample Factory APPO bundles. Each bundle contains one selected checkpoint. These are source-preserving copies with recorded experiment status; this import does not imply new evaluation or acceptance. - -[Model inventory and verification](reports/standard-pipeline-2208875f92b2/README.md). HalfCheetah lacks profile QwenGR00T/QwenPI v3 VLAs; Humanoid lacks all three profile VLAs. +[Benchmark release inventory](reports/benchmark-release/README.md) diff --git a/latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/README.md b/latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/README.md new file mode 100644 index 0000000000000000000000000000000000000000..3408c61a6a8160febdcb83e84927e11aec16b1b6 --- /dev/null +++ b/latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/README.md @@ -0,0 +1,7 @@ +# memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0 + +Deadly Corridor · sample-factory · rollout teacher + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/artifact.json b/latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..d03a36c782913dce5c3c3605ab889c4c62c86d0a --- /dev/null +++ b/latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/artifact.json @@ -0,0 +1,49 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "memory/rollout_teachers/deadly_corridor" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/small-policy/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/train/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/checkpoint_p0/checkpoint_000048836_25004032.pth" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/train/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/train/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/git.diff" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/train/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/sf_log.txt" + } + ], + "asset_state": "teacher_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "memory/rollout_teachers/deadly_corridor", + "framework": "Sample Factory", + "weight_path": "memory/rollout_teachers/deadly_corridor/train/memory_rollout_teacher:deadly_corridor:teacher:fixed_l6:fs4:obs8p75:stride4:seed0/checkpoint_p0/checkpoint_000048836_25004032.pth", + "source_repo": "latency-sensitive-bench/paper-experiment-models", + "source_revision": "7a26e7762131275e85f5b2398fd0406affb820b8", + "source_weight_sha256": "3a5d74e953be4b01ee57ea2cc6012317e3d25c046d34a89e62258b0a9844e124", + "train_step": 48836, + "env_steps": 25004032, + "selection": "latest in config and final checkpoint in training log", + "evidence_repo": "latency-sensitive-bench/memory-data", + "evidence_revision": "49da56bd92b9842fb457418aba760a1b03379258" + } + ], + "training_data": [] +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/README.md b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/README.md new file mode 100644 index 0000000000000000000000000000000000000000..42b9ec719256e2f6667f60fb977a83325071d04d --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/README.md @@ -0,0 +1,7 @@ +# deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/artifact.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..ae041934fdfab9a9037f8ef44c2d152cac0c5216 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/artifact.json @@ -0,0 +1,75 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex", + "framework": "QwenOFT", + "weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "config_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/config.full.yaml", + "normalization_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "f3b8a14cb9feeecdddba3d73f011c49d93034b4d2f20a73acc59b7ea66ba395d", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 8, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "deadly_corridor_fix_latency_6_1000ep_7k2steps_kv_memory_flex" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fixed_latency_6_1000ep_7k2steps", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + } + ] +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/README.md b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ef5f6337f12927533a5475681f0c1c3037c85946 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/README.md @@ -0,0 +1,7 @@ +# deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/artifact.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..f91c6c7d8efb6589edbf3d0c2e176804afa227fe --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/artifact.json @@ -0,0 +1,75 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi", + "framework": "QwenOFT", + "weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "config_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/config.full.yaml", + "normalization_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "d7b311bfda72620b79f10575c5acc9b52d1c722e3ee42b20c2878603797e0530", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "deadly_corridor_fix_latency_6_1000ep_7k2steps_plain_multi" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fixed_latency_6_1000ep_7k2steps", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + } + ] +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/README.md b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a2991d65f834de075d0f21b02520d46a00dd641e --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/artifact.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..927aa629c00a1e70913255955068f7a2ab24e33f --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,85 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/eval/post_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/eval_inputs/deadly_corridor_fixed_latency_6.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "1dfd3f99e14743813ff4a86047bc66f0a4e81a6e8a4c5970c617d1813721e121", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "deadly_corridor_fix_latency_6_1000ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fixed_latency_6_1000ep_7k2steps", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + } + ] +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/README.md b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d7522d7988d5a582c9e44637455a3bb5efb841ba --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/README.md @@ -0,0 +1,7 @@ +# deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/artifact.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..570bb92a9c4d3cd70d5ab2a36a134c6cbd07a616 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/artifact.json @@ -0,0 +1,75 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch", + "framework": "QwenOFT", + "weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "config_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/config.full.yaml", + "normalization_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "6ae6507a0c59ba77f90eb1e823c920092f3ad51ad8bb4da7689ccb48b303bb90", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + }, + "image_mode": "stitch", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "deadly_corridor_fix_latency_6_1000ep_7k2steps_stitch" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fixed_latency_6_1000ep_7k2steps", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + } + ] +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/README.md b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0c7ec8d0d8341768b48f575d7782a3cc4dea1fbb --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/README.md @@ -0,0 +1,7 @@ +# openvla_deadly_corridor_fix_latency_2 + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/artifact.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..240b8ef43c4d732660855f9541dc85d3b33d697c --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_deadly_corridor_fix_latency_2" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_2/checkpoints/steps_500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_deadly_corridor_fix_latency_2", + "source_revision": "0a60d6c8a400b48e4d9737b36ef3a44814d52d2a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_deadly_corridor_fix_latency_2", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_deadly_corridor_fix_latency_2", + "training_step": 500, + "seed": 42, + "training_data_root": "/inspire/hdd/global_user/liumingyu-253208120284/lzj/starvla/data/deadly_corridor_fix_latency_2", + "training_data_mix": "deadly_corridor_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fix_latency_2", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "deadly_corridor_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "deadly_corridor_train", + "mixed_converted_name": "deadly_corridor_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_deadly_corridor_fix_latency_2/checkpoints/best_model.safetensors", + "identity": "630d55bc258d986e0de4b04d9223f418855bb94c2d310c13ae808912553d6c7f", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_deadly_corridor_fix_latency_2/checkpoints/steps_500_model.safetensors", + "identity": "ad5f732f19bbfda68d6686e39e875f07957d08980bf6325dea7f85eccb4abc70", + "source_path": "checkpoints/steps_500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fix_latency_2", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/README.md b/latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d2a4d49ed70d6f0e34d73be504d9a918e232025c --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/README.md @@ -0,0 +1,7 @@ +# wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce + +Deadly Corridor · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/artifact.json b/latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..db573e106d5e1f6c7f25859c38cd980d5641d5cb --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/artifact.json @@ -0,0 +1,65 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/deadly-corridor/vla/starvla-wanoft-h8/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/checkpoints/steps_1000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/eval/fixed_6/eval_latency_6/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/eval/fixed_6/queue_eval_latency_6.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/eval/fixed_6/queue_eval_results.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce", + "framework": "WanOFT", + "weight_path": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/checkpoints/steps_1000_pytorch_model.pt", + "config_path": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/config.full.yaml", + "normalization_path": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce/checkpoints/steps_1000_pytorch_model.pt", + "source_weight_sha256": "a8bd1415990de5a8616a203b15d58f105dbcc57e5ac10ffff0ea70e2220d9c5c", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "wan_oft_deadly_corridor_fix_latency_6_context5_standard_sft_1000_effbs128_224_currentbce" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/deadly-corridor/deadly_corridor_fixed_latency_6_1000ep_7k2steps", + "config_name": "deadly_corridor_fixed_latency_6_1000ep_7k2steps" + } + ] +} diff --git a/latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/README.md b/latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/README.md new file mode 100644 index 0000000000000000000000000000000000000000..24ac8b2420613e2e757da8fc185bcb419c310b15 --- /dev/null +++ b/latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/README.md @@ -0,0 +1,7 @@ +# memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0 + +Demon Attack · sample-factory · rollout teacher + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/artifact.json b/latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..67afce7d0826910c31c0835c88d85ad4da4213b8 --- /dev/null +++ b/latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/artifact.json @@ -0,0 +1,49 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "memory/rollout_teachers/demon_attack" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/small-policy/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/train/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/checkpoint_p0/checkpoint_000048880_25026560.pth" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/train/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/train/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/git.diff" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/train/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/sf_log.txt" + } + ], + "asset_state": "teacher_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "memory/rollout_teachers/demon_attack", + "framework": "Sample Factory", + "weight_path": "memory/rollout_teachers/demon_attack/train/memory_rollout_teacher:demon_attack:teacher:fixed_l6:fs4:obs15:stride4:seed0/checkpoint_p0/checkpoint_000048880_25026560.pth", + "source_repo": "latency-sensitive-bench/paper-experiment-models", + "source_revision": "7a26e7762131275e85f5b2398fd0406affb820b8", + "source_weight_sha256": "6fe7bedd8b32d6c7299f01bd97f1f72f787c7edf7d3dbf519bf90c9384871150", + "train_step": 48880, + "env_steps": 25026560, + "selection": "latest in config and final checkpoint in training log", + "evidence_repo": "latency-sensitive-bench/memory-data", + "evidence_revision": "49da56bd92b9842fb457418aba760a1b03379258" + } + ], + "training_data": [] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..612ba4e8bd6d278898b9f59a829d15d7328f97e3 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_2_200ep + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..5e260bfaa4089d2fb29415b8882de87a85e975ab --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "demon_attack_fix_latency_2_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_2_200ep/checkpoints/steps_7000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_200ep", + "source_revision": "04fece441aa5b6f5a6a015264a6aa5df3dc4c29a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "demon_attack_fix_latency_2_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "demon_attack_fix_latency_2_200ep", + "training_step": 7000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_2_200ep", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "demon_attack_fix_latency_2_200ep/checkpoints/steps_7000_model.safetensors", + "identity": "2ab6d4d4eb795a9a09da9c63d489771039440689877642e3f9ec48dd9112119a", + "source_path": "demon_attack_fix_latency_2_200ep/checkpoints/steps_7000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/demon_attack", + "paper/figures/fig6_latency_robustness/demon_attack" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..663d3f672c8567f36135267952d030e6110adc34 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_4_200ep + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..31dcc60dc535851654b0cc15f60d70993cddc629 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "demon_attack_fix_latency_4_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_4_200ep/checkpoints/steps_7000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_200ep", + "source_revision": "04fece441aa5b6f5a6a015264a6aa5df3dc4c29a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "demon_attack_fix_latency_4_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "demon_attack_fix_latency_4_200ep", + "training_step": 7000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_4_200ep", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "demon_attack_fix_latency_4_200ep/checkpoints/steps_7000_model.safetensors", + "identity": "62e9b2a2b38f2b52cedc248c6e135b2593b212d1965fc3efae42e8130b57b373", + "source_path": "demon_attack_fix_latency_4_200ep/checkpoints/steps_7000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/demon_attack", + "paper/figures/fig6_latency_robustness/demon_attack" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6cf2963028d9c99659a462a231418a7cd1ba7cd4 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_6_200ep + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..fa18ed3df3cc4a3247f9ad7b0c83f15a0d201515 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "demon_attack_fix_latency_6_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep/checkpoints/steps_7000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_200ep", + "source_revision": "04fece441aa5b6f5a6a015264a6aa5df3dc4c29a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "demon_attack_fix_latency_6_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "demon_attack_fix_latency_6_200ep", + "training_step": 7000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_6_200ep", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "demon_attack_fix_latency_6_200ep/checkpoints/steps_7000_model.safetensors", + "identity": "375ff9f24479e219e8197bbc2eee3e71e61e1853e0c80eefe438ca947d5c5b1b", + "source_path": "demon_attack_fix_latency_6_200ep/checkpoints/steps_7000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/demon_attack", + "paper/figures/fig6_latency_robustness/demon_attack" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6fb4cf4807bc43119b17e5209a105418d9a841a1 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..4e73474f421c7bde986c5b95569e61e50b74334c --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/artifact.json @@ -0,0 +1,80 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/eval/post_train/step_4000.cap3600.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex", + "framework": "QwenOFT", + "weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "config_path": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/config.full.yaml", + "normalization_path": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "eef33f41a8d202aced4f896622ec2b872f253e3b453d2fe2497c66cfb05c2cc4", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 8, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "demon_attack_fix_latency_6_200ep_7k2steps_kv_memory_flex" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f295e9c4d601c1ce8a18eaa24f788a44c977fdc6 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_6_200ep_7k2steps_plain_multi + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..1044e6b32df56ec94aa10c8e0ff701314c3bab8a --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/artifact.json @@ -0,0 +1,85 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/eval/post_train/step_4000.cap3600.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/eval/post_train/step_4000.pre_framealign.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi", + "framework": "QwenOFT", + "weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "config_path": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/config.full.yaml", + "normalization_path": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "08ba825529c75663280016705b5583bcd4837fd749a7c9ee4b4eb1f75e266eca", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "demon_attack_fix_latency_6_200ep_7k2steps_plain_multi" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..c3155d5b7325baf8df1b4698f520cb859bb431af --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_6_200ep_7k2steps_single_baseline + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..dc43173ab1d453b00ee69bd232265bc2ed963834 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,80 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/eval/post_train/step_4000.cap3600.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "d5319537cca52cfb0e05705561bb0d883fd3fc6e19314fb386ae3b82659a910c", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "demon_attack_fix_latency_6_200ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/README.md new file mode 100644 index 0000000000000000000000000000000000000000..da6b7a8dcbf7595bb5bf7647cd94bfca7044a47d --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_6_200ep_7k2steps_stitch + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..0b82c97bbbf61d7625c156f16688af4492e3a098 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/artifact.json @@ -0,0 +1,85 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "demon_attack_fix_latency_6_200ep_7k2steps_stitch" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_6_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/eval/post_train/step_4000.cap3600.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/eval/post_train/step_4000.pre_framealign.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_6_200ep_7k2steps_stitch/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "demon_attack_fix_latency_6_200ep_7k2steps_stitch", + "framework": "QwenOFT", + "weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "config_path": "demon_attack_fix_latency_6_200ep_7k2steps_stitch/config.full.yaml", + "normalization_path": "demon_attack_fix_latency_6_200ep_7k2steps_stitch/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "demon_attack_fix_latency_6_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "5bb27757308c5a506424a7c353c5611c3fd4f6cb319a8f9df0c82f696c6c9db9", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + }, + "image_mode": "stitch", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "demon_attack_fix_latency_6_200ep_7k2steps_stitch" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f5a2b6354f8c758225b14a5964862ea14a9430d3 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_8_200ep + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..5145f2b833c64687d4ee87a030c13defde9f02b6 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "demon_attack_fix_latency_8_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_8_200ep/checkpoints/steps_7000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_200ep", + "source_revision": "04fece441aa5b6f5a6a015264a6aa5df3dc4c29a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "demon_attack_fix_latency_8_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "demon_attack_fix_latency_8_200ep", + "training_step": 7000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_8_200ep", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "demon_attack_fix_latency_8_200ep/checkpoints/steps_7000_model.safetensors", + "identity": "0d52d791412c41f17f607087c988d99267ef09ac097022b3fc3d97ebbb5b0b00", + "source_path": "demon_attack_fix_latency_8_200ep/checkpoints/steps_7000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/demon_attack", + "paper/figures/fig6_latency_robustness/demon_attack" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/README.md new file mode 100644 index 0000000000000000000000000000000000000000..aeec6fea5906f6cbaeeda553474f6ec11e0af36b --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15 + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..599e66be5c635abcbd72d576ed8095af39c83933 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/artifact.json @@ -0,0 +1,122 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/eval/mid_train/step_5000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/eval/post_train/step_5000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_revision": "cd810024c06f961c95449785f19ab73ccfd20de1", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "training_step": 5000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_subdir": null + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 40, + "latency_filter": [ + 6 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15/checkpoints/steps_5000_model.safetensors", + "identity": "2506bd51c7054d65cc980738979dd484de07b4af15da554185cbc28ede512de5", + "source_path": "checkpoints/steps_5000_model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_subdir": null + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..74525f8460e21f72079b2ca0c1ef48b1504c8233 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2 + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..ab155044e1cc0d228bb2b76a14e670cc3e898b76 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2/artifact.json @@ -0,0 +1,78 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2" + }, + "checkpoint": [], + "evaluation": [], + "asset_state": "metadata_only", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "source_revision": "ac5ff1109c46295299d74b5e30a56f1e27e930a6", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "artifact_state": "metadata_only", + "framework": "QwenOFT", + "run_id": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "training_step": 4000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/memory-rollouts", + "revision": "189c04468447ed75c3203b44844b52d11832def7", + "path": "demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "evidence": "Original training config directly names this retained dataset repository." + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/memory-rollouts", + "config_name": null, + "source_subdir": "demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "raw_frames", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 200, + "latency_filter": [ + 6 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/memory-rollouts", + "revision": "189c04468447ed75c3203b44844b52d11832def7", + "path": "demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "evidence": "Original training config directly names this retained dataset repository." + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/README.md new file mode 100644 index 0000000000000000000000000000000000000000..4c1d61064a10472ca842741981e07e081e864c01 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3 + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..7748a7913ddf6ce808a664bfe2464de1c744dd86 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/artifact.json @@ -0,0 +1,132 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/post_train/step_4000.max3600.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/post_train/step_4000.part0of2.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/eval/post_train/step_4000.part1of2.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3", + "source_revision": "f8d64aff21628adbdd0d1aaddd1bc7067bc2a23c", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3", + "training_step": 4000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/memory-rollouts", + "revision": "189c04468447ed75c3203b44844b52d11832def7", + "path": "demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "evidence": "Original training config directly names this retained dataset repository." + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/memory-rollouts", + "config_name": null, + "source_subdir": "demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "raw_frames", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 200, + "latency_filter": [ + 6 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2_exp3/checkpoints/steps_4000_model.safetensors", + "identity": "834ce01a8a1ec7ec4da4b6ea2765d7dbc249af3f80e59c571744dcee4242a13e", + "source_path": "checkpoints/steps_4000_model.safetensors", + "size": 9784896958 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/memory-rollouts", + "revision": "189c04468447ed75c3203b44844b52d11832def7", + "path": "demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2", + "evidence": "Original training config directly names this retained dataset repository." + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_2" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..b4260212090cbb04412e09340ba19b4eb1988050 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2 + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..80c2173c24f54e8757b0d9f1c8c4a58f12fbafcb --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/artifact.json @@ -0,0 +1,122 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/eval/mid_train/step_5000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/eval/post_train/step_5000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2", + "source_revision": "c4cbca936a122c2d55b1edee3959656f8174c5b7", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2", + "training_step": 5000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_subdir": null + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 40, + "latency_filter": [ + 6 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_exp2/checkpoints/steps_5000_model.safetensors", + "identity": "2506bd51c7054d65cc980738979dd484de07b4af15da554185cbc28ede512de5", + "source_path": "checkpoints/steps_5000_model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_subdir": null + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/README.md new file mode 100644 index 0000000000000000000000000000000000000000..28b5dc227ddebc8b78c4255cea715d22b45752b0 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1 + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..731a3319561c10e674145b73a80b55b916ca5fec --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/artifact.json @@ -0,0 +1,122 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/eval/mid_train/step_5000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/eval/post_train/step_5000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1", + "source_revision": "bc733d64e32512a8fc76f9be022eb5b0d7c0e2cb", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1", + "training_step": 5000, + "seed": 1, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_subdir": null + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 40, + "latency_filter": [ + 6 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_demon_attack_fixed_latency_6_200ep_7k2steps_ghost15_seed1/checkpoints/steps_5000_model.safetensors", + "identity": "2506bd51c7054d65cc980738979dd484de07b4af15da554185cbc28ede512de5", + "source_path": "checkpoints/steps_5000_model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15", + "source_subdir": null + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps_ghost15" + } + ] +} diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/README.md b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..e45f5e28d887a60d32b4781fc72dd954247c08cb --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/README.md @@ -0,0 +1,7 @@ +# openvla_demon_attack_fix_latency_2_small + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/artifact.json b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..efc6fdd529a016066b8c6becb52cbb6388bd831d --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/artifact.json @@ -0,0 +1,78 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_demon_attack_fix_latency_2_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_2_small/checkpoints/steps_2000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_fix_latency_2_small", + "source_revision": "bb91ae40a1bb9b9f6a93be031193de603da86ecd", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_demon_attack_fix_latency_2_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_demon_attack_fix_latency_2_small", + "training_step": 2000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_2_small", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_demon_attack_fix_latency_2_small/checkpoints/best_model.safetensors", + "identity": "adb853d16890002501675633ddefc2126fd3f102a63cf6ece7a5e56d0c8c662c", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_demon_attack_fix_latency_2_small/checkpoints/steps_2000_model.safetensors", + "identity": "adb853d16890002501675633ddefc2126fd3f102a63cf6ece7a5e56d0c8c662c", + "source_path": "checkpoints/steps_2000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/README.md b/latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d746a873c0e2ba8219dea963706e61d03f5261a8 --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/README.md @@ -0,0 +1,7 @@ +# wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce + +Demon Attack · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/artifact.json b/latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..5f84bca30a2e84cab573fcf9f5af21927007fd5f --- /dev/null +++ b/latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/artifact.json @@ -0,0 +1,85 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/demon-attack/vla/starvla-wanoft-h8/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/eval/fixed_6/eval_latency_6/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/eval/fixed_6/queue_eval_latency_6.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/eval/fixed_6/queue_eval_results.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce", + "framework": "WanOFT", + "weight_path": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/checkpoints/steps_5000_pytorch_model.pt", + "config_path": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/config.full.yaml", + "normalization_path": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce/checkpoints/steps_5000_pytorch_model.pt", + "source_weight_sha256": "791c66ee0313063e3eeec01a3eeb788e642f8e22952e232caa6562998d0754dd", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "wan_oft_demon_attack_fix_latency_6_context5_standard_sft_5000_effbs128_224_currentce" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/demon-attack/demon_attack_fixed_latency_6_200ep_7k2steps", + "config_name": "demon_attack_fixed_latency_6_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/README.md b/latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/README.md new file mode 100644 index 0000000000000000000000000000000000000000..401f68d185ac04d6c787f5ea466b34dbb38e9bc5 --- /dev/null +++ b/latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/README.md @@ -0,0 +1,7 @@ +# memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0 + +Flappy Bird · sample-factory · rollout teacher + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/artifact.json b/latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..aeb7b855b303d7910f89b725bedae12b52cef494 --- /dev/null +++ b/latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/artifact.json @@ -0,0 +1,49 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "memory/rollout_teachers/flappy" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/small-policy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/train/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/checkpoint_p0/checkpoint_000048840_25034752.pth" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/train/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/train/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/git.diff" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/train/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/sf_log.txt" + } + ], + "asset_state": "teacher_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "memory/rollout_teachers/flappy", + "framework": "Sample Factory", + "weight_path": "memory/rollout_teachers/flappy/train/memory_rollout_teacher:flappy:teacher:fixed_l3:fs4:obs30:stride1:seed0/checkpoint_p0/checkpoint_000048840_25034752.pth", + "source_repo": "latency-sensitive-bench/paper-experiment-models", + "source_revision": "7a26e7762131275e85f5b2398fd0406affb820b8", + "source_weight_sha256": "73335509ad209b093377f3c98b5da814a2ae29f23ba9c5ef2d363cae3766cbd2", + "train_step": 48840, + "env_steps": 25034752, + "selection": "latest in config and final checkpoint in training log", + "evidence_repo": "latency-sensitive-bench/memory-data", + "evidence_revision": "49da56bd92b9842fb457418aba760a1b03379258" + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/README.md new file mode 100644 index 0000000000000000000000000000000000000000..32a850b3489d2742725b1be087c37b22c5db0cec --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/README.md @@ -0,0 +1,7 @@ +# flappy_curriculum_cumulative_40ep_per_latency + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..cb6385de9cd1f70fb2fb15cc3be85d47b6a5614e --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_curriculum_cumulative_40ep_per_latency" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_cumulative_40ep_per_latency/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_curriculum_cumulative_40ep_per_latency", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_curriculum_cumulative_40ep_per_latency", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_mixed_latency_40ep_per_lat", + "training_data_mix": "flappy_mixed_latency_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_mixed_latency_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_curriculum_cumulative_40ep_per_latency/checkpoints/steps_5000_model.safetensors", + "identity": "ad4903982cf1ec6f29c8ae4edea25e10b42353d56e0755b5ed123f5df3351754", + "source_path": "flappy_curriculum_cumulative_40ep_per_latency/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/README.md new file mode 100644 index 0000000000000000000000000000000000000000..3a0728808e08890488aabf79dbdd02370e02ba9c --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/README.md @@ -0,0 +1,7 @@ +# flappy_curriculum_exclusive_40ep_per_latency + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..ff759d985e46ffe566582b863c17ec7a68f7eb5a --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_curriculum_exclusive_40ep_per_latency" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_curriculum_exclusive_40ep_per_latency/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_curriculum_exclusive_40ep_per_latency", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_curriculum_exclusive_40ep_per_latency", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_mixed_latency_40ep_per_lat", + "training_data_mix": "flappy_mixed_latency_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_mixed_latency_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_curriculum_exclusive_40ep_per_latency/checkpoints/steps_5000_model.safetensors", + "identity": "4a6b8739fd13e8de24eaacf3f1e133ea6792e01f66cd919ed980d744c0a52e4e", + "source_path": "flappy_curriculum_exclusive_40ep_per_latency/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..948677e0a42e1e23f93635d2914e222aeb923fa9 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_1_200ep + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..5b81f781be0ef3697609c53e7d4b6460590edac0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/artifact.json @@ -0,0 +1,71 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_fix_latency_1_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_1_200ep/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_fix_latency_1_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_fix_latency_1_200ep", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_1_200ep", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_fix_latency_1_200ep/checkpoints/steps_5000_model.safetensors", + "identity": "cff93451f827160bf275aa8bc0bcab83cbd71645ef0605b2e8869bd3691a3bb6", + "source_path": "flappy_fix_latency_1_200ep/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6fea70d6f21a55262831c66e8d6e65965f61952a --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_2_200ep + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..0f1be941c38093b2d704bacee403bfb90215935c --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/artifact.json @@ -0,0 +1,72 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_fix_latency_2_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_fix_latency_2_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_fix_latency_2_200ep", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_2_200ep", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_fix_latency_2_200ep/checkpoints/steps_5000_model.safetensors", + "identity": "aabcc8e3a22947e22d161d324de0e613a03300f39fa1dd28bffa2df927e994cf", + "source_path": "flappy_fix_latency_2_200ep/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/README.md new file mode 100644 index 0000000000000000000000000000000000000000..899f9a6da55aede1f0063f810a5e2c17fd1839f7 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_2_200ep_7k2steps_kv_memory + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..15529f0048627cb69ebe8aaf9e644c6726c6247d --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/artifact.json @@ -0,0 +1,240 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory/checkpoints/steps_4000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_1000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_1250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_1500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_1750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_2000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_2250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_2500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_2750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_3000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_3250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_3500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_3750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/mid_train/step_750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/latency_bench_eval/post_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_2_200ep_7k2steps_kv_memory", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory/checkpoints/steps_4000_pytorch_model.pt", + "config_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory/config.full.yaml", + "normalization_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory/checkpoints/steps_4000_state/pytorch_model/mp_rank_00_model_states.pt", + "source_weight_sha256": "a9b3853e369b9453e194e05846a8be58e62292ce743a75ce7304d0e9068031d2", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 8, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_2_200ep_7k2steps_kv_memory" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_7k2steps", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/README.md new file mode 100644 index 0000000000000000000000000000000000000000..fa3f4b1f578bc2ab975fe10a406f3f5bc000e936 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..e66664467fee4b96fc50074e497fc3cb9f57d193 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/latency_bench_eval/post_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/config.full.yaml", + "normalization_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "2f1e4d04270984d96f78074912c082fa8275d3680797a7a2eff02666dcfc1f25", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 8, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_2_200ep_7k2steps_kv_memory_flex" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_7k2steps", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/README.md new file mode 100644 index 0000000000000000000000000000000000000000..740fae79857103464d178556c08482ac04b89180 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_2_200ep_7k2steps_plain_multi + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..b026ad259947459ab6310cf4dfd87ffdc68ef2a7 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_2_200ep_7k2steps_plain_multi" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/latency_bench_eval/post_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_plain_multi/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_2_200ep_7k2steps_plain_multi", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_2_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_2_200ep_7k2steps_plain_multi/config.full.yaml", + "normalization_path": "flappy_fix_latency_2_200ep_7k2steps_plain_multi/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_2_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "639393e062693dbb9e2bd801367353cd11256a4e8a7419e02186d360b5b249b2", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_2_200ep_7k2steps_plain_multi" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_7k2steps", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..e245d8cfdfcd3e5bd8df9c601e99245ff3fe315a --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_2_200ep_7k2steps_single_baseline + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..db0d8b6d6aea7294468ad38310203cfe428ee6f2 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_2_200ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_2_200ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_2_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_2_200ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "flappy_fix_latency_2_200ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_2_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "a92627efd00d16c3a3a392b46cc366ec06956b0e651f81951b1be5435cc75f37", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_2_200ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_7k2steps", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6c30d11a1bb23d9c4727782fff74a5201c91de38 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_2_200ep_7k2steps_stitch + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..0434dcbc0cb461bdc0b5d2165fc1d954f15408b7 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_2_200ep_7k2steps_stitch" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_2_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3250/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_500/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_750/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/latency_bench_eval/post_train/step_4000/flappy/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_2_200ep_7k2steps_stitch/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_2_200ep_7k2steps_stitch", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_2_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_2_200ep_7k2steps_stitch/config.full.yaml", + "normalization_path": "flappy_fix_latency_2_200ep_7k2steps_stitch/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_2_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "d29f05e6565500aecc785f04b687b107f703021b59947f2d261d167fb932c3f5", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + }, + "image_mode": "stitch", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_2_200ep_7k2steps_stitch" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_7k2steps", + "config_name": "flappy_fix_latency_2_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..11b65921f9523191c29363366c4b690041429eb1 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_3_200ep + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..d5b380cc79d6ada01bf730c6b7cd8e4d189d84f6 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/artifact.json @@ -0,0 +1,72 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_fix_latency_3_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_fix_latency_3_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_fix_latency_3_200ep", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_3_200ep", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_fix_latency_3_200ep/checkpoints/steps_5000_model.safetensors", + "identity": "5da01d6eeb7a46861da90b0a8dbc364a33f92d542d25079bcc86280b10617418", + "source_path": "flappy_fix_latency_3_200ep/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ab484e4626d783ff46af99f6815baf6d3f1dbbe9 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..1ce1096d81b02104a9184c1cdc67bb436faf579c --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_1750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_2750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_3750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/mid_train/step_750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/latency_bench_eval/post_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/config.full.yaml", + "normalization_path": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "4a08664f7263fd473ee6a198aa3150af14d7923f0ad78b0959966a5c08b33461", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 8, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_3_200ep_7k2steps_kv_memory_flex" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9bf76b19a0c1333dd2fd53fe5fa5f3a0b025ea90 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_3_200ep_7k2steps_plain_multi + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..1eea5fcfaa3386462ebf1e111bfd049e4f4be75d --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_3_200ep_7k2steps_plain_multi" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_1750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_2750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_3750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/mid_train/step_750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/latency_bench_eval/post_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_plain_multi/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_3_200ep_7k2steps_plain_multi", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_3_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_3_200ep_7k2steps_plain_multi/config.full.yaml", + "normalization_path": "flappy_fix_latency_3_200ep_7k2steps_plain_multi/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_3_200ep_7k2steps_plain_multi/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "4af03df4b8cb41decacc38f0d37c3e64af1673acc3b170d4b622dad8f320628e", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + }, + "image_mode": "multiframe", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_3_200ep_7k2steps_plain_multi" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7feac983ceb62549a53309eed031b320cc8dd2e5 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_3_200ep_7k2steps_single_baseline + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..93cdb43aafde94a392f59c972609cb7064e5e3d4 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_3_200ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_3_200ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_3_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_3_200ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "flappy_fix_latency_3_200ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_3_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "b0793b41322a4270daf848c5d82e63d09a0cb5c7e43ccf91b29527e574619783", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_3_200ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/README.md new file mode 100644 index 0000000000000000000000000000000000000000..39857e47ad6d73fde34ad1f6ed0c24f219375a10 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_3_200ep_7k2steps_stitch + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..a2f2e6eeffb1fad534f3cd0302b6cd80a0082596 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/artifact.json @@ -0,0 +1,420 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_3_200ep_7k2steps_stitch" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_3_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/post_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/eval_inputs/flappy_fixed_latency_3.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_1750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_2750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3250/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_3750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_500/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/mid_train/step_750/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/latency_bench_eval/post_train/step_4000/flappy/latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_3_200ep_7k2steps_stitch/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_3_200ep_7k2steps_stitch", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_3_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_3_200ep_7k2steps_stitch/config.full.yaml", + "normalization_path": "flappy_fix_latency_3_200ep_7k2steps_stitch/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_3_200ep_7k2steps_stitch/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "0c5786df9eb04f1b88390377df8327452fe6c9bb3d759912c491fbdf5cb299dd", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + }, + "image_mode": "stitch", + "num_obs_frames": 4, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_3_200ep_7k2steps_stitch" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..5d36d165bf808f68edc707dc61b004d4704d92d8 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_4_200ep + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..3644466a0ee188931d45637bcec5c368d434fd49 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/artifact.json @@ -0,0 +1,72 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_fix_latency_4_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_4_200ep/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_fix_latency_4_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_fix_latency_4_200ep", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_4_200ep", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_fix_latency_4_200ep/checkpoints/steps_5000_model.safetensors", + "identity": "99f4d57787efbfd68e5fc68789f72e7e77060c0061568006a8ce8c721c73b305", + "source_path": "flappy_fix_latency_4_200ep/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/README.md new file mode 100644 index 0000000000000000000000000000000000000000..8b84e1e45097cec4c242888fa674c85c4e2d4907 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..51f747d5601ee59272f6c76de502f7df4e1bf8e7 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/artifact.json @@ -0,0 +1,97 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/eval/post_train/step_5000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_revision": "9915dcb0dc644aa2ed14172f298386f53b03c6f6", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "training_step": 5000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_subdir": null + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 200, + "latency_filter": [ + 3 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15/checkpoints/steps_5000_model.safetensors", + "identity": "91931822871b9c3af55f58a7254963b512e478bda6d7514dd2a194ac27311cd5", + "source_path": "checkpoints/steps_5000_model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_subdir": null + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..bfdd5f0aa1ce306fb729dae9c822045ffac5cfb4 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..b60d2293b02d7130cf2169a5f7b44db8b4e322b5 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/artifact.json @@ -0,0 +1,97 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/eval/post_train/step_5000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2", + "source_revision": "73ea97c064d206c3a0dcf6ffdbf2ecfe97bc2c3c", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2", + "training_step": 5000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_subdir": null + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 200, + "latency_filter": [ + 3 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_exp2/checkpoints/steps_5000_model.safetensors", + "identity": "91931822871b9c3af55f58a7254963b512e478bda6d7514dd2a194ac27311cd5", + "source_path": "checkpoints/steps_5000_model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_subdir": null + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/README.md new file mode 100644 index 0000000000000000000000000000000000000000..cf2ebad25a76aaccf81394510455e26c935b9979 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..2b07da2253d9c3b2402b3a79be612bb170d3ec26 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/artifact.json @@ -0,0 +1,97 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/eval/post_train/step_5000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1", + "source_revision": "004ec237af79c10de5c0d4376cf8a13542ac01f8", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1", + "training_step": 5000, + "seed": 1, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_subdir": null + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "config_name": null, + "source_subdir": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 200, + "latency_filter": [ + 3 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost15_seed1/checkpoints/steps_5000_model.safetensors", + "identity": "91931822871b9c3af55f58a7254963b512e478bda6d7514dd2a194ac27311cd5", + "source_path": "checkpoints/steps_5000_model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15", + "source_subdir": null + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost15" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/README.md new file mode 100644 index 0000000000000000000000000000000000000000..5abb24ee74183970bb9c0e6fdbabcd79a3623be3 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..f1469c116af05bd4d6669aaa604f812720ed7f09 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/artifact.json @@ -0,0 +1,97 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/eval/post_train/step_4000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3", + "source_revision": "5dfd90884fc4e3817bc51f574d6fc9cea2d7948b", + "target_repo": "latency-sensitive-bench/memory-models", + "target_run": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3", + "training_step": 4000, + "seed": 42, + "training_data_root": "/lustre/fsw/portfolios/nvr/projects/nvr_lacr_llm/users/zihwang/latency-sensitive-bench/playground/Datasets/rl_games", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/memory-rollouts", + "revision": "189c04468447ed75c3203b44844b52d11832def7", + "path": "flappy_fixed_latency_3_200ep_7k2steps_ghost7", + "evidence": "Original training config directly names this retained dataset repository." + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/memory-rollouts", + "config_name": null, + "source_subdir": "flappy_fixed_latency_3_200ep_7k2steps_ghost7", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": 200, + "latency_filter": [ + 3 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_bridge_flappy_fixed_latency_3_200ep_7k2steps_ghost7_exp3/checkpoints/steps_4000_model.safetensors", + "identity": "553d47011aa16cef4609610ea44342630fb4b0b7aabc65da5fe58d133915f40a", + "source_path": "checkpoints/steps_4000_model.safetensors", + "size": 9784896958 + } + ], + "paper_locations": [ + "paper/tables/table7_memory_comparison" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/memory-rollouts", + "revision": "189c04468447ed75c3203b44844b52d11832def7", + "path": "flappy_fixed_latency_3_200ep_7k2steps_ghost7", + "evidence": "Original training config directly names this retained dataset repository." + } + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps_ghost7" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..fca02c3af2433075e8d488214b4a7b6810282ff0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_flappy_latency_mixed_exp2 + +Flappy Bird · Latency in prompt · step 5000 + +[Settings, paper evidence and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-in-prompt/flappy) + +The source weights and paper evaluation match are preserved. The public source does not include the original full configuration or normalization statistics. This directory is a preserved weight source, not a verified standalone inference bundle. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..6009f63fb2787b97378073c40c45c3578a1e6af8 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/artifact.json @@ -0,0 +1,51 @@ +{ + "condition": "with_latency_prompt", + "checkpoint": { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2/checkpoints/model.safetensors" + }, + "evaluation": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/openvla_bridge_flappy_latency_mixed_exp2/eval/post_train/step_5000.json" + }, + "source": { + "condition": "with_latency_prompt", + "repo": "talha15032/openvla_bridge_flappy_latency_mixed_exp2", + "revision": "610906e990beefa44c23d438a3b50a556d152900", + "eval_path": "eval/post_train/step_5000.json", + "paper_episode_values_match": true, + "paper_episodes": 100, + "weight_path": "steps_5000_state/model.safetensors", + "weight_identity": "95065673334e0cfd68585fa569393bfee903194b78fa88f27c224c6f7bc222a5", + "weight_bytes": 9138230516, + "training_mode": "mixed_latency", + "training_latencies": [ + 0, + 1, + 2, + 3, + 4 + ], + "training_episodes_per_latency": 40, + "checkpoint_step": 5000, + "training_protocol_evidence": "Recorded paper manifest and matching current source script. Full original model configuration and normalization statistics are absent from the public source bundle.", + "training_data": { + "repo": "latency-sensitive-bench/latency-transfer-data", + "revision": "5305590625c28a6922e73f46d18d52738762c906", + "paths": [ + "training/flappy_200ep/flappy_fix_latency_0_200ep", + "training/flappy_200ep/flappy_fix_latency_1_200ep", + "training/flappy_200ep/flappy_fix_latency_2_200ep", + "training/flappy_200ep/flappy_fix_latency_3_200ep", + "training/flappy_200ep/flappy_fix_latency_4_200ep" + ], + "selection": "40 training episodes from each latency partition", + "binding_scope": "Current maintained training recipe and retained raw source. The exact historical training-time dataset revision is not recorded in the public model bundle." + }, + "source_script": "third_party/starVLA/examples/rl_games/bash_scripts/openvla/bridge/mixed/flappy.sh", + "status": "evaluation_source_confirmed; weights remain at original fixed source revision; full inference metadata incomplete" + }, + "inference_metadata_status": "Original config and normalization statistics are absent from the pinned public source. Independent native loading remains unverified." +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0c089e684eafc52e00a9e63349170faae5bdfbe2 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/README.md @@ -0,0 +1,7 @@ +# openvla_bridge_flappy_latency_mixed_exp2_no_latency_information + +Flappy Bird · Latency in prompt · step 5000 + +[Settings, paper evidence and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-in-prompt/flappy) + +The source weights and paper evaluation match are preserved. The public source does not include the original full configuration or normalization statistics. This directory is a preserved weight source, not a verified standalone inference bundle. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..4465ac8d3f78347ac5c672aaa277c35e18e9c081 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/artifact.json @@ -0,0 +1,51 @@ +{ + "condition": "no_latency_information", + "checkpoint": { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/checkpoints/model.safetensors" + }, + "evaluation": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information/eval/post_train/step_5000.json" + }, + "source": { + "condition": "no_latency_information", + "repo": "talha15032/openvla_bridge_flappy_latency_mixed_exp2_no_latency_information", + "revision": "bf8368f6adc45ce8f4c31e1bbe25fc3f9432b68d", + "eval_path": "eval/post_train/step_5000.json", + "paper_episode_values_match": true, + "paper_episodes": 100, + "weight_path": "steps_5000_state/model.safetensors", + "weight_identity": "b52998921f751d61d64f1f6299e28f3964ab3c5d8dcc6b1b3aee44a8c59ca340", + "weight_bytes": 9138230516, + "training_mode": "mixed_latency", + "training_latencies": [ + 0, + 1, + 2, + 3, + 4 + ], + "training_episodes_per_latency": 40, + "checkpoint_step": 5000, + "training_protocol_evidence": "Recorded paper manifest and matching current source script. Full original model configuration and normalization statistics are absent from the public source bundle.", + "training_data": { + "repo": "latency-sensitive-bench/latency-transfer-data", + "revision": "5305590625c28a6922e73f46d18d52738762c906", + "paths": [ + "training/flappy_200ep/flappy_fix_latency_0_200ep", + "training/flappy_200ep/flappy_fix_latency_1_200ep", + "training/flappy_200ep/flappy_fix_latency_2_200ep", + "training/flappy_200ep/flappy_fix_latency_3_200ep", + "training/flappy_200ep/flappy_fix_latency_4_200ep" + ], + "selection": "40 training episodes from each latency partition", + "binding_scope": "Current maintained training recipe and retained raw source. The exact historical training-time dataset revision is not recorded in the public model bundle." + }, + "source_script": "third_party/starVLA/examples/rl_games/bash_scripts/openvla/bridge/latency_ablation/flappy_bird_mixed_latency.sh", + "status": "evaluation_source_confirmed; weights remain at original fixed source revision; full inference metadata incomplete" + }, + "inference_metadata_status": "Original config and normalization statistics are absent from the pinned public source. Independent native loading remains unverified." +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0c20c2b781b19c610cbfbfc39a289ed36a9fc4dd --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_1_small + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..17c584bb1488b7234279c32b3ed037f3c86af0e9 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_1_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small/checkpoints/steps_1500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_1_small", + "source_revision": "7842a04e6ba8ed4dc56ccdae235735a807a77b5d", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_1_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_1_small", + "training_step": 1500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_1_small", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_1_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_1_small/checkpoints/best_model.safetensors", + "identity": "758b4ccf870860c583af4da492b2d6782333e847759afd7314ba4547610fb04e", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_flappy_fix_latency_1_small/checkpoints/steps_1500_model.safetensors", + "identity": "0ecf34ec797bcb608fabd5fee1b994bf8480dfd6d75eb302b9a5380cf9d69151", + "source_path": "checkpoints/steps_1500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_1_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/README.md new file mode 100644 index 0000000000000000000000000000000000000000..8a5a3a73b5445bf5fa113c1c26559f2a43b8c8d6 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_1_small_tmp + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..c24c89ae899e99fbe626206e1da3a9b982a4b503 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/artifact.json @@ -0,0 +1,78 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_1_small_tmp" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_1_small_tmp/checkpoints/steps_1500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_1_small_tmp", + "source_revision": "8fba8c350f6f10d70119807a546218cfa82e9150", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_1_small_tmp", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_1", + "training_step": 1500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_1", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_1_small_tmp/checkpoints/best_model.safetensors", + "identity": "6abc545e6b1abb272b7829b5676f412eebf33cfe0f21ba56368eb619c4610d83", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_flappy_fix_latency_1_small_tmp/checkpoints/steps_1500_model.safetensors", + "identity": "76504a6629f8a781c5cfdbeac1795bdc6e221ba0d1931eabaf12d5bf5d56ccc1", + "source_path": "checkpoints/steps_1500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..13d726da1341041ed467618a24dd5b9224383a14 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_2 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..f7279b05c793b9350bfb42009abebf561091bda8 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/artifact.json @@ -0,0 +1,77 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_2" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/checkpoints/steps_5000_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_2", + "source_revision": "8135db54520527abe4a62f002323a307d1a47c2d", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_2", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_2", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_2", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_2/checkpoints/steps_5000_pytorch_model.pt", + "identity": "7568ff3e7f09a67fd0681287750fbe5c34c99617b08bb440997c97b247c5d651", + "source_path": "checkpoints/steps_5000_pytorch_model.pt", + "size": 9138476048 + }, + { + "path": "historical/openvla_flappy_fix_latency_2/checkpoints/steps_5000_model.safetensors", + "identity": "0fc5f0a0cd58f4f9726d53639e193c48efd5afd512fb0d0f644d239eb4079668", + "source_path": "checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..135f6722a9352a8dc6c8193b0c8b7649f0f5e1ce --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_2_small + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..85d4b2b9c0f07db003f1e8f8d60dc42942f340d8 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_2_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_2_small/checkpoints/steps_1500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_2_small", + "source_revision": "87c38afed1a227e92aac9efeb0921f57d556df8a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_2_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_2_small", + "training_step": 1500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_2_small", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_2_small/checkpoints/best_model.safetensors", + "identity": "ee25ed102ecbf335b536f8260de7ae96b84cfa7158024c19e6dc51d32cd9ed27", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_flappy_fix_latency_2_small/checkpoints/steps_1500_model.safetensors", + "identity": "3aa56b5e881919571003b6155387d5ea5dfe50181e8f57b788e15e94dc643c73", + "source_path": "checkpoints/steps_1500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..76e9004dedd2cd37bd5de911c481ed0222f7bb46 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_3_small + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..62cc24c13f564a0e49e3e65b6a9d7cf4a1846dc0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_3_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_3_small/checkpoints/steps_1500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_3_small", + "source_revision": "86b4d314c377420fa8eb9ac7aa268695c2ab4f40", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_3_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_3_small", + "training_step": 1500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_3_small", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_3_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_3_small/checkpoints/best_model.safetensors", + "identity": "dfd6f77b285dff5a628b8a2e8aee12d2c8f11c2196464ed2505f7e138ec2a828", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_flappy_fix_latency_3_small/checkpoints/steps_1500_model.safetensors", + "identity": "8b61cc767d8a885e8bf0e3fecc383c8e3233beb3f3ad2544e6f7c8c4265ae614", + "source_path": "checkpoints/steps_1500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_3_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0e590b6390dcb363de5724110442ae42e68d4e9b --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_4_exp + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..a77314356bb5587e9b597457053cdc24f1d4db68 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/artifact.json @@ -0,0 +1,77 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_4_exp" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/checkpoints/steps_2500_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_exp/checkpoints/steps_2500_pytorch_model.pt" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_4_exp", + "source_revision": "213bab6c06b13c73fee875412d99e40ea4a4dd70", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_4_exp", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_4_exp", + "training_step": 2500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_4", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_4_exp/checkpoints/steps_2500_pytorch_model.pt", + "identity": "1610fb6615308c849e844acde8d98b233fe6f6c4d01657d184fd4bd9945e55f5", + "source_path": "checkpoints/steps_2500_pytorch_model.pt", + "size": 9138476048 + }, + { + "path": "historical/openvla_flappy_fix_latency_4_exp/checkpoints/steps_2500_model.safetensors", + "identity": "b4a6e3ade6d47371998034284d7a0ccbeb1f92b9713a657eb6041604cdbe9235", + "source_path": "checkpoints/steps_2500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..753d4f9ef16fde3507b51497ea1c59671a2db451 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_4_small + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..955015ca0de4ef18b59d007aa50cec4277365f79 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_4_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_4_small/checkpoints/steps_1500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_4_small", + "source_revision": "371074f07545d1d4e5533afc4d457a63d03bf9da", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_4_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_4_small", + "training_step": 1500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_4_small", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_4_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_4_small/checkpoints/best_model.safetensors", + "identity": "f893f7feb006ca5eee1dbc84ac222cf50c7b0dddcb7d572cf6d52135605bb0b9", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_flappy_fix_latency_4_small/checkpoints/steps_1500_model.safetensors", + "identity": "e3af4a77ec1ff9bb6d5cc58129a8e8b5b060d80d978228564561e188791d8029", + "source_path": "checkpoints/steps_1500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_4_small", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/README.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/README.md new file mode 100644 index 0000000000000000000000000000000000000000..cea7ba6bbcab1c9bde12df4b3bedce00a9c054f3 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_6 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/artifact.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..100dcd2d42191d4bd7cf7b8b917fda9de04045cf --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/artifact.json @@ -0,0 +1,77 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_6" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/checkpoints/steps_5000_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_6/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_6", + "source_revision": "7a2e2f81dda9fc9de068fa15b1eac442ed6fb51f", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_6", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_6", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_6", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_6/checkpoints/steps_5000_pytorch_model.pt", + "identity": "26943b407d9280f89a461ca674615037f1e2ddda896fe00b88015ff5ce22a9f0", + "source_path": "checkpoints/steps_5000_pytorch_model.pt", + "size": 9138476048 + }, + { + "path": "historical/openvla_flappy_fix_latency_6/checkpoints/steps_5000_model.safetensors", + "identity": "a24aac79d3d3fd72e0a8e782c0fd1f5b3294ee74eca4bbce1cbff73128b72122", + "source_path": "checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/README.md b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..50fed2078e3af285fc6beb9a44d35942be6bf489 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/README.md @@ -0,0 +1,7 @@ +# wan_oft_flappy_fix_latency_1_context5_2000_effbs128 + +Flappy Bird · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/artifact.json b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..0c391c9ce2916f9ad58c07cdda7174f0289134a7 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/artifact.json @@ -0,0 +1,80 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "wan_oft_flappy_fix_latency_1_context5_2000_effbs128" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_1_context5_2000_effbs128/eval/post_train/step_2000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_flappy_fix_latency_1_context5_2000_effbs128", + "module": "latency-transfer", + "train_latency_raw_frames": 1, + "train_seed": 42, + "training_step": 2000, + "training_started_at_utc": "2026-06-29T14:50:14.198406Z", + "git_commit": "972037fae54ada9fe8f64e4ac8878b2f79feba1f", + "wandb_run_id": "3kohuetr", + "checkpoint": "wan_oft_flappy_fix_latency_1_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt", + "checkpoint_identity": "ca01daf3e577fedcc238882e531c748f6a5c178934ca7cc7209da18eb24ff03f", + "source_model_revision": "1ba5d2f489f03f9cb07a637ca680fa4f2f8e51a4", + "dataset": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_1_200ep_context5", + "source_repo": "latency-sensitive-bench/flappy_200ep_context5", + "source_revision": "19d800e5f7ec9d5311f508dbfe4a50463261616c", + "train_rows": 297483, + "train_episodes": 180, + "val_rows": 72000, + "val_episodes": 20, + "pairing_evidence": "Training local path, row count, trajectory count, action statistics and fixed raw-frame latency" + }, + "paper_consumers": [ + "fig12_wanoft_latency_grid" + ] + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_1_200ep_context5" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/README.md b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..3ee21620559ccd3a31f846dc1205fe7c815c30cd --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/README.md @@ -0,0 +1,7 @@ +# wan_oft_flappy_fix_latency_2_context5_2000_effbs128 + +Flappy Bird · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/artifact.json b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..ed413e0b647bf67b34b23a6064370e9b9fe8b094 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/artifact.json @@ -0,0 +1,62 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "wan_oft_flappy_fix_latency_2_context5_2000_effbs128" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_2_context5_2000_effbs128/eval/post_train/step_2000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_flappy_fix_latency_2_context5_2000_effbs128", + "module": "latency-transfer", + "train_latency_raw_frames": 2, + "train_seed": 42, + "training_step": 2000, + "training_started_at_utc": "2026-07-01T09:28:56.645196Z", + "git_commit": "972037fae54ada9fe8f64e4ac8878b2f79feba1f", + "wandb_run_id": "2z7am65s", + "checkpoint": "wan_oft_flappy_fix_latency_2_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt", + "checkpoint_identity": "328140abd4682a66cb56cdbe03524f6c7af534e2c9ac2bd23ac1c9691363957a", + "source_model_revision": "1ba5d2f489f03f9cb07a637ca680fa4f2f8e51a4", + "dataset": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_context5", + "source_repo": "latency-sensitive-bench/flappy_200ep_context5", + "source_revision": "19d800e5f7ec9d5311f508dbfe4a50463261616c", + "train_rows": 330734, + "train_episodes": 180, + "val_rows": 72000, + "val_episodes": 20, + "pairing_evidence": "Training local path, row count, trajectory count, action statistics and fixed raw-frame latency" + }, + "paper_consumers": [ + "fig12_wanoft_latency_grid", + "fig13_wanoft_timing_controls", + "fig14_wanoft_timing_buckets" + ] + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_2_200ep_context5" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/README.md b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..91dc310daa9fe023bb61e7450b4235550901e044 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/README.md @@ -0,0 +1,7 @@ +# wan_oft_flappy_fix_latency_3_context5_2000_effbs128 + +Flappy Bird · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/artifact.json b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..d151164e34a708d1cfbd6e23257a07bd3533c758 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/artifact.json @@ -0,0 +1,80 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "wan_oft_flappy_fix_latency_3_context5_2000_effbs128" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_2000_effbs128/eval/post_train/step_2000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_flappy_fix_latency_3_context5_2000_effbs128", + "module": "latency-transfer", + "train_latency_raw_frames": 3, + "train_seed": 42, + "training_step": 2000, + "training_started_at_utc": "2026-07-01T16:06:06.153675Z", + "git_commit": "972037fae54ada9fe8f64e4ac8878b2f79feba1f", + "wandb_run_id": "jlutvxqx", + "checkpoint": "wan_oft_flappy_fix_latency_3_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt", + "checkpoint_identity": "d5940d8ea363b40611f391103534da4d13dd7204e43024079c56eb2f34322bcc", + "source_model_revision": "1ba5d2f489f03f9cb07a637ca680fa4f2f8e51a4", + "dataset": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_3_200ep_context5", + "source_repo": "latency-sensitive-bench/flappy_200ep_context5", + "source_revision": "19d800e5f7ec9d5311f508dbfe4a50463261616c", + "train_rows": 316258, + "train_episodes": 180, + "val_rows": 72000, + "val_episodes": 20, + "pairing_evidence": "Training local path, row count, trajectory count, action statistics and fixed raw-frame latency" + }, + "paper_consumers": [ + "fig12_wanoft_latency_grid" + ] + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_3_200ep_context5" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/README.md b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/README.md new file mode 100644 index 0000000000000000000000000000000000000000..b95207f61782750aceb4af599af5cc614608bc4e --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/README.md @@ -0,0 +1,7 @@ +# wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce + +Flappy Bird · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/artifact.json b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..e7830df58533b5ab1768784a28e27ab31f759895 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/artifact.json @@ -0,0 +1,65 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/checkpoints/steps_2000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/eval/fixed_3/eval_latency_3/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/eval/fixed_3/queue_eval_latency_3.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/eval/fixed_3/queue_eval_results.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce", + "framework": "WanOFT", + "weight_path": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/checkpoints/steps_2000_pytorch_model.pt", + "config_path": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/config.full.yaml", + "normalization_path": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce/checkpoints/steps_2000_pytorch_model.pt", + "source_weight_sha256": "898e0b3e19c6d07987be834b503b315688dda724ffa25e5642c16c1cb316c660", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "wan_oft_flappy_fix_latency_3_context5_standard_sft_2000_effbs128_224_currentce" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fixed_latency_3_200ep_7k2steps", + "config_name": "flappy_fixed_latency_3_200ep_7k2steps" + } + ] +} diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/README.md b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..4205f55c0f8e561a0dbc1eb3727234bf97cd4fe9 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/README.md @@ -0,0 +1,7 @@ +# wan_oft_flappy_fix_latency_4_context5_2000_effbs128 + +Flappy Bird · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/artifact.json b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..51d82aa4639666bab9d7be8820848246f0ab2b29 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/artifact.json @@ -0,0 +1,80 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "wan_oft_flappy_fix_latency_4_context5_2000_effbs128" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_4_context5_2000_effbs128/eval/post_train/step_2000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_flappy_fix_latency_4_context5_2000_effbs128", + "module": "latency-transfer", + "train_latency_raw_frames": 4, + "train_seed": 42, + "training_step": 2000, + "training_started_at_utc": "2026-07-01T17:12:40.334458Z", + "git_commit": "972037fae54ada9fe8f64e4ac8878b2f79feba1f", + "wandb_run_id": "hvokzu73", + "checkpoint": "wan_oft_flappy_fix_latency_4_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt", + "checkpoint_identity": "05a1c777b84de4f2b66a2db1a13eb7b1aa326167eabb30ddf50bd149d6e68118", + "source_model_revision": "1ba5d2f489f03f9cb07a637ca680fa4f2f8e51a4", + "dataset": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_4_200ep_context5", + "source_repo": "latency-sensitive-bench/flappy_200ep_context5", + "source_revision": "19d800e5f7ec9d5311f508dbfe4a50463261616c", + "train_rows": 311588, + "train_episodes": 180, + "val_rows": 72000, + "val_episodes": 20, + "pairing_evidence": "Training local path, row count, trajectory count, action statistics and fixed raw-frame latency" + }, + "paper_consumers": [ + "fig12_wanoft_latency_grid" + ] + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "latency-aware/flappy/flappy_fix_latency_4_200ep_context5" + } + ] +} diff --git a/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/README.md b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..1525c3cbc1d6fd218302155affa4526fbaeacccb --- /dev/null +++ b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/README.md @@ -0,0 +1,7 @@ +# openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2 + +Inverted Pendulum · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-in-prompt/archive/inverted-pendulum) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/artifact.json b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..a5d979947798ce0cbc3bab972b53b5e2268041f6 --- /dev/null +++ b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/artifact.json @@ -0,0 +1,140 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-information-models", + "revision": "1d10c3297eec77e34c6e5fbe36c1c9f061badfef", + "path": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/eval/post_train/step_5000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/resolved_config.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/resolved_config.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/resolved_config.yaml" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2", + "source_revision": "177c11a86ee64432264ae8f1e577ecfd6c767918", + "target_repo": "latency-sensitive-bench/latency-information-models", + "target_run": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2", + "training_step": 5000, + "seed": 42, + "training_data_root": "/home/ubuntu/talha/playground/Datasets/rl_games", + "training_data_mix": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps" + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/inverted_pendulum_200ep_2", + "config_name": null, + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps", + "converted_name": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": true, + "skip_verification": true, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "max_steps_per_episode": 1000, + "episodes_per_latency": null, + "episodes_per_latency_by_latency": "0:90,2:1000,4:1000", + "latency_filter": [ + 0, + 2, + 4 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp2/checkpoints/steps_5000_pytorch_model.pt", + "identity": "28b6ca56206555b081f481247649ac06762debcb7cf74776b2ca240338d14974", + "source_path": "checkpoints/steps_5000_pytorch_model.pt", + "size": 9138435024 + } + ], + "paper_locations": [ + "paper/tables/table2_latency_information_ablation" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps" + } + } + ], + "training_data": [] +} diff --git a/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/README.md b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f9c691800de886b8394fb55ab681b8f2298253fb --- /dev/null +++ b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/README.md @@ -0,0 +1,7 @@ +# openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3 + +Inverted Pendulum · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-in-prompt/archive/inverted-pendulum) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/artifact.json b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..01d0ef5db0e4c68ea334e184ad998a5f66c03cb2 --- /dev/null +++ b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/artifact.json @@ -0,0 +1,140 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-information-models", + "revision": "1d10c3297eec77e34c6e5fbe36c1c9f061badfef", + "path": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/eval/post_train/step_5000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/resolved_config.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/resolved_config.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/resolved_config.yaml" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3", + "source_revision": "1967aac3fadf905a7f693e388cbf2aebdd40e5de", + "target_repo": "latency-sensitive-bench/latency-information-models", + "target_run": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3", + "training_step": 5000, + "seed": 42, + "training_data_root": "/home/ubuntu/talha/playground/Datasets/rl_games", + "training_data_mix": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps" + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/inverted_pendulum_200ep_2", + "config_name": null, + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps", + "converted_name": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": true, + "skip_verification": true, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "max_steps_per_episode": 1000, + "episodes_per_latency": null, + "episodes_per_latency_by_latency": "0:90,2:1000,4:1000", + "latency_filter": [ + 0, + 2, + 4 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_exp3/checkpoints/steps_5000_pytorch_model.pt", + "identity": "2f41824572db70b0517fffee698dfbd6a2274494eb6aecac82e6f91edf47a219", + "source_path": "checkpoints/steps_5000_pytorch_model.pt", + "size": 9138435024 + } + ], + "paper_locations": [ + "paper/tables/table2_latency_information_ablation" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps" + } + } + ], + "training_data": [] +} diff --git a/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/README.md b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..4d1e883fd1f34049c73a0efe7bdd4fbcc2a76d4a --- /dev/null +++ b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/README.md @@ -0,0 +1,7 @@ +# openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2 + +Inverted Pendulum · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-in-prompt/archive/inverted-pendulum) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/artifact.json b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..12100c03d3e3a12e8169797ad63bb58849d8e178 --- /dev/null +++ b/latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/artifact.json @@ -0,0 +1,140 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-information-models", + "revision": "1d10c3297eec77e34c6e5fbe36c1c9f061badfef", + "path": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "latency-aware/inverted-pendulum/vla/starvla-qwenoft-h1/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/eval/post_train/step_5000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_0/resolved_config.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_2/resolved_config.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/observation_attempts.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/inverted-pendulum/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/latency_bench_eval/post_train/step_5000/gymnasium/latency_4/resolved_config.yaml" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2", + "source_revision": "bd6274f20b7ff2f36c0b758058ef8ce2e996d7f6", + "target_repo": "latency-sensitive-bench/latency-information-models", + "target_run": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2", + "training_step": 5000, + "seed": 42, + "training_data_root": "/home/ubuntu/talha/playground/Datasets/rl_games", + "training_data_mix": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps" + }, + "original_dataset_config": { + "source_hf": "latency-sensitive-bench/inverted_pendulum_200ep_2", + "config_name": null, + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps", + "converted_name": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": true, + "skip_verification": true, + "target_latency_unit": "observation_steps", + "verify_rows": 200, + "max_episodes": null, + "max_steps_per_episode": 1000, + "episodes_per_latency": null, + "episodes_per_latency_by_latency": "0:90,2:1000,4:1000", + "latency_filter": [ + 0, + 2, + 4 + ], + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "openvla_inverted_pendulum_mixed_latency_024_og_l0_native_no_latency_prompt_exp2/checkpoints/steps_5000_pytorch_model.pt", + "identity": "d275016d512fe0fe071d5c2c9f7362fba939c89cb357e461d8b59e96c31ba62f", + "source_path": "checkpoints/steps_5000_pytorch_model.pt", + "size": 9138435024 + } + ], + "paper_locations": [ + "paper/tables/table2_latency_information_ablation" + ], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "source_subdir": "inverted_pendulum_fixed_latency_0_100ep_1ksteps_og,inverted_pendulum_fixed_latency_2_1000ep_7k2steps,inverted_pendulum_fixed_latency_4_1000ep_7k2steps" + } + } + ], + "training_data": [] +} diff --git a/reports/benchmark-release/README.md b/reports/benchmark-release/README.md new file mode 100644 index 0000000000000000000000000000000000000000..fbfd4282d355291192ed86d5a5c762ebb1dc81dc --- /dev/null +++ b/reports/benchmark-release/README.md @@ -0,0 +1,74 @@ +--- +license: other +--- +# Latency-Sensitive Bench models + +Inference-ready teachers and VLA policies for the supported benchmark tasks. + +## Layout + +- `zero-latency//small-policy/` and `zero-latency//vla/`: models trained without latency. +- `latency-aware//small-policy/` and `latency-aware//vla/`: models trained for latency. Profile and fixed-2 training conditions are identified by run ID and bundle provenance. + +MIKASA H8 conditioned inference bundles use `latency-aware/mikasa-intercept-grab-fast/vla/starvla--h8//`, where `` is `qwenoft`, `qwenpi_v3`, or `qwengr00t`. Each model has an SFT run `h8-conditioned-seed-reset-20260922-{profile|fixed-2}` and a DAgger run `h8-conditioned-dagger-20260923-{profile|fixed-2}`. + +## Hist8 VLA release: 18 models, 27 formal evaluations + +Flappy, Demon Attack and Deadly Corridor × QwenOFT, QwenGR00T and QwenPI v3 × zero/profile training. Each model is the evaluated step-5000 checkpoint, initialized from Qwen3-VL-4B revision `ebb281ec70b05090aa6165b016eac8ec08e71b17` with seed 42. + +Input is one current RGB image plus eight causal decision histories, passed as raw 0/1 transport state. Action horizon is **1**; the model directory suffix `h1` refers to this action horizon, while `hist8` in each run ID refers to input history. + +### Formal scores + +Each value is the mean of 100 episodes, seeds 1,000,000–1,000,099; parallel 32 and capacity 1. Flappy/Demon/Deadly caps are 3,600/7,200/3,600 raw frames at FPS/decision Hz 10/10, 60/15, 35/8.75. Deadly uses the corrected sf-render-v4 view (160×120 RGB with HUD). + +| Game / model | zero→zero | zero→profile | profile→profile | +|---|---:|---:|---:| +| flappy / qwenoft | 439.13 | 19.19 | 373.78 | +| flappy / qwengr00t | 428.48 | 6.62 | 414.52 | +| flappy / qwenpi_v3 | 408.72 | 5.58 | 344.48 | +| demon_attack / qwenoft | 2362.15 | 584.35 | 1452.80 | +| demon_attack / qwengr00t | 2370.90 | 231.90 | 1203.10 | +| demon_attack / qwenpi_v3 | 2350.95 | 112.30 | 907.30 | +| deadly_corridor / qwenoft | 2098.67 | 742.35 | 2091.23 | +| deadly_corridor / qwengr00t | 2100.46 | 341.76 | 2090.82 | +| deadly_corridor / qwenpi_v3 | 2104.09 | 140.94 | 1450.83 | + +### Downloadable models + +| Training | Game / model | Run | +|---|---|---| +| profile | flappy / qwenoft | [flappy-qwenoft-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-hist8-scratch-s42-v2) | +| profile | demon_attack / qwenoft | [demon_attack-qwenoft-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/demon-attack/vla/starvla-qwenoft-h1/demon_attack-qwenoft-hist8-scratch-s42-v2) | +| profile | flappy / qwengr00t | [flappy-qwengr00t-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/flappy/vla/starvla-qwengr00t-h1/flappy-qwengr00t-hist8-scratch-s42-v2) | +| profile | demon_attack / qwengr00t | [demon_attack-qwengr00t-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/demon-attack/vla/starvla-qwengr00t-h1/demon_attack-qwengr00t-hist8-scratch-s42-v2) | +| profile | demon_attack / qwenpi_v3 | [demon_attack-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/demon-attack/vla/starvla-qwenpi_v3-h1/demon_attack-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6) | +| zero | flappy / qwenoft | [flappy-qwenoft-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-hist8-zero-scratch-s42-v2) | +| profile | flappy / qwenpi_v3 | [flappy-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/flappy/vla/starvla-qwenpi_v3-h1/flappy-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6) | +| zero | demon_attack / qwenoft | [demon_attack-qwenoft-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack-qwenoft-hist8-zero-scratch-s42-v2) | +| zero | flappy / qwengr00t | [flappy-qwengr00t-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/flappy/vla/starvla-qwengr00t-h1/flappy-qwengr00t-hist8-zero-scratch-s42-v2) | +| zero | flappy / qwenpi_v3 | [flappy-qwenpi_v3-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/flappy/vla/starvla-qwenpi_v3-h1/flappy-qwenpi_v3-hist8-zero-scratch-s42-v2) | +| zero | deadly_corridor / qwenoft | [deadly_corridor-qwenoft-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-hist8-zero-scratch-s42-v2) | +| zero | demon_attack / qwengr00t | [demon_attack-qwengr00t-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/demon-attack/vla/starvla-qwengr00t-h1/demon_attack-qwengr00t-hist8-zero-scratch-s42-v2) | +| profile | deadly_corridor / qwenpi_v3 | [deadly_corridor-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/deadly-corridor/vla/starvla-qwenpi_v3-h1/deadly_corridor-qwenpi_v3-hist8-scratch-s42-v2-vlm3e-6) | +| profile | deadly_corridor / qwenoft | [deadly_corridor-qwenoft-hist8-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-hist8-scratch-s42-v2) | +| profile | deadly_corridor / qwengr00t | [deadly_corridor-qwengr00t-hist8-scratch-s42-v2-workers16](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/latency-aware/deadly-corridor/vla/starvla-qwengr00t-h1/deadly_corridor-qwengr00t-hist8-scratch-s42-v2-workers16) | +| zero | demon_attack / qwenpi_v3 | [demon_attack-qwenpi_v3-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/demon-attack/vla/starvla-qwenpi_v3-h1/demon_attack-qwenpi_v3-hist8-zero-scratch-s42-v2) | +| zero | deadly_corridor / qwengr00t | [deadly_corridor-qwengr00t-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/deadly-corridor/vla/starvla-qwengr00t-h1/deadly_corridor-qwengr00t-hist8-zero-scratch-s42-v2) | +| zero | deadly_corridor / qwenpi_v3 | [deadly_corridor-qwenpi_v3-hist8-zero-scratch-s42-v2](https://huggingface.co/latency-sensitive-bench/benchmark-models/tree/main/zero-latency/deadly-corridor/vla/starvla-qwenpi_v3-h1/deadly_corridor-qwenpi_v3-hist8-zero-scratch-s42-v2) | + +Every model contains its README, checkpoint, inference/training configuration, statistics, task contract, collection/training provenance, exact latency profile and episode-level evaluation evidence. The original detailed JSON index remains in the benchmark publication receipt. + +### Interpretation and reproduction + +The same zero-trained checkpoint is used in both evaluation environments. Each architecture has its own latency profile, so this is not a same-latency cross-architecture ranking. Zero and profile training use different teachers and collected data; differences do not isolate learning rate or a single training factor. Results use one training seed. + +Common training: two GPUs, bf16 ZeRO-2, batch 64/rank, accumulation 1, global batch 128, 16 workers/rank, prefetch 4, backbone LR 3e-6, action-head LR 1e-4, 100-step warmup, cosine min_lr_rate 1/30, 5,000 steps. Architecture-specific losses and other parameter groups are retained. + +Hist8 corrects the earlier -1/1 history input mismatch. Deadly also corrects an evaluation rendering mismatch; invalid older results are not included in this release. Profiles are preserved byte-for-byte with their sampling sidecars. Use each model task contract and pinned base-model revision when replaying. + +## Standard-Pipeline import: seven Gymnasium tasks + +AirRaid, Ant, HalfCheetah, Hopper, Humanoid, InvertedPendulum and Walker2d: 37 H1 VLA bundles (QwenOFT, QwenGR00T, QwenPI v3) and 28 Sample Factory APPO bundles. Each bundle contains one selected checkpoint. These are source-preserving copies with recorded experiment status; this import does not imply new evaluation or acceptance. + +[Model inventory and verification](reports/standard-pipeline-2208875f92b2/README.md). HalfCheetah lacks profile QwenGR00T/QwenPI v3 VLAs; Humanoid lacks all three profile VLAs. diff --git a/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/README.md b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..50c31f2a86d486a2760e64a1f193dce22224e010 --- /dev/null +++ b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/artifact.json b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..b5ec77b1d6e1d9bea38bb4dbac1afdcba37e7c87 --- /dev/null +++ b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,70 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/deadly-corridor/deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "0557aba023581368ee883695e9535217afb5c126d083aa825e12c797c7bd3ddf", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "deadly_corridor_fixed_latency_0_1000ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "deadly_corridor_fix_latency_0_1000ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "zero-latency/deadly-corridor/deadly_corridor_fixed_latency_0_1000ep_7k2steps", + "config_name": "deadly_corridor_fixed_latency_0_1000ep_7k2steps" + } + ] +} diff --git a/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/README.md b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92103c7c096d347c8aeb31d407b4dea06cd43d36 --- /dev/null +++ b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/README.md @@ -0,0 +1,7 @@ +# openvla_deadly_corridor_fix_latency_0 + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/artifact.json b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..bb0d9d961fe34092b36b1cb3e346c2ac10b6975b --- /dev/null +++ b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_deadly_corridor_fix_latency_0" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0/checkpoints/steps_500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_deadly_corridor_fix_latency_0", + "source_revision": "9f8c6842151405a82a3e852d2a00e27885db1e93", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_deadly_corridor_fix_latency_0", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_deadly_corridor_fix_latency_0", + "training_step": 500, + "seed": 42, + "training_data_root": "/inspire/hdd/global_user/liumingyu-253208120284/lzj/starvla/data/deadly_corridor_fix_latency_0", + "training_data_mix": "deadly_corridor_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "deadly_corridor_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "deadly_corridor_train", + "mixed_converted_name": "deadly_corridor_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_deadly_corridor_fix_latency_0/checkpoints/best_model.safetensors", + "identity": "1f2b7941653b956dbb5c368b9da60e5538e04216c34371254ef6cfa5c34bce74", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_deadly_corridor_fix_latency_0/checkpoints/steps_500_model.safetensors", + "identity": "b1478d01499d37121fee35d4f9f1dc090f6e87192797f50556388c94970c3fc1", + "source_path": "checkpoints/steps_500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/README.md b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/README.md new file mode 100644 index 0000000000000000000000000000000000000000..c9a09e8e2795e653946ac6a5125e319fcb6321d9 --- /dev/null +++ b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/README.md @@ -0,0 +1,7 @@ +# openvla_deadly_corridor_fix_latency_0_full + +Deadly Corridor · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/deadly-corridor) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/artifact.json b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..0a55b07be97490d37e73e20777af4c619eb8da5d --- /dev/null +++ b/zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/artifact.json @@ -0,0 +1,87 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_deadly_corridor_fix_latency_0_full" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/deadly-corridor/vla/starvla-qwenoft-h1/openvla_deadly_corridor_fix_latency_0_full/checkpoints/steps_2500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_deadly_corridor_fix_latency_0_full", + "source_revision": "9281fd061894c14d3641ee9724e108cd33baf494", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_deadly_corridor_fix_latency_0_full", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_deadly_corridor_fix_latency_0", + "training_step": 2500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/deadly_corridor_fix_latency_0", + "training_data_mix": "deadly_corridor_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "deadly_corridor_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "deadly_corridor_train", + "mixed_converted_name": "deadly_corridor_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_deadly_corridor_fix_latency_0_full/checkpoints/best_model.safetensors", + "identity": "5de650dbc0c50201489cb391a18f95b76bc1b9fb5a3df03ee965050b90284e84", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_deadly_corridor_fix_latency_0_full/checkpoints/steps_2500_model.safetensors", + "identity": "dfdb46c577b89bc4dcdef272e2b9e1d590ac0331c23632284bae3980d419bd5e", + "source_path": "checkpoints/steps_2500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/README.md b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..856313af532471cfffbc4ee5bf358e934e153ee1 --- /dev/null +++ b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_0_200ep + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/artifact.json b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..bd3ec608b4b8efac27abaca263ac6856367bc83f --- /dev/null +++ b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/artifact.json @@ -0,0 +1,73 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "demon_attack_fix_latency_0_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep/checkpoints/steps_7000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_200ep", + "source_revision": "04fece441aa5b6f5a6a015264a6aa5df3dc4c29a", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "demon_attack_fix_latency_0_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "demon_attack_fix_latency_0_200ep", + "training_step": 7000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_0_200ep", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "source_subdir": null, + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "skip_verification": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "demon_attack_fix_latency_0_200ep/checkpoints/steps_7000_model.safetensors", + "identity": "25a15d0460a63a7ca036e9300d39a4f57b8f9e656c6ff42941348b6a570bf767", + "source_path": "demon_attack_fix_latency_0_200ep/checkpoints/steps_7000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/demon_attack", + "paper/figures/fig6_latency_robustness/demon_attack" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/README.md b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a27664e25d568ae2c76a230b3317bdc5a0d92375 --- /dev/null +++ b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# demon_attack_fix_latency_0_200ep_7k2steps_single_baseline + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/artifact.json b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..783140a7a5fd285a3e816be02eb11086f881a8a7 --- /dev/null +++ b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,70 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/demon-attack/vla/starvla-qwenoft-h1/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/demon-attack/demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "4f7856234a1dca6d6c678270ec3958d2c603cee4f58f5f2fd9b138ba074bebcc", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "demon_attack_fixed_latency_0_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "demon_attack_fix_latency_0_200ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "zero-latency/demon-attack/demon_attack_fixed_latency_0_200ep_7k2steps", + "config_name": "demon_attack_fixed_latency_0_200ep_7k2steps" + } + ] +} diff --git a/zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/README.md b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..bb5b42e5cd6fde0d169500b9cf81032c96692ba2 --- /dev/null +++ b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/README.md @@ -0,0 +1,7 @@ +# openvla_demon_attack_fix_latency_0_small + +Demon Attack · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/demon-attack) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/artifact.json b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..91a3d5be522174d8c72a9ca214d1a1519f948379 --- /dev/null +++ b/zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/artifact.json @@ -0,0 +1,76 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_demon_attack_fix_latency_0_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/demon-attack/vla/starvla-qwenoft-h1/openvla_demon_attack_fix_latency_0_small/checkpoints/best_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_demon_attack_fix_latency_0_small", + "source_revision": "aeefbbd9676d95973aea2a6f63c433807a44d05b", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_demon_attack_fix_latency_0_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_demon_attack_fix_latency_0", + "training_step": 2500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/demon_attack_fix_latency_0", + "training_data_mix": "demon_attack_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "demon_attack_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "demon_attack_train", + "mixed_converted_name": "demon_attack_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_demon_attack_fix_latency_0_small/checkpoints/best_model.safetensors", + "identity": "0eda74a2340c7b2df580ca0589505c47ba32f7244b54a15878f61f86fa922213", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/README.md b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f7f7078a85741aa87a3a26a4deb1bf55ad9f91a8 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_0_200ep + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/artifact.json b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..958e01cb25d5c5bc3f3c214d5f1d89ef2ed2435f --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/artifact.json @@ -0,0 +1,71 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "flappy_fix_latency_0_200ep" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep/checkpoints/steps_5000_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_200ep", + "source_revision": "7918bf5c0aeb830a01352f5378acc6a4d4cf601c", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "flappy_fix_latency_0_200ep", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "flappy_fix_latency_0_200ep", + "training_step": 5000, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_0_200ep", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "config_name": null, + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "flappy_fix_latency_0_200ep/checkpoints/steps_5000_model.safetensors", + "identity": "7e22c0b5c6118ff5bac5536d076ef7c8a51cf14cfa8b86752464321ac7311d39", + "source_path": "flappy_fix_latency_0_200ep/checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [ + "paper/exps/openvla/flappy", + "paper/figures/fig6_latency_robustness/flappy" + ], + "paper_identity_confirmed": true, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/README.md b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7710f5a5bd573abe2cdf620c85edf5b36f0a35ab --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/README.md @@ -0,0 +1,7 @@ +# flappy_fix_latency_0_200ep_7k2steps_single_baseline + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/visual-history/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/artifact.json b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..7a30b5ad9a81f9fb3534b057d2f310cb03d81b69 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/artifact.json @@ -0,0 +1,410 @@ +{ + "source": { + "repo": "latency-sensitive-bench/memory-models", + "revision": "289306dcd0bd72502011d98e71b0648d149071bc", + "path": "flappy_fix_latency_0_200ep_7k2steps_single_baseline" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-qwenoft-h1/flappy_fix_latency_0_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_1000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_1250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_1500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_1750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_2000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_2250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_2500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_2750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_3000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_3250.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_3500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_3750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_500.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/mid_train/step_750.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/eval/post_train/step_4000.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/hydra/.hydra/hydra.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/hydra/.hydra/overrides.yaml" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/hydra/train_starvla_hydra.log" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1000/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1250/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1500/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_1750/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2000/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2250/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_250/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2500/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_2750/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3000/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3250/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3500/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_3750/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_4000/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_500/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/mid_train/step_750/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/_progress/rank_0.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/_progress/rank_1.json" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/latency_bench_eval/post_train/step_4000/flappy/latency_0/episode_metrics.jsonl" + }, + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/flappy_fix_latency_0_200ep_7k2steps_single_baseline/summary.jsonl" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "run": "flappy_fix_latency_0_200ep_7k2steps_single_baseline", + "framework": "QwenOFT", + "weight_path": "flappy_fix_latency_0_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "config_path": "flappy_fix_latency_0_200ep_7k2steps_single_baseline/config.full.yaml", + "normalization_path": "flappy_fix_latency_0_200ep_7k2steps_single_baseline/dataset_statistics.json", + "source_repo": "latency-sensitive-bench/memory", + "source_revision": "19560a283b9b2f0e94f4ee134e8913c7ac6ec441", + "source_weight_path": "flappy_fix_latency_0_200ep_7k2steps_single_baseline/checkpoints/steps_4000_model.safetensors", + "source_weight_sha256": "e6df7c64a760d025417d84c2bbc1dfe6e2567790a4c18bacd9e634f81c676eeb", + "training_data": { + "status": "source confirmed", + "repo": "latency-sensitive-bench/memory-data", + "revision": "49da56bd92b9842fb457418aba760a1b03379258", + "config_name": "flappy_fixed_latency_0_200ep_7k2steps" + }, + "image_mode": "single", + "num_obs_frames": 1, + "evaluation_repo": "latency-sensitive-bench/memory-data", + "evaluation_revision": "49da56bd92b9842fb457418aba760a1b03379258", + "evidence_directory": "flappy_fix_latency_0_200ep_7k2steps_single_baseline" + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "zero-latency/flappy/flappy_fixed_latency_0_200ep_7k2steps", + "config_name": "flappy_fixed_latency_0_200ep_7k2steps" + } + ] +} diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/README.md b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/README.md new file mode 100644 index 0000000000000000000000000000000000000000..b76e6537af89c38703c9bc368d2c9ed85688b548 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_0 + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/artifact.json b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..94e1d1c1d872ae16a8db98ff944a57ee0f6cbc90 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/artifact.json @@ -0,0 +1,77 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_0" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/checkpoints/steps_5000_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0/checkpoints/steps_5000_pytorch_model.pt" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_0", + "source_revision": "06f64d1a03905ad0f04a447c57afb970c29a559e", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_0", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_0", + "training_step": 5000, + "seed": 42, + "training_data_root": "/inspire/hdd/global_user/liumingyu-253208120284/lzj/starvla/data/flappy_fix_latency_0_parquet", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": null, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_0/checkpoints/steps_5000_pytorch_model.pt", + "identity": "a7581596728be2fb496d24bea5ec1c8586a5fcbf8ee99f1e9d5c90f11288f70b", + "source_path": "checkpoints/steps_5000_pytorch_model.pt", + "size": 9138476289 + }, + { + "path": "historical/openvla_flappy_fix_latency_0/checkpoints/steps_5000_model.safetensors", + "identity": "57b21a1d51a95495f08c60166b561aed0183fd4b680b665a440b9b0694aff4c8", + "source_path": "checkpoints/steps_5000_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "unpublished_training_data": "The recorded local training directory has no recovered exact HF source. No replacement dataset is assumed." + } + ], + "training_data": [] +} diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/README.md b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/README.md new file mode 100644 index 0000000000000000000000000000000000000000..dd1575b3505703fd427b0a40e3d0f969f98230a4 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/README.md @@ -0,0 +1,7 @@ +# openvla_flappy_fix_latency_0_small + +Flappy Bird · qwenoft · H1 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/archive/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/artifact.json b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..a067992af502bf1b3e65f42c275d3ec14f6fe435 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/artifact.json @@ -0,0 +1,88 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "historical/openvla_flappy_fix_latency_0_small" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/checkpoints/best_model.safetensors" + }, + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-qwenoft-h1/openvla_flappy_fix_latency_0_small/checkpoints/steps_1500_model.safetensors" + } + ], + "evaluation": [], + "asset_state": "complete_bundle", + "paper_source_confirmed": false, + "original_training_bindings": [ + { + "source_repo": "latency-sensitive-bench/openvla_flappy_fix_latency_0_small", + "source_revision": "05586f38706727534a0f388ae0ece46541776d65", + "target_repo": "latency-sensitive-bench/latency-transfer-models", + "target_run": "historical/openvla_flappy_fix_latency_0_small", + "artifact_state": "complete_checkpoint_bundle", + "framework": "QwenOFT", + "run_id": "openvla_flappy_fix_latency_0_small", + "training_step": 1500, + "seed": 42, + "training_data_root": "/workspace/starVLA/data/flappy_fix_latency_0_small", + "training_data_mix": "flappy_train__bridge", + "dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + }, + "original_dataset_config": { + "source_hf": "", + "converted_name": "flappy_train", + "single_source_hf": "", + "mixed_source_hf": "", + "single_converted_name": "flappy_train", + "mixed_converted_name": "flappy_mixed_latency_train", + "single_latency_filter": null, + "mixed_latency_filter": null, + "force_download": false, + "setup_force": false, + "verify_rows": 200, + "max_episodes": null, + "episodes_per_latency": null, + "latency_filter": null, + "debug_subset": { + "enabled": false, + "max_episodes": 5, + "suffix": "debug" + } + }, + "weights": [ + { + "path": "historical/openvla_flappy_fix_latency_0_small/checkpoints/best_model.safetensors", + "identity": "9117254092707a50375a50e48626afdfbc3e7ac1ea51320a09ec901cd53f0e37", + "source_path": "checkpoints/best_state/model.safetensors", + "size": 9138230516 + }, + { + "path": "historical/openvla_flappy_fix_latency_0_small/checkpoints/steps_1500_model.safetensors", + "identity": "bd2796b952df7654f0376b226923307644f24d90232731083919241d2a269f62", + "source_path": "checkpoints/steps_1500_state/model.safetensors", + "size": 9138230516 + } + ], + "paper_locations": [], + "paper_identity_confirmed": false, + "interpretation": "The paper experiment type is known; an exact plotted checkpoint binding is recorded only when its immutable source manifest establishes it.", + "training_recipe_dataset_reference": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "", + "evidence": "Training config uses the exact source dataset name as its local data-root directory. Original metadata and statistics are retained separately." + } + } + ], + "training_data": [] +} diff --git a/zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/README.md b/zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..aacecc0d4b7e42644bee0c38534acb4a88a16471 --- /dev/null +++ b/zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/README.md @@ -0,0 +1,7 @@ +# wan_oft_flappy_fix_latency_0_context5_2000_effbs128 + +Flappy Bird · wanoft · H8 + +[Experiment settings and asset manifest](https://huggingface.co/datasets/latency-sensitive-bench/benchmark-datasets/tree/main/experiments/latency-transfer/flappy) + +Download this run at the fixed revision in the task manifest. Keep the checkpoint, original configuration and normalization statistics together. Original local paths describe source provenance. `artifact.json` records the current asset references. diff --git a/zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/artifact.json b/zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/artifact.json new file mode 100644 index 0000000000000000000000000000000000000000..c0d0120a39654927e5d9e7b35421047cd1b4c38e --- /dev/null +++ b/zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/artifact.json @@ -0,0 +1,60 @@ +{ + "source": { + "repo": "latency-sensitive-bench/latency-transfer-models", + "revision": "857ab8492f7514a420397a9644fcc8e0272fe6b2", + "path": "wan_oft_flappy_fix_latency_0_context5_2000_effbs128" + }, + "checkpoint": [ + { + "repo": "latency-sensitive-bench/benchmark-models", + "revision": "a4099f759f71ffa72d967be3695c8e054cfc377c", + "path": "zero-latency/flappy/vla/starvla-wanoft-h8/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt" + } + ], + "evaluation": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "evaluations/flappy/wan_oft_flappy_fix_latency_0_context5_2000_effbs128/eval/post_train/step_2000.json" + } + ], + "asset_state": "complete_bundle", + "paper_source_confirmed": true, + "original_training_bindings": [ + { + "run": "wan_oft_flappy_fix_latency_0_context5_2000_effbs128", + "module": "latency-transfer", + "train_latency_raw_frames": 0, + "train_seed": 42, + "training_step": 2000, + "training_started_at_utc": "2026-06-28T13:44:26.129154Z", + "git_commit": "f9acb69ef3d99d2a8b620103161e15eebc4c3091", + "wandb_run_id": "ky48y0be", + "checkpoint": "wan_oft_flappy_fix_latency_0_context5_2000_effbs128/checkpoints/steps_2000_pytorch_model.pt", + "checkpoint_identity": "daa8a61baa149139792808dcfc5406df766d39113161b92d0baa13a9bd2402f4", + "source_model_revision": "1ba5d2f489f03f9cb07a637ca680fa4f2f8e51a4", + "dataset": { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "zero-latency/flappy/flappy_fix_latency_0_200ep_context5", + "source_repo": "latency-sensitive-bench/flappy_200ep_context5", + "source_revision": "19d800e5f7ec9d5311f508dbfe4a50463261616c", + "train_rows": 279882, + "train_episodes": 180, + "val_rows": 72000, + "val_episodes": 20, + "pairing_evidence": "Training local path, row count, trajectory count, action statistics and fixed raw-frame latency" + }, + "paper_consumers": [ + "fig12_wanoft_latency_grid" + ] + } + ], + "training_data": [ + { + "repo": "latency-sensitive-bench/benchmark-datasets", + "revision": "15d17287246bd19940c837d6c8a5327564f969c5", + "path": "zero-latency/flappy/flappy_fix_latency_0_200ep_context5" + } + ] +}