CAMP RMBench policies (inference weights only)
Browse files- .gitattributes +18 -0
- README.md +232 -37
- cover_blocks/memory/best_model.pt +3 -0
- cover_blocks/memory/normalizer.pt +3 -0
- cover_blocks/policy.ckpt +3 -0
- observe_and_pickup/memory/best_model.pt +3 -0
- observe_and_pickup/memory/normalizer.pt +3 -0
- observe_and_pickup/policy.ckpt +3 -0
- press_button/memory/best_model.pt +3 -0
- press_button/memory/normalizer.pt +3 -0
- press_button/policy.ckpt +3 -0
- previews/battery_try.gif +3 -0
- previews/battery_try.mp4 +3 -0
- previews/blocks_ranking_try.gif +3 -0
- previews/blocks_ranking_try.mp4 +3 -0
- previews/cover_blocks.gif +3 -0
- previews/cover_blocks.mp4 +3 -0
- previews/observe_and_pickup.gif +3 -0
- previews/observe_and_pickup.mp4 +3 -0
- previews/press_button.gif +3 -0
- previews/press_button.mp4 +3 -0
- previews/put_back_block.gif +3 -0
- previews/put_back_block.mp4 +3 -0
- previews/rearrange_blocks.gif +3 -0
- previews/rearrange_blocks.mp4 +3 -0
- previews/swap_T.gif +3 -0
- previews/swap_T.mp4 +3 -0
- previews/swap_blocks.gif +3 -0
- previews/swap_blocks.mp4 +3 -0
- swap_T/memory/best_model.pt +3 -0
- swap_T/memory/normalizer.pt +3 -0
- swap_T/policy.ckpt +3 -0
- swap_blocks/memory/best_model.pt +3 -0
- swap_blocks/memory/normalizer.pt +3 -0
- swap_blocks/policy.ckpt +3 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,21 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
previews/battery_try.gif filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
previews/battery_try.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
previews/blocks_ranking_try.gif filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
previews/blocks_ranking_try.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
previews/cover_blocks.gif filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
previews/cover_blocks.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
previews/observe_and_pickup.gif filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
previews/observe_and_pickup.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
previews/press_button.gif filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
previews/press_button.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
previews/put_back_block.gif filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
previews/put_back_block.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
previews/rearrange_blocks.gif filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
previews/rearrange_blocks.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
previews/swap_T.gif filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
previews/swap_T.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
previews/swap_blocks.gif filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
previews/swap_blocks.mp4 filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -1,64 +1,259 @@
|
|
| 1 |
---
|
| 2 |
license: mit
|
|
|
|
| 3 |
tags:
|
| 4 |
- robotics
|
| 5 |
- imitation-learning
|
| 6 |
- diffusion-policy
|
| 7 |
- memory
|
|
|
|
| 8 |
- rmbench
|
| 9 |
- robotwin
|
| 10 |
---
|
| 11 |
|
| 12 |
-
|
| 13 |
|
| 14 |
-
|
| 15 |
-
([CAMP](https://robo-camp.github.io/), code: https://github.com/ucsdarclab/CAMP), trained on the
|
| 16 |
-
[RMBench](https://github.com/robotwin-Platform/rmbench) (RoboTwin 2.0, Aloha-AgileX) benchmark from the 50 released
|
| 17 |
-
demonstrations per task. Evaluated with RMBench's own protocol (`demo_clean`, seeds from 100000 validated by the
|
| 18 |
-
scripted expert, per-task step limits, 100 episodes).
|
| 19 |
|
| 20 |
-
|
| 21 |
-
|---|---|---|
|
| 22 |
-
| rearrange_blocks | 100 / 100 | `rearrange_blocks/` |
|
| 23 |
-
| blocks_ranking_try | 100 / 100 | `blocks_ranking_try/` |
|
| 24 |
-
| put_back_block | 100 / 100 | `put_back_block/` |
|
| 25 |
-
| battery_try | 97 / 100 | `battery_try/` |
|
| 26 |
|
| 27 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
|
| 29 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
|
|
|
| 35 |
|
| 36 |
-
|
|
|
|
|
|
|
|
|
|
| 37 |
|
| 38 |
-
|
| 39 |
-
96x128 + 14-D joint state, hidden 128, action subsampling 4.
|
| 40 |
-
- Stage 2: Diffusion Policy (head camera 240x320, 14-D joint targets, `n_obs_steps=1`, 8-step action chunks) conditioned on the
|
| 41 |
-
memory through a 32-D projection; memory frozen for 400 epochs, then jointly finetuned (200 epochs; put_back_block 600).
|
| 42 |
-
The checkpoint reported per task is the best one over evaluated epochs.
|
| 43 |
|
| 44 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
|
| 46 |
-
```bash
|
| 47 |
-
# inside the CAMP + RoboTwin evaluation image (see scripts/rmbench/eval in the CAMP repo)
|
| 48 |
-
python scripts/rmbench/eval/rmbench_eval.py eval --task rearrange_blocks --ckpt policy --episodes 100 \
|
| 49 |
-
--ckpt_root <this repo>/stage2 --stage1_root <this repo>/stage1
|
| 50 |
```
|
| 51 |
-
|
| 52 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 53 |
`get_model / eval / reset_model` interface.
|
| 54 |
|
| 55 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
|
| 57 |
```bibtex
|
| 58 |
-
@article{
|
| 59 |
-
title
|
| 60 |
-
author
|
| 61 |
-
journal
|
| 62 |
-
year
|
| 63 |
}
|
| 64 |
```
|
|
|
|
| 1 |
---
|
| 2 |
license: mit
|
| 3 |
+
pipeline_tag: robotics
|
| 4 |
tags:
|
| 5 |
- robotics
|
| 6 |
- imitation-learning
|
| 7 |
- diffusion-policy
|
| 8 |
- memory
|
| 9 |
+
- bimanual-manipulation
|
| 10 |
- rmbench
|
| 11 |
- robotwin
|
| 12 |
---
|
| 13 |
|
| 14 |
+
<h1 align="center">CAMP on RMBench</h1>
|
| 15 |
|
| 16 |
+
<p align="center"><b>Remember what you did?<br>Learning Behavioral Memories for Partially Observable Object Manipulation</b></p>
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
|
| 18 |
+
<p align="center">Kuancheng Wang, Seungho Yeom, Jinglin Cao, Yuheng Zhi, Nikhil Shinde, Michael Yip</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
|
| 20 |
+
<p align="center">
|
| 21 |
+
<a href="https://arxiv.org/abs/2606.21188"><img alt="arXiv" src="https://img.shields.io/badge/arXiv-2606.21188-b31b1b?logo=arxiv&logoColor=white"></a>
|
| 22 |
+
|
| 23 |
+
<a href="https://github.com/KuanchengWang/CAMP"><img alt="Code" src="https://img.shields.io/badge/GitHub-CAMP-181717?logo=github&logoColor=white"></a>
|
| 24 |
+
|
| 25 |
+
<a href="https://robo-camp.github.io/"><img alt="Project page" src="https://img.shields.io/badge/Project-Page-2f6fe4?logo=googlechrome&logoColor=white"></a>
|
| 26 |
+
|
| 27 |
+
<a href="https://github.com/RoboTwin-Platform/RMBench"><img alt="RMBench" src="https://img.shields.io/badge/Benchmark-RMBench-6a3fb5"></a>
|
| 28 |
+
</p>
|
| 29 |
|
| 30 |
+
<p align="center">
|
| 31 |
+
<img src="previews/rearrange_blocks.gif" width="32%">
|
| 32 |
+
<img src="previews/put_back_block.gif" width="32%">
|
| 33 |
+
<img src="previews/battery_try.gif" width="32%">
|
| 34 |
+
</p>
|
| 35 |
|
| 36 |
+
**CAMP** (Compressed Action Memory Policy) gives a visuomotor policy a *behavioral memory*. A recurrent memory
|
| 37 |
+
is pretrained to reconstruct a compressed (DCT) summary of the robot's own past actions, so its hidden state
|
| 38 |
+
has to encode what the robot already did; a Diffusion Policy is then conditioned on that state. This lets the
|
| 39 |
+
policy track task progress and learn from its own failed attempts, which a memoryless policy cannot do when the
|
| 40 |
+
current image does not determine the next action.
|
| 41 |
|
| 42 |
+
This repository holds the CAMP policies for all nine tasks of **[RMBench](https://github.com/RoboTwin-Platform/RMBench)**,
|
| 43 |
+
a memory-dependent bimanual manipulation benchmark built on RoboTwin 2.0 (Aloha-AgileX dual-arm robot). Every
|
| 44 |
+
policy is trained from the 50 demonstrations RMBench releases per task and evaluated with RMBench's own protocol.
|
| 45 |
+
The files contain inference weights only.
|
| 46 |
|
| 47 |
+
## 📊 Results
|
|
|
|
|
|
|
|
|
|
|
|
|
| 48 |
|
| 49 |
+
Success rate over **100 test episodes** per task. We follow RMBench's evaluation protocol: the `demo_clean`
|
| 50 |
+
configuration, test seeds from 100000 that the scripted expert can solve, and the per-task step limit.
|
| 51 |
+
|
| 52 |
+
| Task | Memory needed for | Step limit | Success |
|
| 53 |
+
|:--|:--|:-:|:-:|
|
| 54 |
+
| [`rearrange_blocks`](#rearrange_blocks) | task progress | 700 | **100%** |
|
| 55 |
+
| [`blocks_ranking_try`](#blocks_ranking_try) | learning from failure | 3500 | **100%** |
|
| 56 |
+
| [`put_back_block`](#put_back_block) | task progress | 500 | **100%** |
|
| 57 |
+
| [`battery_try`](#battery_try) | learning from failure | 1000 | **97%** |
|
| 58 |
+
| [`swap_T`](#swap_t) | task progress | 600 | 24% |
|
| 59 |
+
| [`swap_blocks`](#swap_blocks) | task progress | 1000 | 19% |
|
| 60 |
+
| [`cover_blocks`](#cover_blocks) | task progress | 1500 | 17% |
|
| 61 |
+
| [`observe_and_pickup`](#observe_and_pickup) | a past observation | 250 | 9% |
|
| 62 |
+
| [`press_button`](#press_button) | counting | 1500 | 5% |
|
| 63 |
+
|
| 64 |
+
## 🎬 Tasks
|
| 65 |
+
|
| 66 |
+
Each preview is a **demonstration by RMBench's scripted expert**: the first of the 50 released demonstrations of
|
| 67 |
+
the task, re-rendered so each task and its success condition can be seen clearly. It is filmed from the head
|
| 68 |
+
camera the policy is trained on, with the same pose and field of view, at 1920×1440 instead of 320×240. Every
|
| 69 |
+
preview passes the task's success check. Long demos are sped up to at most about 24 s. The `observe_and_pickup`
|
| 70 |
+
preview holds the first frame and plays at half speed, because the target is visible for only that one frame.
|
| 71 |
+
The full-resolution MP4 files are in [`previews/`](previews). Success rates in the table above come only from the
|
| 72 |
+
100-episode policy evaluations.
|
| 73 |
+
|
| 74 |
+
<a id="rearrange_blocks"></a>
|
| 75 |
+
### rearrange_blocks: 100%
|
| 76 |
+
|
| 77 |
+
<img src="previews/rearrange_blocks.gif" width="70%">
|
| 78 |
+
|
| 79 |
+
Two blocks sit on mats next to a button, and one mat is empty. The robot moves the first block onto the empty
|
| 80 |
+
mat and presses the button. It then moves the second block off its mat to the spot between the mats.
|
| 81 |
+
|
| 82 |
+
- **Memory:** whether the button has already been pressed. The scene looks the same before and after the press.
|
| 83 |
+
- **Success:** the first block is within 3 cm of the target mat and the second block within 3 cm of the spot
|
| 84 |
+
between the mats. The button has been pressed exactly once and the gripper is open.
|
| 85 |
+
|
| 86 |
+
<a id="blocks_ranking_try"></a>
|
| 87 |
+
### blocks_ranking_try: 100%
|
| 88 |
+
|
| 89 |
+
<img src="previews/blocks_ranking_try.gif" width="70%">
|
| 90 |
+
|
| 91 |
+
Three colored cubes stand in a row in a random order, next to a check button. The robot does not know the target
|
| 92 |
+
order. It presses the button to test the current arrangement. If the arrangement is rejected, it swaps two cubes
|
| 93 |
+
and tests again, working through the orders until the button accepts one.
|
| 94 |
+
|
| 95 |
+
- **Memory:** which arrangements have already been tried and rejected. Repeating a rejected arrangement never
|
| 96 |
+
succeeds, and the scene does not show the history.
|
| 97 |
+
- **Success:** the three cubes stand next to each other in the correct left-to-right order and the button has
|
| 98 |
+
been pressed.
|
| 99 |
+
|
| 100 |
+
<a id="put_back_block"></a>
|
| 101 |
+
### put_back_block: 100%
|
| 102 |
+
|
| 103 |
+
<img src="previews/put_back_block.gif" width="70%">
|
| 104 |
+
|
| 105 |
+
A block starts on a mat. The robot moves the block to the center of the table and presses the button. It then
|
| 106 |
+
puts the block back on the mat it came from.
|
| 107 |
+
|
| 108 |
+
- **Memory:** the mat the block started on. Once the block is in the center, the image no longer shows where
|
| 109 |
+
it came from.
|
| 110 |
+
- **Success:** the button has been pressed once with the block in the center. The block then rests within
|
| 111 |
+
3 cm of its original mat and the gripper is open.
|
| 112 |
+
|
| 113 |
+
<a id="battery_try"></a>
|
| 114 |
+
### battery_try: 97%
|
| 115 |
+
|
| 116 |
+
<img src="previews/battery_try.gif" width="70%">
|
| 117 |
+
|
| 118 |
+
Two batteries must go into a slot whose correct polarity is hidden. The dashboard needle shows whether the current
|
| 119 |
+
combination is correct. If it is not, the robot takes a battery out and re-inserts it the other way round.
|
| 120 |
+
|
| 121 |
+
- **Memory:** which orientations have already been tried.
|
| 122 |
+
- **Success:** both batteries are seated in the slot in the correct orientation and the dashboard turns on.
|
| 123 |
+
|
| 124 |
+
<a id="swap_t"></a>
|
| 125 |
+
### swap_T: 24%
|
| 126 |
+
|
| 127 |
+
<img src="previews/swap_T.gif" width="70%">
|
| 128 |
+
|
| 129 |
+
Two T-shaped blocks lie on the table. The robot picks them up and places each one at the other's initial position
|
| 130 |
+
and orientation.
|
| 131 |
+
|
| 132 |
+
- **Memory:** both initial poses. Once the robot moves a block, its original pose is no longer visible.
|
| 133 |
+
- **Success:** each block is within 2.5 cm and 15° of the other block's initial pose, both are resting on the
|
| 134 |
+
table, and both grippers are open.
|
| 135 |
+
|
| 136 |
+
<a id="swap_blocks"></a>
|
| 137 |
+
### swap_blocks: 19%
|
| 138 |
+
|
| 139 |
+
<img src="previews/swap_blocks.gif" width="70%">
|
| 140 |
+
|
| 141 |
+
Two blocks are in two of three trays. The robot may move one block at a time and each tray holds at most one
|
| 142 |
+
block. It swaps the two blocks using the spare tray as a buffer, then presses the button.
|
| 143 |
+
|
| 144 |
+
- **Memory:** where each block started and which step of the three-move swap comes next.
|
| 145 |
+
- **Success:** each block is inside the tray the other block started in, the button has been pressed once
|
| 146 |
+
and the gripper is open.
|
| 147 |
+
|
| 148 |
+
<a id="cover_blocks"></a>
|
| 149 |
+
### cover_blocks: 17%
|
| 150 |
+
|
| 151 |
+
<img src="previews/cover_blocks.gif" width="70%">
|
| 152 |
+
|
| 153 |
+
A red, a green and a blue block are arranged randomly together with three identical lids. The robot covers the
|
| 154 |
+
blocks from left to right. Then it lifts the lids again in the order red, green, blue.
|
| 155 |
+
|
| 156 |
+
- **Memory:** which block is under which lid. The lids are identical, so the colors are hidden once covered.
|
| 157 |
+
- **Success:** the covering and uncovering sequence matches the required order exactly, with no wrong lid
|
| 158 |
+
lifted.
|
| 159 |
+
|
| 160 |
+
<a id="observe_and_pickup"></a>
|
| 161 |
+
### observe_and_pickup: 9%
|
| 162 |
+
|
| 163 |
+
<img src="previews/observe_and_pickup.gif" width="70%">
|
| 164 |
+
|
| 165 |
+
A target object is shown on a shelf. A wall then drops in front of the shelf and hides it. The robot must pick
|
| 166 |
+
up the matching object from several distractors on the table.
|
| 167 |
+
|
| 168 |
+
- **Memory:** the target object's identity. During the evaluation it is visible only in the first frame,
|
| 169 |
+
before the robot moves.
|
| 170 |
+
- **Success:** the arms stay still while the target is shown, and the correct object is then lifted off the
|
| 171 |
+
table.
|
| 172 |
+
- **Note:** this is the only task whose memory uses every frame (action subsampling 1 instead of 4). Otherwise
|
| 173 |
+
the single frame that shows the target would be skipped.
|
| 174 |
+
|
| 175 |
+
<a id="press_button"></a>
|
| 176 |
+
### press_button: 5%
|
| 177 |
+
|
| 178 |
+
<img src="previews/press_button.gif" width="70%">
|
| 179 |
+
|
| 180 |
+
Two number cards lie on the table. The robot presses the left button as many times as the left card shows and
|
| 181 |
+
the middle button as many times as the right card shows, then presses the right button to confirm.
|
| 182 |
+
|
| 183 |
+
- **Memory:** how many presses each button has received so far. A button looks the same after every press.
|
| 184 |
+
- **Success:** both press counts match the cards exactly and the confirm button has been pressed.
|
| 185 |
+
|
| 186 |
+
## 📦 Files
|
| 187 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 188 |
```
|
| 189 |
+
<task>/
|
| 190 |
+
├── policy.ckpt # CAMP policy: EMA weights of the memory-conditioned Diffusion Policy + resolved config
|
| 191 |
+
└── memory/
|
| 192 |
+
├── best_model.pt # Stage-1 behavioral-memory LSTM (weights + architecture args)
|
| 193 |
+
└── normalizer.pt # its input normalizer
|
| 194 |
+
previews/<task>.gif | .mp4 # expert demonstrations shown above (1920×1440 MP4)
|
| 195 |
+
```
|
| 196 |
+
|
| 197 |
+
The checkpoints hold only what inference needs. There is no optimizer or scheduler state and no training
|
| 198 |
+
bookkeeping.
|
| 199 |
+
|
| 200 |
+
## 🧠 Training recipe
|
| 201 |
+
|
| 202 |
+
| | |
|
| 203 |
+
|:--|:--|
|
| 204 |
+
| Data | the 50 RMBench demonstrations per task (`demo_clean`) |
|
| 205 |
+
| Observation | head camera 240×320 (random crop 216×288) and the 14-D joint state; `n_obs_steps = 1` |
|
| 206 |
+
| Action | 14-D absolute joint targets in 8-step chunks, normalized to a per-joint range |
|
| 207 |
+
| Stage 1: memory | LSTM (hidden 128) pretrained to reconstruct the DCT coefficients of its past actions; action subsampling 4 (1 for `observe_and_pickup`) |
|
| 208 |
+
| Stage 2: policy | Diffusion Policy conditioned on the memory through a 32-D projection. The memory is frozen for 400 epochs, then memory and policy are finetuned jointly (200 epochs; 600 for `put_back_block`) |
|
| 209 |
+
| Augmentation | joint noise 0.01, image noise 0.02, brightness and contrast jitter 0.15 |
|
| 210 |
+
| Checkpoint | the best of the evaluated epochs per task (every 100 epochs) |
|
| 211 |
+
|
| 212 |
+
## 🚀 Usage
|
| 213 |
+
|
| 214 |
+
The evaluation code is in [`scripts/rmbench/eval`](https://github.com/KuanchengWang/CAMP) of the CAMP repository.
|
| 215 |
+
It includes a Docker launcher for RMBench and the `policy_CAMP` adapter, which implements RMBench's
|
| 216 |
`get_model / eval / reset_model` interface.
|
| 217 |
|
| 218 |
+
```bash
|
| 219 |
+
# 1. download the checkpoints
|
| 220 |
+
huggingface-cli download harrywang01/CAMP-RMBench-Checkpoints --local-dir camp_rmbench
|
| 221 |
+
|
| 222 |
+
# 2. arrange one task in the layout the evaluator expects
|
| 223 |
+
T=swap_T
|
| 224 |
+
mkdir -p ckpts/stage2/$T/checkpoints ckpts/stage1/$T
|
| 225 |
+
cp camp_rmbench/$T/policy.ckpt ckpts/stage2/$T/checkpoints/policy.ckpt
|
| 226 |
+
cp camp_rmbench/$T/memory/* ckpts/stage1/$T/
|
| 227 |
+
|
| 228 |
+
# 3. evaluate on 100 RMBench test seeds (run from the CAMP repository)
|
| 229 |
+
STAGE1=$PWD/ckpts/stage1 STAGE2=$PWD/ckpts/stage2 \
|
| 230 |
+
scripts/rmbench/eval/docker_rmbench.sh python /workspace/scripts/rmbench/eval/rmbench_eval.py \
|
| 231 |
+
eval --task $T --ckpt policy --episodes 100
|
| 232 |
+
```
|
| 233 |
+
|
| 234 |
+
## 📝 Citation
|
| 235 |
+
|
| 236 |
+
If you find CAMP useful, please cite:
|
| 237 |
+
|
| 238 |
+
```bibtex
|
| 239 |
+
@misc{wang2026rememberdidlearningbehavioral,
|
| 240 |
+
title={Remember what you did?: Learning Behavioral Memories for Partially Observable Object Manipulation},
|
| 241 |
+
author={Kuancheng Wang and Seungho Yeom and Jinglin Cao and Yuheng Zhi and Nikhil Shinde and Michael Yip},
|
| 242 |
+
year={2026},
|
| 243 |
+
eprint={2606.21188},
|
| 244 |
+
archivePrefix={arXiv},
|
| 245 |
+
primaryClass={cs.RO},
|
| 246 |
+
url={https://arxiv.org/abs/2606.21188},
|
| 247 |
+
}
|
| 248 |
+
```
|
| 249 |
+
|
| 250 |
+
The tasks, demonstrations and evaluation protocol come from RMBench:
|
| 251 |
|
| 252 |
```bibtex
|
| 253 |
+
@article{chen2026rmbench,
|
| 254 |
+
title={RMBench: Memory-Dependent Robotic Manipulation Benchmark with Insights into Policy Design},
|
| 255 |
+
author={Chen, Tianxing and Wang, Yuran and Li, Mingleyang and Qin, Yan and Shi, Hao and Li, Zixuan and Hu, Yifan and Zhang, Yingsheng and Wang, Kaixuan and Chen, Yue and others},
|
| 256 |
+
journal={arXiv preprint arXiv:2603.01229},
|
| 257 |
+
year={2026}
|
| 258 |
}
|
| 259 |
```
|
cover_blocks/memory/best_model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:310eefdf1527761cdc5d0d17a885523978997822ff12940de62b952bf9173672
|
| 3 |
+
size 48734773
|
cover_blocks/memory/normalizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cd9b1c7ea6f50c407f6fb6766628b54d6c9dc095fffa5eea81e814df37da28a5
|
| 3 |
+
size 8309
|
cover_blocks/policy.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fbf68fe509fffb08d26603ceaa9565794fb568acb0635f038ce75181456d9b51
|
| 3 |
+
size 1097430247
|
observe_and_pickup/memory/best_model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:28e9d0cbde5261dcb5db9843eb48378604792760f3b9c1c668a7afbd1b161560
|
| 3 |
+
size 48729653
|
observe_and_pickup/memory/normalizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bd157c67c3dfa5b7c37279ff1c42ca2556b6af41bbc0b281461a3dddf33478d9
|
| 3 |
+
size 8309
|
observe_and_pickup/policy.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cdaa20569b783a534020493c924dfe63ae9fd4586f3b16b05242a38e5c060553
|
| 3 |
+
size 1097425127
|
press_button/memory/best_model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cdad186e5d9f671d57e4507917adc0fc2047f9cb47c3bcd92d9a0d353844304f
|
| 3 |
+
size 48720693
|
press_button/memory/normalizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:73c6a817fc6bd6e6179f20db4e149ae34532eba65ccc37c33edcf163a440f58e
|
| 3 |
+
size 8309
|
press_button/policy.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:70c7559de1fc188835c314a92c2a1818cc11e9cbe44d115fbd5e26e8ebb793fd
|
| 3 |
+
size 1097416167
|
previews/battery_try.gif
ADDED
|
Git LFS Details
|
previews/battery_try.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f3dc5915063b0d23c335b4278381f3faa454d30f53a9c3a024338bea98665487
|
| 3 |
+
size 28824307
|
previews/blocks_ranking_try.gif
ADDED
|
Git LFS Details
|
previews/blocks_ranking_try.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cbb123f6f46e8c57e7a32b705a0c0841b42951baf08ac25749aef85d784ebc35
|
| 3 |
+
size 38522031
|
previews/cover_blocks.gif
ADDED
|
Git LFS Details
|
previews/cover_blocks.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:01ee1e2729098e71aecfdb555322fb53e89d66534f88ca6af1466bc26b127718
|
| 3 |
+
size 36582415
|
previews/observe_and_pickup.gif
ADDED
|
Git LFS Details
|
previews/observe_and_pickup.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9572a46112cd54ca015024293153cc777ad4c8225ca22f46316bd441d7bcff
|
| 3 |
+
size 3417549
|
previews/press_button.gif
ADDED
|
Git LFS Details
|
previews/press_button.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e68293a0703c4891c6fb965e0bf4ac466ce8c828ce4981396de06768af71ed8f
|
| 3 |
+
size 18874704
|
previews/put_back_block.gif
ADDED
|
Git LFS Details
|
previews/put_back_block.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6ed47e7919f3f4feb1280583fdef210a6c807ee46523ccd3a516fd5b6dfd60a2
|
| 3 |
+
size 12927871
|
previews/rearrange_blocks.gif
ADDED
|
Git LFS Details
|
previews/rearrange_blocks.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d2f1ada2d67f99e7238859ed5f1f05239f3ad8c72cbfd774a3e4519fd3657ad9
|
| 3 |
+
size 14340436
|
previews/swap_T.gif
ADDED
|
Git LFS Details
|
previews/swap_T.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5ab1b9460864271e520af8ac7e4a4803d9d4f90dbb555092f0cef0ac06278fe0
|
| 3 |
+
size 15869974
|
previews/swap_blocks.gif
ADDED
|
Git LFS Details
|
previews/swap_blocks.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:58f81403a2d7f51f9f270b5c459c0e4bd16b3c448fd26cde3c54c2f28d17f649
|
| 3 |
+
size 28768290
|
swap_T/memory/best_model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:581f7051efa0a70c836a1389fb6601aadf85aa2672dc1bb6b2002db3a511672b
|
| 3 |
+
size 48712373
|
swap_T/memory/normalizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:088e5465951f9686fd2f86143bc82fc9ef420eef9bf76d82bd5773501b868ed0
|
| 3 |
+
size 8309
|
swap_T/policy.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2bae46d18a01c7c6243318a630ce10b1dadca4547d3440eac960954c733758ba
|
| 3 |
+
size 1097407783
|
swap_blocks/memory/best_model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c5df042c66b31b6042909eda92be345895a640a5ecf6428ebabbb9f14950f4a7
|
| 3 |
+
size 48720693
|
swap_blocks/memory/normalizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:13e9c2bb3cb95fc6f16bb7da9f0ece7f9cd2b8839b01f195f66a582a94090007
|
| 3 |
+
size 8309
|
swap_blocks/policy.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:028b7b2dd4296ff094cf937685786b441dcb78c8f70e428893495cad2164adb4
|
| 3 |
+
size 1097416103
|