Add files using upload-large-folder tool
Browse files- README.md +31 -0
- config.json +41 -0
- model.pt +3 -0
README.md
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
tags:
|
| 3 |
+
- robotics
|
| 4 |
+
- human-robot-interaction
|
| 5 |
+
- flow-matching
|
| 6 |
+
- robot-learning
|
| 7 |
+
---
|
| 8 |
+
|
| 9 |
+
# HRI-diff large, H50, 100k steps
|
| 10 |
+
|
| 11 |
+
This is the 768-wide HRI-diff joint checkpoint trained for 100,000 steps on
|
| 12 |
+
HITBench tomato-to-bowl. `model.pt` includes the conditioning encoder, trust
|
| 13 |
+
flow, action decoder, frozen trust-tube tokenizer, and latent statistics.
|
| 14 |
+
|
| 15 |
+
The policy takes RGB-D at causal offsets `[-9, -6, -3, 0]`, current 8-D robot
|
| 16 |
+
state, and separate regular and interaction prompts. It predicts 50 continuous
|
| 17 |
+
actions. In the closed-loop benchmark it executes one action and replans at
|
| 18 |
+
every step using 20 flow integration steps.
|
| 19 |
+
|
| 20 |
+
Evaluation artifacts: https://huggingface.co/datasets/nhatcm/hri_diff_large_eval
|
| 21 |
+
|
| 22 |
+
```python
|
| 23 |
+
from huggingface_hub import snapshot_download
|
| 24 |
+
from hitbench.hitbench.models.hri_diff.training import HRIDiffPhaseTwo
|
| 25 |
+
|
| 26 |
+
path = snapshot_download("nhatcm/hri_diff_large")
|
| 27 |
+
model = HRIDiffPhaseTwo.from_pretrained(path, map_location="cpu")
|
| 28 |
+
```
|
| 29 |
+
|
| 30 |
+
The flow-matching dependency is licensed CC BY-NC; check its terms before
|
| 31 |
+
commercial use.
|
config.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"hri": {
|
| 3 |
+
"observation_dim": 768,
|
| 4 |
+
"motion_dim": 768,
|
| 5 |
+
"instruction_dim": 768,
|
| 6 |
+
"proprio_dim": 8,
|
| 7 |
+
"action_dim": 7,
|
| 8 |
+
"trust_dim": 512,
|
| 9 |
+
"hidden_dim": 768,
|
| 10 |
+
"trust_tokens": 50,
|
| 11 |
+
"action_horizon": 50,
|
| 12 |
+
"num_heads": 12,
|
| 13 |
+
"context_layers": 3,
|
| 14 |
+
"flow_layers": 12,
|
| 15 |
+
"decoder_layers": 8,
|
| 16 |
+
"mlp_ratio": 4.0,
|
| 17 |
+
"dropout": 0.1,
|
| 18 |
+
"smooth_l1_beta": 1.0,
|
| 19 |
+
"bound_actions": false
|
| 20 |
+
},
|
| 21 |
+
"conditioning": {
|
| 22 |
+
"feature_dim": 768,
|
| 23 |
+
"spatial_grid": 8,
|
| 24 |
+
"motion_grid": 4,
|
| 25 |
+
"max_prompt_bytes": 256,
|
| 26 |
+
"prompt_heads": 12,
|
| 27 |
+
"dropout": 0.1
|
| 28 |
+
},
|
| 29 |
+
"tokenizer": {
|
| 30 |
+
"horizon": 50,
|
| 31 |
+
"mask_size": 64,
|
| 32 |
+
"trust_dim": 512,
|
| 33 |
+
"hidden_dim": 1024,
|
| 34 |
+
"depth_scale_m": 3.0
|
| 35 |
+
},
|
| 36 |
+
"training_contract": {
|
| 37 |
+
"trust_tail": "forward-fill-last-causal-target",
|
| 38 |
+
"decoder_trust": "linear-flow-clean-estimate-curriculum",
|
| 39 |
+
"action_history": "two-pass-self-conditioning-curriculum"
|
| 40 |
+
}
|
| 41 |
+
}
|
model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7c4fe040382609645a65b7b4967b162ac7a570597f1fa201a5c5cc123611c220
|
| 3 |
+
size 1192855103
|