Skip to content

Commit 49d426c

Browse files
committed
Add GR00T-H
Signed-off-by: Nigel Nelson <nigeln@nvidia.com>
1 parent 9c4d59f commit 49d426c

94 files changed

Lines changed: 11465 additions & 302 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎README.md‎

Lines changed: 18 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -1,18 +1,30 @@
11
<div align="center">
22

3-
<img src="media/header_compress.png" width="800" alt="NVIDIA Isaac GR00T N1.6 Header">
3+
<img src="media/gr00t-h-header.png" width="800" alt="NVIDIA GR00T-H Header">
44

55
<!-- --- -->
66

77
<p style="font-size: 1.2em;">
8-
<a href="https://developer.nvidia.com/isaac/gr00t"><strong>Website</strong></a> |
9-
<a href="https://huggingface.co/nvidia/GR00T-N1.6-3B"><strong>Model</strong></a> |
10-
<a href="https://huggingface.co/datasets/nvidia/PhysicalAI-Robotics-GR00T-X-Embodiment-Sim"><strong>Dataset</strong></a> |
11-
<a href="https://arxiv.org/abs/2503.14734"><strong>Paper</strong></a> |
12-
<a href="https://research.nvidia.com/labs/gear/gr00t-n1_6/"><strong>Research Blog</strong></a>
8+
<a href="https://huggingface.co/nvidia/GR00T-H"><strong>Model</strong></a> |
9+
<a href="https://huggingface.co/datasets/nvidia/PhysicalAI-Robotics-Open-H-Embodiment"><strong>Open-H Dataset</strong></a> |
1310
</p>
1411
</div>
1512

13+
> GR00T-H is a variant of GR00T N1.6 post-trained on the [Open-H dataset](open_h/README.md) for healthcare robotics autonomy. This repository is intended for developers working with the Open-H dataset or building on GR00T for healthcare applications. For general robotics use cases, the upstream [Isaac-GR00T](https://github.com/NVIDIA/Isaac-GR00T) project is a better starting point.
14+
15+
## GR00T-H
16+
17+
The primary differences from upstream Isaac-GR00T live in [`open_h/`](open_h/README.md):
18+
- Per-embodiment modality configs converting 16 healthcare robot datasets to a common action representation
19+
- Multi-embodiment training config and dataset preparation tooling
20+
- Extensions to the data pipeline (clutch-aware filtering, motion scaling, step filtering)
21+
22+
*The rest of this README documents the base GR00T N1.6 model, which GR00T-H builds on.*
23+
24+
---
25+
26+
---
27+
1628
## NVIDIA Isaac GR00T
1729

1830
<div align="center">

‎gr00t/configs/base_config.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -48,6 +48,11 @@ def load(self, path: Path):
4848
def load_dict(self, data: dict):
4949
if "model" in data:
5050
self.model = self.model.__class__(**data["model"])
51+
# YAML loads sequences as lists, but some model fields expect tuples.
52+
for attr in ("image_crop_size", "image_target_size"):
53+
val = getattr(self.model, attr, None)
54+
if isinstance(val, list):
55+
setattr(self.model, attr, tuple(val))
5156
if "data" in data:
5257
self.data = DataConfig(**data["data"])
5358
# Ensure nested datasets are converted to dataclass instances

‎gr00t/configs/data/data_config.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -31,6 +31,11 @@ class SingleDatasetConfig:
3131
# If not provided, falls back to dataset_paths for evaluation
3232
val_dataset_path: Optional[str] = None
3333

34+
# Optional split-based episode filtering (from meta/info.json splits)
35+
# If include_splits is set, only those splits are used. Then exclude_splits is applied.
36+
exclude_splits: List[str] | None = None
37+
include_splits: List[str] | None = None
38+
3439

3540
@dataclass
3641
class DataConfig:

‎gr00t/configs/finetune_config.py‎

Lines changed: 56 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -31,6 +31,18 @@ class FinetuneConfig:
3131
If None, use the pre-registered modality config in `gr00t/configs/data/embodiment_configs.py`.
3232
"""
3333

34+
include_splits: list[str] | None = None
35+
"""
36+
Optional allowlist of dataset splits (from meta/info.json) to include.
37+
If provided, only these splits are used for training and stats.
38+
"""
39+
40+
exclude_splits: list[str] | None = None
41+
"""
42+
Optional denylist of dataset splits (from meta/info.json) to exclude.
43+
Applied after include_splits (if set). Useful for skipping fail episodes.
44+
"""
45+
3446
# --- Model Tuning Flags ---
3547
tune_llm: bool = False
3648
"""If True, fine-tune the language model (LLM) backbone during training."""
@@ -49,6 +61,12 @@ class FinetuneConfig:
4961
Dropout probability applied to state inputs for regularization during training.
5062
"""
5163

64+
state_dropout_prob_per_embodiment: dict[str, float] | None = None
65+
"""
66+
Per-embodiment state dropout overrides. Keys are embodiment tag strings,
67+
values are dropout probabilities in [0.0, 1.0].
68+
"""
69+
5270
# --- Data Augmentation ---
5371
random_rotation_angle: int | None = None
5472
"""Maximum rotation angle (in degrees) for random rotation augmentation of input images."""
@@ -85,6 +103,23 @@ class FinetuneConfig:
85103
If None, no extra augmentations are applied.
86104
"""
87105

106+
image_size: tuple[int, int] | None = None
107+
"""
108+
Intermediate padded size as (height, width) for resize with padding.
109+
Images are resized (preserving aspect ratio) and padded to this size.
110+
Should be >= image_crop_size to allow for cropping augmentation.
111+
Example: (540, 720) to pad all images to 540×720 before cropping.
112+
If None, uses shortest_image_edge with aspect-preserving crops (default albumentations).
113+
"""
114+
115+
image_crop_size: tuple[int, int] | None = None
116+
"""
117+
Final output size as (height, width) after cropping. Only used when image_size is set.
118+
This is the resolution the model actually sees.
119+
Example: (480, 640) means final images are 480×640.
120+
If None, defaults to image_size (no cropping, padded size is final size).
121+
"""
122+
88123
# --- Training Configuration ---
89124
global_batch_size: int = 64
90125
"""Total effective batch size across all GPUs and accumulation steps."""
@@ -134,3 +169,24 @@ class FinetuneConfig:
134169

135170
num_shards_per_epoch: int = int(1e5)
136171
"""Number of shards to use for the dataset. reduce this number if vram is limited."""
172+
173+
# --- Statistics Calculation Flags ---
174+
calculate_norm_stats: bool = False
175+
"""
176+
If True, only calculate normalization statistics and exit without training.
177+
Uses skip_video=True for fast iteration. Statistics will be saved to
178+
the norm_stats_output_path or the dataset's meta directory.
179+
"""
180+
181+
norm_stats_output_path: str | None = None
182+
"""
183+
Path to save calculated normalization statistics. If None, saves to
184+
the dataset's meta/percentile_stats.json file.
185+
"""
186+
187+
stats_num_workers: int | None = None
188+
"""
189+
Number of parallel workers for statistics calculation. If None, uses CPU count.
190+
Set to 1 to disable parallelism. Only used when calculate_norm_stats=True.
191+
"""
192+

‎gr00t/configs/model/gr00t_n1d6.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -98,6 +98,7 @@ class Gr00tN1d6Config(PretrainedConfig):
9898

9999
# State Augmentation parameters
100100
state_dropout_prob: float = 0.0 # State dropout probability
101+
state_dropout_prob_per_embodiment: dict[str, float] | None = None # Per-embodiment overrides
101102
state_additive_noise_scale: float = 0.0 # Scale for additive Gaussian noise on state features
102103

103104
# Multi-embodiment parameters

0 commit comments

Comments
 (0)