Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
104 changes: 104 additions & 0 deletions examples/dynamo/run_groot_export.py

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

You can just put the example of the exporter tool with the tool unless you want them rendered. Then we need to use the sphinx gallery format

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sure, I added another PR on this stack with rough docs, still need to add inference instructions

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do you have the inference side of these examples? (Ideally both python and C++ deployments)

Original file line number Diff line number Diff line change
@@ -0,0 +1,104 @@
#!/usr/bin/env python3
"""Smoke EdgeExporter on GR00T (4 engines: vision, language, context_projection, action).

Pass the LeRobot GrootPolicy, not policy._groot_model — prepare_sample_inputs
needs GrootEagleEncodeStep / embodiment_id from the policy wrapper.
"""

from __future__ import annotations

import argparse
from pathlib import Path

_REPO_ROOT = Path(__file__).resolve().parents[2] # TensorRT/
_TRT_PY = _REPO_ROOT / "py"

import torch # noqa: E402
import torch_tensorrt # noqa: E402

_src_pkg = str(_TRT_PY / "torch_tensorrt")
if _src_pkg not in list(torch_tensorrt.__path__):
torch_tensorrt.__path__.append(_src_pkg)

from lerobot.configs import FeatureType, PolicyFeature
from lerobot.policies.groot import GrootPolicy
from lerobot.policies.groot.configuration_groot import GrootConfig
from lerobot.utils.constants import ACTION, OBS_STATE
from torch_tensorrt.hf.exporters import EdgeConfig, EdgeExporter
from torch_tensorrt.hf.exporters.plugin.plugin_utils import load_plugins_for_trt
from torch_tensorrt.hf.exporters.utils import configure_thor_pytorch, force_hf_attention


def load_groot(device: torch.device) -> GrootPolicy:
config = GrootConfig(
base_model_path="nvidia/GR00T-N1.5-3B",
device=str(device),
embodiment_tag="new_embodiment",
chunk_size=50,
n_action_steps=50,
max_state_dim=64,
max_action_dim=32,
image_size=(224, 224),
tokenizer_assets_repo="lerobot/eagle2hg-processor-groot-n1p5",
input_features={
"observation.images.image": PolicyFeature(
type=FeatureType.VISUAL, shape=(3, 224, 224)
),
"observation.images.image2": PolicyFeature(
type=FeatureType.VISUAL, shape=(3, 224, 224)
),
OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(7,)),
},
output_features={ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(32,))},
)
return GrootPolicy(config).to(device).eval()


def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument(
"--compile", action="store_true", help="Build TRT engines (default: dryrun)"
)
parser.add_argument("--engine-dir", default="/tmp/groot_edge_exporter")
args = parser.parse_args()

configure_thor_pytorch()
load_plugins_for_trt()

device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
dtype = torch.float16

policy = load_groot(device)
model = policy._groot_model.to(device=device, dtype=dtype).eval()
eagle = model.backbone.eagle_model
force_hf_attention(eagle.vision_model, "eager")
force_hf_attention(eagle.language_model, "eager")

exporter = EdgeExporter()
config = EdgeConfig(
model_type="groot",
engine_dir=args.engine_dir,
max_seq_len=968,
dryrun=not args.compile,
skip_runtime_export=False,
)

# Spec tokenizes libero via Eagle chat template because we pass the policy.
sample_inputs = {"device": device, "dtype": dtype}
program = exporter.export(policy, sample_inputs, config=config)

print("engines:", exporter.engines)
print("runtime keys:", sorted(exporter.sample))

with torch.no_grad():
if hasattr(program, "module"):
velocity = program.module()(**exporter.sample)
else:
velocity = program(**exporter.sample)

out = velocity[0] if isinstance(velocity, (tuple, list)) else velocity
print("velocity", tuple(out.shape), "mean", float(out.float().mean()))


if __name__ == "__main__":
main()
98 changes: 98 additions & 0 deletions examples/dynamo/run_nemotron_export.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,98 @@
#!/usr/bin/env python3
"""Smoke EdgeExporter on Nemotron-H (one language engine: attn + mamba + MoE).

Pass the HF causal LM. Collation is tokenizer → input_ids; the spec embeds and
pads to max_seq_len. apply_mamba_stub() must run before from_pretrained.
"""

from __future__ import annotations

import argparse
from pathlib import Path

_REPO_ROOT = Path(__file__).resolve().parents[2] # TensorRT/
_TRT_PY = _REPO_ROOT / "py"

import torch # noqa: E402
import torch_tensorrt # noqa: E402

_src_pkg = str(_TRT_PY / "torch_tensorrt")
if _src_pkg not in list(torch_tensorrt.__path__):
torch_tensorrt.__path__.append(_src_pkg)

from torch_tensorrt.hf.exporters import EdgeConfig, EdgeExporter
from torch_tensorrt.hf.exporters.mamba_stub import apply as apply_mamba_stub
from torch_tensorrt.hf.exporters.plugin.plugin_utils import load_plugins_for_trt
from torch_tensorrt.hf.exporters.utils import configure_thor_pytorch
from transformers import AutoModelForCausalLM, AutoTokenizer


def load_nemotron(checkpoint: str, device: torch.device, dtype: torch.dtype):
apply_mamba_stub()
model = (
AutoModelForCausalLM.from_pretrained(
checkpoint,
trust_remote_code=True,
torch_dtype=dtype,
)
.to(device=device, dtype=dtype)
.eval()
)
tokenizer = AutoTokenizer.from_pretrained(checkpoint, trust_remote_code=True)
if tokenizer.pad_token_id is None:
tokenizer.pad_token = tokenizer.eos_token
return model, tokenizer


def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument(
"--compile", action="store_true", help="Build TRT engines (default: dryrun)"
)
parser.add_argument("--engine-dir", default="/tmp/nemotron_edge_exporter")
parser.add_argument(
"--checkpoint",
default="nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16",
)
parser.add_argument("--prompt", default="Hello.")
parser.add_argument("--max-seq-len", type=int, default=128)
args = parser.parse_args()

configure_thor_pytorch()
load_plugins_for_trt()

device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
dtype = torch.float16

model, tokenizer = load_nemotron(args.checkpoint, device, dtype)
encoded = tokenizer(args.prompt, return_tensors="pt")
sample_inputs = {
"input_ids": encoded["input_ids"].to(device),
"attention_mask": encoded["attention_mask"].to(device),
}

exporter = EdgeExporter()
config = EdgeConfig(
model_type="nemotron_h",
engine_dir=args.engine_dir,
max_seq_len=args.max_seq_len,
dryrun=not args.compile,
skip_runtime_export=False,
)
program = exporter.export(model, sample_inputs, config=config)

print("engines:", exporter.engines)
print("runtime keys:", sorted(exporter.sample))

with torch.no_grad():
if hasattr(program, "module"):
out = program.module()(**exporter.sample)
else:
out = program(**exporter.sample)

logits = out[0] if isinstance(out, (tuple, list)) else out
print("logits", tuple(logits.shape), "mean", float(logits.float().mean()))


if __name__ == "__main__":
main()
108 changes: 108 additions & 0 deletions examples/dynamo/run_pi05_export.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
#!/usr/bin/env python3
from __future__ import annotations

import argparse
from pathlib import Path

_REPO_ROOT = Path(__file__).resolve().parents[2] # TensorRT/
_TRT_PY = _REPO_ROOT / "py"

import torch # noqa: E402
import torch_tensorrt # noqa: E402

_src_pkg = str(_TRT_PY / "torch_tensorrt")
if _src_pkg not in list(torch_tensorrt.__path__):
torch_tensorrt.__path__.append(_src_pkg)

from lerobot.configs import FeatureType, PolicyFeature
from lerobot.policies.pi05 import PI05Policy
from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
from torch_tensorrt.hf.exporters import EdgeConfig, EdgeExporter
from torch_tensorrt.hf.exporters.plugin.plugin_utils import load_plugins_for_trt
from torch_tensorrt.hf.exporters.utils import configure_thor_pytorch, force_hf_attention


def load_pi05(device: torch.device) -> PI05Policy:
policy = PI05Policy.from_pretrained("lerobot/pi05_libero_base").eval()
cfg = policy.config
cfg.device = str(device)
cfg.chunk_size = 50
cfg.n_action_steps = 50
cfg.max_state_dim = 32
cfg.max_action_dim = 32
cfg.input_features = {
f"{OBS_IMAGES}.image": PolicyFeature(
type=FeatureType.VISUAL, shape=(3, 224, 224)
),
f"{OBS_IMAGES}.image2": PolicyFeature(
type=FeatureType.VISUAL, shape=(3, 224, 224)
),
f"{OBS_IMAGES}.image3": PolicyFeature(
type=FeatureType.VISUAL, shape=(3, 224, 224)
),
f"{OBS_IMAGES}.image4": PolicyFeature(
type=FeatureType.VISUAL, shape=(3, 224, 224)
),
OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(32,)),
}
cfg.output_features = {ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(32,))}
cfg.empty_cameras = 0
cfg.validate_features()
return policy


def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument(
"--compile", action="store_true", help="Build TRT engines (default: dryrun)"
)
parser.add_argument("--engine-dir", default="/tmp/pi05_edge_exporter")
args = parser.parse_args()

configure_thor_pytorch()
load_plugins_for_trt()

device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
dtype = torch.float16

policy = load_pi05(device)
# Weights on GPU; spec still needs the policy object for the preprocessor.
policy.model.to(device=device, dtype=dtype).eval()
paligemma = policy.model.paligemma_with_expert.paligemma.model
force_hf_attention(paligemma.vision_tower, "eager")
force_hf_attention(paligemma.language_model, "eager")
force_hf_attention(policy.model.paligemma_with_expert.gemma_expert.model, "eager")

exporter = EdgeExporter()
config = EdgeConfig(
model_type="pi05", # optional; inferred from paligemma_with_expert
engine_dir=args.engine_dir,
max_seq_len=968,
dryrun=not args.compile, # True = no TRT, still writes config.json + runtime graph
skip_runtime_export=False, # False = also torch.export the stitched execute_engine graph
# components=("vision",), # uncomment to export only vision
)

# Spec loads libero + preprocessor because we pass the policy, not a tensor dict.
sample_inputs = {"device": device, "dtype": dtype}

program = exporter.export(policy, sample_inputs, config=config)

print("engines:", exporter.engines)

# Runtime kwargs are tensors only (pixel_values, lang_embeds, rope, KVs, …).
runtime_kwargs = exporter.sample
print("runtime keys:", sorted(runtime_kwargs))

with torch.no_grad():
if hasattr(program, "module"):
velocity = program.module()(**runtime_kwargs)
else:
velocity = program(**runtime_kwargs)

out = velocity[0] if isinstance(velocity, (tuple, list)) else velocity
print("velocity", tuple(out.shape), "mean", float(out.float().mean()))


if __name__ == "__main__":
main()
4 changes: 4 additions & 0 deletions py/torch_tensorrt/hf/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
"""HuggingFace-facing export helpers for Torch-TensorRT.

Use ``from torch_tensorrt.hf.exporters import EdgeExporter, EdgeConfig``.
"""
26 changes: 26 additions & 0 deletions py/torch_tensorrt/hf/exporters/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
from torch_tensorrt.hf.exporters.config import EdgeConfig
from torch_tensorrt.hf.exporters.exporter import EdgeExporter
from torch_tensorrt.hf.exporters.models.groot.spec import ( # noqa: F401
GrootSpec as _GrootSpec,
)
from torch_tensorrt.hf.exporters.models.nemotron.spec import ( # noqa: F401
NemotronSpec as _NemotronSpec,
)
from torch_tensorrt.hf.exporters.models.pi05.spec import ( # noqa: F401
Pi05Spec as _Pi05Spec,
)
from torch_tensorrt.hf.exporters.spec import (
ComponentBundle,
EdgeSpec,
get_edge_spec,
register_edge_spec,
)

__all__ = [
"ComponentBundle",
"EdgeConfig",
"EdgeExporter",
"EdgeSpec",
"get_edge_spec",
"register_edge_spec",
]
Loading
Loading