From 90e731ec175a6d9598d6266b4b9400026a7ec080 Mon Sep 17 00:00:00 2001 From: Filipe Martins <293984334+filipemartinsubrobotics@users.noreply.github.com> Date: Fri, 2 Oct 2026 08:47:26 +0000 Subject: [PATCH 1/2] Add Cosmos3-Edge variant of the LIBERO-10 action-policy SFT recipe action_policy_libero_edge is action_policy_libero_nano on the public nvidia/Cosmos3-Edge base. It differs only in the model config (EDGE_MODEL_CONFIG, the dense Nemotron-2B-Dense-VL backbone) and one extra trainable key, k_norm_und_for_gen, which trains with the generation pathway as in vision_sft_edge. Adds the experiment, a run TOML and launcher mirroring the Nano preset A (HSDP 2x8, global batch 2048), and a docs section with the one-node setting (replicate 1, grad_accum 2) and notes for running outside the Docker image (NPP and an AV1-capable FFmpeg for torchcodec). Validated with a 50-iteration single-node smoke run on 8x RTX PRO 6000 Blackwell: about 79 s/iteration at global batch 2048, loss 15.3 -> 3.6, DCP checkpoint saved. A full 2000-iteration run and the closed-loop libero_10 success rate have not been measured for Edge yet. Signed-off-by: Filipe Martins <293984334+filipemartinsubrobotics@users.noreply.github.com> --- cosmos_framework/configs/base/config.py | 1 + .../action_policy_libero_edge.py | 239 ++++++++++++++++++ docs/action_policy_libero_posttrain.md | 43 ++++ ...launch_sft_action_policy_libero_10_edge.sh | 46 ++++ .../action_policy_libero_10_edge.toml | 47 ++++ 5 files changed, 376 insertions(+) create mode 100644 cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py create mode 100755 examples/launch_sft_action_policy_libero_10_edge.sh create mode 100644 examples/toml/sft_config/action_policy_libero_10_edge.toml diff --git a/cosmos_framework/configs/base/config.py b/cosmos_framework/configs/base/config.py index 9e950946b..73b37e11b 100644 --- a/cosmos_framework/configs/base/config.py +++ b/cosmos_framework/configs/base/config.py @@ -96,6 +96,7 @@ def make_config() -> Config: # vision_sft_nano_mapstyle_dataloader — the CosmosDataLoader variant — in the same module.) import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_droid_nano # noqa: F401 import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_libero_all_nano # noqa: F401 + import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_libero_edge # noqa: F401 import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_libero_nano # noqa: F401 import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_robocasa_nano # noqa: F401 import cosmos_framework.configs.base.experiment.action.posttrain_config.action_fd_droid_posttrain # noqa: F401 diff --git a/cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py b/cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py new file mode 100644 index 000000000..3d164da3a --- /dev/null +++ b/cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py @@ -0,0 +1,239 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: OpenMDW-1.1 + +"""``action_policy_libero_edge`` — Cosmos3-Edge LIBERO-10 action-policy SFT recipe. + +Edge sibling of ``action_policy_libero_nano``: same data pipeline +(``LIBEROLeRobotDataset``, frame-wise-relative rot6d, ``quantile_rot``, +concat_view third-person + wrist), optimizer, scheduler and trainer blocks. +The only differences are the model config (``EDGE_MODEL_CONFIG``, the dense +Nemotron-2B-Dense-VL backbone) and one extra trainable key, +``k_norm_und_for_gen``, which trains with the generation pathway as in +``vision_sft_edge``. Starts from the public ``nvidia/Cosmos3-Edge`` base. +Train on ``libero_10`` alone (``LIBERO_ROOT``). +See docs/action_policy_libero_posttrain.md. +""" + +import copy + +from hydra.core.config_store import ConfigStore + +from cosmos_framework.configs.base.experiment.sft.models.edge_model_config import EDGE_MODEL_CONFIG +from cosmos_framework.data.generator.action.datasets.action_sft_dataset import get_action_libero_sft_dataset +from cosmos_framework.data.generator.joint_dataloader import ( + PackingDataLoader, + RankPartitionedDataLoader, +) +from cosmos_framework.utils.lazy_config import LazyCall as L +from cosmos_framework.utils.lazy_config import LazyDict + +cs = ConfigStore.instance() + + +def _action_policy_libero_edge_model_config() -> dict: + """LIBERO model config (Edge): capped packed tokens, selective activation + checkpointing, fresh diffusion-expert init, 10x vision flow-matching loss. + Keep ``encode_exact_durations=[17, 61, 73]``, as in the Nano recipe.""" + cfg = copy.deepcopy(EDGE_MODEL_CONFIG) # action_gen=True, max_action_dim=64 + # Cap the packed sequence. Uncapped (-1) + a large max_samples_per_batch packs + # one very long sequence and OOMs even on H200; 74000 keeps the GA-validated bound. + cfg["max_num_tokens_after_packing"] = 74000 + cfg["activation_checkpointing"]["mode"] = "selective" + cfg["diffusion_expert_config"]["load_weights_from_pretrained"] = False + cfg["rectified_flow_training_config"]["loss_scale"] = 10.0 + cfg["rectified_flow_training_config"]["image_loss_scale"] = None + cfg["tokenizer"]["encode_exact_durations"] = [17, 61, 73] # match Cosmos3 base + reference SFT (do NOT reduce) + return cfg + + +action_policy_libero_edge = LazyDict( + dict( + defaults=[ + {"override /model": "mot_fsdp"}, + {"override /data_train": None}, + {"override /data_val": None}, + # FusedAdam with fp32 master_weights + eps 1e-8 (bf16 params + eps 1e-6 + # diverged on the action loss). + {"override /optimizer": "fusedadamw"}, + {"override /scheduler": "lambdalinear"}, # linear LR decay + {"override /checkpoint": "s3"}, + { + "override /callbacks": [ + "basic", + "optimization", + "job_monitor", + ] + }, + {"override /ema": "power"}, + {"override /tokenizer": "wan2pt2_tokenizer"}, + {"override /sound_tokenizer": None}, + {"override /vlm_config": None}, + {"override /ckpt_type": "dcp"}, + "_self_", + ], + job=dict( + project="cosmos3", + group="action_sft", + name="action_policy_libero_edge", + wandb_mode="disabled", + ), + model=dict( + config=_action_policy_libero_edge_model_config(), + ), + optimizer=dict( + betas=[0.9, 0.99], + eps=1.0e-08, + fused=True, # popped by build_optimizer for FusedAdam (fused by construction) + # Train the generation + action heads. + keys_to_select=[ + "moe_gen", + "time_embedder", + "vae2llm", + "llm2vae", + "k_norm_und_for_gen", # Edge: und-K norm trains with the gen pathway (as vision_sft_edge) + "action2llm", + "llm2action", + "action_modality_embed", + ], + lr=5.0e-05, + lr_multipliers={ + "action2llm": 5.0, + "llm2action": 5.0, + "action_modality_embed": 5.0, + }, + optimizer_type="FusedAdam", + weight_decay=0.05, + ), + scheduler=dict( + lr_scheduler_type="LambdaLinear", + cycle_lengths=[100], # smoke: 100 iters (real run sets via TOML, GA=10000) + f_max=[1.0], + f_min=[0.0], + f_start=[1.0e-06], + verbosity_interval=0, + warm_up_steps=[0], # smoke (real run sets via TOML, GA=2000) + ), + trainer=dict( + distributed_parallelism="fsdp", + grad_accum_iter=1, # real run sets via TOML (GA=2) + logging_iter=1, + max_iter=100, # smoke + max_val_iter=None, + run_validation=False, + run_validation_on_start=False, + save_zero_checkpoint=False, + seed=42, + timeout_period=999999999, + validation_iter=100, + compile_config=dict(recompile_limit=8, use_duck_shape=False), + cudnn=dict(benchmark=True, deterministic=False), + ddp=dict(broadcast_buffers=True, find_unused_parameters=False, static_graph=True), + grad_scaler_args=dict(enabled=False), + callbacks=dict( + dataloader_speed=dict(every_n=100, save_s3=False, step_size=1), + device_monitor=dict( + every_n=200, log_memory_detail=True, save_s3=False, step_size=1, upload_every_n_mul=5 + ), + grad_clip=dict(clip_norm=1.0, force_finite=True), + heart_beat=dict(every_n=200, save_s3=False, step_size=1, update_interval_in_minute=20), + iter_speed=dict(every_n=1, hit_thres=50, save_s3=False, save_s3_every_log_n=500), + low_precision=dict(update_iter=1), + manual_gc=dict(every_n=5, gc_level=1, warm_up=1), + param_count=dict(save_s3=False), + skip_nan_step=dict(max_consecutive_nan=100), + training_stats=dict(log_freq=100), + ), + ), + checkpoint=dict( + broadcast_via_filesystem=False, + dcp_async_mode_enabled=False, + enable_gcs_patch_in_boto3=True, + keys_not_to_resume=[], + # Skip net_ema (EMA warm-starts from net, see dcp.py) and the action + # heads, so they init fresh from the base (the public Cosmos3-Edge base + # has no LIBERO-trained action heads). + keys_to_skip_loading=[ + "net_ema.", + "action2llm", + "llm2action", + "action_modality_embed", + "action_pos_embed", + ], + load_ema_to_reg=False, + load_path="???", # Cosmos3-Edge DCP dir; supply via TOML/env + load_training_state=False, + only_load_scheduler_state=False, + save_iter=100, + strict_resume=False, # base init: tolerate key set differences + verbose=True, + hf_export=dict( + enabled=False, + export_every_n=1, + hf_repo_id=None, + upload_to_object_store=dict(bucket="", credentials="", enabled=False), + ), + jit=dict(device="cuda", dtype="bfloat16", enabled=False, input_shape=None, strict=True), + load_from_object_store=dict(bucket="", credentials="", enabled=False), + save_to_object_store=dict(bucket="", credentials="", enabled=False), + ), + dataloader_train=L(PackingDataLoader)( + audio_sample_rate=48000, + dataset_name="action_libero", + max_samples_per_batch=128, # peak-mem bound (256 OOMs on H200); global = 128 x DP8 x grad_accum2 = 2048 + max_sequence_length=None, # None disables token packing (TOML can't express null) + patch_spatial=2, + sound_latent_fps=0, + tokenizer_spatial_compression_factor=16, + tokenizer_temporal_compression_factor=4, + dataloader=L(RankPartitionedDataLoader)( + batch_size=1, + in_order=False, + num_workers=4, + persistent_workers=True, + pin_memory=True, + prefetch_factor=4, + sampler=None, + # Shuffling is handled by the dataset (iterable_shuffle=True below): + # ActionIterableShuffleDataset streams rank x worker-sharded, episode-order- + # shuffled, sequential-within-episode. + datasets=dict( + libero=dict( + ratio=1, + dataset=L(get_action_libero_sft_dataset)( + # Local LeRobot dir for the libero_10 suite ONLY. Use the + # 20 FPS nvidia/LIBERO_LeRobot_v3 (matches the bundled stats + 20 Hz eval): + # hf download nvidia/LIBERO_LeRobot_v3 --repo-type dataset \ + # --include 'libero_10/**' --local-dir # LIBERO_ROOT=/libero_10 + root="${oc.env:LIBERO_ROOT}", + fps=20, # metadata only (FPS-agnostic loader reads native fps from info.json) + chunk_length=16, + image_size=256, # concat_view -> 256x512 + mode="wam", + camera_mode="concat_view", + action_space="frame_wise_relative", + rotation_space="6d", + pose_coordinate_frame="native", + action_normalization="quantile_rot", + val_ratio=0.01, + iterable_shuffle=True, + episode_shuffle_seed=42, + resolution=None, + max_action_dim="${model.config.max_action_dim}", + cfg_dropout_rate=0.1, + format_prompt_as_json=True, # structured JSON prompts (set False for plain-text) + tokenizer_config="${model.config.vlm_config.tokenizer}", + ), + ), + ), + ), + ), + dataloader_val=None, + upload_reproducible_setup=False, + ), + flags={"allow_objects": True}, +) + + +for _item in [action_policy_libero_edge]: + _name = [k for k, v in globals().items() if v is _item][0] + cs.store(group="experiment", package="_global_", name=_name, node=_item) diff --git a/docs/action_policy_libero_posttrain.md b/docs/action_policy_libero_posttrain.md index 934b4b642..c5a37fdf3 100644 --- a/docs/action_policy_libero_posttrain.md +++ b/docs/action_policy_libero_posttrain.md @@ -146,3 +146,46 @@ Eval parity — the client/server already handle these; verify if accuracy is lo rotates them back. - **Normalization** — start the server with `--action-normalization quantile_rot` and the bundled rot6d stats, or actions come out at the wrong scale. + +## 5. Cosmos3-Edge variant + +`action_policy_libero_edge` is the same libero_10 recipe on the public +`nvidia/Cosmos3-Edge` base (dense Nemotron-2B-Dense-VL backbone). It differs +from `action_policy_libero_nano` only in the model config (`EDGE_MODEL_CONFIG`) +and one extra trainable key, `k_norm_und_for_gen`, as in `vision_sft_edge`. + +| Piece | Path | +| ---------- | ------------------------------------------------------------------------------------------------ | +| Experiment | `cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py` | +| Run TOML | `examples/toml/sft_config/action_policy_libero_10_edge.toml` | +| Launch | `examples/launch_sft_action_policy_libero_10_edge.sh` | + +```bash +python -m cosmos_framework.scripts.convert_model_to_dcp \ + -o examples/checkpoints/Cosmos3-Edge \ + --checkpoint-path Cosmos3-Edge + +export BASE_CHECKPOINT_PATH=examples/checkpoints/Cosmos3-Edge +export LIBERO_ROOT=/LIBERO_LeRobot_v3/libero_10 +bash examples/launch_sft_action_policy_libero_10_edge.sh # HSDP 2x8, as the Nano preset A +``` + +**One 8-GPU node:** set `data_parallel_replicate_degree = 1` and +`grad_accum_iter = 2` in the TOML; the global batch stays 2048. + +**Validated so far:** a 50-iteration single-node smoke run (8× RTX PRO 6000 +Blackwell 96 GB, replicate 1, grad_accum 2, global batch 2048) trained at about +79 s/iteration; loss went from 15.3 (iteration 1) to 3.6 (iteration 50) and the +DCP checkpoint saved. A full 2000-iteration run and the closed-loop libero_10 +success rate have not been measured for Edge yet. + +**Outside the Docker image:** the data loader decodes LIBERO's AV1 videos with +`torchcodec`, which dlopens NPP and FFmpeg shared libraries that the container +provides on the system path. In a plain `uv sync` environment: + +- NPP: add `.venv/lib/python3.13/site-packages/nvidia/cu13/lib` to `LD_LIBRARY_PATH` + (otherwise `libnppicc.so.13: cannot open shared object file`). +- FFmpeg with an AV1 decoder (`dav1d`): install system FFmpeg, or point + `LD_LIBRARY_PATH` at an FFmpeg 8 build that includes `libdav1d`. The FFmpeg + bundled in `opencv-python` has no AV1 decoder and fails with + `Could not push packet to decoder: Function not implemented`. diff --git a/examples/launch_sft_action_policy_libero_10_edge.sh b/examples/launch_sft_action_policy_libero_10_edge.sh new file mode 100755 index 000000000..da82d84db --- /dev/null +++ b/examples/launch_sft_action_policy_libero_10_edge.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: OpenMDW-1.1 + +# Structured-TOML launch for action_policy_libero_edge — Cosmos3-Edge LIBERO +# action-policy SFT (HSDP, full SFT). Drives cosmos_framework.scripts.train +# against examples/toml/sft_config/action_policy_libero_10_edge.toml. +# +# Point LIBERO_ROOT at the libero_10 suite ONLY. Use the 20 FPS +# nvidia/LIBERO_LeRobot_v3. The default recipe is HSDP 2x8 (global batch 2048); +# set NNODES/NODE_RANK/MASTER_ADDR per node. +# See docs/action_policy_libero_posttrain.md. +# +# Required env vars: +# LIBERO_ROOT local LIBERO-10 LeRobot dataset dir, e.g. /libero_10 (no default) +# Optional env vars (defaults below; override to relocate data/checkpoints): +# BASE_CHECKPOINT_PATH default: examples/checkpoints/Cosmos3-Edge +# WAN_VAE_PATH default: examples/checkpoints/wan22_vae/Wan2.2_VAE.pth +# HF_TOKEN if any tokenizer download requires gated HF access +# OUTPUT_ROOT default: outputs/train +# +# Pre-sync the 20 FPS suite once: +# hf download nvidia/LIBERO_LeRobot_v3 --repo-type dataset --include 'libero_10/**' --local-dir +# export LIBERO_ROOT=/libero_10 +# +# Usage (HSDP 2x8; set NNODES/NODE_RANK/MASTER_ADDR per node): +# LIBERO_ROOT=/libero_10 bash examples/launch_sft_action_policy_libero_10_edge.sh + +TOML_FILE="examples/toml/sft_config/action_policy_libero_10_edge.toml" +: "${BASE_CHECKPOINT_PATH:=examples/checkpoints/Cosmos3-Edge}" + +# LIBEROLeRobotDataset reads ${oc.env:LIBERO_ROOT} directly (a LOCAL LeRobot dir); +# export it so torchrun (launched in this shell) inherits it. +export LIBERO_ROOT="${LIBERO_ROOT:-}" + +EXTRA_DATASET_CHECK='[[ -f "$LIBERO_ROOT/meta/info.json" ]] || { echo "ERROR: LIBERO_ROOT must be a local LeRobot dir containing meta/info.json (got: '\''$LIBERO_ROOT'\''). Pre-sync: hf download nvidia/LIBERO_LeRobot_v3 --repo-type dataset --include '\''libero_10/**'\'' --local-dir (then LIBERO_ROOT=/libero_10). See docs/action_policy_libero_posttrain.md" >&2; exit 1; }' + +# Extra Hydra overrides from the environment: a space-separated string word-split into +# the TAIL_OVERRIDES array. An exported string survives `bash ` (a child +# process), unlike a TAIL_OVERRIDES array set in your shell. Use it for smoke runs, +# e.g. EXTRA_TAIL_OVERRIDES="trainer.max_iter=5 job.wandb_mode=offline". +TAIL_OVERRIDES=( + ${EXTRA_TAIL_OVERRIDES:-} +) + +source "$(dirname "${BASH_SOURCE[0]}")/_sft_launcher_common.sh" diff --git a/examples/toml/sft_config/action_policy_libero_10_edge.toml b/examples/toml/sft_config/action_policy_libero_10_edge.toml new file mode 100644 index 000000000..4f117c70e --- /dev/null +++ b/examples/toml/sft_config/action_policy_libero_10_edge.toml @@ -0,0 +1,47 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: OpenMDW-1.1 + +# LIBERO-10 action-policy SFT run config for the `action_policy_libero_edge` +# experiment. Train on libero_10 alone (HSDP 2x8, global batch 2048). +# Env: LIBERO_ROOT, BASE_CHECKPOINT_PATH, WAN_VAE_PATH, IMAGINAIRE_OUTPUT_ROOT. +# See docs/action_policy_libero_sft.md. + +[job] +task = "vfm" +experiment = "action_policy_libero_edge" +project = "cosmos3_action_libero" +group = "action_sft" +name = "action_policy_libero_10_edge" +wandb_mode = "online" + +[model] +precision = "bfloat16" +max_num_tokens_after_packing = 74000 + +[model.parallelism] +data_parallel_shard_degree = 8 +data_parallel_replicate_degree = 2 # HSDP 2x8 = 16 ranks (2 nodes); minimum for gbs 2048 at grad_accum 1 + # One 8-GPU node: replicate 1 + [trainer] grad_accum_iter = 2 keeps gbs 2048 + +[model.activation_checkpointing] +mode = "selective" +save_ops_regex = ["fmha"] + +[model.tokenizer] +vae_path = "${oc.env:WAN_VAE_PATH}" + +[optimizer] +lr = 5.0e-05 + +[scheduler] +cycle_lengths = [16000] +warm_up_steps = [500] + +[trainer] +max_iter = 2000 +logging_iter = 50 +grad_accum_iter = 1 # global batch = max_samples 128 x (shard 8 x replicate 2) x 1 = 2048 + +[checkpoint] +load_path = "${oc.env:BASE_CHECKPOINT_PATH}" +save_iter = 500 From 0c60e9800886b0e1279b8d1c607d2bdd4639d8c4 Mon Sep 17 00:00:00 2001 From: Filipe Martins <293984334+filipemartinsubrobotics@users.noreply.github.com> Date: Fri, 2 Oct 2026 10:16:40 +0000 Subject: [PATCH 2/2] docs: align the Edge recipe table (rumdl fmt) Signed-off-by: Filipe Martins <293984334+filipemartinsubrobotics@users.noreply.github.com> --- docs/action_policy_libero_posttrain.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/action_policy_libero_posttrain.md b/docs/action_policy_libero_posttrain.md index c5a37fdf3..135badfd3 100644 --- a/docs/action_policy_libero_posttrain.md +++ b/docs/action_policy_libero_posttrain.md @@ -154,11 +154,11 @@ Eval parity — the client/server already handle these; verify if accuracy is lo from `action_policy_libero_nano` only in the model config (`EDGE_MODEL_CONFIG`) and one extra trainable key, `k_norm_und_for_gen`, as in `vision_sft_edge`. -| Piece | Path | -| ---------- | ------------------------------------------------------------------------------------------------ | +| Piece | Path | +| ---------- | ----------------------------------------------------------------------------------------------- | | Experiment | `cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py` | -| Run TOML | `examples/toml/sft_config/action_policy_libero_10_edge.toml` | -| Launch | `examples/launch_sft_action_policy_libero_10_edge.sh` | +| Run TOML | `examples/toml/sft_config/action_policy_libero_10_edge.toml` | +| Launch | `examples/launch_sft_action_policy_libero_10_edge.sh` | ```bash python -m cosmos_framework.scripts.convert_model_to_dcp \