Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cosmos_framework/configs/base/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -96,6 +96,7 @@ def make_config() -> Config:
# vision_sft_nano_mapstyle_dataloader — the CosmosDataLoader variant — in the same module.)
import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_droid_nano # noqa: F401
import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_libero_all_nano # noqa: F401
import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_libero_edge # noqa: F401
import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_libero_nano # noqa: F401
import cosmos_framework.configs.base.experiment.action.posttrain_config.action_policy_robocasa_nano # noqa: F401
import cosmos_framework.configs.base.experiment.action.posttrain_config.action_fd_droid_posttrain # noqa: F401
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,239 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: OpenMDW-1.1

"""``action_policy_libero_edge`` — Cosmos3-Edge LIBERO-10 action-policy SFT recipe.

Edge sibling of ``action_policy_libero_nano``: same data pipeline
(``LIBEROLeRobotDataset``, frame-wise-relative rot6d, ``quantile_rot``,
concat_view third-person + wrist), optimizer, scheduler and trainer blocks.
The only differences are the model config (``EDGE_MODEL_CONFIG``, the dense
Nemotron-2B-Dense-VL backbone) and one extra trainable key,
``k_norm_und_for_gen``, which trains with the generation pathway as in
``vision_sft_edge``. Starts from the public ``nvidia/Cosmos3-Edge`` base.
Train on ``libero_10`` alone (``LIBERO_ROOT``).
See docs/action_policy_libero_posttrain.md.
"""

import copy

from hydra.core.config_store import ConfigStore

from cosmos_framework.configs.base.experiment.sft.models.edge_model_config import EDGE_MODEL_CONFIG
from cosmos_framework.data.generator.action.datasets.action_sft_dataset import get_action_libero_sft_dataset
from cosmos_framework.data.generator.joint_dataloader import (
PackingDataLoader,
RankPartitionedDataLoader,
)
from cosmos_framework.utils.lazy_config import LazyCall as L
from cosmos_framework.utils.lazy_config import LazyDict

cs = ConfigStore.instance()


def _action_policy_libero_edge_model_config() -> dict:
"""LIBERO model config (Edge): capped packed tokens, selective activation
checkpointing, fresh diffusion-expert init, 10x vision flow-matching loss.
Keep ``encode_exact_durations=[17, 61, 73]``, as in the Nano recipe."""
cfg = copy.deepcopy(EDGE_MODEL_CONFIG) # action_gen=True, max_action_dim=64
# Cap the packed sequence. Uncapped (-1) + a large max_samples_per_batch packs
# one very long sequence and OOMs even on H200; 74000 keeps the GA-validated bound.
cfg["max_num_tokens_after_packing"] = 74000
cfg["activation_checkpointing"]["mode"] = "selective"
cfg["diffusion_expert_config"]["load_weights_from_pretrained"] = False
cfg["rectified_flow_training_config"]["loss_scale"] = 10.0
cfg["rectified_flow_training_config"]["image_loss_scale"] = None
cfg["tokenizer"]["encode_exact_durations"] = [17, 61, 73] # match Cosmos3 base + reference SFT (do NOT reduce)
return cfg


action_policy_libero_edge = LazyDict(
dict(
defaults=[
{"override /model": "mot_fsdp"},
{"override /data_train": None},
{"override /data_val": None},
# FusedAdam with fp32 master_weights + eps 1e-8 (bf16 params + eps 1e-6
# diverged on the action loss).
{"override /optimizer": "fusedadamw"},
{"override /scheduler": "lambdalinear"}, # linear LR decay
{"override /checkpoint": "s3"},
{
"override /callbacks": [
"basic",
"optimization",
"job_monitor",
]
},
{"override /ema": "power"},
{"override /tokenizer": "wan2pt2_tokenizer"},
{"override /sound_tokenizer": None},
{"override /vlm_config": None},
{"override /ckpt_type": "dcp"},
"_self_",
],
job=dict(
project="cosmos3",
group="action_sft",
name="action_policy_libero_edge",
wandb_mode="disabled",
),
model=dict(
config=_action_policy_libero_edge_model_config(),
),
optimizer=dict(
betas=[0.9, 0.99],
eps=1.0e-08,
fused=True, # popped by build_optimizer for FusedAdam (fused by construction)
# Train the generation + action heads.
keys_to_select=[
"moe_gen",
"time_embedder",
"vae2llm",
"llm2vae",
"k_norm_und_for_gen", # Edge: und-K norm trains with the gen pathway (as vision_sft_edge)
"action2llm",
"llm2action",
"action_modality_embed",
],
lr=5.0e-05,
lr_multipliers={
"action2llm": 5.0,
"llm2action": 5.0,
"action_modality_embed": 5.0,
},
optimizer_type="FusedAdam",
weight_decay=0.05,
),
scheduler=dict(
lr_scheduler_type="LambdaLinear",
cycle_lengths=[100], # smoke: 100 iters (real run sets via TOML, GA=10000)
f_max=[1.0],
f_min=[0.0],
f_start=[1.0e-06],
verbosity_interval=0,
warm_up_steps=[0], # smoke (real run sets via TOML, GA=2000)
),
trainer=dict(
distributed_parallelism="fsdp",
grad_accum_iter=1, # real run sets via TOML (GA=2)
logging_iter=1,
max_iter=100, # smoke
max_val_iter=None,
run_validation=False,
run_validation_on_start=False,
save_zero_checkpoint=False,
seed=42,
timeout_period=999999999,
validation_iter=100,
compile_config=dict(recompile_limit=8, use_duck_shape=False),
cudnn=dict(benchmark=True, deterministic=False),
ddp=dict(broadcast_buffers=True, find_unused_parameters=False, static_graph=True),
grad_scaler_args=dict(enabled=False),
callbacks=dict(
dataloader_speed=dict(every_n=100, save_s3=False, step_size=1),
device_monitor=dict(
every_n=200, log_memory_detail=True, save_s3=False, step_size=1, upload_every_n_mul=5
),
grad_clip=dict(clip_norm=1.0, force_finite=True),
heart_beat=dict(every_n=200, save_s3=False, step_size=1, update_interval_in_minute=20),
iter_speed=dict(every_n=1, hit_thres=50, save_s3=False, save_s3_every_log_n=500),
low_precision=dict(update_iter=1),
manual_gc=dict(every_n=5, gc_level=1, warm_up=1),
param_count=dict(save_s3=False),
skip_nan_step=dict(max_consecutive_nan=100),
training_stats=dict(log_freq=100),
),
),
checkpoint=dict(
broadcast_via_filesystem=False,
dcp_async_mode_enabled=False,
enable_gcs_patch_in_boto3=True,
keys_not_to_resume=[],
# Skip net_ema (EMA warm-starts from net, see dcp.py) and the action
# heads, so they init fresh from the base (the public Cosmos3-Edge base
# has no LIBERO-trained action heads).
keys_to_skip_loading=[
"net_ema.",
"action2llm",
"llm2action",
"action_modality_embed",
"action_pos_embed",
],
load_ema_to_reg=False,
load_path="???", # Cosmos3-Edge DCP dir; supply via TOML/env
load_training_state=False,
only_load_scheduler_state=False,
save_iter=100,
strict_resume=False, # base init: tolerate key set differences
verbose=True,
hf_export=dict(
enabled=False,
export_every_n=1,
hf_repo_id=None,
upload_to_object_store=dict(bucket="", credentials="", enabled=False),
),
jit=dict(device="cuda", dtype="bfloat16", enabled=False, input_shape=None, strict=True),
load_from_object_store=dict(bucket="", credentials="", enabled=False),
save_to_object_store=dict(bucket="", credentials="", enabled=False),
),
dataloader_train=L(PackingDataLoader)(
audio_sample_rate=48000,
dataset_name="action_libero",
max_samples_per_batch=128, # peak-mem bound (256 OOMs on H200); global = 128 x DP8 x grad_accum2 = 2048
max_sequence_length=None, # None disables token packing (TOML can't express null)
patch_spatial=2,
sound_latent_fps=0,
tokenizer_spatial_compression_factor=16,
tokenizer_temporal_compression_factor=4,
dataloader=L(RankPartitionedDataLoader)(
batch_size=1,
in_order=False,
num_workers=4,
persistent_workers=True,
pin_memory=True,
prefetch_factor=4,
sampler=None,
# Shuffling is handled by the dataset (iterable_shuffle=True below):
# ActionIterableShuffleDataset streams rank x worker-sharded, episode-order-
# shuffled, sequential-within-episode.
datasets=dict(
libero=dict(
ratio=1,
dataset=L(get_action_libero_sft_dataset)(
# Local LeRobot dir for the libero_10 suite ONLY. Use the
# 20 FPS nvidia/LIBERO_LeRobot_v3 (matches the bundled stats + 20 Hz eval):
# hf download nvidia/LIBERO_LeRobot_v3 --repo-type dataset \
# --include 'libero_10/**' --local-dir <dir> # LIBERO_ROOT=<dir>/libero_10
root="${oc.env:LIBERO_ROOT}",
fps=20, # metadata only (FPS-agnostic loader reads native fps from info.json)
chunk_length=16,
image_size=256, # concat_view -> 256x512
mode="wam",
camera_mode="concat_view",
action_space="frame_wise_relative",
rotation_space="6d",
pose_coordinate_frame="native",
action_normalization="quantile_rot",
val_ratio=0.01,
iterable_shuffle=True,
episode_shuffle_seed=42,
resolution=None,
max_action_dim="${model.config.max_action_dim}",
cfg_dropout_rate=0.1,
format_prompt_as_json=True, # structured JSON prompts (set False for plain-text)
tokenizer_config="${model.config.vlm_config.tokenizer}",
),
),
),
),
),
dataloader_val=None,
upload_reproducible_setup=False,
),
flags={"allow_objects": True},
)


for _item in [action_policy_libero_edge]:
_name = [k for k, v in globals().items() if v is _item][0]
cs.store(group="experiment", package="_global_", name=_name, node=_item)
43 changes: 43 additions & 0 deletions docs/action_policy_libero_posttrain.md
Original file line number Diff line number Diff line change
Expand Up @@ -146,3 +146,46 @@ Eval parity — the client/server already handle these; verify if accuracy is lo
rotates them back.
- **Normalization** — start the server with `--action-normalization quantile_rot`
and the bundled rot6d stats, or actions come out at the wrong scale.

## 5. Cosmos3-Edge variant

`action_policy_libero_edge` is the same libero_10 recipe on the public
`nvidia/Cosmos3-Edge` base (dense Nemotron-2B-Dense-VL backbone). It differs
from `action_policy_libero_nano` only in the model config (`EDGE_MODEL_CONFIG`)
and one extra trainable key, `k_norm_und_for_gen`, as in `vision_sft_edge`.

| Piece | Path |
| ---------- | ----------------------------------------------------------------------------------------------- |
| Experiment | `cosmos_framework/configs/base/experiment/action/posttrain_config/action_policy_libero_edge.py` |
| Run TOML | `examples/toml/sft_config/action_policy_libero_10_edge.toml` |
| Launch | `examples/launch_sft_action_policy_libero_10_edge.sh` |

```bash
python -m cosmos_framework.scripts.convert_model_to_dcp \
-o examples/checkpoints/Cosmos3-Edge \
--checkpoint-path Cosmos3-Edge

export BASE_CHECKPOINT_PATH=examples/checkpoints/Cosmos3-Edge
export LIBERO_ROOT=<nfs>/LIBERO_LeRobot_v3/libero_10
bash examples/launch_sft_action_policy_libero_10_edge.sh # HSDP 2x8, as the Nano preset A
```

**One 8-GPU node:** set `data_parallel_replicate_degree = 1` and
`grad_accum_iter = 2` in the TOML; the global batch stays 2048.

**Validated so far:** a 50-iteration single-node smoke run (8× RTX PRO 6000
Blackwell 96 GB, replicate 1, grad_accum 2, global batch 2048) trained at about
79 s/iteration; loss went from 15.3 (iteration 1) to 3.6 (iteration 50) and the
DCP checkpoint saved. A full 2000-iteration run and the closed-loop libero_10
success rate have not been measured for Edge yet.

**Outside the Docker image:** the data loader decodes LIBERO's AV1 videos with
`torchcodec`, which dlopens NPP and FFmpeg shared libraries that the container
provides on the system path. In a plain `uv sync` environment:

- NPP: add `.venv/lib/python3.13/site-packages/nvidia/cu13/lib` to `LD_LIBRARY_PATH`
(otherwise `libnppicc.so.13: cannot open shared object file`).
- FFmpeg with an AV1 decoder (`dav1d`): install system FFmpeg, or point
`LD_LIBRARY_PATH` at an FFmpeg 8 build that includes `libdav1d`. The FFmpeg
bundled in `opencv-python` has no AV1 decoder and fails with
`Could not push packet to decoder: Function not implemented`.
46 changes: 46 additions & 0 deletions examples/launch_sft_action_policy_libero_10_edge.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
#!/usr/bin/env bash
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: OpenMDW-1.1

# Structured-TOML launch for action_policy_libero_edge — Cosmos3-Edge LIBERO
# action-policy SFT (HSDP, full SFT). Drives cosmos_framework.scripts.train
# against examples/toml/sft_config/action_policy_libero_10_edge.toml.
#
# Point LIBERO_ROOT at the libero_10 suite ONLY. Use the 20 FPS
# nvidia/LIBERO_LeRobot_v3. The default recipe is HSDP 2x8 (global batch 2048);
# set NNODES/NODE_RANK/MASTER_ADDR per node.
# See docs/action_policy_libero_posttrain.md.
#
# Required env vars:
# LIBERO_ROOT local LIBERO-10 LeRobot dataset dir, e.g. <dir>/libero_10 (no default)
# Optional env vars (defaults below; override to relocate data/checkpoints):
# BASE_CHECKPOINT_PATH default: examples/checkpoints/Cosmos3-Edge
# WAN_VAE_PATH default: examples/checkpoints/wan22_vae/Wan2.2_VAE.pth
# HF_TOKEN if any tokenizer download requires gated HF access
# OUTPUT_ROOT default: outputs/train
#
# Pre-sync the 20 FPS suite once:
# hf download nvidia/LIBERO_LeRobot_v3 --repo-type dataset --include 'libero_10/**' --local-dir <dir>
# export LIBERO_ROOT=<dir>/libero_10
#
# Usage (HSDP 2x8; set NNODES/NODE_RANK/MASTER_ADDR per node):
# LIBERO_ROOT=<dir>/libero_10 bash examples/launch_sft_action_policy_libero_10_edge.sh

TOML_FILE="examples/toml/sft_config/action_policy_libero_10_edge.toml"
: "${BASE_CHECKPOINT_PATH:=examples/checkpoints/Cosmos3-Edge}"

# LIBEROLeRobotDataset reads ${oc.env:LIBERO_ROOT} directly (a LOCAL LeRobot dir);
# export it so torchrun (launched in this shell) inherits it.
export LIBERO_ROOT="${LIBERO_ROOT:-}"

EXTRA_DATASET_CHECK='[[ -f "$LIBERO_ROOT/meta/info.json" ]] || { echo "ERROR: LIBERO_ROOT must be a local LeRobot dir containing meta/info.json (got: '\''$LIBERO_ROOT'\''). Pre-sync: hf download nvidia/LIBERO_LeRobot_v3 --repo-type dataset --include '\''libero_10/**'\'' --local-dir <dir> (then LIBERO_ROOT=<dir>/libero_10). See docs/action_policy_libero_posttrain.md" >&2; exit 1; }'

# Extra Hydra overrides from the environment: a space-separated string word-split into
# the TAIL_OVERRIDES array. An exported string survives `bash <wrapper>` (a child
# process), unlike a TAIL_OVERRIDES array set in your shell. Use it for smoke runs,
# e.g. EXTRA_TAIL_OVERRIDES="trainer.max_iter=5 job.wandb_mode=offline".
TAIL_OVERRIDES=(
${EXTRA_TAIL_OVERRIDES:-}
)

source "$(dirname "${BASH_SOURCE[0]}")/_sft_launcher_common.sh"
47 changes: 47 additions & 0 deletions examples/toml/sft_config/action_policy_libero_10_edge.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: OpenMDW-1.1

# LIBERO-10 action-policy SFT run config for the `action_policy_libero_edge`
# experiment. Train on libero_10 alone (HSDP 2x8, global batch 2048).
# Env: LIBERO_ROOT, BASE_CHECKPOINT_PATH, WAN_VAE_PATH, IMAGINAIRE_OUTPUT_ROOT.
# See docs/action_policy_libero_sft.md.

[job]
task = "vfm"
experiment = "action_policy_libero_edge"
project = "cosmos3_action_libero"
group = "action_sft"
name = "action_policy_libero_10_edge"
wandb_mode = "online"

[model]
precision = "bfloat16"
max_num_tokens_after_packing = 74000

[model.parallelism]
data_parallel_shard_degree = 8
data_parallel_replicate_degree = 2 # HSDP 2x8 = 16 ranks (2 nodes); minimum for gbs 2048 at grad_accum 1
# One 8-GPU node: replicate 1 + [trainer] grad_accum_iter = 2 keeps gbs 2048

[model.activation_checkpointing]
mode = "selective"
save_ops_regex = ["fmha"]

[model.tokenizer]
vae_path = "${oc.env:WAN_VAE_PATH}"

[optimizer]
lr = 5.0e-05

[scheduler]
cycle_lengths = [16000]
warm_up_steps = [500]

[trainer]
max_iter = 2000
logging_iter = 50
grad_accum_iter = 1 # global batch = max_samples 128 x (shard 8 x replicate 2) x 1 = 2048

[checkpoint]
load_path = "${oc.env:BASE_CHECKPOINT_PATH}"
save_iter = 500