Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
17 commits
Select commit Hold shift + click to select a range
1890758
build: declare training dependencies and optional storage backends
ramanathan831 Oct 1, 2026
3b56fb8
build(docker): support immutable non-root training images
ramanathan831 Oct 1, 2026
663d8cf
fix(runtime): restore public exports and respect process resources
ramanathan831 Oct 1, 2026
54ae603
feat(data): add resumable video batching and worker prefetch
ramanathan831 Oct 1, 2026
4474f70
fix(video): preserve shared metadata, frame clocks, and modality
ramanathan831 Oct 1, 2026
1618fd8
fix(edge): support padded mRoPE batches and available attention kernels
ramanathan831 Oct 1, 2026
ac0ad64
fix(kv-cache): preserve FP8 decoding across CUDA architectures
ramanathan831 Oct 1, 2026
daaf089
feat(reasoner): integrate LoRA training and Qwen patch embedding
ramanathan831 Oct 1, 2026
5d18a8c
perf(reasoner): optimize vision attention and repeated-video validation
ramanathan831 Oct 1, 2026
f80de9e
feat(training): add epoch scheduling and structured lifecycle reporting
ramanathan831 Oct 1, 2026
e2c40f4
feat(sft): add dataset-neutral Nano and Edge video recipes
ramanathan831 Oct 1, 2026
76380f5
feat(training): add opt-in gradient-spike rollback and healthy LR rec…
ramanathan831 Oct 1, 2026
9f655f8
feat(export): add VLM checkpoint export with worktree-aware provenance
ramanathan831 Oct 1, 2026
0ec0e93
ci: build the exact pinned formatter before running pre-commit
ramanathan831 Oct 1, 2026
434d281
refactor(training): consolidate native video SFT components
ramanathan831 Oct 1, 2026
1898157
docs(examples): generalize video SFT recipes and launcher
ramanathan831 Oct 1, 2026
e4481fc
fix(build): distinguish default builds from verified source images
ramanathan831 Oct 1, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .dockerignore
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
.venv
.git
**/__pycache__
**/*.pyc
/checkpoints
/datasets
/output
Expand Down
18 changes: 18 additions & 0 deletions .github/workflows/pre-commit.yml
Original file line number Diff line number Diff line change
Expand Up @@ -19,4 +19,22 @@ jobs:
- uses: actions/setup-python@v6
- uses: astral-sh/setup-uv@v7
- run: uvx pre-commit@4.5.1 run -a -c ci/.pre-commit-config-base.yaml
# The pinned hook requires rumdl 0.1.62, which has no PyPI distribution.
# Build the genuine release without changing the hook or formatter version.
- name: Build pinned rumdl wheel
env:
RUSTUP_HOME: ${{ runner.temp }}/rumdl-rustup
CARGO_HOME: ${{ runner.temp }}/rumdl-cargo
RUSTUP_TOOLCHAIN: '1.94.0'
run: |
git init "$RUNNER_TEMP/rumdl-source"
git -C "$RUNNER_TEMP/rumdl-source" fetch --depth=1 https://github.com/rvben/rumdl.git 8e22c9b16f49c7209355106f731500cc1aacef20
git -C "$RUNNER_TEMP/rumdl-source" checkout --detach FETCH_HEAD
rustup toolchain install "$RUSTUP_TOOLCHAIN" --profile minimal --no-self-update
uvx maturin@1.15.0 build --release --locked \
--manifest-path "$RUNNER_TEMP/rumdl-source/Cargo.toml" \
--out "$RUNNER_TEMP/rumdl-wheels"
uvx --no-index --find-links "$RUNNER_TEMP/rumdl-wheels" rumdl@0.1.62 --version
- run: uvx pre-commit@4.5.1 run -a
env:
PIP_FIND_LINKS: ${{ runner.temp }}/rumdl-wheels
39 changes: 36 additions & 3 deletions Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,25 @@ ARG CUDA_VERSION=13.0.2
ARG BASE_IMAGE=nvidia/cuda:${CUDA_VERSION}-cudnn-devel-ubuntu24.04
FROM ${BASE_IMAGE}

ARG SOURCE_COMMIT
ARG SOURCE_TREE
ARG SOURCE_DIRTY=1
ARG BUILD_TIMESTAMP
ARG REQUIRE_SOURCE_PROVENANCE=0
ARG BASE_IMAGE
ARG CUDA_VERSION
LABEL org.opencontainers.image.revision="${SOURCE_COMMIT}" \
org.opencontainers.image.created="${BUILD_TIMESTAMP}" \
com.nvidia.cosmos.source-tree="${SOURCE_TREE}" \
com.nvidia.cosmos.backend="cosmos-framework"
ENV SOURCE_COMMIT="${SOURCE_COMMIT}" \
SOURCE_TREE="${SOURCE_TREE}" \
SOURCE_DIRTY="${SOURCE_DIRTY}" \
BUILD_TIMESTAMP="${BUILD_TIMESTAMP}" \
REQUIRE_SOURCE_PROVENANCE="${REQUIRE_SOURCE_PROVENANCE}" \
PROVENANCE_BASE_IMAGE="${BASE_IMAGE}" \
CUDA_VERSION="${CUDA_VERSION}"

# Set the DEBIAN_FRONTEND environment variable to avoid interactive prompts during apt operations.
ENV DEBIAN_FRONTEND=noninteractive

Expand All @@ -28,7 +47,8 @@ COPY --from=ghcr.io/astral-sh/uv:0.12.2 /uv /uvx /usr/local/bin/
# Copy from the cache instead of linking since it's a mounted volume
ENV UV_LINK_MODE=copy
# Cache python downloads
ENV UV_PYTHON_CACHE_DIR=/root/.cache/uv/python
ENV UV_PYTHON_CACHE_DIR=/opt/uv-python-cache \
UV_PYTHON_INSTALL_DIR=/opt/uv-python

# Install just: https://just.systems/man/en/pre-built-binaries.html
RUN curl --proto '=https' --tlsv1.2 -sSf https://just.systems/install.sh | bash -s -- --to /usr/local/bin --tag 1.46.0
Expand All @@ -40,7 +60,8 @@ WORKDIR /workspace
# Install python
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=type=bind,source=.python-version,target=.python-version \
uv python install
uv python install && \
chmod -R a+rX /opt/uv-python /opt/uv-python-cache

# Install into virtual environment
RUN echo "$CUDA_VERSION" | sed -E 's/^([0-9]+)\.([0-9]+).*/cu\1\2/' > /root/.cuda-name
Expand All @@ -49,7 +70,7 @@ RUN --mount=type=cache,target=/root/.cache/uv \
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
--mount=type=bind,source=.python-version,target=.python-version \
--mount=type=bind,source=packages,target=packages \
uv sync --locked --no-install-project --no-editable --all-extras --group=$(cat /root/.cuda-name) --group=vllm
uv sync --locked --no-install-project --no-editable --all-extras --group=$(cat /root/.cuda-name)-train
ENV PATH="/workspace/.venv/bin:$PATH"

# Set to 0 to skip the apex build, which is by far the slowest layer. apex is optional:
Expand All @@ -66,6 +87,18 @@ RUN --mount=type=cache,target=/root/.cache/uv \
echo "INSTALL_APEX=$INSTALL_APEX, skipping apex"; \
fi

# Package the exact source state after the expensive dependency layers so source
# edits do not rebuild apex. Managed platforms do not require a host bind mount.
COPY . /workspace
RUN --mount=type=cache,target=/root/.cache/uv \
uv pip install --no-deps .

RUN /workspace/.venv/bin/python /workspace/docker/write_image_provenance.py && \
chmod a+rx /workspace /workspace/docker /workspace/docker/entrypoint.sh && \
chmod -R a+rX /opt/cosmos /workspace/.venv /workspace/cosmos_framework /workspace && \
test -x /workspace/docker/entrypoint.sh && \
test -x /workspace/.venv/bin/python

# Triton bundled ptxas doesn't support latest GPU architectures
ENV TRITON_PTXAS_PATH="/usr/local/cuda/bin/ptxas"

Expand Down
2 changes: 1 addition & 1 deletion cosmos_framework/callbacks/iter_speed.py
Original file line number Diff line number Diff line change
Expand Up @@ -114,7 +114,7 @@ def on_training_step_end(
) -> None:
if self.hit_counter < self.hit_thres:
log.info(
f"Iteration {iteration}: "
f"[RANK {log.RANK}] Iteration {iteration}: "
f"Hit counter: {self.hit_counter + 1}/{self.hit_thres} | "
f"Loss: {loss.detach().item():.4f} | "
f"Time: {time.time() - self.last_hit_time:.2f}s",
Expand Down
Loading