Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions doc/workloads/osu.rst
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,16 @@ Test-in-Scenario example:
iterations = 10
message_size = "1024"

Comparison Reports
------------------

The OSU comparison v2 report labels the X axis with the measured message sizes
in bytes. Sizes are displayed at equally spaced positions, including zero-byte
messages, rather than using default logarithmic ticks.
When runs measure different size ranges, charts and tables use the sorted union
of measured sizes. Missing measurements appear as chart gaps and ``n/a`` table
cells.

API Documentation
-----------------

Expand Down
9 changes: 9 additions & 0 deletions src/cloudai/workloads/osu_bench/osu_comparison_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,6 +69,12 @@ def build_sections(self, cmp_groups: list[GroupedTestRuns]) -> list[ComparisonSe
sections: list[ComparisonSection] = []
for group in cmp_groups:
dfs = [self.extract_data_as_df(item.tr) for item in group.items]
if not dfs or any(df.empty for df in dfs):
continue

# Keep chart and table rows aligned across different message-size ranges.
sizes = sorted({size for df in dfs for size in df["size"]})
dfs = [df.set_index("size").reindex(sizes).rename_axis("size").reset_index() for df in dfs]
Comment thread
coderabbitai[bot] marked this conversation as resolved.

if self._has_metric(dfs, "avg_lat"):
sections.append(
Expand All @@ -79,6 +85,7 @@ def build_sections(self, cmp_groups: list[GroupedTestRuns]) -> list[ComparisonSe
info_columns=list(self.INFO_COLUMNS),
data_columns=["avg_lat"],
y_axis_label="Time (us)",
x_axis_type="indexed_category",
Comment thread
coderabbitai[bot] marked this conversation as resolved.
)
)
if self._has_metric(dfs, "mb_sec"):
Expand All @@ -90,6 +97,7 @@ def build_sections(self, cmp_groups: list[GroupedTestRuns]) -> list[ComparisonSe
info_columns=list(self.INFO_COLUMNS),
data_columns=["mb_sec"],
y_axis_label="Bandwidth (MB/s)",
x_axis_type="indexed_category",
)
)
if self._has_metric(dfs, "messages_sec"):
Expand All @@ -101,6 +109,7 @@ def build_sections(self, cmp_groups: list[GroupedTestRuns]) -> list[ComparisonSe
info_columns=list(self.INFO_COLUMNS),
data_columns=["messages_sec"],
y_axis_label="Messages/s",
x_axis_type="indexed_category",
)
)

Expand Down
121 changes: 121 additions & 0 deletions tests/report_generation_strategy/test_osu_comparison_report.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,121 @@
# SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES
# Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

from pathlib import Path
from unittest.mock import Mock

import pandas as pd
import pytest

from cloudai.core import TestRun, TestScenario
from cloudai.report_generator.comparison_report import ComparisonReportConfig
from cloudai.report_generator.groups import GroupedTestRuns, TRGroupItem
from cloudai.systems.slurm import SlurmSystem
from cloudai.workloads.osu_bench.osu_comparison_report import OSUBenchComparisonReport


@pytest.mark.parametrize("metric", ["avg_lat", "mb_sec", "messages_sec"])
@pytest.mark.parametrize("sizes", [[0, 1, 2, 4, 8], [2, 4, 8, 1024, 4194304]])
def test_v2_axis_labels_match_measured_sizes(
metric: str, sizes: list[int], tmp_path: Path, slurm_system: SlurmSystem
) -> None:
values = [float(index + 1) for index in range(len(sizes))]
pd.DataFrame({"size": sizes, metric: values}).to_csv(tmp_path / "osu_bench.csv", index=False)
tr = TestRun(name="osu", test=Mock(), num_nodes=2, nodes=[], output_path=tmp_path)
report = OSUBenchComparisonReport(
slurm_system,
TestScenario(name="osu", test_runs=[]),
tmp_path,
ComparisonReportConfig(enable=True, group_by=[]),
)
group = GroupedTestRuns(name="all-in-one", items=[TRGroupItem(name="case-a", tr=tr)])

sections = report.build_sections([group])
assert len(sections) == 1
chart = report._build_sections_v2(sections)[0]["chart"]

assert chart["x_axis_type"] == "indexed_category"
assert chart["labels"] == [str(size) for size in sizes]
assert chart["datasets"][0]["data"] == values


@pytest.mark.parametrize(
("metric", "reverse_runs"),
[("avg_lat", False), ("mb_sec", False), ("messages_sec", False), ("mb_sec", True)],
)
def test_comparison_aligns_different_size_ranges(
metric: str, reverse_runs: bool, tmp_path: Path, slurm_system: SlurmSystem
) -> None:
runs = [
("small", [1, 2, 4], [10.0, 20.0, 40.0]),
("large", [2, 4, 8], [200.0, 400.0, 800.0]),
]
if reverse_runs:
runs.reverse()
items = []
for name, sizes, values in runs:
output = tmp_path / name
output.mkdir()
pd.DataFrame({"size": sizes, metric: values}).to_csv(output / "osu_bench.csv", index=False)
tr = TestRun(name=name, test=Mock(), num_nodes=2, nodes=[], output_path=output)
items.append(TRGroupItem(name=name, tr=tr))
report = OSUBenchComparisonReport(
slurm_system,
TestScenario(name="osu", test_runs=[]),
tmp_path,
ComparisonReportConfig(enable=True, group_by=[]),
)

sections = report.build_sections([GroupedTestRuns(name="all-in-one", items=items)])
assert len(sections) == 1
section = sections[0]
payload = report._build_sections_v2(sections)[0]

assert payload["chart"]["labels"] == ["1", "2", "4", "8"]
assert {dataset["label"]: dataset["data"] for dataset in payload["chart"]["datasets"]} == {
"small": [10.0, 20.0, 40.0, None],
"large": [None, 200.0, 400.0, 800.0],
}
for df in section.dfs:
assert df["size"].tolist() == [1, 2, 4, 8]
expected_values = {
"small": ["10.0", "20.0", "40.0", "n/a"],
"large": ["n/a", "200.0", "400.0", "800.0"],
}
for column, item in enumerate(items):
assert [row["data_cells"][column] for row in payload["table"]["rows"]] == expected_values[item.name]


@pytest.mark.parametrize("missing_output", [False, True])
def test_comparison_skips_empty_run(missing_output: bool, tmp_path: Path, slurm_system: SlurmSystem) -> None:
items = []
for name in ["valid", "empty"]:
output = tmp_path / name
output.mkdir()
if name == "valid":
pd.DataFrame({"size": [1, 2], "mb_sec": [10.0, 20.0]}).to_csv(output / "osu_bench.csv", index=False)
elif not missing_output:
pd.DataFrame(columns=["size", "mb_sec"]).to_csv(output / "osu_bench.csv", index=False)
tr = TestRun(name=name, test=Mock(), num_nodes=2, nodes=[], output_path=output)
items.append(TRGroupItem(name=name, tr=tr))
report = OSUBenchComparisonReport(
slurm_system,
TestScenario(name="osu", test_runs=[]),
tmp_path,
ComparisonReportConfig(enable=True, group_by=[]),
)

assert report.build_sections([GroupedTestRuns(name="all-in-one", items=items)]) == []
Loading