From 3bd24d15b9576a1153154f2c9350242de93a2693 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Wed, 7 Oct 2026 18:42:45 +0200 Subject: [PATCH 1/5] [JUnit] Preserve reports on job submission failure Signed-off-by: Ivan Podkidyshev --- doc/reporting.rst | 8 ++++- src/cloudai/_core/base_runner.py | 8 ++++- src/cloudai/configurator/cloudai_gym.py | 4 ++- src/cloudai/core.py | 2 ++ src/cloudai/reporter.py | 23 +++++++++---- tests/test_base_runner.py | 19 +++++++++++ tests/test_cloudaigym.py | 18 ++++++++++- tests/test_handlers.py | 36 +++++++++++++++++++++ tests/test_reporter.py | 43 ++++++++++++++++++++++++- 9 files changed, 149 insertions(+), 12 deletions(-) diff --git a/doc/reporting.rst b/doc/reporting.rst index 30ad69743..4cff00b45 100644 --- a/doc/reporting.rst +++ b/doc/reporting.rst @@ -104,10 +104,16 @@ Enabling or disabling a report needs to be done in the system configuration: junit = { enable = true } The ``junit`` scenario reporter is disabled by default. When enabled, it writes ``junit.xml`` in the scenario results -directory. It emits one test case for every regular test iteration and every DSE step, including pass/fail status, +directory. It emits one test case for each existing regular test iteration and DSE step, including pass/fail status, failure details, scheduler duration when available, and the contents of ``stdout.txt`` and ``stderr.txt``. The artifact can be consumed directly by Jenkins, GitLab, GitHub Actions, and other CI systems that support JUnit XML. +If job submission fails, CloudAI generates enabled reports before exiting with a nonzero status. +The JUnit report preserves completed executions and records the failed submission as an ``error``, including +its exception type and submission details. Reports include only existing test-run directories; unattempted +iterations and DSE steps are absent. The failed directory retains ``submission-error.txt`` so subsequent +``generate-report`` calls preserve the submission error. + Speed-of-Light comparisons -------------------------- diff --git a/src/cloudai/_core/base_runner.py b/src/cloudai/_core/base_runner.py index 0a75918c7..16bde18c8 100644 --- a/src/cloudai/_core/base_runner.py +++ b/src/cloudai/_core/base_runner.py @@ -54,6 +54,8 @@ class BaseRunner(ABC): new tests and ensuring a graceful termination of all running tests. """ + SUBMISSION_ERROR_FILE_NAME = "submission-error.txt" + def __init__(self, mode: str, system: System, test_scenario: TestScenario, output_path: Path): """ Initialize the BaseRunner with a system object, test scenario, and monitor interval. @@ -162,7 +164,11 @@ def submit_test(self, tr: TestRun): self.update_run_output(job) except JobSubmissionError as e: logging.error(e) - exit(1) + try: + (tr.output_path / self.SUBMISSION_ERROR_FILE_NAME).write_text(f"{type(e).__name__}: {e}") + except OSError: + logging.exception("Failed to persist submission error for %s", tr.name) + raise def on_job_submit(self, tr: TestRun) -> None: return diff --git a/src/cloudai/configurator/cloudai_gym.py b/src/cloudai/configurator/cloudai_gym.py index a472da9b5..f818278e8 100644 --- a/src/cloudai/configurator/cloudai_gym.py +++ b/src/cloudai/configurator/cloudai_gym.py @@ -20,7 +20,7 @@ from pathlib import Path from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple, cast -from cloudai.core import METRIC_ERROR, BaseRunner, Registry, TestRun +from cloudai.core import METRIC_ERROR, BaseRunner, JobSubmissionError, Registry, TestRun from cloudai.util.lazy_imports import lazy from .base_agent import RewardOverrides @@ -195,6 +195,8 @@ def step(self, action: Any) -> Tuple[list, float, bool, dict]: try: self.runner.run() + except JobSubmissionError: + raise except Exception as e: logging.error(f"Error running step {self.test_run.step}: {e}") diff --git a/src/cloudai/core.py b/src/cloudai/core.py index 2a6e1dc58..55c251afe 100644 --- a/src/cloudai/core.py +++ b/src/cloudai/core.py @@ -25,6 +25,7 @@ from ._core.exceptions import ( JobFailureError, JobIdRetrievalError, + JobSubmissionError, MissingTestError, SystemConfigParsingError, TestConfigParsingError, @@ -104,6 +105,7 @@ "JobFailureError", "JobIdRetrievalError", "JobStatusResult", + "JobSubmissionError", "JsonGenStrategy", "MetricErrorSentinel", "MetricValue", diff --git a/src/cloudai/reporter.py b/src/cloudai/reporter.py index 918fe26fb..a21f6f428 100644 --- a/src/cloudai/reporter.py +++ b/src/cloudai/reporter.py @@ -31,7 +31,7 @@ from cloudai.report_generator.dse_report import build_dse_summaries, load_trajectory_dataframe from cloudai.report_generator.util import load_system_metadata -from .core import CommandGenStrategy, Reporter, TestRun, case_name +from .core import BaseRunner, CommandGenStrategy, Reporter, TestRun, case_name from .models.scenario import TestRunDetails @@ -148,15 +148,21 @@ class JUnitReporter(Reporter): def generate(self) -> None: self.load_test_runs() - results = [(tr, tr.test.was_run_successful(tr), self._duration(tr.output_path)) for tr in self.trs] - failures = sum(not status.is_successful for _, status, _ in results) - durations = [duration for _, _, duration in results if duration is not None] + results = [] + for tr in self.trs: + error_path = tr.output_path / BaseRunner.SUBMISSION_ERROR_FILE_NAME + error = error_path.read_text() if error_path.is_file() else None + status = tr.test.was_run_successful(tr) if error is None else None + results.append((tr, status, self._duration(tr.output_path), error)) + failures = sum(status is not None and not status.is_successful for _, status, _, _ in results) + errors = sum(error is not None for _, _, _, error in results) + durations = [duration for _, _, duration, _ in results if duration is not None] suite_attributes = { "name": self.test_scenario.name, "tests": str(len(results)), "failures": str(failures), - "errors": "0", + "errors": str(errors), "skipped": "0", } if durations: @@ -164,13 +170,16 @@ def generate(self) -> None: root = ET.Element("testsuites", suite_attributes) suite = ET.SubElement(root, "testsuite", suite_attributes) - for tr, status, duration in results: + for tr, status, duration, error in results: attributes = {"name": case_name(tr), "classname": self.test_scenario.name} if duration is not None: attributes["time"] = self._format_duration(duration) testcase = ET.SubElement(suite, "testcase", attributes) - if not status.is_successful: + if error is not None: + error_element = ET.SubElement(testcase, "error", {"message": self._xml_text(error)}) + error_element.text = self._xml_text(error) + elif status is not None and not status.is_successful: message = status.error_message or "Test run failed" failure = ET.SubElement(testcase, "failure", {"message": self._xml_text(message)}) failure.text = self._xml_text(message) diff --git a/tests/test_base_runner.py b/tests/test_base_runner.py index 2a671a851..9f3137411 100644 --- a/tests/test_base_runner.py +++ b/tests/test_base_runner.py @@ -24,6 +24,7 @@ from cloudai.core import ( BaseJob, BaseRunner, + JobIdRetrievalError, JobStatusResult, System, TestDefinition, @@ -159,3 +160,21 @@ def test_end_post_comp(self, runner: MyRunner, tr_main: TestRun): runner.handle_dependencies(BaseJob(tr_main, 0)) assert len(runner.killed_by_dependency) == 1 assert runner.killed_by_dependency[0].test_run == tr_dep + + +def test_submission_error_is_preserved(runner: MyRunner, monkeypatch: pytest.MonkeyPatch) -> None: + error = JobIdRetrievalError("tr-name", "sbatch test.sh", "", "submission timeout", "No job ID") + + def submit(tr: TestRun) -> BaseJob: + raise error + + monkeypatch.setattr(runner, "_submit_test", submit) + tr = runner.test_scenario.test_runs[0] + with pytest.raises(JobIdRetrievalError) as caught: + runner.submit_test(tr) + assert caught.value is error + assert runner.jobs == [] + persisted = (tr.output_path / runner.SUBMISSION_ERROR_FILE_NAME).read_text() + assert "JobIdRetrievalError" in persisted + assert "submission timeout" in persisted + assert "sbatch test.sh" in persisted diff --git a/tests/test_cloudaigym.py b/tests/test_cloudaigym.py index a7aa60720..164af22b0 100644 --- a/tests/test_cloudaigym.py +++ b/tests/test_cloudaigym.py @@ -27,7 +27,7 @@ Trajectory, ) from cloudai.configurator.env_params import EnvParamSpec, ObsLeafDescriptor -from cloudai.core import BaseRunner, RewardOverrides, Runner, TestRun, TestScenario +from cloudai.core import BaseRunner, JobIdRetrievalError, RewardOverrides, Runner, TestRun, TestScenario from cloudai.systems.slurm import SlurmRunner, SlurmSystem from cloudai.util import flatten_dict from cloudai.workloads.nemo_run import ( @@ -934,3 +934,19 @@ def test_reset_reports_the_regime_step_will_apply_on_info(self, tmp_path: Path) assert obs == env.define_observation_space(), "reset's flat obs stays the metrics placeholder" assert info["env_params"] == {"ball_speed": upcoming}, "reset peeks step+1 and reports the regime" assert env.encode_env_params(info["env_params"]) == {"ball_speed": [1, 2, 3].index(upcoming)} + + +def test_step_propagates_submission_error(setup_env: tuple[TestRun, BaseRunner], monkeypatch: pytest.MonkeyPatch): + test_run, runner = setup_env + test_run.test.cmd_args.data.global_batch_size = 8 + env = CloudAIGymEnv(test_run=test_run, runner=runner, rewards=RewardOverrides()) + agent = GridSearchAgent(env, GridSearchAgent.get_config_class()()) + _, action = agent.select_action() + error = JobIdRetrievalError(test_run.name, "sbatch test.sh", "", "timeout", "No job ID") + monkeypatch.setattr(runner, "run", MagicMock(side_effect=error)) + observation = MagicMock() + monkeypatch.setattr(env, "get_observation", observation) + with pytest.raises(JobIdRetrievalError) as caught: + env.step(action) + assert caught.value is error + observation.assert_not_called() diff --git a/tests/test_handlers.py b/tests/test_handlers.py index 87a21513e..020ffc6f2 100644 --- a/tests/test_handlers.py +++ b/tests/test_handlers.py @@ -32,6 +32,8 @@ BaseAgentConfig, GitRepo, InstallStatusResult, + JobIdRetrievalError, + JobStatusResult, Parser, Registry, RewardOverrides, @@ -761,3 +763,37 @@ def test_verify_test_scenarios_allows_env_params_with_dse( good = TestScenario(name="s", test_runs=[dse_tr]) monkeypatch.setattr(Parser, "parse_test_scenario", lambda *a, **k: good) assert verify_test_scenarios([Path("dummy.toml")], [], [], []) == 0 + + +@pytest.mark.parametrize("dse", [True, False]) +def test_submission_failure_generates_reports_before_propagating( + slurm_system: SlurmSystem, + base_tr: TestRun, + dse_tr: TestRun, + custom_run_agent_name: str, + monkeypatch: pytest.MonkeyPatch, + dse: bool, +) -> None: + tr = dse_tr if dse else base_tr + error = JobIdRetrievalError(tr.name, "sbatch test.sh", "", "timeout", "No job ID") + tr.test.agent = custom_run_agent_name + scenario = TestScenario(name="scenario", test_runs=[tr]) + runner = Runner(mode="run", system=slurm_system, test_scenario=scenario, runner_class=SlurmRunner) + + def check_final_output(*args) -> None: + output = runner.runner.experiment_output.snapshot() + assert output.status == "failed" + assert output.finish is not None + assert runner.runner.test_scenario.test_runs == [tr] + + reports = MagicMock(side_effect=check_final_output) + monkeypatch.setattr("cloudai.handlers.generate_reports", reports) + monkeypatch.setattr("cloudai.handlers._ensure_installation", MagicMock()) + if dse: + CustomRunStubAgent.run_raises = error + else: + monkeypatch.setattr(runner, "run", MagicMock(side_effect=error)) + with pytest.raises(JobIdRetrievalError) as caught: + execute_experiment(runner, []) + assert caught.value is error + reports.assert_called_once_with(slurm_system, scenario, runner.runner.scenario_root) diff --git a/tests/test_reporter.py b/tests/test_reporter.py index d3bdad65e..46a521009 100644 --- a/tests/test_reporter.py +++ b/tests/test_reporter.py @@ -26,7 +26,7 @@ import toml from cloudai import TestRun, TestScenario -from cloudai.core import CommandGenStrategy, Registry, Reporter, System +from cloudai.core import BaseRunner, CommandGenStrategy, JobStatusResult, Registry, Reporter, System from cloudai.handlers import generate_reports from cloudai.models.scenario import ReportConfig, TestRunDetails from cloudai.report_generator.dse_report import build_dse_summaries @@ -688,3 +688,44 @@ def test_dse_reporter( assert (slurm_system.output_path / "single-dse-scenario-dse-report.html").exists() assert (slurm_system.output_path / dse_case.name / "0" / f"{dse_case.name}.toml").exists() + + +@pytest.mark.parametrize("dse", [False, True]) +def test_junit_submission_error_keeps_completed_runs_only( + slurm_system: SlurmSystem, + benchmark_tr: TestRun, + dse_tr: TestRun, + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + dse: bool, +) -> None: + tr = dse_tr if dse else benchmark_tr + tr.iterations = 1 if dse else 32 + tr.test.agent_steps = 32 if dse else tr.test.agent_steps + run_dirs = [tmp_path / tr.name / "0" / str(i + 1) if dse else tmp_path / tr.name / str(i) for i in range(12)] + for run_dir in run_dirs: + run_dir.mkdir(parents=True) + failed_dir = run_dirs[-1] + (failed_dir / BaseRunner.SUBMISSION_ERROR_FILE_NAME).write_text("JobIdRetrievalError: submission timeout") + checked = [] + + def status(self, tr): + checked.append(tr.output_path) + return JobStatusResult(True) + + monkeypatch.setattr(type(tr.test), "was_run_successful", status) + reporter = JUnitReporter(slurm_system, TestScenario(name="scenario", test_runs=[tr]), tmp_path, ReportConfig()) + reporter.generate() + suite = ET.parse(tmp_path / "junit.xml").getroot().find("testsuite") + assert suite is not None + assert suite.attrib == {"name": "scenario", "tests": "12", "failures": "0", "errors": "1", "skipped": "0"} + cases = suite.findall("testcase") + assert len(cases) == 12 + assert all(case.find("error") is None and case.find("failure") is None for case in cases[:11]) + error = cases[11].find("error") + assert error is not None + assert "JobIdRetrievalError" in error.attrib["message"] + assert error.text is not None + assert "submission timeout" in error.text + assert len(checked) == 11 + assert failed_dir not in checked From 23b0afa6860e26fce884d5417985262628269e2e Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Wed, 7 Oct 2026 20:24:18 +0200 Subject: [PATCH 2/5] docs: remove submission failure reporting paragraph Signed-off-by: Ivan Podkidyshev --- doc/reporting.rst | 6 ------ 1 file changed, 6 deletions(-) diff --git a/doc/reporting.rst b/doc/reporting.rst index 4cff00b45..0bfdba85e 100644 --- a/doc/reporting.rst +++ b/doc/reporting.rst @@ -108,12 +108,6 @@ directory. It emits one test case for each existing regular test iteration and D failure details, scheduler duration when available, and the contents of ``stdout.txt`` and ``stderr.txt``. The artifact can be consumed directly by Jenkins, GitLab, GitHub Actions, and other CI systems that support JUnit XML. -If job submission fails, CloudAI generates enabled reports before exiting with a nonzero status. -The JUnit report preserves completed executions and records the failed submission as an ``error``, including -its exception type and submission details. Reports include only existing test-run directories; unattempted -iterations and DSE steps are absent. The failed directory retains ``submission-error.txt`` so subsequent -``generate-report`` calls preserve the submission error. - Speed-of-Light comparisons -------------------------- From 74e15d2d4ec50b9c4af746120284eeef7584b4e9 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Wed, 7 Oct 2026 20:33:36 +0200 Subject: [PATCH 3/5] test: remove completed-run submission error accounting test Signed-off-by: Ivan Podkidyshev --- tests/test_reporter.py | 43 +----------------------------------------- 1 file changed, 1 insertion(+), 42 deletions(-) diff --git a/tests/test_reporter.py b/tests/test_reporter.py index 46a521009..d3bdad65e 100644 --- a/tests/test_reporter.py +++ b/tests/test_reporter.py @@ -26,7 +26,7 @@ import toml from cloudai import TestRun, TestScenario -from cloudai.core import BaseRunner, CommandGenStrategy, JobStatusResult, Registry, Reporter, System +from cloudai.core import CommandGenStrategy, Registry, Reporter, System from cloudai.handlers import generate_reports from cloudai.models.scenario import ReportConfig, TestRunDetails from cloudai.report_generator.dse_report import build_dse_summaries @@ -688,44 +688,3 @@ def test_dse_reporter( assert (slurm_system.output_path / "single-dse-scenario-dse-report.html").exists() assert (slurm_system.output_path / dse_case.name / "0" / f"{dse_case.name}.toml").exists() - - -@pytest.mark.parametrize("dse", [False, True]) -def test_junit_submission_error_keeps_completed_runs_only( - slurm_system: SlurmSystem, - benchmark_tr: TestRun, - dse_tr: TestRun, - tmp_path: Path, - monkeypatch: pytest.MonkeyPatch, - dse: bool, -) -> None: - tr = dse_tr if dse else benchmark_tr - tr.iterations = 1 if dse else 32 - tr.test.agent_steps = 32 if dse else tr.test.agent_steps - run_dirs = [tmp_path / tr.name / "0" / str(i + 1) if dse else tmp_path / tr.name / str(i) for i in range(12)] - for run_dir in run_dirs: - run_dir.mkdir(parents=True) - failed_dir = run_dirs[-1] - (failed_dir / BaseRunner.SUBMISSION_ERROR_FILE_NAME).write_text("JobIdRetrievalError: submission timeout") - checked = [] - - def status(self, tr): - checked.append(tr.output_path) - return JobStatusResult(True) - - monkeypatch.setattr(type(tr.test), "was_run_successful", status) - reporter = JUnitReporter(slurm_system, TestScenario(name="scenario", test_runs=[tr]), tmp_path, ReportConfig()) - reporter.generate() - suite = ET.parse(tmp_path / "junit.xml").getroot().find("testsuite") - assert suite is not None - assert suite.attrib == {"name": "scenario", "tests": "12", "failures": "0", "errors": "1", "skipped": "0"} - cases = suite.findall("testcase") - assert len(cases) == 12 - assert all(case.find("error") is None and case.find("failure") is None for case in cases[:11]) - error = cases[11].find("error") - assert error is not None - assert "JobIdRetrievalError" in error.attrib["message"] - assert error.text is not None - assert "submission timeout" in error.text - assert len(checked) == 11 - assert failed_dir not in checked From 7eb96cd1505b8790003e20243cee7694d4dbd602 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Wed, 7 Oct 2026 20:36:59 +0200 Subject: [PATCH 4/5] test: remove redundant submission failure finalization coverage Signed-off-by: Ivan Podkidyshev --- tests/test_handlers.py | 35 ----------------------------------- 1 file changed, 35 deletions(-) diff --git a/tests/test_handlers.py b/tests/test_handlers.py index 020ffc6f2..c0a823b23 100644 --- a/tests/test_handlers.py +++ b/tests/test_handlers.py @@ -32,7 +32,6 @@ BaseAgentConfig, GitRepo, InstallStatusResult, - JobIdRetrievalError, JobStatusResult, Parser, Registry, @@ -763,37 +762,3 @@ def test_verify_test_scenarios_allows_env_params_with_dse( good = TestScenario(name="s", test_runs=[dse_tr]) monkeypatch.setattr(Parser, "parse_test_scenario", lambda *a, **k: good) assert verify_test_scenarios([Path("dummy.toml")], [], [], []) == 0 - - -@pytest.mark.parametrize("dse", [True, False]) -def test_submission_failure_generates_reports_before_propagating( - slurm_system: SlurmSystem, - base_tr: TestRun, - dse_tr: TestRun, - custom_run_agent_name: str, - monkeypatch: pytest.MonkeyPatch, - dse: bool, -) -> None: - tr = dse_tr if dse else base_tr - error = JobIdRetrievalError(tr.name, "sbatch test.sh", "", "timeout", "No job ID") - tr.test.agent = custom_run_agent_name - scenario = TestScenario(name="scenario", test_runs=[tr]) - runner = Runner(mode="run", system=slurm_system, test_scenario=scenario, runner_class=SlurmRunner) - - def check_final_output(*args) -> None: - output = runner.runner.experiment_output.snapshot() - assert output.status == "failed" - assert output.finish is not None - assert runner.runner.test_scenario.test_runs == [tr] - - reports = MagicMock(side_effect=check_final_output) - monkeypatch.setattr("cloudai.handlers.generate_reports", reports) - monkeypatch.setattr("cloudai.handlers._ensure_installation", MagicMock()) - if dse: - CustomRunStubAgent.run_raises = error - else: - monkeypatch.setattr(runner, "run", MagicMock(side_effect=error)) - with pytest.raises(JobIdRetrievalError) as caught: - execute_experiment(runner, []) - assert caught.value is error - reports.assert_called_once_with(slurm_system, scenario, runner.runner.scenario_root) From ed8946effc50a05b9ee2b4d92173d5de563a8634 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Wed, 7 Oct 2026 21:34:19 +0200 Subject: [PATCH 5/5] Store CloudAI execution errors in test-run metadata Signed-off-by: Ivan Podkidyshev --- src/cloudai/_core/base_runner.py | 27 +++++++++++++++++++++------ src/cloudai/models/scenario.py | 8 ++++++++ src/cloudai/reporter.py | 18 ++++++++++++------ tests/test_base_runner.py | 18 +++++++++++++----- tests/test_handlers.py | 1 - tests/test_reporter.py | 12 +++++++++++- 6 files changed, 65 insertions(+), 19 deletions(-) diff --git a/src/cloudai/_core/base_runner.py b/src/cloudai/_core/base_runner.py index 16bde18c8..43ded3286 100644 --- a/src/cloudai/_core/base_runner.py +++ b/src/cloudai/_core/base_runner.py @@ -21,6 +21,8 @@ from pathlib import Path from typing import Dict, List +import toml + import cloudai.models.output import cloudai.output @@ -54,8 +56,6 @@ class BaseRunner(ABC): new tests and ensuring a graceful termination of all running tests. """ - SUBMISSION_ERROR_FILE_NAME = "submission-error.txt" - def __init__(self, mode: str, system: System, test_scenario: TestScenario, output_path: Path): """ Initialize the BaseRunner with a system object, test scenario, and monitor interval. @@ -164,12 +164,27 @@ def submit_test(self, tr: TestRun): self.update_run_output(job) except JobSubmissionError as e: logging.error(e) - try: - (tr.output_path / self.SUBMISSION_ERROR_FILE_NAME).write_text(f"{type(e).__name__}: {e}") - except OSError: - logging.exception("Failed to persist submission error for %s", tr.name) + self.record_execution_error(tr, e) raise + def record_execution_error(self, tr: TestRun, error: JobSubmissionError) -> None: + """Preserve a CloudAI failure in the run dump for report regeneration.""" + from cloudai.models.scenario import ExecutionError, TestRunDetails + + path = tr.output_path / CommandGenStrategy.TEST_RUN_DUMP_FILE_NAME + try: + if path.is_file(): + details = toml.load(path) + else: + details = TestRunDetails.from_test_run(tr, test_cmd="", full_cmd=error.command).model_dump( + exclude_none=True + ) + details["execution_error"] = ExecutionError(type=type(error).__name__, message=str(error)).model_dump() + with path.open("w") as stream: + toml.dump(details, stream) + except (OSError, toml.TomlDecodeError): + logging.exception("Failed to persist execution error for %s", tr.name) + def on_job_submit(self, tr: TestRun) -> None: return diff --git a/src/cloudai/models/scenario.py b/src/cloudai/models/scenario.py index 543a0cf3c..249ed3ae4 100644 --- a/src/cloudai/models/scenario.py +++ b/src/cloudai/models/scenario.py @@ -265,6 +265,13 @@ def parse_reports(cls, value: dict[str, Any] | None) -> dict[str, ReportConfig] return parse_reports_spec(value) +class ExecutionError(BaseModel): + """Unrecoverable CloudAI execution error, distinct from a workload failure.""" + + type: str + message: str + + class TestRunDetails(BaseModel): """ Model for test run dump. @@ -285,6 +292,7 @@ class TestRunDetails(BaseModel): test_cmd: str full_cmd: str test_definition: Any + execution_error: ExecutionError | None = None @field_serializer("output_path") def _path_serializer(self, v: Path) -> str: diff --git a/src/cloudai/reporter.py b/src/cloudai/reporter.py index a21f6f428..9f217f46f 100644 --- a/src/cloudai/reporter.py +++ b/src/cloudai/reporter.py @@ -31,8 +31,8 @@ from cloudai.report_generator.dse_report import build_dse_summaries, load_trajectory_dataframe from cloudai.report_generator.util import load_system_metadata -from .core import BaseRunner, CommandGenStrategy, Reporter, TestRun, case_name -from .models.scenario import TestRunDetails +from .core import CommandGenStrategy, Reporter, TestRun, case_name +from .models.scenario import ExecutionError, TestRunDetails @dataclass @@ -150,8 +150,11 @@ def generate(self) -> None: results = [] for tr in self.trs: - error_path = tr.output_path / BaseRunner.SUBMISSION_ERROR_FILE_NAME - error = error_path.read_text() if error_path.is_file() else None + dump_path = tr.output_path / CommandGenStrategy.TEST_RUN_DUMP_FILE_NAME + details = toml.load(dump_path) if dump_path.is_file() else {} + error = ( + ExecutionError.model_validate(details["execution_error"]) if details.get("execution_error") else None + ) status = tr.test.was_run_successful(tr) if error is None else None results.append((tr, status, self._duration(tr.output_path), error)) failures = sum(status is not None and not status.is_successful for _, status, _, _ in results) @@ -177,8 +180,11 @@ def generate(self) -> None: testcase = ET.SubElement(suite, "testcase", attributes) if error is not None: - error_element = ET.SubElement(testcase, "error", {"message": self._xml_text(error)}) - error_element.text = self._xml_text(error) + message = self._xml_text(f"{error.type}: {error.message}") + error_element = ET.SubElement( + testcase, "error", {"type": self._xml_text(error.type), "message": message} + ) + error_element.text = message elif status is not None and not status.is_successful: message = status.error_message or "Test run failed" failure = ET.SubElement(testcase, "failure", {"message": self._xml_text(message)}) diff --git a/tests/test_base_runner.py b/tests/test_base_runner.py index 9f3137411..77ec71309 100644 --- a/tests/test_base_runner.py +++ b/tests/test_base_runner.py @@ -19,6 +19,7 @@ from typing import cast import pytest +import toml from pydantic import ConfigDict from cloudai.core import ( @@ -162,7 +163,8 @@ def test_end_post_comp(self, runner: MyRunner, tr_main: TestRun): assert runner.killed_by_dependency[0].test_run == tr_dep -def test_submission_error_is_preserved(runner: MyRunner, monkeypatch: pytest.MonkeyPatch) -> None: +@pytest.mark.parametrize("existing_dump", [False, True]) +def test_submission_error_is_preserved(runner: MyRunner, monkeypatch: pytest.MonkeyPatch, existing_dump: bool) -> None: error = JobIdRetrievalError("tr-name", "sbatch test.sh", "", "submission timeout", "No job ID") def submit(tr: TestRun) -> BaseJob: @@ -170,11 +172,17 @@ def submit(tr: TestRun) -> BaseJob: monkeypatch.setattr(runner, "_submit_test", submit) tr = runner.test_scenario.test_runs[0] + if existing_dump: + tr.output_path = runner.get_job_output_path(tr) + with (tr.output_path / "test-run.toml").open("w") as stream: + toml.dump({"name": tr.name, "full_cmd": "original command"}, stream) with pytest.raises(JobIdRetrievalError) as caught: runner.submit_test(tr) assert caught.value is error assert runner.jobs == [] - persisted = (tr.output_path / runner.SUBMISSION_ERROR_FILE_NAME).read_text() - assert "JobIdRetrievalError" in persisted - assert "submission timeout" in persisted - assert "sbatch test.sh" in persisted + details = toml.load(tr.output_path / "test-run.toml") + assert details["name"] == tr.name + assert details["full_cmd"] == ("original command" if existing_dump else error.command) + assert details["execution_error"]["type"] == "JobIdRetrievalError" + assert "submission timeout" in details["execution_error"]["message"] + assert "sbatch test.sh" in details["execution_error"]["message"] diff --git a/tests/test_handlers.py b/tests/test_handlers.py index c0a823b23..87a21513e 100644 --- a/tests/test_handlers.py +++ b/tests/test_handlers.py @@ -32,7 +32,6 @@ BaseAgentConfig, GitRepo, InstallStatusResult, - JobStatusResult, Parser, Registry, RewardOverrides, diff --git a/tests/test_reporter.py b/tests/test_reporter.py index d3bdad65e..3d63d4489 100644 --- a/tests/test_reporter.py +++ b/tests/test_reporter.py @@ -475,6 +475,11 @@ def was_run_successful(tr: TestRun): return JobStatusResult(successful, message) monkeypatch.setattr(type(benchmark_tr.test), "was_run_successful", lambda self, tr: was_run_successful(tr)) + details = TestRunDetails.from_test_run(benchmark_tr, test_cmd="benchmark", full_cmd="sbatch test.sh") + dump = details.model_dump(exclude_none=True) + dump["execution_error"] = {"type": "JobIdRetrievalError", "message": "submission timeout"} + with (run_dirs[2] / "test-run.toml").open("w") as stream: + toml.dump(dump, stream) reporter = JUnitReporter( slurm_system, TestScenario(name="test-scenario", test_runs=[benchmark_tr]), @@ -489,7 +494,7 @@ def was_run_successful(tr: TestRun): "name": "test-scenario", "tests": "3", "failures": "1", - "errors": "0", + "errors": "1", "skipped": "0", "time": "6.000", } @@ -505,6 +510,11 @@ def was_run_successful(tr: TestRun): assert failure is not None assert failure.attrib["message"] == "benchmark failed" assert cases[1].findtext("system-err") == "stderr 1\n" + error = cases[2].find("error") + assert error is not None + assert error.attrib == {"type": "JobIdRetrievalError", "message": "JobIdRetrievalError: submission timeout"} + assert error.text == "JobIdRetrievalError: submission timeout" + assert cases[2].find("failure") is None def _write_slurm_job(step_dir: Path, elapsed_time_sec: int) -> None: