From 9993a96a416991608c408a5b7b3099a3a69e5d33 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=EA=B9=80=EC=9E=AC=EC=96=B5?= Date: Sun, 20 Sep 2026 10:31:29 +0900 Subject: [PATCH 1/3] fix https://github.com/NVIDIA/TensorRT-LLM/issues/19362: name the act fusion in the skip reason MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit test_cute_dsl_nvfp4 and its 4-GPU variant skip on anything other than SM 100 or 103 with: CuTe DSL blockscaling mm supports SM 100 and 103 only That reads as though CuTe DSL blockscaling matmul as a whole stops at SM 103, which is no longer where the line is. #18761 and #18765 added SM107 CuTe DSL dense GEMM and BMM ops. What these two tests need and do not have is the NVFP4 GEMM with fused SwiGLU, gated separately in cute_dsl_custom_ops.py: CuteDSL NVFP4 SwiGLU backend requires SM 100 (B200) or SM 103 (B300) CuteDSL NVFP4 SwiGLU FP4Out requires SM 100 or SM 103 Say that instead. The docstrings on both tests already describe them as "GEMM+SwiGLU fusion for shared experts", so the message now agrees with them. This is the second of the two options in #19362 and changes no behaviour: the same host states skip, for the same reason, with the reason stated accurately. Porting the act-fusion variants to SM107 is the other option and is not this. Signed-off-by: 김재억 --- tests/integration/defs/accuracy/test_llm_api_pytorch.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch.py b/tests/integration/defs/accuracy/test_llm_api_pytorch.py index 34688c88513e..bb5222a1d514 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch.py @@ -1052,7 +1052,9 @@ def test_cute_dsl_nvfp4( """Test NVFP4 with CuTe DSL blockscaling mm (GEMM+SwiGLU fusion for shared experts).""" sm_version = get_sm_version() if sm_version not in (100, 103): - pytest.skip("CuTe DSL blockscaling mm supports SM 100 and 103 only") + pytest.skip( + "CuTe DSL NVFP4 GEMM+SwiGLU act fusion supports SM 100 and 103 only" + ) kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.9) torch_compile_config = _get_default_torch_compile_config(torch_compile) @@ -1104,7 +1106,9 @@ def test_cute_dsl_nvfp4_4gpus( """Test NVFP4 4 GPUs with CuTe DSL blockscaling mm (GEMM+SwiGLU fusion for shared experts).""" sm_version = get_sm_version() if sm_version not in (100, 103): - pytest.skip("CuTe DSL blockscaling mm supports SM 100 and 103 only") + pytest.skip( + "CuTe DSL NVFP4 GEMM+SwiGLU act fusion supports SM 100 and 103 only" + ) kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.9) torch_compile_config = _get_default_torch_compile_config(torch_compile) From eb66cfcb33d9da30bf8d6442af3fd9daa00aa987 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=EA=B9=80=EC=9E=AC=EC=96=B5?= Date: Fri, 25 Sep 2026 14:43:22 +0900 Subject: [PATCH 2/3] fix https://github.com/NVIDIA/TensorRT-LLM/issues/19362: name the MoE backend in the skip reason too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both tests also run with MoeConfig(backend="CUTEDSL"), which is restricted to SM 100/103 independently of the act-fusion GEMM (see the "{moe_backend} backend supports SM 100 and 103 only" skips elsewhere in this file). The two restrictions cover the same SM set today, but if they ever diverge the skip reason would again name only half of the story. Name both, as suggested in review. Signed-off-by: 김재억 --- tests/integration/defs/accuracy/test_llm_api_pytorch.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch.py b/tests/integration/defs/accuracy/test_llm_api_pytorch.py index bb5222a1d514..8e14efbbd32e 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch.py @@ -1053,8 +1053,8 @@ def test_cute_dsl_nvfp4( sm_version = get_sm_version() if sm_version not in (100, 103): pytest.skip( - "CuTe DSL NVFP4 GEMM+SwiGLU act fusion supports SM 100 and 103 only" - ) + "CuTe DSL NVFP4 GEMM+SwiGLU act fusion and CUTEDSL MoE backend " + "support SM 100 and 103 only") kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.9) torch_compile_config = _get_default_torch_compile_config(torch_compile) @@ -1107,8 +1107,8 @@ def test_cute_dsl_nvfp4_4gpus( sm_version = get_sm_version() if sm_version not in (100, 103): pytest.skip( - "CuTe DSL NVFP4 GEMM+SwiGLU act fusion supports SM 100 and 103 only" - ) + "CuTe DSL NVFP4 GEMM+SwiGLU act fusion and CUTEDSL MoE backend " + "support SM 100 and 103 only") kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.9) torch_compile_config = _get_default_torch_compile_config(torch_compile) From abb4b1993f90404f2f10987d42286f90e397f769 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=EA=B9=80=EC=9E=AC=EC=96=B5?= Date: Mon, 28 Sep 2026 19:47:03 +0900 Subject: [PATCH 3/3] fix https://github.com/NVIDIA/TensorRT-LLM/issues/19362: drop the MoE backend from the skip reason MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The previous commit said the CUTEDSL MoE backend supports SM 100 and 103 only. It does not: CuteDslFusedMoE accepts NVFP4 on SM 107 when Rubin support in CuTe DSL is present ("NVFP4 - SM in {100, 103, 107}" in fused_moe_cute_dsl.py). The SM 100/103 gate in these two tests is justified by the NVFP4 GEMM+SwiGLU act fusion alone, which raises on any other SM. The "{moe_backend} backend supports SM 100 and 103 only" skips elsewhere in this file are test-level gates, not the backend contract, so they were the wrong evidence for naming the backend here. Restore the wording from the first commit. Signed-off-by: 김재억 --- tests/integration/defs/accuracy/test_llm_api_pytorch.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch.py b/tests/integration/defs/accuracy/test_llm_api_pytorch.py index 8e14efbbd32e..bb5222a1d514 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch.py @@ -1053,8 +1053,8 @@ def test_cute_dsl_nvfp4( sm_version = get_sm_version() if sm_version not in (100, 103): pytest.skip( - "CuTe DSL NVFP4 GEMM+SwiGLU act fusion and CUTEDSL MoE backend " - "support SM 100 and 103 only") + "CuTe DSL NVFP4 GEMM+SwiGLU act fusion supports SM 100 and 103 only" + ) kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.9) torch_compile_config = _get_default_torch_compile_config(torch_compile) @@ -1107,8 +1107,8 @@ def test_cute_dsl_nvfp4_4gpus( sm_version = get_sm_version() if sm_version not in (100, 103): pytest.skip( - "CuTe DSL NVFP4 GEMM+SwiGLU act fusion and CUTEDSL MoE backend " - "support SM 100 and 103 only") + "CuTe DSL NVFP4 GEMM+SwiGLU act fusion supports SM 100 and 103 only" + ) kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.9) torch_compile_config = _get_default_torch_compile_config(torch_compile)