From 4fab178dd6fa5e41efcbc40a9affd520f851bab8 Mon Sep 17 00:00:00 2001 From: Guokai Ma Date: Sun, 27 Sep 2026 23:21:13 +0800 Subject: [PATCH] Skip fp16-config tests on accelerators without fp16 support MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every fp16-config test that reaches deepspeed.initialize crashes its sanity check ("Type fp16 is not supported on your device.") on accelerators whose is_fp16_supported() is false — on CPU that maps to the AVX512-FP16 capability of the host, and GitHub's ubuntu-24.04 runners are hardware-heterogeneous. #8398 added skipifs for the first three files; the multi-rank CPU run in #8381 flushed out six more: - checkpoint/test_universal_checkpoint.py: the fp16 parametrizations fail inside the baseline DistributedFixture's distributed run, which reports a setup ERROR for every dependent test instead of a skip. - v1/zero/test_zero_coalesce_grad_reduction.py: TestCoalesceFP16 forces an fp16 config. - runtime/test_no_sync_ctxt.py: the dtype=float16 parametrizations of the three dtype-parametrized methods; stages 2/3 additionally never reach their expected no_sync AssertionError on such hosts. - checkpoint/test_moe_checkpoint.py: the whole class hardcodes fp16. - runtime/zero/test_zero_offloadpp.py: TestZeroPartialOffloadConfigSweep hardcodes fp16. - checkpoint/test_pipeline.py: fp16 is enabled for zero_stage > 0; skip only that parametrization so zero_stage=0 keeps running. With this, every fp16-config test in the suite guards on is_fp16_supported(): the gap is a skip, not a failure, on any accelerator. Verified on an fp16-incapable CPU (the affected parametrizations skip; adjacent non-fp16 ones keep passing) and by the multi-rank CPU run in #8381. Signed-off-by: Guokai Ma --- tests/unit/checkpoint/test_moe_checkpoint.py | 2 ++ tests/unit/checkpoint/test_pipeline.py | 5 +++++ tests/unit/checkpoint/test_universal_checkpoint.py | 5 +++++ tests/unit/runtime/test_no_sync_ctxt.py | 13 +++++++++++++ tests/unit/runtime/zero/test_zero_offloadpp.py | 2 ++ .../v1/zero/test_zero_coalesce_grad_reduction.py | 1 + 6 files changed, 28 insertions(+) diff --git a/tests/unit/checkpoint/test_moe_checkpoint.py b/tests/unit/checkpoint/test_moe_checkpoint.py index 89878b5d8fa9..a1c326c13f17 100644 --- a/tests/unit/checkpoint/test_moe_checkpoint.py +++ b/tests/unit/checkpoint/test_moe_checkpoint.py @@ -8,12 +8,14 @@ from unit.common import DistributedTest from unit.simple_model import * +from deepspeed.accelerator import get_accelerator from unit.checkpoint.common import checkpoint_correctness_verification import pytest +@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator") class TestMoECheckpoint(DistributedTest): world_size = 4 diff --git a/tests/unit/checkpoint/test_pipeline.py b/tests/unit/checkpoint/test_pipeline.py index c6c228ccada7..26d0e51978da 100644 --- a/tests/unit/checkpoint/test_pipeline.py +++ b/tests/unit/checkpoint/test_pipeline.py @@ -8,6 +8,7 @@ from unit.simple_model import * from unit.checkpoint.common import checkpoint_correctness_verification from unit.util import skip_on_arch +from deepspeed.accelerator import get_accelerator import pytest @@ -18,6 +19,10 @@ class TestPipelineCheckpoint(DistributedTest): @pytest.mark.parametrize("zero_stage", [0, 1]) def test_checkpoint_pipe_engine(self, zero_stage, tmpdir): skip_on_arch(min_arch=7) + # fp16 is only enabled for zero_stage > 0; skip that parametrization on + # accelerators without fp16 support instead of failing the sanity check. + if zero_stage > 0 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_batch_size": 2, diff --git a/tests/unit/checkpoint/test_universal_checkpoint.py b/tests/unit/checkpoint/test_universal_checkpoint.py index 27e151103cc4..5a9af80d746c 100644 --- a/tests/unit/checkpoint/test_universal_checkpoint.py +++ b/tests/unit/checkpoint/test_universal_checkpoint.py @@ -162,6 +162,11 @@ class _baseline(DistributedFixture): world_size = None def run(self, tmpdir, ds_config, zero_stage, dtype, load_optim, use_torch_adam): + # fp16 configs crash deepspeed.initialize's sanity check on accelerators + # without fp16 support, surfacing as a setup error for every dependent + # test instead of a skip. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") hidden_dim = 10 train_save_convert(ds_config, hidden_dim, load_optim, use_torch_adam, dtype, tmpdir, self.world_size) diff --git a/tests/unit/runtime/test_no_sync_ctxt.py b/tests/unit/runtime/test_no_sync_ctxt.py index 8c6497013809..b490b79e4f97 100644 --- a/tests/unit/runtime/test_no_sync_ctxt.py +++ b/tests/unit/runtime/test_no_sync_ctxt.py @@ -13,6 +13,7 @@ import deepspeed import deepspeed.comm as dist +from deepspeed.accelerator import get_accelerator from deepspeed.utils import safe_get_full_grad @@ -22,6 +23,10 @@ class TestNoSyncCtxt(DistributedTest): @pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32]) @pytest.mark.parametrize("zero_stage", [0, 1, 2, 3]) def test_zero_stage(self, zero_stage, dtype): + # The fp16 parametrization crashes initialize's sanity check on accelerators + # without fp16 support (#8398's hardware lottery); skip instead of failing. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 1, @@ -65,6 +70,10 @@ def test_zero_stage(self, zero_stage, dtype): @pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32]) @pytest.mark.parametrize("zero_stage", [0, 1]) def test_engine_step(self, zero_stage, dtype): + # The fp16 parametrization crashes initialize's sanity check on accelerators + # without fp16 support (#8398's hardware lottery); skip instead of failing. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 1, @@ -107,6 +116,10 @@ def test_engine_step(self, zero_stage, dtype): @pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32]) @pytest.mark.parametrize("zero_stage", [0, 1]) def test_multiple_ctxts(self, zero_stage, dtype): + # The fp16 parametrization crashes initialize's sanity check on accelerators + # without fp16 support (#8398's hardware lottery); skip instead of failing. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 1, diff --git a/tests/unit/runtime/zero/test_zero_offloadpp.py b/tests/unit/runtime/zero/test_zero_offloadpp.py index 32e7ccc4f9ae..6dcc6d72df6b 100644 --- a/tests/unit/runtime/zero/test_zero_offloadpp.py +++ b/tests/unit/runtime/zero/test_zero_offloadpp.py @@ -10,6 +10,7 @@ import deepspeed import torch from deepspeed.runtime.zero.offload_config import DeepSpeedZeroOffloadOptimizerConfig +from deepspeed.accelerator import get_accelerator import torch.nn as nn @@ -33,6 +34,7 @@ def test_zero_partial_offload_config(): #Large sweep along hidden dim, num_layers of different sizes +@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator") @pytest.mark.parametrize("h_dim", [1024]) @pytest.mark.parametrize("n_layers", [4, 8]) class TestZeroPartialOffloadConfigSweep(DistributedTest): diff --git a/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py b/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py index 439f048787c2..baa754ef8166 100644 --- a/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py +++ b/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py @@ -174,6 +174,7 @@ def test_cpu_offload_bit_exact(self, zero_stage, offload_optimizer, offload_para # --------------------------------------------------------------------------- # FP16 + dynamic loss scaling # --------------------------------------------------------------------------- +@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator") @pytest.mark.parametrize("zero_stage", [1, 2, 3]) class TestCoalesceFP16(DistributedTest): world_size = 2