diff --git a/tests/unit/checkpoint/test_moe_checkpoint.py b/tests/unit/checkpoint/test_moe_checkpoint.py index 89878b5d8fa9..a1c326c13f17 100644 --- a/tests/unit/checkpoint/test_moe_checkpoint.py +++ b/tests/unit/checkpoint/test_moe_checkpoint.py @@ -8,12 +8,14 @@ from unit.common import DistributedTest from unit.simple_model import * +from deepspeed.accelerator import get_accelerator from unit.checkpoint.common import checkpoint_correctness_verification import pytest +@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator") class TestMoECheckpoint(DistributedTest): world_size = 4 diff --git a/tests/unit/checkpoint/test_pipeline.py b/tests/unit/checkpoint/test_pipeline.py index c6c228ccada7..26d0e51978da 100644 --- a/tests/unit/checkpoint/test_pipeline.py +++ b/tests/unit/checkpoint/test_pipeline.py @@ -8,6 +8,7 @@ from unit.simple_model import * from unit.checkpoint.common import checkpoint_correctness_verification from unit.util import skip_on_arch +from deepspeed.accelerator import get_accelerator import pytest @@ -18,6 +19,10 @@ class TestPipelineCheckpoint(DistributedTest): @pytest.mark.parametrize("zero_stage", [0, 1]) def test_checkpoint_pipe_engine(self, zero_stage, tmpdir): skip_on_arch(min_arch=7) + # fp16 is only enabled for zero_stage > 0; skip that parametrization on + # accelerators without fp16 support instead of failing the sanity check. + if zero_stage > 0 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_batch_size": 2, diff --git a/tests/unit/checkpoint/test_universal_checkpoint.py b/tests/unit/checkpoint/test_universal_checkpoint.py index 27e151103cc4..5a9af80d746c 100644 --- a/tests/unit/checkpoint/test_universal_checkpoint.py +++ b/tests/unit/checkpoint/test_universal_checkpoint.py @@ -162,6 +162,11 @@ class _baseline(DistributedFixture): world_size = None def run(self, tmpdir, ds_config, zero_stage, dtype, load_optim, use_torch_adam): + # fp16 configs crash deepspeed.initialize's sanity check on accelerators + # without fp16 support, surfacing as a setup error for every dependent + # test instead of a skip. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") hidden_dim = 10 train_save_convert(ds_config, hidden_dim, load_optim, use_torch_adam, dtype, tmpdir, self.world_size) diff --git a/tests/unit/runtime/test_no_sync_ctxt.py b/tests/unit/runtime/test_no_sync_ctxt.py index 8c6497013809..b490b79e4f97 100644 --- a/tests/unit/runtime/test_no_sync_ctxt.py +++ b/tests/unit/runtime/test_no_sync_ctxt.py @@ -13,6 +13,7 @@ import deepspeed import deepspeed.comm as dist +from deepspeed.accelerator import get_accelerator from deepspeed.utils import safe_get_full_grad @@ -22,6 +23,10 @@ class TestNoSyncCtxt(DistributedTest): @pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32]) @pytest.mark.parametrize("zero_stage", [0, 1, 2, 3]) def test_zero_stage(self, zero_stage, dtype): + # The fp16 parametrization crashes initialize's sanity check on accelerators + # without fp16 support (#8398's hardware lottery); skip instead of failing. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 1, @@ -65,6 +70,10 @@ def test_zero_stage(self, zero_stage, dtype): @pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32]) @pytest.mark.parametrize("zero_stage", [0, 1]) def test_engine_step(self, zero_stage, dtype): + # The fp16 parametrization crashes initialize's sanity check on accelerators + # without fp16 support (#8398's hardware lottery); skip instead of failing. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 1, @@ -107,6 +116,10 @@ def test_engine_step(self, zero_stage, dtype): @pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32]) @pytest.mark.parametrize("zero_stage", [0, 1]) def test_multiple_ctxts(self, zero_stage, dtype): + # The fp16 parametrization crashes initialize's sanity check on accelerators + # without fp16 support (#8398's hardware lottery); skip instead of failing. + if dtype == torch.float16 and not get_accelerator().is_fp16_supported(): + pytest.skip("fp16 is not supported on this accelerator") config_dict = { "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 1, diff --git a/tests/unit/runtime/zero/test_zero_offloadpp.py b/tests/unit/runtime/zero/test_zero_offloadpp.py index 32e7ccc4f9ae..6dcc6d72df6b 100644 --- a/tests/unit/runtime/zero/test_zero_offloadpp.py +++ b/tests/unit/runtime/zero/test_zero_offloadpp.py @@ -10,6 +10,7 @@ import deepspeed import torch from deepspeed.runtime.zero.offload_config import DeepSpeedZeroOffloadOptimizerConfig +from deepspeed.accelerator import get_accelerator import torch.nn as nn @@ -33,6 +34,7 @@ def test_zero_partial_offload_config(): #Large sweep along hidden dim, num_layers of different sizes +@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator") @pytest.mark.parametrize("h_dim", [1024]) @pytest.mark.parametrize("n_layers", [4, 8]) class TestZeroPartialOffloadConfigSweep(DistributedTest): diff --git a/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py b/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py index 439f048787c2..baa754ef8166 100644 --- a/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py +++ b/tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py @@ -174,6 +174,7 @@ def test_cpu_offload_bit_exact(self, zero_stage, offload_optimizer, offload_para # --------------------------------------------------------------------------- # FP16 + dynamic loss scaling # --------------------------------------------------------------------------- +@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator") @pytest.mark.parametrize("zero_stage", [1, 2, 3]) class TestCoalesceFP16(DistributedTest): world_size = 2