Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions tests/unit/checkpoint/test_moe_checkpoint.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,12 +8,14 @@

from unit.common import DistributedTest
from unit.simple_model import *
from deepspeed.accelerator import get_accelerator

from unit.checkpoint.common import checkpoint_correctness_verification

import pytest


@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator")
class TestMoECheckpoint(DistributedTest):
world_size = 4

Expand Down
5 changes: 5 additions & 0 deletions tests/unit/checkpoint/test_pipeline.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
from unit.simple_model import *
from unit.checkpoint.common import checkpoint_correctness_verification
from unit.util import skip_on_arch
from deepspeed.accelerator import get_accelerator

import pytest

Expand All @@ -18,6 +19,10 @@ class TestPipelineCheckpoint(DistributedTest):
@pytest.mark.parametrize("zero_stage", [0, 1])
def test_checkpoint_pipe_engine(self, zero_stage, tmpdir):
skip_on_arch(min_arch=7)
# fp16 is only enabled for zero_stage > 0; skip that parametrization on
# accelerators without fp16 support instead of failing the sanity check.
if zero_stage > 0 and not get_accelerator().is_fp16_supported():
pytest.skip("fp16 is not supported on this accelerator")

config_dict = {
"train_batch_size": 2,
Expand Down
5 changes: 5 additions & 0 deletions tests/unit/checkpoint/test_universal_checkpoint.py
Original file line number Diff line number Diff line change
Expand Up @@ -162,6 +162,11 @@ class _baseline(DistributedFixture):
world_size = None

def run(self, tmpdir, ds_config, zero_stage, dtype, load_optim, use_torch_adam):
# fp16 configs crash deepspeed.initialize's sanity check on accelerators
# without fp16 support, surfacing as a setup error for every dependent
# test instead of a skip.
if dtype == torch.float16 and not get_accelerator().is_fp16_supported():
pytest.skip("fp16 is not supported on this accelerator")
hidden_dim = 10
train_save_convert(ds_config, hidden_dim, load_optim, use_torch_adam, dtype, tmpdir, self.world_size)

Expand Down
13 changes: 13 additions & 0 deletions tests/unit/runtime/test_no_sync_ctxt.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@

import deepspeed
import deepspeed.comm as dist
from deepspeed.accelerator import get_accelerator
from deepspeed.utils import safe_get_full_grad


Expand All @@ -22,6 +23,10 @@ class TestNoSyncCtxt(DistributedTest):
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32])
@pytest.mark.parametrize("zero_stage", [0, 1, 2, 3])
def test_zero_stage(self, zero_stage, dtype):
# The fp16 parametrization crashes initialize's sanity check on accelerators
# without fp16 support (#8398's hardware lottery); skip instead of failing.
if dtype == torch.float16 and not get_accelerator().is_fp16_supported():

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should we have a check for bf16 here as well?

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Thanks for the comments. Currently CPU with accelerated instruction set (AVX or AMX) has BF16 support. We can add bf16 skip check when we need to run validation on accelerators without bf16 support.

pytest.skip("fp16 is not supported on this accelerator")
config_dict = {
"train_micro_batch_size_per_gpu": 1,
"gradient_accumulation_steps": 1,
Expand Down Expand Up @@ -65,6 +70,10 @@ def test_zero_stage(self, zero_stage, dtype):
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32])
@pytest.mark.parametrize("zero_stage", [0, 1])
def test_engine_step(self, zero_stage, dtype):
# The fp16 parametrization crashes initialize's sanity check on accelerators
# without fp16 support (#8398's hardware lottery); skip instead of failing.
if dtype == torch.float16 and not get_accelerator().is_fp16_supported():
pytest.skip("fp16 is not supported on this accelerator")
config_dict = {
"train_micro_batch_size_per_gpu": 1,
"gradient_accumulation_steps": 1,
Expand Down Expand Up @@ -107,6 +116,10 @@ def test_engine_step(self, zero_stage, dtype):
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16, torch.float32])
@pytest.mark.parametrize("zero_stage", [0, 1])
def test_multiple_ctxts(self, zero_stage, dtype):
# The fp16 parametrization crashes initialize's sanity check on accelerators
# without fp16 support (#8398's hardware lottery); skip instead of failing.
if dtype == torch.float16 and not get_accelerator().is_fp16_supported():
pytest.skip("fp16 is not supported on this accelerator")
config_dict = {
"train_micro_batch_size_per_gpu": 1,
"gradient_accumulation_steps": 1,
Expand Down
2 changes: 2 additions & 0 deletions tests/unit/runtime/zero/test_zero_offloadpp.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
import deepspeed
import torch
from deepspeed.runtime.zero.offload_config import DeepSpeedZeroOffloadOptimizerConfig
from deepspeed.accelerator import get_accelerator

import torch.nn as nn

Expand All @@ -33,6 +34,7 @@ def test_zero_partial_offload_config():


#Large sweep along hidden dim, num_layers of different sizes
@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator")
@pytest.mark.parametrize("h_dim", [1024])
@pytest.mark.parametrize("n_layers", [4, 8])
class TestZeroPartialOffloadConfigSweep(DistributedTest):
Expand Down
1 change: 1 addition & 0 deletions tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py
Original file line number Diff line number Diff line change
Expand Up @@ -174,6 +174,7 @@ def test_cpu_offload_bit_exact(self, zero_stage, offload_optimizer, offload_para
# ---------------------------------------------------------------------------
# FP16 + dynamic loss scaling
# ---------------------------------------------------------------------------
@pytest.mark.skipif(not get_accelerator().is_fp16_supported(), reason="fp16 is not supported on this accelerator")
@pytest.mark.parametrize("zero_stage", [1, 2, 3])
class TestCoalesceFP16(DistributedTest):
world_size = 2
Expand Down
Loading