ColossalAI/colossalai/testing/comparison.py

from typing import Any, List, OrderedDict

import torch
import torch.distributed as dist
from torch import Tensor
from torch.distributed import ProcessGroup
from torch.testing import assert_close
from torch.utils._pytree import tree_flatten


def assert_equal(a: Tensor, b: Tensor):
    assert torch.all(a == b), f"expected a and b to be equal but they are not, {a} vs {b}"


def assert_not_equal(a: Tensor, b: Tensor):
    assert not torch.all(a == b), f"expected a and b to be not equal but they are, {a} vs {b}"


def assert_close_loose(a: Tensor, b: Tensor, rtol: float = 1e-3, atol: float = 1e-3):
    assert_close(
        a,
        b,
        rtol=rtol,
        atol=atol,
    )


def assert_equal_in_group(tensor: Tensor, process_group: ProcessGroup = None):
    # all gather tensors from different ranks
    world_size = dist.get_world_size(process_group)
    tensor_list = [torch.empty_like(tensor) for _ in range(world_size)]
    dist.all_gather(tensor_list, tensor, group=process_group)

    # check if they are equal one by one
    for i in range(world_size - 1):
        a = tensor_list[i]
        b = tensor_list[i + 1]
        assert torch.all(a == b), f"expected tensors on rank {i} and {i + 1} to be equal but they are not, {a} vs {b}"


def check_state_dict_equal(
    d1: OrderedDict,
    d2: OrderedDict,
    ignore_device: bool = True,
    ignore_dtype: bool = False,
):
    assert len(list(d1.keys())) == len(
        list(d2.keys())
    ), f"Number of keys unequal: {len(list(d1.keys()))} vs {len(list(d2.keys()))}"
    for k, v1 in d1.items():
        assert k in d2
        v2 = d2[k]
        if isinstance(v1, dict):
            assert isinstance(v2, dict)
            check_state_dict_equal(v1, v2, ignore_device)
        elif isinstance(v1, list):
            assert isinstance(v2, list)
            for v1_i, v2_i in zip(v1, v2):
                if isinstance(v1_i, torch.Tensor):
                    assert isinstance(v2_i, torch.Tensor)
                    if not ignore_device:
                        v1_i = v1_i.to("cpu")
                        v2_i = v2_i.to("cpu")
                    if ignore_dtype:
                        v1_i = v1_i.to(v2_i.dtype)
                    assert_close_loose(v1_i, v2_i)
                elif isinstance(v1_i, dict):
                    assert isinstance(v2_i, dict)
                    check_state_dict_equal(v1_i, v2_i, ignore_device)
                else:
                    assert v1_i == v2_i, f"{v1_i} not equals to {v2_i}"
        elif isinstance(v1, torch.Tensor):
            assert isinstance(v2, torch.Tensor)
            if not ignore_device:
                v1 = v1.to("cpu")
                v2 = v2.to("cpu")
            if ignore_dtype:
                v1 = v1.to(v2.dtype)
            assert_close_loose(v1, v2)
        else:
            assert v1 == v2, f"{v1} not equals to {v2}"


def check_state_dict_equal_pytree(d1: OrderedDict, d2: OrderedDict, ignore_device: bool = True):
    flat_d1, _ = tree_flatten(d1)
    flat_d2, _ = tree_flatten(d2)
    assert len(flat_d1) == len(flat_d2)
    for v1, v2 in zip(flat_d1, flat_d2):
        if isinstance(v1, torch.Tensor):
            assert isinstance(v2, torch.Tensor)
            if not ignore_device:
                v1 = v1.to("cpu")
                v2 = v2.to("cpu")
            assert_close_loose(v1, v2)
        else:
            assert v1 == v2, f"{v1} not equals to {v2}"


def assert_hf_output_close(
    out1: Any,
    out2: Any,
    ignore_keys: List[str] = None,
    track_name: str = "",
    atol=1e-5,
    rtol=1e-5,
):
    """
    Check if two outputs from huggingface are equal.

    Args:
        out1 (Any): the first output
        out2 (Any): the second output
        ignore_keys (List[str]): the keys to ignore when comparing two dicts
        track_name (str): the name of the value compared, used to track the path
    """
    if isinstance(out1, dict) and isinstance(out2, dict):
        # if two values are dict
        # we recursively check the keys
        assert set(out1.keys()) == set(out2.keys())
        for k in out1.keys():
            if ignore_keys is not None and k in ignore_keys:
                continue
            assert_hf_output_close(
                out1[k],
                out2[k],
                track_name=f"{track_name}.{k}",
                ignore_keys=ignore_keys,
                atol=atol,
                rtol=rtol,
            )
    elif isinstance(out1, (list, tuple)) and isinstance(out2, (list, tuple)):
        # if two values are list
        # we recursively check the elements
        assert len(out1) == len(out2)
        for i in range(len(out1)):
            assert_hf_output_close(
                out1[i],
                out2[i],
                track_name=f"{track_name}.{i}",
                ignore_keys=ignore_keys,
                atol=atol,
                rtol=rtol,
            )
    elif isinstance(out1, Tensor) and isinstance(out2, Tensor):
        if out1.shape != out2.shape:
            raise AssertionError(f"{track_name}: shape mismatch: {out1.shape} vs {out2.shape}")
        assert_close(
            out1, out2, atol=atol, rtol=rtol
        ), f"{track_name}: tensor value mismatch\nvalue 1: {out1}\nvalue 2: {out2}, \nmean error: {torch.abs(out1 - out2).mean()}"
    else:
        assert out1 == out2, f"{track_name}: value mismatch.\nout1: {out1}\nout2: {out2}"
[shardformer] supported T5 and its variants (#4045) 1 year ago			`from typing import Any, List, OrderedDict`
[booster] add tests for ddp and low level zero's checkpointio (#3715) * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update booster tutorials#3717, fix recursive check 2 years ago
added testing module (#435) 3 years ago			`import torch`
			`import torch.distributed as dist`
			`from torch import Tensor`
			`from torch.distributed import ProcessGroup`
[amp] add gradient clipping for unit tests (#2283) * [amp] add gradient clipping in unit tests * fix bugs 2 years ago			`from torch.testing import assert_close`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`from torch.utils._pytree import tree_flatten`
added testing module (#435) 3 years ago

			`def assert_equal(a: Tensor, b: Tensor):`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert torch.all(a == b), f"expected a and b to be equal but they are not, {a} vs {b}"`
added testing module (#435) 3 years ago
[NFC] polish colossalai/testing/comparison.py code style. (#1558) 2 years ago
added testing module (#435) 3 years ago			`def assert_not_equal(a: Tensor, b: Tensor):`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert not torch.all(a == b), f"expected a and b to be not equal but they are, {a} vs {b}"`
added testing module (#435) 3 years ago
[NFC] polish colossalai/testing/comparison.py code style. (#1558) 2 years ago
optimized context test time consumption (#446) 3 years ago			`def assert_close_loose(a: Tensor, b: Tensor, rtol: float = 1e-3, atol: float = 1e-3):`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert_close(`
			`a,`
			`b,`
			`rtol=rtol,`
			`atol=atol,`
			`)`
added testing module (#435) 3 years ago
[NFC] polish colossalai/testing/comparison.py code style. (#1558) 2 years ago
added testing module (#435) 3 years ago			`def assert_equal_in_group(tensor: Tensor, process_group: ProcessGroup = None):`
			`# all gather tensors from different ranks`
			`world_size = dist.get_world_size(process_group)`
			`tensor_list = [torch.empty_like(tensor) for _ in range(world_size)]`
			`dist.all_gather(tensor_list, tensor, group=process_group)`

			`# check if they are equal one by one`
			`for i in range(world_size - 1):`
			`a = tensor_list[i]`
[NFC] polish colossalai/testing/comparison.py code style. (#1558) 2 years ago			`b = tensor_list[i + 1]`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert torch.all(a == b), f"expected tensors on rank {i} and {i + 1} to be equal but they are not, {a} vs {b}"`
[booster] add tests for ddp and low level zero's checkpointio (#3715) * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update booster tutorials#3717, fix recursive check 2 years ago

[shardformer] update colo attention to support custom mask (#5510) * [feature] refactor colo attention (#5462) * [extension] update api * [feature] add colo attention * [feature] update sdpa * [feature] update npu attention * [feature] update flash-attn * [test] add flash attn test * [test] update flash attn test * [shardformer] update modeling to fit colo attention (#5465) * [misc] refactor folder structure * [shardformer] update llama flash-attn * [shardformer] fix llama policy * [devops] update tensornvme install * [test] update llama test * [shardformer] update colo attn kernel dispatch * [shardformer] update blip2 * [shardformer] update chatglm * [shardformer] update gpt2 * [shardformer] update gptj * [shardformer] update opt * [shardformer] update vit * [shardformer] update colo attention mask prep * [shardformer] update whisper * [test] fix shardformer tests (#5514) * [test] fix shardformer tests * [test] fix shardformer tests 8 months ago			`def check_state_dict_equal(`
			`d1: OrderedDict,`
			`d2: OrderedDict,`
			`ignore_device: bool = True,`
			`ignore_dtype: bool = False,`
			`):`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert len(list(d1.keys())) == len(`
			`list(d2.keys())`
			`), f"Number of keys unequal: {len(list(d1.keys()))} vs {len(list(d2.keys()))}"`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`for k, v1 in d1.items():`
			`assert k in d2`
			`v2 = d2[k]`
			`if isinstance(v1, dict):`
			`assert isinstance(v2, dict)`
			`check_state_dict_equal(v1, v2, ignore_device)`
			`elif isinstance(v1, list):`
			`assert isinstance(v2, list)`
			`for v1_i, v2_i in zip(v1, v2):`
			`if isinstance(v1_i, torch.Tensor):`
			`assert isinstance(v2_i, torch.Tensor)`
[booster] add tests for ddp and low level zero's checkpointio (#3715) * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update booster tutorials#3717, fix recursive check 2 years ago			`if not ignore_device:`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`v1_i = v1_i.to("cpu")`
			`v2_i = v2_i.to("cpu")`
[gemini] support amp o3 for gemini (#4872) * [gemini] support no reuse fp16 chunk * [gemini] support no master weight for optim * [gemini] support no master weight for gemini ddp * [test] update gemini tests * [test] update gemini tests * [plugin] update gemini plugin * [test] fix gemini checkpointio test * [test] fix gemini checkpoint io 1 year ago			`if ignore_dtype:`
			`v1_i = v1_i.to(v2_i.dtype)`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`assert_close_loose(v1_i, v2_i)`
			`elif isinstance(v1_i, dict):`
			`assert isinstance(v2_i, dict)`
			`check_state_dict_equal(v1_i, v2_i, ignore_device)`
[booster] add tests for ddp and low level zero's checkpointio (#3715) * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update booster tutorials#3717, fix recursive check 2 years ago			`else:`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`assert v1_i == v2_i, f"{v1_i} not equals to {v2_i}"`
			`elif isinstance(v1, torch.Tensor):`
			`assert isinstance(v2, torch.Tensor)`
[booster] add tests for ddp and low level zero's checkpointio (#3715) * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update booster tutorials#3717, fix recursive check 2 years ago			`if not ignore_device:`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`v1 = v1.to("cpu")`
			`v2 = v2.to("cpu")`
[gemini] support amp o3 for gemini (#4872) * [gemini] support no reuse fp16 chunk * [gemini] support no master weight for optim * [gemini] support no master weight for gemini ddp * [test] update gemini tests * [test] update gemini tests * [plugin] update gemini plugin * [test] fix gemini checkpointio test * [test] fix gemini checkpoint io 1 year ago			`if ignore_dtype:`
			`v1 = v1.to(v2.dtype)`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`assert_close_loose(v1, v2)`
[booster] add tests for ddp and low level zero's checkpointio (#3715) * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update tests for booster * [booster] update booster tutorials#3717, fix recursive check 2 years ago			`else:`
Next commit [checkpointio] Unsharded Optimizer Checkpoint for Gemini Plugin (#4141) * [checkpointio] unsharded optimizer checkpoint for Gemini plugin * [checkpointio] unsharded optimizer checkpoint for Gemini using all_gather 1 year ago			`assert v1 == v2, f"{v1} not equals to {v2}"`


			`def check_state_dict_equal_pytree(d1: OrderedDict, d2: OrderedDict, ignore_device: bool = True):`
			`flat_d1, _ = tree_flatten(d1)`
			`flat_d2, _ = tree_flatten(d2)`
			`assert len(flat_d1) == len(flat_d2)`
			`for v1, v2 in zip(flat_d1, flat_d2):`
			`if isinstance(v1, torch.Tensor):`
			`assert isinstance(v2, torch.Tensor)`
			`if not ignore_device:`
			`v1 = v1.to("cpu")`
			`v2 = v2.to("cpu")`
			`assert_close_loose(v1, v2)`
			`else:`
			`assert v1 == v2, f"{v1} not equals to {v2}"`
[shardformer] supported T5 and its variants (#4045) 1 year ago

[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`def assert_hf_output_close(`
[shardformer] update colo attention to support custom mask (#5510) * [feature] refactor colo attention (#5462) * [extension] update api * [feature] add colo attention * [feature] update sdpa * [feature] update npu attention * [feature] update flash-attn * [test] add flash attn test * [test] update flash attn test * [shardformer] update modeling to fit colo attention (#5465) * [misc] refactor folder structure * [shardformer] update llama flash-attn * [shardformer] fix llama policy * [devops] update tensornvme install * [test] update llama test * [shardformer] update colo attn kernel dispatch * [shardformer] update blip2 * [shardformer] update chatglm * [shardformer] update gpt2 * [shardformer] update gptj * [shardformer] update opt * [shardformer] update vit * [shardformer] update colo attention mask prep * [shardformer] update whisper * [test] fix shardformer tests (#5514) * [test] fix shardformer tests * [test] fix shardformer tests 8 months ago			`out1: Any,`
			`out2: Any,`
			`ignore_keys: List[str] = None,`
			`track_name: str = "",`
			`atol=1e-5,`
			`rtol=1e-5,`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`):`
[shardformer] supported T5 and its variants (#4045) 1 year ago			`"""`
			`Check if two outputs from huggingface are equal.`

			`Args:`
			`out1 (Any): the first output`
			`out2 (Any): the second output`
			`ignore_keys (List[str]): the keys to ignore when comparing two dicts`
			`track_name (str): the name of the value compared, used to track the path`
			`"""`
			`if isinstance(out1, dict) and isinstance(out2, dict):`
			`# if two values are dict`
			`# we recursively check the keys`
			`assert set(out1.keys()) == set(out2.keys())`
			`for k in out1.keys():`
			`if ignore_keys is not None and k in ignore_keys:`
			`continue`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert_hf_output_close(`
[shardformer] update colo attention to support custom mask (#5510) * [feature] refactor colo attention (#5462) * [extension] update api * [feature] add colo attention * [feature] update sdpa * [feature] update npu attention * [feature] update flash-attn * [test] add flash attn test * [test] update flash attn test * [shardformer] update modeling to fit colo attention (#5465) * [misc] refactor folder structure * [shardformer] update llama flash-attn * [shardformer] fix llama policy * [devops] update tensornvme install * [test] update llama test * [shardformer] update colo attn kernel dispatch * [shardformer] update blip2 * [shardformer] update chatglm * [shardformer] update gpt2 * [shardformer] update gptj * [shardformer] update opt * [shardformer] update vit * [shardformer] update colo attention mask prep * [shardformer] update whisper * [test] fix shardformer tests (#5514) * [test] fix shardformer tests * [test] fix shardformer tests 8 months ago			`out1[k],`
			`out2[k],`
			`track_name=f"{track_name}.{k}",`
			`ignore_keys=ignore_keys,`
			`atol=atol,`
			`rtol=rtol,`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`)`
[shardformer] supported T5 and its variants (#4045) 1 year ago			`elif isinstance(out1, (list, tuple)) and isinstance(out2, (list, tuple)):`
			`# if two values are list`
			`# we recursively check the elements`
			`assert len(out1) == len(out2)`
			`for i in range(len(out1)):`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`assert_hf_output_close(`
[shardformer] update colo attention to support custom mask (#5510) * [feature] refactor colo attention (#5462) * [extension] update api * [feature] add colo attention * [feature] update sdpa * [feature] update npu attention * [feature] update flash-attn * [test] add flash attn test * [test] update flash attn test * [shardformer] update modeling to fit colo attention (#5465) * [misc] refactor folder structure * [shardformer] update llama flash-attn * [shardformer] fix llama policy * [devops] update tensornvme install * [test] update llama test * [shardformer] update colo attn kernel dispatch * [shardformer] update blip2 * [shardformer] update chatglm * [shardformer] update gpt2 * [shardformer] update gptj * [shardformer] update opt * [shardformer] update vit * [shardformer] update colo attention mask prep * [shardformer] update whisper * [test] fix shardformer tests (#5514) * [test] fix shardformer tests * [test] fix shardformer tests 8 months ago			`out1[i],`
			`out2[i],`
			`track_name=f"{track_name}.{i}",`
			`ignore_keys=ignore_keys,`
			`atol=atol,`
			`rtol=rtol,`
[misc] update pre-commit and run all files (#4752) * [misc] update pre-commit * [misc] run pre-commit * [misc] remove useless configuration files * [misc] ignore cuda for clang-format 1 year ago			`)`
[shardformer] supported T5 and its variants (#4045) 1 year ago			`elif isinstance(out1, Tensor) and isinstance(out2, Tensor):`
			`if out1.shape != out2.shape:`
			`raise AssertionError(f"{track_name}: shape mismatch: {out1.shape} vs {out2.shape}")`
[shardformer] update colo attention to support custom mask (#5510) * [feature] refactor colo attention (#5462) * [extension] update api * [feature] add colo attention * [feature] update sdpa * [feature] update npu attention * [feature] update flash-attn * [test] add flash attn test * [test] update flash attn test * [shardformer] update modeling to fit colo attention (#5465) * [misc] refactor folder structure * [shardformer] update llama flash-attn * [shardformer] fix llama policy * [devops] update tensornvme install * [test] update llama test * [shardformer] update colo attn kernel dispatch * [shardformer] update blip2 * [shardformer] update chatglm * [shardformer] update gpt2 * [shardformer] update gptj * [shardformer] update opt * [shardformer] update vit * [shardformer] update colo attention mask prep * [shardformer] update whisper * [test] fix shardformer tests (#5514) * [test] fix shardformer tests * [test] fix shardformer tests 8 months ago			`assert_close(`
[shardformer] supported T5 and its variants (#4045) 1 year ago			`out1, out2, atol=atol, rtol=rtol`
[shardformer] adapted T5 and LLaMa test to use kit (#4049) * [shardformer] adapted T5 and LLaMa test to use kit * polish code 1 year ago			`), f"{track_name}: tensor value mismatch\nvalue 1: {out1}\nvalue 2: {out2}, \nmean error: {torch.abs(out1 - out2).mean()}"`
[shardformer] supported T5 and its variants (#4045) 1 year ago			`else:`
			`assert out1 == out2, f"{track_name}: value mismatch.\nout1: {out1}\nout2: {out2}"`