Fix a couple of bugs and add tests from vLLM

Browse files

Files changed (7) hide show

ext-torch/__init__.py +35 -29
ext-torch/torch_binding.cpp +4 -0
tests/__init__.py +0 -0
tests/kernels/__init__.py +0 -0
tests/kernels/allclose_default.py +14 -0
tests/kernels/test_activation.py +139 -0
tests/kernels/utils.py +73 -0

ext-torch/__init__.py CHANGED Viewed

@@ -6,36 +6,42 @@ except ImportError as e:
     # Fallback for local development.
     try:
         import _activation
         ops = torch.ops._activition
     except ImportError:
         raise e
-def silu_and_mul(out: torch.Tensor, x: torch.Tensor) -> None:
-    ops.silu_and_mul(out, x)
-def gelu_and_mul(out: torch.Tensor, x: torch.Tensor) -> None:
-    ops.gelu_and_mul(out, x)
-def gelu_tanh_and_mul(out: torch.Tensor, x: torch.Tensor) -> None:
-    ops.gelu_tanh_and_mul(out, x)
-def fatrelu_and_mul(out: torch.Tensor,
-                    x: torch.Tensor,
-                    threshold: float = 0.0) -> None:
-    ops.fatrelu_and_mul(out, x, threshold)
-def gelu_fast(out: torch.Tensor, x: torch.Tensor) -> None:
-    ops.gelu_fast(out, x)
-def gelu_new(out: torch.Tensor, x: torch.Tensor) -> None:
-    ops.gelu_new(out, x)
-def gelu_quick(out: torch.Tensor, x: torch.Tensor) -> None:
     ops.gelu_quick(out, x)

     # Fallback for local development.
     try:
         import _activation
         ops = torch.ops._activition
     except ImportError:
         raise e
+def silu_and_mul(out: torch.Tensor, x: torch.Tensor) -> None:
+    ops.silu_and_mul(out, x)
+    return out
+def gelu_and_mul(out: torch.Tensor, x: torch.Tensor) -> None:
+    ops.gelu_and_mul(out, x)
+    return out
+def gelu_tanh_and_mul(out: torch.Tensor, x: torch.Tensor) -> None:
+    ops.gelu_tanh_and_mul(out, x)
+    return out
+def fatrelu_and_mul(out: torch.Tensor, x: torch.Tensor, threshold: float = 0.0) -> None:
+    ops.fatrelu_and_mul(out, x, threshold)
+    return out
+def gelu_fast(out: torch.Tensor, x: torch.Tensor) -> None:
+    ops.gelu_fast(out, x)
+    return out
+def gelu_new(out: torch.Tensor, x: torch.Tensor) -> None:
+    ops.gelu_new(out, x)
+    return out
+def gelu_quick(out: torch.Tensor, x: torch.Tensor) -> None:
     ops.gelu_quick(out, x)
+    return out

ext-torch/torch_binding.cpp CHANGED Viewed

@@ -28,6 +28,10 @@ TORCH_LIBRARY_EXPAND(TORCH_EXTENSION_NAME, ops) {
   // Approximate GELU implementation.
   ops.def("gelu_fast(Tensor! out, Tensor input) -> ()");
   ops.impl("gelu_fast", torch::kCUDA, &gelu_fast);
 }
 REGISTER_EXTENSION(TORCH_EXTENSION_NAME)

   // Approximate GELU implementation.
   ops.def("gelu_fast(Tensor! out, Tensor input) -> ()");
   ops.impl("gelu_fast", torch::kCUDA, &gelu_fast);
+  // Quick GELU implementation.
+  ops.def("gelu_quick(Tensor! out, Tensor input) -> ()");
+  ops.impl("gelu_quick", torch::kCUDA, &gelu_quick);
 }
 REGISTER_EXTENSION(TORCH_EXTENSION_NAME)

tests/__init__.py ADDED Viewed

File without changes

tests/kernels/__init__.py ADDED Viewed

File without changes

tests/kernels/allclose_default.py ADDED Viewed

	@@ -0,0 +1,14 @@

+import torch
+# Reference default values of atol and rtol are from
+# https://github.com/pytorch/pytorch/blob/6d96beb6bec24d73ee3f080bac54d2104068f675/test/test_transformers.py#L67
+default_atol = {torch.float16: 1e-3, torch.bfloat16: 1e-3, torch.float: 1e-5}
+default_rtol = {torch.float16: 1e-3, torch.bfloat16: 1.6e-2, torch.float: 1.3e-6}
+def get_default_atol(output) -> float:
+    return default_atol[output.dtype]
+def get_default_rtol(output) -> float:
+    return default_rtol[output.dtype]

tests/kernels/test_activation.py ADDED Viewed

	@@ -0,0 +1,139 @@

+import math
+import random
+from typing import Type
+import activation
+import pytest
+import torch
+import torch.nn.functional as F
+from .utils import opcheck
+from .allclose_default import get_default_atol, get_default_rtol
+DTYPES = [torch.half, torch.bfloat16, torch.float]
+NUM_TOKENS = [7, 83, 2048]  # Arbitrary values for testing
+D = [512, 13824]  # Arbitrary values for testing
+SEEDS = [0]
+CUDA_DEVICES = [f"cuda:{i}" for i in range(1 if torch.cuda.device_count() == 1 else 2)]
+def gelu_fast(x: torch.Tensor) -> torch.Tensor:
+    return 0.5 * x * (1.0 + torch.tanh(x * 0.7978845608 * (1.0 + 0.044715 * x * x)))
+def gelu_new(x: torch.Tensor) -> torch.Tensor:
+    c = math.sqrt(2.0 / math.pi)
+    return 0.5 * x * (1.0 + torch.tanh(c * (x + 0.044715 * torch.pow(x, 3.0))))
+def gelu_quick(x: torch.Tensor) -> torch.Tensor:
+    return x * torch.sigmoid(1.702 * x)
+def fatrelu_and_mul(x: torch.Tensor, threshold: float) -> torch.Tensor:
+    d = x.shape[-1] // 2
+    x1 = x[..., :d]
+    x2 = x[..., d:]
+    x1 = F.threshold(x1, threshold, 0.0)
+    return x1 * x2
+def silu_and_mul(x: torch.Tensor) -> torch.Tensor:
+    d = x.shape[-1] // 2
+    return F.silu(x[..., :d]) * x[..., d:]
+def gelu_and_mul(x: torch.Tensor, approximate: str) -> torch.Tensor:
+    d = x.shape[-1] // 2
+    return F.gelu(x[..., :d], approximate=approximate) * x[..., d:]
+@pytest.mark.parametrize("activation_name", ["silu", "gelu", "gelu_tanh", "fatrelu"])
+@pytest.mark.parametrize("num_tokens", NUM_TOKENS)
+@pytest.mark.parametrize("d", D)
+@pytest.mark.parametrize("dtype", DTYPES)
+@pytest.mark.parametrize("seed", SEEDS)
+@pytest.mark.parametrize("device", CUDA_DEVICES)
+@torch.inference_mode()
+def test_act_and_mul(
+    activation_name: str,
+    num_tokens: int,
+    d: int,
+    dtype: torch.dtype,
+    seed: int,
+    device: str,
+) -> None:
+    random.seed(seed)
+    torch.manual_seed(seed)
+    torch.set_default_device(device)
+    x = torch.randn(num_tokens, 2 * d, dtype=dtype)
+    if activation_name == "silu":
+        torch_fn = silu_and_mul
+        fn = activation.silu_and_mul
+        op = activation.ops.silu_and_mul
+    elif activation_name == "gelu":
+        torch_fn = lambda x: gelu_and_mul(x, "none")
+        fn = activation.gelu_and_mul
+        op = activation.ops.gelu_and_mul
+    elif activation_name == "gelu_tanh":
+        torch_fn = lambda x: gelu_and_mul(x, "tanh")
+        fn = activation.gelu_tanh_and_mul
+        op = activation.ops.gelu_tanh_and_mul
+    elif activation_name == "fatrelu":
+        threshold = random.uniform(0, 1)
+        torch_fn = lambda x: fatrelu_and_mul(x, threshold)
+        fn = lambda out, x: activation.fatrelu_and_mul(out, x, threshold)
+        op = activation.ops.fatrelu_and_mul
+    out_shape = x.shape[:-1] + (x.shape[-1] // 2,)
+    out = torch.empty(out_shape, dtype=x.dtype, device=x.device)
+    out = fn(out, x)
+    ref_out = torch_fn(x)
+    # The SiLU, GELU and FatReLU implementations are equivalent to the native
+    # PyTorch implementations, so we can do exact comparison.
+    torch.testing.assert_close(out, ref_out, atol=0.0, rtol=0.0)
+    d = x.shape[-1] // 2
+    output_shape = x.shape[:-1] + (d,)
+    out = torch.empty(output_shape, dtype=x.dtype, device=x.device)
+    if activation_name == "fatrelu":
+        opcheck(op, (out, x, threshold))
+    else:
+        opcheck(op, (out, x))
+@pytest.mark.parametrize(
+    "activation_fns",
+    [
+        (gelu_fast, activation.gelu_fast, activation.ops.gelu_fast),
+        (gelu_new, activation.gelu_new, activation.ops.gelu_new),
+        (gelu_quick, activation.gelu_quick, activation.ops.gelu_quick),
+    ],
+)
+@pytest.mark.parametrize("num_tokens", NUM_TOKENS)
+@pytest.mark.parametrize("d", D)
+@pytest.mark.parametrize("dtype", DTYPES)
+@pytest.mark.parametrize("seed", SEEDS)
+@pytest.mark.parametrize("device", CUDA_DEVICES)
+@torch.inference_mode()
+def test_activation(
+    activation_fns,
+    num_tokens: int,
+    d: int,
+    dtype: torch.dtype,
+    seed: int,
+    device: str,
+) -> None:
+    torch.manual_seed(seed)
+    torch.set_default_device(device)
+    x = torch.randn(num_tokens, d, dtype=dtype)
+    torch_fn, fn, op = activation_fns
+    out = fn(torch.empty_like(x), x)
+    ref_out = torch_fn(x)
+    torch.testing.assert_close(
+        out, ref_out, atol=get_default_atol(out), rtol=get_default_rtol(out)
+    )
+    out = torch.empty_like(x)
+    opcheck(op, (out, x))

tests/kernels/utils.py ADDED Viewed

	@@ -0,0 +1,73 @@

+"""Kernel test utils"""
+import itertools
+import random
+import unittest
+from numbers import Number
+from typing import Any, Dict, List, NamedTuple, Optional, Sequence, Tuple, Union
+import pytest
+import torch
+from torch._prims_common import TensorLikeType
+# For now, disable "test_aot_dispatch_dynamic" since there are some
+# bugs related to this test in PyTorch 2.4.
+DEFAULT_OPCHECK_TEST_UTILS: Tuple[str, ...] = (
+    "test_schema",
+    "test_autograd_registration",
+    "test_faketensor",
+)
+ALL_OPCHECK_TEST_UTILS: Tuple[str, ...] = (
+    "test_schema",
+    "test_autograd_registration",
+    "test_faketensor",
+    "test_aot_dispatch_dynamic",
+)
+# Copied/modified from torch._refs.__init__.py
+def fp8_allclose(
+    a: TensorLikeType,
+    b: TensorLikeType,
+    rtol: float = 1e-05,
+    atol: float = 1e-08,
+    equal_nan: bool = False,
+) -> bool:
+    """
+    Reference implementation of torch.allclose
+    """
+    torch._refs._check_close_args(name="torch.allclose", a=a, b=b, rtol=rtol, atol=atol)
+    return bool(
+        torch.all(
+            torch.isclose(
+                a.double(), b.double(), rtol=rtol, atol=atol, equal_nan=equal_nan
+            )
+        ).item()
+    )
+# A special version of op check that has a restricted default set of test_utils
+# and a patched version of allclose that supports fp8 types.
+def opcheck(
+    op: Union[
+        torch._ops.OpOverload,
+        torch._ops.OpOverloadPacket,
+        torch._library.custom_ops.CustomOpDef,
+    ],
+    args: Tuple[Any, ...],
+    kwargs: Optional[Dict[str, Any]] = None,
+    *,
+    test_utils: Union[str, Sequence[str]] = ALL_OPCHECK_TEST_UTILS,
+    raise_exception: bool = True,
+    cond: bool = True
+) -> Dict[str, str]:
+    with unittest.mock.patch("torch.allclose", new=fp8_allclose):
+        return (
+            torch.library.opcheck(
+                op, args, kwargs, test_utils=test_utils, raise_exception=raise_exception
+            )
+            if cond
+            else {}
+        )