PyPI - liger-kernel - Versions diffs - 0.1.0__py3-none-any.whl → 0.3.0__py3-none-any.whl - Mend

liger-kernel 0.1.0py3-none-any.whl → 0.3.0py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (39) hide show

liger_kernel/env_report.py +46 -0
liger_kernel/ops/cross_entropy.py +130 -63
liger_kernel/ops/experimental/embedding.py +143 -0
liger_kernel/ops/fused_linear_cross_entropy.py +203 -126
liger_kernel/ops/geglu.py +54 -42
liger_kernel/ops/kl_div.py +247 -0
liger_kernel/ops/layer_norm.py +236 -0
liger_kernel/ops/rms_norm.py +220 -84
liger_kernel/ops/rope.py +91 -84
liger_kernel/ops/swiglu.py +48 -41
liger_kernel/ops/utils.py +12 -0
liger_kernel/transformers/__init__.py +22 -0
liger_kernel/transformers/auto_model.py +33 -0
liger_kernel/transformers/cross_entropy.py +11 -1
liger_kernel/transformers/experimental/embedding.py +28 -0
liger_kernel/transformers/functional.py +19 -0
liger_kernel/transformers/fused_linear_cross_entropy.py +8 -2
liger_kernel/transformers/geglu.py +4 -2
liger_kernel/transformers/kl_div.py +13 -0
liger_kernel/transformers/layer_norm.py +30 -0
liger_kernel/transformers/model/gemma.py +138 -0
liger_kernel/transformers/model/llama.py +1 -1
liger_kernel/transformers/model/mistral.py +138 -0
liger_kernel/transformers/model/mixtral.py +158 -0
liger_kernel/transformers/model/phi3.py +136 -0
liger_kernel/transformers/model/qwen2.py +135 -0
liger_kernel/transformers/model/qwen2_vl.py +172 -0
liger_kernel/transformers/monkey_patch.py +605 -14
liger_kernel/transformers/rms_norm.py +23 -4
liger_kernel/transformers/swiglu.py +24 -0
liger_kernel/transformers/trainer_integration.py +2 -45
liger_kernel-0.3.0.dist-info/METADATA +388 -0
liger_kernel-0.3.0.dist-info/RECORD +42 -0
{liger_kernel-0.1.0.dist-info → liger_kernel-0.3.0.dist-info}/WHEEL +1 -1
liger_kernel-0.1.0.dist-info/METADATA +0 -16
liger_kernel-0.1.0.dist-info/RECORD +0 -27
{liger_kernel-0.1.0.dist-info → liger_kernel-0.3.0.dist-info}/LICENSE +0 -0
{liger_kernel-0.1.0.dist-info → liger_kernel-0.3.0.dist-info}/NOTICE +0 -0
{liger_kernel-0.1.0.dist-info → liger_kernel-0.3.0.dist-info}/top_level.txt +0 -0

liger_kernel/ops/rms_norm.py CHANGED Viewed

@@ -1,26 +1,61 @@
+"""
+This file incorporates code from Unsloth licensed under the Apache License, Version 2.0.
+See the original Unsloth repository at https://github.com/unslothai/unsloth.
+The following line
+https://github.com/linkedin/Liger-Kernel/blob/7382a8761f9af679482b968f9348013d933947c7/src/liger_kernel/ops/rms_norm.py#L30
+is based on code from Unsloth, located at:
+https://github.com/unslothai/unsloth/blob/fd753fed99ed5f10ef8a9b7139588d9de9ddecfb/unsloth/kernels/rms_layernorm.py#L22
+Modifications made by Yanning Chen, 2024.
+"""
+import operator
 import torch
 import triton
 import triton.language as tl
-from liger_kernel.ops.utils import calculate_settings, ensure_contiguous
+from liger_kernel.ops.utils import (
+    calculate_settings,
+    compare_version,
+    ensure_contiguous,
+)
+if compare_version("triton", operator.ge, "3.0.0"):
+    try:
+        # typical import path with dispatch available
+        from triton.language.extra.libdevice import rsqrt
+    except ModuleNotFoundError:
+        # for working with NGC containers
+        from triton.language.extra.cuda.libdevice import rsqrt
+else:
+    from triton.language.math import rsqrt
+_CASTING_MODE_NONE = tl.constexpr(-1)
+_CASTING_MODE_LLAMA = tl.constexpr(0)
+_CASTING_MODE_GEMMA = tl.constexpr(1)
 @triton.jit
-def _rms_norm_forward(
+def _rms_norm_forward_kernel(
     Y_ptr,
     Y_row_stride,
     X_ptr,
     X_row_stride,
     W_ptr,
     W_row_stride,
-    r_ptr,
-    r_row_stride,
+    RSTD_ptr,
+    RSTD_row_stride,
     n_cols,
     eps,
+    offset,
+    casting_mode: tl.constexpr,  # constexpr so the `if` blocks can be optimized out
     BLOCK_SIZE: tl.constexpr,
 ):
     """
-    y_i = (x_i / (RMS)) * wi, RMS = sqrt(sum(x_i^2) / N)
+    y_i = (x_i / (RMS)) * (offset + wi), RMS = sqrt(sum(x_i^2) / N)
     Reference:
     1. https://triton-lang.org/main/getting-started/tutorials/05-layer-norm.html
@@ -34,42 +69,59 @@ def _rms_norm_forward(
     Y_ptr += row_idx * Y_row_stride
     X_ptr += row_idx * X_row_stride
-    r_ptr += row_idx * r_row_stride
+    RSTD_ptr += row_idx * RSTD_row_stride
     X_row = tl.load(X_ptr + col_offsets, mask=mask, other=0)
+    X_row_dtype = X_row.dtype
     W_row = tl.load(W_ptr + col_offsets, mask=mask, other=0)
+    # On Llama, only rstd is computed on fp32
+    if casting_mode == _CASTING_MODE_LLAMA:
+        X_row = X_row.to(tl.float32)
+    # Gemma computes everything on fp32, and then casts back the output to the original dtype
+    if casting_mode == _CASTING_MODE_GEMMA:
+        W_row = W_row.to(tl.float32)
+        X_row = X_row.to(tl.float32)
     mean_square = tl.sum(X_row * X_row, axis=0) / n_cols
-    inv_rms = tl.math.rsqrt(mean_square + eps)
+    rstd = rsqrt(mean_square + eps)
     # We can save time by caching rms with minimal memory overhead
     # because rms is much smaller compared to X_row, as rms is for each row.
     # However, on the computation side, it can save 4 operations (*, sum, /, sqrt).
-    tl.store(r_ptr, inv_rms)
+    tl.store(RSTD_ptr, rstd)
+    X_row = X_row * rstd
+    # On Llama, the multiplication with the weight is done on the original dtype
+    if casting_mode == _CASTING_MODE_LLAMA:
+        X_row = X_row.to(X_row_dtype)
-    Y_row = X_row * inv_rms * W_row
+    Y_row = X_row * (offset + W_row)
     tl.store(Y_ptr + col_offsets, Y_row, mask=mask)
 @triton.jit
-def _rms_norm_backward(
+def _rms_norm_backward_kernel(
     dY_ptr,
     dY_row_stride,
     X_ptr,
     X_row_stride,
     W_ptr,
     W_row_stride,
-    r_ptr,
-    r_row_stride,
+    RSTD_ptr,
+    RSTD_row_stride,
     dW_ptr,
     dW_row_stride,
     n_cols,
-    eps,
+    offset,
+    casting_mode: tl.constexpr,
     BLOCK_SIZE: tl.constexpr,
 ):
     """
-    dx = (1 / RMS) * [dy * w  - (1 / N) * (1 / RMS^2) * ((dy * w) dot x) * x]. * means element-wise multiplication, whileas dot means dot product
+    dx = (1 / RMS) * [dy * (w + offset - (1 / N) * (1 / RMS^2) * ((dy * (w + offset)) dot x) * x]. * means element-wise multiplication, whileas dot means dot product
     dw = sum(dy * (x / RMS)). summation over BxT dimension
     """
@@ -79,75 +131,175 @@ def _rms_norm_backward(
     dY_ptr += row_idx * dY_row_stride
     X_ptr += row_idx * X_row_stride
-    r_ptr += row_idx * r_row_stride
+    RSTD_ptr += row_idx * RSTD_row_stride
     dW_ptr += row_idx * dW_row_stride
     dY_row = tl.load(dY_ptr + col_offsets, mask=mask, other=0)
     X_row = tl.load(X_ptr + col_offsets, mask=mask, other=0)
     W_row = tl.load(W_ptr + col_offsets, mask=mask, other=0)
+    original_x_dtype = X_row.dtype
     # Get cached rms
-    inv_rms_row = tl.load(r_ptr)
-    dX_row = (inv_rms_row) * (
-        dY_row * W_row
-        - (1 / n_cols)
-        * inv_rms_row
-        * inv_rms_row
-        * tl.sum(dY_row * W_row * X_row, axis=0)
-        * X_row
+    rstd_row = tl.load(RSTD_ptr)
+    W_row = W_row + offset
+    X_row = X_row.to(tl.float32)
+    # Different bacward graphs for different casting modes
+    if casting_mode == _CASTING_MODE_LLAMA:
+        m = (dY_row * W_row).to(tl.float32)
+    elif casting_mode == _CASTING_MODE_GEMMA:
+        dY_row, W_row = (
+            dY_row.to(tl.float32),
+            W_row.to(tl.float32),
+        )
+    m = dY_row * W_row
+    dX_row = rstd_row * m
+    dX_row += (rstd_row) * (
+        -(1 / n_cols) * rstd_row * rstd_row * tl.sum(m * X_row, axis=0) * X_row
     )
-    tl.store(dY_ptr + col_offsets, dX_row, mask=mask)
     # calculate the gradient of W
-    dW_row = dY_row * X_row * inv_rms_row
+    if casting_mode == _CASTING_MODE_LLAMA:
+        dW_row = dY_row * (X_row * rstd_row).to(original_x_dtype)
+    else:
+        # here X_row is already in fp32 (see previous if block)
+        dW_row = dY_row * (X_row * rstd_row)
+    tl.store(dY_ptr + col_offsets, dX_row, mask=mask)
     tl.store(dW_ptr + col_offsets, dW_row, mask=mask)
+_str_to_casting_mode = {
+    "llama": _CASTING_MODE_LLAMA.value,
+    "gemma": _CASTING_MODE_GEMMA.value,
+    "none": _CASTING_MODE_NONE.value,
+}
+def rms_norm_forward(X, W, eps, offset, casting_mode):
+    if not isinstance(casting_mode, int):
+        assert (
+            casting_mode in _str_to_casting_mode
+        ), f"Invalid casting mode: {casting_mode}"
+        casting_mode = _str_to_casting_mode[casting_mode]
+    else:
+        assert (
+            casting_mode in _str_to_casting_mode.values()
+        ), f"Invalid casting mode: {casting_mode}"
+    shape = X.shape
+    dim = shape[-1]
+    X = X.view(-1, dim)
+    n_rows, n_cols = X.shape
+    BLOCK_SIZE, num_warps = calculate_settings(n_cols)
+    Y = torch.empty((n_rows, n_cols), dtype=X.dtype, device=X.device)
+    # RSTD is to cache rstd for each row
+    # RSTD is always computed/stored in fp32 if we are using Llama or Gemma casting mode
+    rstd_dtype = (
+        torch.float32
+        if casting_mode in (_CASTING_MODE_LLAMA.value, _CASTING_MODE_GEMMA.value)
+        else X.dtype
+    )
+    RSTD = torch.empty(n_rows, dtype=rstd_dtype, device=X.device)
+    # Check constraints.
+    assert (
+        X.shape[1] == W.shape[0]
+    ), "Incompatible hidden size dimension between tensor1.shape[1] and tensor2.shape[0]"
+    _rms_norm_forward_kernel[(n_rows,)](
+        Y,
+        Y.stride(0),
+        X,
+        X.stride(0),
+        W,
+        W.stride(0),
+        RSTD,
+        RSTD.stride(0),
+        n_cols,
+        eps,
+        offset,
+        casting_mode,
+        BLOCK_SIZE=BLOCK_SIZE,
+        num_warps=num_warps,
+    )
+    return Y.view(*shape), X, RSTD, BLOCK_SIZE, num_warps, casting_mode
+def rms_norm_backward(dY, X, W, RSTD, offset, casting_mode, BLOCK_SIZE, num_warps):
+    shape = dY.shape
+    dim = shape[-1]
+    dY = dY.view(-1, dim)
+    n_rows, n_cols = dY.shape
+    dW = torch.empty_like(
+        X,
+        dtype=(torch.float32 if casting_mode == _CASTING_MODE_GEMMA.value else W.dtype),
+    )
+    # Here we use dY to store the value of dX to save memory
+    _rms_norm_backward_kernel[(n_rows,)](
+        dY,
+        dY.stride(0),
+        X,
+        X.stride(0),
+        W,
+        W.stride(0),
+        RSTD,
+        RSTD.stride(0),
+        dW,
+        dW.stride(0),
+        n_cols,
+        offset,
+        casting_mode,
+        BLOCK_SIZE=BLOCK_SIZE,
+        num_warps=num_warps,
+    )
+    dX = dY.view(*shape)
+    dW = torch.sum(dW, dim=0).to(W.dtype)
+    return dX, dW
 class LigerRMSNormFunction(torch.autograd.Function):
+    """
+    Performs RMSNorm (Root Mean Square Normalization), which normalizes the input tensor `X` using the
+    weight tensor `W`, with an optional offset and casting mode.
+    Some models use an 'offset' to shift the weight tensor `W` by a constant value. For example, Gemma
+    uses an offset of 1.0, so the computation becomes `(X / RMS(X)) * (W + 1.0)` instead of the usual
+    `(X / RMS(X)) * W`. You can pass the offset value as an argument to the forward function.
+    In addition, different models cast their inputs at different places during RMSNorm computation. For
+    example, Gemma casts everything to fp32 nefore starting the computation, while Llama casts only the
+    inverse RMS to fp32. You can specify the casting mode using the `casting_mode` argument. We currently
+    support the following casting modes (they match HuggingFace Transformers' implementations):
+    - 'llama': matches the Llama implementation, where only the inverse RMS is computed on fp32.
+    - 'gemma': matches the Gemma implementation, where everything is cast to fp32, then computed, then cast back to the original dtype.
+    - 'none': no casting is done. The computation is done in the original dtype. This saves memory and is slightly faster, but has more error w.r.t. the original implementation.
+    """
     @staticmethod
     @ensure_contiguous
-    def forward(ctx, X, W, eps):
+    def forward(ctx, X, W, eps, offset=0.0, casting_mode="llama"):
         """
         X: (B, T, H) or (BxT, H)
         W: (H,)
         """
-        shape = X.shape
-        dim = shape[-1]
-        X = X.view(-1, dim)
-        n_rows, n_cols = X.shape
-        BLOCK_SIZE, num_warps = calculate_settings(n_cols)
-        Y = torch.empty((n_rows, n_cols), dtype=X.dtype, device=X.device)
-        # r is to cache (1/rms) for each row
-        r = torch.empty(n_rows, dtype=X.dtype, device=X.device)
-        # Check constraints.
-        assert (
-            X.shape[1] == W.shape[0]
-        ), "Incompatible hidden size dimension between tensor1.shape[1] and tensor2.shape[0]"
-        _rms_norm_forward[(n_rows,)](
-            Y,
-            Y.stride(0),
-            X,
-            X.stride(0),
-            W,
-            W.stride(0),
-            r,
-            r.stride(0),
-            n_cols,
-            eps,
-            BLOCK_SIZE=BLOCK_SIZE,
-            num_warps=num_warps,
+        Y, X, RSTD, BLOCK_SIZE, num_warps, casting_mode = rms_norm_forward(
+            X, W, eps, offset, casting_mode
         )
-        ctx.eps = eps
+        ctx.offset = offset
+        ctx.casting_mode = casting_mode
         ctx.BLOCK_SIZE = BLOCK_SIZE
         ctx.num_warps = num_warps
-        ctx.save_for_backward(X, W, r)
-        return Y.view(*shape)
+        ctx.save_for_backward(X, W, RSTD)
+        return Y
     @staticmethod
     @ensure_contiguous
@@ -155,31 +307,15 @@ class LigerRMSNormFunction(torch.autograd.Function):
         """
         Y: (B, T, H) or (BxT, H)
         """
-        shape = dY.shape
-        dim = shape[-1]
-        dY = dY.view(-1, dim)
-        X, W, r = ctx.saved_tensors
-        n_rows, n_cols = dY.shape
-        dW = torch.zeros_like(X)
-        # Here we use dY to store the value of dX to save memory
-        _rms_norm_backward[(n_rows,)](
+        X, W, RSTD = ctx.saved_tensors
+        dX, dW = rms_norm_backward(
             dY,
-            dY.stride(0),
             X,
-            X.stride(0),
             W,
-            W.stride(0),
-            r,
-            r.stride(0),
-            dW,
-            dW.stride(0),
-            n_cols,
-            ctx.eps,
-            BLOCK_SIZE=ctx.BLOCK_SIZE,
-            num_warps=ctx.num_warps,
+            RSTD,
+            ctx.offset,
+            ctx.casting_mode,
+            ctx.BLOCK_SIZE,
+            ctx.num_warps,
         )
-        dX = dY.view(*shape)
-        dW = torch.sum(dW, dim=0)
-        return dX, dW, None
+        return dX, dW, None, None, None

liger_kernel/ops/rope.py CHANGED Viewed

@@ -13,8 +13,8 @@ def _triton_rope(
     cos_row_stride,
     sin,
     sin_row_stride,
+    sl,
     bs: tl.constexpr,
-    sl: tl.constexpr,
     n_qh: tl.constexpr,
     n_kh: tl.constexpr,
     hd: tl.constexpr,
@@ -117,6 +117,92 @@ def _triton_rope(
         tl.store(k_ptr + second_half_k_offsets, new_k_tile_2, mask=second_k_mask)
+def rope_forward(q, k, cos, sin):
+    # transpose it back to the physical shape because Triton looks at the physical storage
+    # note: q and k are incontiguous before the transformation and will become contiguous after transpose
+    q = q.transpose(1, 2)
+    k = k.transpose(1, 2)
+    batch_size, seq_len, n_q_head, head_dim = q.shape
+    n_kv_head = k.shape[2]
+    pad_hd = triton.next_power_of_2(head_dim)
+    pad_n_q_head = triton.next_power_of_2(n_q_head)
+    pad_n_kv_head = triton.next_power_of_2(n_kv_head)
+    BLOCK_SIZE = max(pad_n_q_head, pad_n_kv_head)
+    n_row = batch_size * seq_len
+    # ensure tensors passed into the kernel are contiguous. It will be no-op if they are already contiguous
+    q = q.contiguous()
+    k = k.contiguous()
+    cos = cos.contiguous()
+    sin = sin.contiguous()
+    _triton_rope[(n_row,)](
+        q,
+        q.stride(1),
+        k,
+        k.stride(1),
+        cos,
+        cos.stride(-2),
+        sin,
+        sin.stride(-2),
+        seq_len,
+        batch_size,
+        n_q_head,
+        n_kv_head,
+        head_dim,
+        pad_n_q_head,
+        pad_n_kv_head,
+        pad_hd,
+        BLOCK_SIZE=BLOCK_SIZE,
+        BACKWARD_PASS=False,
+    )
+    return q.transpose(1, 2), k.transpose(1, 2), cos, sin
+def rope_backward(dq, dk, cos, sin):
+    dq = dq.transpose(1, 2)
+    dk = dk.transpose(1, 2)
+    batch_size, seq_len, n_q_head, head_dim = dq.shape
+    n_kv_head = dk.shape[2]
+    pad_hd = triton.next_power_of_2(head_dim)
+    pad_n_q_head = triton.next_power_of_2(n_q_head)
+    pad_n_kv_head = triton.next_power_of_2(n_kv_head)
+    BLOCK_SIZE = max(pad_n_q_head, pad_n_kv_head)
+    n_row = batch_size * seq_len
+    # ensure dq and dk are contiguous
+    dq = dq.contiguous()
+    dk = dk.contiguous()
+    # backward is similar to forward except swapping few ops
+    _triton_rope[(n_row,)](
+        dq,
+        dq.stride(1),
+        dk,
+        dk.stride(1),
+        cos,
+        cos.stride(-2),
+        sin,
+        sin.stride(-2),
+        seq_len,
+        batch_size,
+        n_q_head,
+        n_kv_head,
+        head_dim,
+        pad_n_q_head,
+        pad_n_kv_head,
+        pad_hd,
+        BLOCK_SIZE=BLOCK_SIZE,
+        BACKWARD_PASS=True,
+    )
+    return dq.transpose(1, 2), dk.transpose(1, 2)
 class LigerRopeFunction(torch.autograd.Function):
     """
     Triton implementation of the Rotary Positional Embedding (RoPE) operation. Please note that
@@ -138,50 +224,9 @@ class LigerRopeFunction(torch.autograd.Function):
         cos size: (1, seq_len, head_dim)
         sin size: (1, seq_len, head_dim)
         """
-        # transpose it back to the physical shape because Triton looks at the physical storage
-        # note: q and k are incontiguous before the transformation and will become contiguous after transpose
-        q = q.transpose(1, 2)
-        k = k.transpose(1, 2)
-        batch_size, seq_len, n_q_head, head_dim = q.shape
-        n_kv_head = k.shape[2]
-        pad_hd = triton.next_power_of_2(head_dim)
-        pad_n_q_head = triton.next_power_of_2(n_q_head)
-        pad_n_kv_head = triton.next_power_of_2(n_kv_head)
-        BLOCK_SIZE = max(pad_n_q_head, pad_n_kv_head)
-        n_row = batch_size * seq_len
-        # ensure tensors passed into the kernel are contiguous. It will be no-op if they are already contiguous
-        q = q.contiguous()
-        k = k.contiguous()
-        cos = cos.contiguous()
-        sin = sin.contiguous()
-        _triton_rope[(n_row,)](
-            q,
-            q.stride(1),
-            k,
-            k.stride(1),
-            cos,
-            cos.stride(-2),
-            sin,
-            sin.stride(-2),
-            batch_size,
-            seq_len,
-            n_q_head,
-            n_kv_head,
-            head_dim,
-            pad_n_q_head,
-            pad_n_kv_head,
-            pad_hd,
-            BLOCK_SIZE=BLOCK_SIZE,
-            BACKWARD_PASS=False,
-        )
+        q, k, cos, sin = rope_forward(q, k, cos, sin)
         ctx.save_for_backward(cos, sin)
-        return q.transpose(1, 2), k.transpose(1, 2)
+        return q, k
     def backward(ctx, dq, dk):
         """
@@ -192,43 +237,5 @@ class LigerRopeFunction(torch.autograd.Function):
         """
         cos, sin = ctx.saved_tensors
-        dq = dq.transpose(1, 2)
-        dk = dk.transpose(1, 2)
-        batch_size, seq_len, n_q_head, head_dim = dq.shape
-        n_kv_head = dk.shape[2]
-        pad_hd = triton.next_power_of_2(head_dim)
-        pad_n_q_head = triton.next_power_of_2(n_q_head)
-        pad_n_kv_head = triton.next_power_of_2(n_kv_head)
-        BLOCK_SIZE = max(pad_n_q_head, pad_n_kv_head)
-        n_row = batch_size * seq_len
-        # ensure dq and dk are contiguous
-        dq = dq.contiguous()
-        dk = dk.contiguous()
-        # backward is similar to forward except swapping few ops
-        _triton_rope[(n_row,)](
-            dq,
-            dq.stride(1),
-            dk,
-            dk.stride(1),
-            cos,
-            cos.stride(-2),
-            sin,
-            sin.stride(-2),
-            batch_size,
-            seq_len,
-            n_q_head,
-            n_kv_head,
-            head_dim,
-            pad_n_q_head,
-            pad_n_kv_head,
-            pad_hd,
-            BLOCK_SIZE=BLOCK_SIZE,
-            BACKWARD_PASS=True,
-        )
-        return dq.transpose(1, 2), dk.transpose(1, 2), None, None, None, None
+        dq, dk = rope_backward(dq, dk, cos, sin)
+        return dq, dk, None, None, None, None

liger_kernel/ops/swiglu.py CHANGED Viewed

@@ -60,54 +60,61 @@ def _swiglu_backward_kernel(
     tl.store(b_ptr + col_offsets, db_row, mask=mask)
+def swiglu_forward(a, b):
+    ori_shape = a.shape
+    n_cols = ori_shape[-1]
+    a = a.view(-1, n_cols)
+    b = b.view(-1, n_cols)
+    c = torch.empty_like(a)
+    n_rows = a.shape[0]
+    BLOCK_SIZE, num_warps = calculate_settings(n_cols)
+    _swiglu_forward_kernel[(n_rows,)](
+        a,
+        b,
+        c,
+        c.stride(-2),
+        n_cols=n_cols,
+        BLOCK_SIZE=BLOCK_SIZE,
+        num_warps=num_warps,
+    )
+    return a, b, c.view(*ori_shape)
+def swiglu_backward(a, b, dc):
+    ori_shape = dc.shape
+    n_cols = ori_shape[-1]
+    dc = dc.view(-1, n_cols)
+    n_rows = dc.shape[0]
+    BLOCK_SIZE, num_warps = calculate_settings(n_cols)
+    _swiglu_backward_kernel[(n_rows,)](
+        dc,
+        a,
+        b,
+        dc.stride(-2),
+        n_cols=n_cols,
+        BLOCK_SIZE=BLOCK_SIZE,
+        num_warps=num_warps,
+    )
+    return a.view(*ori_shape), b.view(*ori_shape)
 class LigerSiLUMulFunction(torch.autograd.Function):
     @staticmethod
     @ensure_contiguous
     def forward(ctx, a, b):
-        ori_shape = a.shape
-        n_cols = ori_shape[-1]
-        a = a.view(-1, n_cols)
-        b = b.view(-1, n_cols)
-        c = torch.zeros_like(a)
-        n_rows = a.shape[0]
-        BLOCK_SIZE, num_warps = calculate_settings(n_cols)
-        _swiglu_forward_kernel[(n_rows,)](
-            a,
-            b,
-            c,
-            c.stride(-2),
-            n_cols=n_cols,
-            BLOCK_SIZE=BLOCK_SIZE,
-            num_warps=num_warps,
-        )
+        a, b, c = swiglu_forward(a, b)
         ctx.save_for_backward(a, b)
-        return c.view(*ori_shape)
+        return c
     @staticmethod
     @ensure_contiguous
     def backward(ctx, dc):
-        ori_shape = dc.shape
-        n_cols = ori_shape[-1]
-        dc = dc.view(-1, n_cols)
         a, b = ctx.saved_tensors
-        n_rows = dc.shape[0]
-        BLOCK_SIZE, num_warps = calculate_settings(n_cols)
-        _swiglu_backward_kernel[(n_rows,)](
-            dc,
-            a,
-            b,
-            dc.stride(-2),
-            n_cols=n_cols,
-            BLOCK_SIZE=BLOCK_SIZE,
-            num_warps=num_warps,
-        )
-        return a.view(*ori_shape), b.view(*ori_shape)
+        a, b = swiglu_backward(a, b, dc)
+        return a, b

liger-kernel 0.1.0__py3-none-any.whl → 0.3.0__py3-none-any.whl

liger-kernel 0.1.0py3-none-any.whl → 0.3.0py3-none-any.whl