PyPI - fa4 - Versions diffs - 4.0.0b3__tar.gz - Mend

fa4 4.0.0b3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (49) hide show

fa4-4.0.0b3/.flake8 +4 -0
fa4-4.0.0b3/AUTHORS +5 -0
fa4-4.0.0b3/LICENSE +29 -0
fa4-4.0.0b3/MANIFEST.in +5 -0
fa4-4.0.0b3/PKG-INFO +57 -0
fa4-4.0.0b3/README.md +26 -0
fa4-4.0.0b3/__init__.py +26 -0
fa4-4.0.0b3/ampere_helpers.py +103 -0
fa4-4.0.0b3/barrier.py +71 -0
fa4-4.0.0b3/benchmark.py +268 -0
fa4-4.0.0b3/blackwell_helpers.py +1089 -0
fa4-4.0.0b3/block_info.py +108 -0
fa4-4.0.0b3/block_sparse_utils.py +1476 -0
fa4-4.0.0b3/block_sparsity.py +440 -0
fa4-4.0.0b3/cache_utils.py +307 -0
fa4-4.0.0b3/compute_block_sparsity.py +378 -0
fa4-4.0.0b3/copy_utils.py +372 -0
fa4-4.0.0b3/cute_dsl_ptxas.py +151 -0
fa4-4.0.0b3/cute_dsl_utils.py +167 -0
fa4-4.0.0b3/dense_gemm_persistent.py +2190 -0
fa4-4.0.0b3/fa4.egg-info/SOURCES.txt +80 -0
fa4-4.0.0b3/fast_math.py +21 -0
fa4-4.0.0b3/flash_bwd.py +1264 -0
fa4-4.0.0b3/flash_bwd_postprocess.py +585 -0
fa4-4.0.0b3/flash_bwd_preprocess.py +361 -0
fa4-4.0.0b3/flash_bwd_sm100.py +3974 -0
fa4-4.0.0b3/flash_bwd_sm90.py +1591 -0
fa4-4.0.0b3/flash_fwd.py +2426 -0
fa4-4.0.0b3/flash_fwd_combine.py +692 -0
fa4-4.0.0b3/flash_fwd_epitile.py +2467 -0
fa4-4.0.0b3/flash_fwd_sm100.py +2842 -0
fa4-4.0.0b3/flash_fwd_sm100_nopipeline.py +1833 -0
fa4-4.0.0b3/flash_launch.py +235 -0
fa4-4.0.0b3/interface.py +1855 -0
fa4-4.0.0b3/mask.py +653 -0
fa4-4.0.0b3/mma_sm100_desc.py +296 -0
fa4-4.0.0b3/named_barrier.py +32 -0
fa4-4.0.0b3/pack_gqa.py +165 -0
fa4-4.0.0b3/paged_kv.py +214 -0
fa4-4.0.0b3/pipeline.py +440 -0
fa4-4.0.0b3/pyproject.toml +64 -0
fa4-4.0.0b3/sass_patch.py +209 -0
fa4-4.0.0b3/seqlen_info.py +138 -0
fa4-4.0.0b3/setup.cfg +4 -0
fa4-4.0.0b3/softmax.py +592 -0
fa4-4.0.0b3/test_flash_fwd_combine.py +125 -0
fa4-4.0.0b3/testing.py +456 -0
fa4-4.0.0b3/tile_scheduler.py +727 -0
fa4-4.0.0b3/utils.py +698 -0

fa4-4.0.0b3/.flake8 ADDED Viewed

@@ -0,0 +1,4 @@
+[flake8]
+max-line-length = 100
+# W503: line break before binary operator
+ignore = E731, E741, F841, W503

fa4-4.0.0b3/AUTHORS ADDED Viewed

@@ -0,0 +1,5 @@
+Tri Dao, tri@tridao.me
+Jay Shah
+Ted Zadouri
+Markus Hoehnerbach
+Vijay Thakkar

fa4-4.0.0b3/LICENSE ADDED Viewed

@@ -0,0 +1,29 @@
+BSD 3-Clause License
+Copyright (c) 2022, the respective contributors, as shown by the AUTHORS file.
+All rights reserved.
+Redistribution and use in source and binary forms, with or without
+modification, are permitted provided that the following conditions are met:
+* Redistributions of source code must retain the above copyright notice, this
+  list of conditions and the following disclaimer.
+* Redistributions in binary form must reproduce the above copyright notice,
+  this list of conditions and the following disclaimer in the documentation
+  and/or other materials provided with the distribution.
+* Neither the name of the copyright holder nor the names of its
+  contributors may be used to endorse or promote products derived from
+  this software without specific prior written permission.
+THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
+AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
+IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
+FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
+DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
+SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
+CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
+OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
+OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.

fa4-4.0.0b3/MANIFEST.in ADDED Viewed

@@ -0,0 +1,5 @@
+global-exclude *.egg-info/*
+prune flash_attn_4.egg-info
+prune flash_attn.egg-info
+prune build
+prune dist

fa4-4.0.0b3/PKG-INFO ADDED Viewed

@@ -0,0 +1,57 @@
+Metadata-Version: 2.4
+Name: fa4
+Version: 4.0.0b3
+Summary: Flash Attention CUTE (CUDA Template Engine) implementation
+Author: Tri Dao
+License: BSD 3-Clause License
+Project-URL: Homepage, https://github.com/Dao-AILab/flash-attention
+Project-URL: Repository, https://github.com/Dao-AILab/flash-attention
+Classifier: Development Status :: 3 - Alpha
+Classifier: License :: OSI Approved :: BSD License
+Classifier: Programming Language :: Python :: 3
+Classifier: Programming Language :: Python :: 3.10
+Classifier: Programming Language :: Python :: 3.11
+Classifier: Programming Language :: Python :: 3.12
+Requires-Python: >=3.10
+Description-Content-Type: text/markdown
+License-File: LICENSE
+License-File: AUTHORS
+Requires-Dist: nvidia-cutlass-dsl>=4.4.1
+Requires-Dist: torch
+Requires-Dist: einops
+Requires-Dist: typing_extensions
+Requires-Dist: apache-tvm-ffi<0.2,>=0.1.5
+Requires-Dist: torch-c-dlpack-ext
+Requires-Dist: quack-kernels>=0.2.10
+Requires-Dist: setuptools
+Provides-Extra: dev
+Requires-Dist: pytest; extra == "dev"
+Requires-Dist: ruff; extra == "dev"
+Dynamic: license-file
+# FlashAttention-4 (CuTeDSL)
+FlashAttention-4 is a CuTeDSL-based implementation of FlashAttention for Hopper and Blackwell GPUs.
+## Installation
+```sh
+pip install flash-attn4
+```
+## Usage
+```python
+from flash_attn.cute import flash_attn_func, flash_attn_varlen_func
+out = flash_attn_func(q, k, v, causal=True)
+```
+## Development
+```sh
+git clone https://github.com/Dao-AILab/flash-attention.git
+cd flash-attention
+pip install -e "flash_attn/cute[dev]"
+pytest tests/cute/
+```

fa4-4.0.0b3/README.md ADDED Viewed

@@ -0,0 +1,26 @@
+# FlashAttention-4 (CuTeDSL)
+FlashAttention-4 is a CuTeDSL-based implementation of FlashAttention for Hopper and Blackwell GPUs.
+## Installation
+```sh
+pip install flash-attn4
+```
+## Usage
+```python
+from flash_attn.cute import flash_attn_func, flash_attn_varlen_func
+out = flash_attn_func(q, k, v, causal=True)
+```
+## Development
+```sh
+git clone https://github.com/Dao-AILab/flash-attention.git
+cd flash-attention
+pip install -e "flash_attn/cute[dev]"
+pytest tests/cute/
+```

fa4-4.0.0b3/__init__.py ADDED Viewed

@@ -0,0 +1,26 @@
+"""Flash Attention CUTE (CUDA Template Engine) implementation."""
+from importlib.metadata import PackageNotFoundError, version
+try:
+    __version__ = version("flash-attn4")
+except PackageNotFoundError:
+    __version__ = "0.0.0"
+import cutlass.cute as cute
+from .interface import (
+    flash_attn_func,
+    flash_attn_varlen_func,
+)
+from flash_attn.cute.cute_dsl_utils import cute_compile_patched
+# Patch cute.compile to optionally dump SASS
+cute.compile = cute_compile_patched
+__all__ = [
+    "flash_attn_func",
+    "flash_attn_varlen_func",
+]

fa4-4.0.0b3/ampere_helpers.py ADDED Viewed

@@ -0,0 +1,103 @@
+# Copyright (c) 2025, Tri Dao.
+from typing import Type, Callable, Optional
+import cutlass
+import cutlass.cute as cute
+def get_smem_layout_atom(dtype: Type[cutlass.Numeric], k_dim: int) -> cute.ComposedLayout:
+    dtype_byte = cutlass.const_expr(dtype.width // 8)
+    bytes_per_row = cutlass.const_expr(k_dim * dtype_byte)
+    smem_k_block_size = (
+        cutlass.const_expr(
+            128
+            if bytes_per_row % 128 == 0
+            else (64 if bytes_per_row % 64 == 0 else (32 if bytes_per_row % 32 == 0 else 16))
+        )
+        // dtype_byte
+    )
+    swizzle_bits = (
+        4
+        if smem_k_block_size == 128
+        else (3 if smem_k_block_size == 64 else (2 if smem_k_block_size == 32 else 1))
+    )
+    swizzle_base = 2 if dtype_byte == 4 else (3 if dtype_byte == 2 else 4)
+    return cute.make_composed_layout(
+        cute.make_swizzle(swizzle_bits, swizzle_base, swizzle_base),
+        0,
+        cute.make_ordered_layout(
+            (8 if cutlass.const_expr(k_dim % 32 == 0) else 16, smem_k_block_size), order=(1, 0)
+        ),
+    )
+@cute.jit
+def gemm(
+    tiled_mma: cute.TiledMma,
+    acc: cute.Tensor,
+    tCrA: cute.Tensor,
+    tCrB: cute.Tensor,
+    tCsA: cute.Tensor,
+    tCsB: cute.Tensor,
+    smem_thr_copy_A: cute.TiledCopy,
+    smem_thr_copy_B: cute.TiledCopy,
+    hook_fn: Optional[Callable] = None,
+    A_in_regs: cutlass.Constexpr[bool] = False,
+    B_in_regs: cutlass.Constexpr[bool] = False,
+    swap_AB: cutlass.Constexpr[bool] = False,
+) -> None:
+    if cutlass.const_expr(swap_AB):
+        gemm(
+            tiled_mma,
+            acc,
+            tCrB,
+            tCrA,
+            tCsB,
+            tCsA,
+            smem_thr_copy_B,
+            smem_thr_copy_A,
+            hook_fn,
+            A_in_regs=B_in_regs,
+            B_in_regs=A_in_regs,
+            swap_AB=False,
+        )
+    else:
+        tCrA_copy_view = smem_thr_copy_A.retile(tCrA)
+        tCrB_copy_view = smem_thr_copy_B.retile(tCrB)
+        if cutlass.const_expr(not A_in_regs):
+            cute.copy(smem_thr_copy_A, tCsA[None, None, 0], tCrA_copy_view[None, None, 0])
+        if cutlass.const_expr(not B_in_regs):
+            cute.copy(smem_thr_copy_B, tCsB[None, None, 0], tCrB_copy_view[None, None, 0])
+        for k in cutlass.range_constexpr(cute.size(tCsA.shape[2])):
+            if k < cute.size(tCsA.shape[2]) - 1:
+                if cutlass.const_expr(not A_in_regs):
+                    cute.copy(
+                        smem_thr_copy_A, tCsA[None, None, k + 1], tCrA_copy_view[None, None, k + 1]
+                    )
+                if cutlass.const_expr(not B_in_regs):
+                    cute.copy(
+                        smem_thr_copy_B, tCsB[None, None, k + 1], tCrB_copy_view[None, None, k + 1]
+                    )
+            cute.gemm(tiled_mma, acc, tCrA[None, None, k], tCrB[None, None, k], acc)
+            if cutlass.const_expr(k == 0 and hook_fn is not None):
+                hook_fn()
+@cute.jit
+def gemm_rs(
+    tiled_mma: cute.TiledMma,
+    acc: cute.Tensor,
+    tCrA: cute.Tensor,
+    tCrB: cute.Tensor,
+    tCsB: cute.Tensor,
+    smem_thr_copy_B: cute.TiledCopy,
+    hook_fn: Optional[Callable] = None,
+) -> None:
+    tCrB_copy_view = smem_thr_copy_B.retile(tCrB)
+    cute.copy(smem_thr_copy_B, tCsB[None, None, 0], tCrB_copy_view[None, None, 0])
+    for k in cutlass.range_constexpr(cute.size(tCrA.shape[2])):
+        if cutlass.const_expr(k < cute.size(tCrA.shape[2]) - 1):
+            cute.copy(smem_thr_copy_B, tCsB[None, None, k + 1], tCrB_copy_view[None, None, k + 1])
+        cute.gemm(tiled_mma, acc, tCrA[None, None, k], tCrB[None, None, k], acc)
+        if cutlass.const_expr(k == 0 and hook_fn is not None):
+            hook_fn()

fa4-4.0.0b3/barrier.py ADDED Viewed

@@ -0,0 +1,71 @@
+import cutlass
+import cutlass.cute as cute
+from cutlass import Int32
+from cutlass.cutlass_dsl import T, dsl_user_op
+from cutlass._mlir.dialects import llvm
+@dsl_user_op
+def ld_acquire(lock_ptr: cute.Pointer, *, loc=None, ip=None) -> cutlass.Int32:
+    lock_ptr_i64 = lock_ptr.toint(loc=loc, ip=ip).ir_value()
+    state = llvm.inline_asm(
+        T.i32(),
+        [lock_ptr_i64],
+        "ld.global.acquire.gpu.b32 $0, [$1];",
+        "=r,l",
+        has_side_effects=True,
+        is_align_stack=False,
+        asm_dialect=llvm.AsmDialect.AD_ATT,
+    )
+    return cutlass.Int32(state)
+@dsl_user_op
+def red_relaxed(
+    lock_ptr: cute.Pointer, val: cutlass.Constexpr[Int32], *, loc=None, ip=None
+) -> None:
+    lock_ptr_i64 = lock_ptr.toint(loc=loc, ip=ip).ir_value()
+    llvm.inline_asm(
+        None,
+        [lock_ptr_i64, Int32(val).ir_value(loc=loc, ip=ip)],
+        "red.relaxed.gpu.global.add.s32 [$0], $1;",
+        "l,r",
+        has_side_effects=True,
+        is_align_stack=False,
+        asm_dialect=llvm.AsmDialect.AD_ATT,
+    )
+@dsl_user_op
+def red_release(
+    lock_ptr: cute.Pointer, val: cutlass.Constexpr[Int32], *, loc=None, ip=None
+) -> None:
+    lock_ptr_i64 = lock_ptr.toint(loc=loc, ip=ip).ir_value()
+    llvm.inline_asm(
+        None,
+        [lock_ptr_i64, Int32(val).ir_value(loc=loc, ip=ip)],
+        "red.release.gpu.global.add.s32 [$0], $1;",
+        "l,r",
+        has_side_effects=True,
+        is_align_stack=False,
+        asm_dialect=llvm.AsmDialect.AD_ATT,
+    )
+@cute.jit
+def wait_eq(lock_ptr: cute.Pointer, thread_idx: int | Int32, flag_offset: int, val: Int32) -> None:
+    flag_ptr = lock_ptr + flag_offset
+    if thread_idx == 0:
+        read_val = Int32(0)
+        while read_val != val:
+            read_val = ld_acquire(flag_ptr)
+@cute.jit
+def arrive_inc(
+    lock_ptr: cute.Pointer, thread_idx: int | Int32, flag_offset: int, val: cutlass.Constexpr[Int32]
+) -> None:
+    flag_ptr = lock_ptr + flag_offset
+    if thread_idx == 0:
+        red_release(flag_ptr, val)
+        # red_relaxed(flag_ptr, val)

fa4-4.0.0b3/benchmark.py ADDED Viewed

@@ -0,0 +1,268 @@
+# Copyright (c) 2023, Tri Dao.
+"""Useful functions for writing test code."""
+import torch
+import torch.utils.benchmark as benchmark
+def benchmark_forward(
+    fn, *inputs, repeats=10, desc="", verbose=True, amp=False, amp_dtype=torch.float16, **kwinputs
+):
+    """Use Pytorch Benchmark on the forward pass of an arbitrary function."""
+    if verbose:
+        print(desc, "- Forward pass")
+    def amp_wrapper(*inputs, **kwinputs):
+        with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+            fn(*inputs, **kwinputs)
+    t = benchmark.Timer(
+        stmt="fn_amp(*inputs, **kwinputs)",
+        globals={"fn_amp": amp_wrapper, "inputs": inputs, "kwinputs": kwinputs},
+        num_threads=torch.get_num_threads(),
+    )
+    m = t.timeit(repeats)
+    if verbose:
+        print(m)
+    return t, m
+def benchmark_backward(
+    fn,
+    *inputs,
+    grad=None,
+    repeats=10,
+    desc="",
+    verbose=True,
+    amp=False,
+    amp_dtype=torch.float16,
+    **kwinputs,
+):
+    """Use Pytorch Benchmark on the backward pass of an arbitrary function."""
+    if verbose:
+        print(desc, "- Backward pass")
+    with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+        y = fn(*inputs, **kwinputs)
+        if type(y) is tuple:
+            y = y[0]
+    if grad is None:
+        grad = torch.randn_like(y)
+    else:
+        if grad.shape != y.shape:
+            raise RuntimeError("Grad shape does not match output shape")
+    def f(*inputs, y, grad):
+        # Set .grad to None to avoid extra operation of gradient accumulation
+        for x in inputs:
+            if isinstance(x, torch.Tensor):
+                x.grad = None
+        y.backward(grad, retain_graph=True)
+    t = benchmark.Timer(
+        stmt="f(*inputs, y=y, grad=grad)",
+        globals={"f": f, "inputs": inputs, "y": y, "grad": grad},
+        num_threads=torch.get_num_threads(),
+    )
+    m = t.timeit(repeats)
+    if verbose:
+        print(m)
+    return t, m
+def benchmark_combined(
+    fn,
+    *inputs,
+    grad=None,
+    repeats=10,
+    desc="",
+    verbose=True,
+    amp=False,
+    amp_dtype=torch.float16,
+    **kwinputs,
+):
+    """Use Pytorch Benchmark on the forward+backward pass of an arbitrary function."""
+    if verbose:
+        print(desc, "- Forward + Backward pass")
+    with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+        y = fn(*inputs, **kwinputs)
+        if type(y) is tuple:
+            y = y[0]
+    if grad is None:
+        grad = torch.randn_like(y)
+    else:
+        if grad.shape != y.shape:
+            raise RuntimeError("Grad shape does not match output shape")
+    def f(grad, *inputs, **kwinputs):
+        for x in inputs:
+            if isinstance(x, torch.Tensor):
+                x.grad = None
+        with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+            y = fn(*inputs, **kwinputs)
+            if type(y) is tuple:
+                y = y[0]
+        y.backward(grad, retain_graph=True)
+    t = benchmark.Timer(
+        stmt="f(grad, *inputs, **kwinputs)",
+        globals={"f": f, "fn": fn, "inputs": inputs, "grad": grad, "kwinputs": kwinputs},
+        num_threads=torch.get_num_threads(),
+    )
+    m = t.timeit(repeats)
+    if verbose:
+        print(m)
+    return t, m
+def benchmark_fwd_bwd(
+    fn,
+    *inputs,
+    grad=None,
+    repeats=10,
+    desc="",
+    verbose=True,
+    amp=False,
+    amp_dtype=torch.float16,
+    **kwinputs,
+):
+    """Use Pytorch Benchmark on the forward+backward pass of an arbitrary function."""
+    return (
+        benchmark_forward(
+            fn,
+            *inputs,
+            repeats=repeats,
+            desc=desc,
+            verbose=verbose,
+            amp=amp,
+            amp_dtype=amp_dtype,
+            **kwinputs,
+        ),
+        benchmark_backward(
+            fn,
+            *inputs,
+            grad=grad,
+            repeats=repeats,
+            desc=desc,
+            verbose=verbose,
+            amp=amp,
+            amp_dtype=amp_dtype,
+            **kwinputs,
+        ),
+    )
+def benchmark_all(
+    fn,
+    *inputs,
+    grad=None,
+    repeats=10,
+    desc="",
+    verbose=True,
+    amp=False,
+    amp_dtype=torch.float16,
+    **kwinputs,
+):
+    """Use Pytorch Benchmark on the forward+backward pass of an arbitrary function."""
+    return (
+        benchmark_forward(
+            fn,
+            *inputs,
+            repeats=repeats,
+            desc=desc,
+            verbose=verbose,
+            amp=amp,
+            amp_dtype=amp_dtype,
+            **kwinputs,
+        ),
+        benchmark_backward(
+            fn,
+            *inputs,
+            grad=grad,
+            repeats=repeats,
+            desc=desc,
+            verbose=verbose,
+            amp=amp,
+            amp_dtype=amp_dtype,
+            **kwinputs,
+        ),
+        benchmark_combined(
+            fn,
+            *inputs,
+            grad=grad,
+            repeats=repeats,
+            desc=desc,
+            verbose=verbose,
+            amp=amp,
+            amp_dtype=amp_dtype,
+            **kwinputs,
+        ),
+    )
+def pytorch_profiler(
+    fn,
+    *inputs,
+    trace_filename=None,
+    backward=False,
+    amp=False,
+    amp_dtype=torch.float16,
+    cpu=False,
+    verbose=True,
+    **kwinputs,
+):
+    """Wrap benchmark functions in Pytorch profiler to see CUDA information."""
+    if backward:
+        with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+            out = fn(*inputs, **kwinputs)
+            if type(out) is tuple:
+                out = out[0]
+            g = torch.randn_like(out)
+    for _ in range(30):  # Warm up
+        if backward:
+            for x in inputs:
+                if isinstance(x, torch.Tensor):
+                    x.grad = None
+        with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+            out = fn(*inputs, **kwinputs)
+            if type(out) is tuple:
+                out = out[0]
+        # Backward should be done outside autocast
+        if backward:
+            out.backward(g, retain_graph=True)
+    activities = ([torch.profiler.ProfilerActivity.CPU] if cpu else []) + [
+        torch.profiler.ProfilerActivity.CUDA
+    ]
+    with torch.profiler.profile(
+        activities=activities,
+        record_shapes=True,
+        # profile_memory=True,
+        with_stack=True,
+    ) as prof:
+        if backward:
+            for x in inputs:
+                if isinstance(x, torch.Tensor):
+                    x.grad = None
+        with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
+            out = fn(*inputs, **kwinputs)
+            if type(out) is tuple:
+                out = out[0]
+        if backward:
+            out.backward(g, retain_graph=True)
+    if verbose:
+        # print(prof.key_averages().table(sort_by="self_cuda_time_total", row_limit=50))
+        print(prof.key_averages().table(row_limit=50))
+    if trace_filename is not None:
+        prof.export_chrome_trace(trace_filename)
+def benchmark_memory(fn, *inputs, desc="", verbose=True, **kwinputs):
+    torch.cuda.empty_cache()
+    torch.cuda.reset_peak_memory_stats()
+    torch.cuda.synchronize()
+    fn(*inputs, **kwinputs)
+    torch.cuda.synchronize()
+    mem = torch.cuda.max_memory_allocated() / ((2**20) * 1000)
+    if verbose:
+        print(f"{desc} max memory: {mem}GB")
+    torch.cuda.empty_cache()
+    return mem