mitsuba-oidn 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mitsuba_oidn-0.1.0/.gitignore +7 -0
- mitsuba_oidn-0.1.0/.gitmodules +3 -0
- mitsuba_oidn-0.1.0/CMakeLists.txt +149 -0
- mitsuba_oidn-0.1.0/PKG-INFO +245 -0
- mitsuba_oidn-0.1.0/README.md +228 -0
- mitsuba_oidn-0.1.0/ext/oidn/.gitignore +97 -0
- mitsuba_oidn-0.1.0/ext/oidn/CHANGELOG.md +431 -0
- mitsuba_oidn-0.1.0/ext/oidn/CMakeLists.txt +210 -0
- mitsuba_oidn-0.1.0/ext/oidn/LICENSE.txt +202 -0
- mitsuba_oidn-0.1.0/ext/oidn/MITSUBA_CHANGES.md +20 -0
- mitsuba_oidn-0.1.0/ext/oidn/README.md +2270 -0
- mitsuba_oidn-0.1.0/ext/oidn/SECURITY.md +20 -0
- mitsuba_oidn-0.1.0/ext/oidn/api/CMakeLists.txt +42 -0
- mitsuba_oidn-0.1.0/ext/oidn/api/api.cpp +1110 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/Config.cmake.in +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/FindLevelZero.cmake +107 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/FindTBB.cmake +493 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_bnns.cmake +14 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_common.cmake +65 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_common_external.cmake +16 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_ispc.cmake +242 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_macros.cmake +171 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_metal.cmake +80 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_package.cmake +99 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_platform.cmake +301 -0
- mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_version.cmake +10 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/CMakeLists.txt +54 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/common.cpp +65 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/common.h +35 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/export.linux.map.in +12 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/export.macos.map.in +7 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/half.cpp +116 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/half.h +31 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/oidn.rc +36 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/oidn_utils.cpp +117 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/oidn_utils.h +25 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/platform.cpp +137 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/platform.h +389 -0
- mitsuba_oidn-0.1.0/ext/oidn/common/timer.h +36 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/CMakeLists.txt +126 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/arena.cpp +103 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/arena.h +95 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/arena_planner.cpp +162 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/arena_planner.h +97 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/autoexposure.h +56 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/buffer.cpp +202 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/buffer.h +153 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/color.cpp +18 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/color.h +220 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/concat_conv.cpp +212 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/concat_conv.h +133 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/context.cpp +56 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/context.h +139 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/conv.cpp +87 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/conv.h +63 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/data.h +45 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/device.cpp +373 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/device.h +202 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/device_factory.h +53 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/engine.cpp +162 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/engine.h +149 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/exception.cpp +17 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/exception.h +36 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/filter.cpp +86 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/filter.h +53 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/graph.cpp +591 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/graph.h +132 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/heap.cpp +92 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/heap.h +69 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/image.cpp +148 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/image.h +121 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/image_accessor.h +249 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/image_copy.h +30 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/input_process.cpp +74 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/input_process.h +53 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/kernel.h +464 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/math.h +92 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/module.cpp +199 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/module.h +53 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/op.cpp +24 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/op.h +53 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/output_process.cpp +54 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/output_process.h +42 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/pool.cpp +37 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/pool.h +38 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/progress.cpp +40 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/progress.h +48 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/record.h +31 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/ref.h +161 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/rt_filter.cpp +131 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/rt_filter.h +25 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/rtlightmap_filter.cpp +77 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/rtlightmap_filter.h +25 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/semaphore.h +21 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/subdevice.cpp +36 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/subdevice.h +39 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor.cpp +186 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor.h +119 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor_accessor.h +92 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor_desc.h +187 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor_layout.h +449 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor_reorder.cpp +101 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tensor_reorder.h +12 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/thread.cpp +335 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/thread.h +197 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tile.h +20 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tza.cpp +138 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/tza.h +13 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/unet_filter.cpp +691 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/unet_filter.h +131 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/upsample.cpp +37 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/upsample.h +37 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/vec.h +279 -0
- mitsuba_oidn-0.1.0/ext/oidn/core/verbose.h +47 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/CMakeLists.txt +190 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/CMakeLists.txt +184 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_common.cpp +65 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_common.h +15 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_conv.cpp +73 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_conv.h +29 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_engine.cpp +24 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_engine.h +20 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_pool.cpp +51 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_pool.h +26 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/color.ispc +172 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/color.isph +41 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_autoexposure.cpp +62 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_autoexposure.h +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_autoexposure.ispc +25 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_common.cpp +190 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_common.h +18 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv.cpp +103 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv.h +27 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv.ispc +237 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv_amx.cpp +158 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv_amx.h +36 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv_amx.ispc +424 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_device.cpp +264 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_device.h +69 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_engine.cpp +217 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_engine.h +74 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_image_copy.cpp +31 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_image_copy.h +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_image_copy.ispc +19 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process.cpp +48 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process.h +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process.isph +136 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process_f16.ispc +10 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process_f32.ispc +10 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_module.cpp +24 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process.cpp +45 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process.h +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process.isph +71 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process_f16.ispc +10 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process_f32.ispc +10 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool.cpp +47 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool.h +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool.isph +35 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool_f16.ispc +14 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool_f32.ispc +14 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample.cpp +83 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample.h +23 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample.isph +34 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample_f16.ispc +14 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample_f32.ispc +14 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/image_accessor.isph +84 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/math.isph +107 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/platform.ispc +38 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/platform.isph +47 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tasking.cpp +51 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tasking.h +170 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tensor_accessor.isph +218 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tile.isph +12 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/vec.isph +693 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/CMakeLists.txt +141 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_conv.cu +58 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_conv.h +13 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_device.cpp +278 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_device.h +65 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_engine.cu +262 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_engine.h +168 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_buffer.cpp +110 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_buffer.h +30 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_semaphore.cpp +67 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_semaphore.h +34 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_module.cpp +46 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/curtn.cpp +1233 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/curtn.h +17 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cutlass_conv.h +333 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cutlass_conv_sm75.cu +25 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cutlass_conv_sm80.cu +25 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_autoexposure.h +267 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_image_copy.h +73 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_input_process.h +280 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_output_process.h +133 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_pool.h +84 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_upsample.h +84 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/CMakeLists.txt +107 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/ck_conv.h +71 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/ck_conv_dl.cpp +217 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/ck_conv_wmma.cpp +222 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_conv.cpp +60 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_conv.h +13 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_device.cpp +252 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_device.h +66 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_engine.cpp +268 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_engine.h +163 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_buffer.cpp +112 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_buffer.h +30 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_semaphore.cpp +58 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_semaphore.h +34 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_module.cpp +41 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/CMakeLists.txt +60 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_buffer.h +50 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_buffer.mm +245 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_common.h +32 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_common.mm +104 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_concat_conv.h +33 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_concat_conv.mm +117 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_conv.h +32 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_conv.mm +118 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_device.h +52 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_device.mm +145 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_engine.h +141 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_engine.mm +211 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_heap.h +35 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_heap.mm +76 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_kernels.metal +79 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_module.mm +39 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/CMakeLists.txt +210 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_common.h +392 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_concat_conv.h +175 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_conv.h +169 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_conv_base.h +213 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_device.cpp +592 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_device.h +101 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_device_table.h +101 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_engine.cpp +231 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_engine.h +174 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_external_buffer.cpp +82 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_external_buffer.h +28 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_input_process.h +176 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_module.cpp +51 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops.h +48 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xe2.cpp +9 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xehpc.cpp +9 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xehpg.cpp +9 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xelp.cpp +9 -0
- mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_output_process.h +128 -0
- mitsuba_oidn-0.1.0/ext/oidn/external/catch.hpp +17937 -0
- mitsuba_oidn-0.1.0/ext/oidn/external/level_zero/ze_intel_gpu.h +610 -0
- mitsuba_oidn-0.1.0/ext/oidn/external/level_zero/ze_stypes.h +69 -0
- mitsuba_oidn-0.1.0/ext/oidn/include/OpenImageDenoise/config.h.in +82 -0
- mitsuba_oidn-0.1.0/ext/oidn/include/OpenImageDenoise/oidn.h +637 -0
- mitsuba_oidn-0.1.0/ext/oidn/include/OpenImageDenoise/oidn.hpp +1293 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/blob_to_cpp.py +91 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/build.py +243 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/build_src.py +32 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/build_weights.py +32 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/common.py +69 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/csan.supp.xml +10 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/protex_scan.sh +59 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/store-files.sh +11 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/test.py +358 -0
- mitsuba_oidn-0.1.0/ext/oidn/scripts/valgrind.supp +120 -0
- mitsuba_oidn-0.1.0/ext/oidn/third-party-programs-DPCPP.txt +110 -0
- mitsuba_oidn-0.1.0/ext/oidn/third-party-programs-oneTBB.txt +198 -0
- mitsuba_oidn-0.1.0/ext/oidn/third-party-programs.txt +636 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/LICENSE.txt +202 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/README.md +7 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_alb.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_alb_large.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb_nrm.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb_nrm_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_calb_cnrm.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_calb_cnrm_large.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_calb_cnrm_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb_nrm.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb_nrm_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_calb_cnrm.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_calb_cnrm_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_small.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_nrm.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rt_nrm_large.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rtlightmap_dir.tza +0 -0
- mitsuba_oidn-0.1.0/ext/oidn/weights/rtlightmap_hdr.tza +0 -0
- mitsuba_oidn-0.1.0/pyproject.toml +83 -0
- mitsuba_oidn-0.1.0/src/ext.cpp +1109 -0
- mitsuba_oidn-0.1.0/src/mitsuba_oidn/__init__.py +76 -0
- mitsuba_oidn-0.1.0/src/mitsuba_oidn/_denoise.py +211 -0
- mitsuba_oidn-0.1.0/tests/test_basic.py +191 -0
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
cmake_minimum_required(VERSION 3.21...3.31)
|
|
2
|
+
|
|
3
|
+
project(mitsuba_oidn LANGUAGES C CXX)
|
|
4
|
+
|
|
5
|
+
if (NOT SKBUILD)
|
|
6
|
+
message(WARNING "This CMake file is meant to be executed through scikit-build-core. "
|
|
7
|
+
"To build the package, run 'pip install .' or, for development, "
|
|
8
|
+
"'pip install --no-build-isolation -ve .'")
|
|
9
|
+
endif()
|
|
10
|
+
|
|
11
|
+
find_package(Python 3.10
|
|
12
|
+
REQUIRED COMPONENTS Interpreter Development.Module
|
|
13
|
+
OPTIONAL_COMPONENTS Development.SABIModule)
|
|
14
|
+
|
|
15
|
+
find_package(nanobind CONFIG REQUIRED)
|
|
16
|
+
|
|
17
|
+
# ------------------------------------------------------------------------------
|
|
18
|
+
# Open Image Denoise (vendored, patched to use an external thread pool)
|
|
19
|
+
# ------------------------------------------------------------------------------
|
|
20
|
+
|
|
21
|
+
set(OIDN_APPS OFF CACHE BOOL "" FORCE)
|
|
22
|
+
set(OIDN_LIBRARY_NAME "mitsuba_oidn" CACHE STRING "" FORCE)
|
|
23
|
+
set(OIDN_API_NAMESPACE "mitsuba_oidn" CACHE STRING "" FORCE)
|
|
24
|
+
set(OIDN_LIBRARY_VERSIONED OFF CACHE BOOL "" FORCE)
|
|
25
|
+
set(OIDN_INSTALL_DEPENDENCIES OFF CACHE BOOL "" FORCE)
|
|
26
|
+
set(OIDN_DEVICE_CPU ON CACHE BOOL "" FORCE)
|
|
27
|
+
|
|
28
|
+
# ISPC binary unpacked into ext/ispc (see README)
|
|
29
|
+
if (NOT ISPC_EXECUTABLE)
|
|
30
|
+
set(_ispc "${CMAKE_CURRENT_SOURCE_DIR}/ext/ispc/bin/ispc")
|
|
31
|
+
if (WIN32)
|
|
32
|
+
set(_ispc "${_ispc}.exe")
|
|
33
|
+
endif()
|
|
34
|
+
if (EXISTS "${_ispc}")
|
|
35
|
+
set(ISPC_EXECUTABLE "${_ispc}" CACHE FILEPATH "Path to the ISPC executable." FORCE)
|
|
36
|
+
endif()
|
|
37
|
+
endif()
|
|
38
|
+
|
|
39
|
+
if (APPLE AND CMAKE_SYSTEM_PROCESSOR MATCHES "arm64")
|
|
40
|
+
option(MITSUBA_OIDN_METAL "Build the Metal device" ON)
|
|
41
|
+
else()
|
|
42
|
+
set(MITSUBA_OIDN_METAL OFF)
|
|
43
|
+
endif()
|
|
44
|
+
|
|
45
|
+
if (NOT APPLE)
|
|
46
|
+
option(MITSUBA_OIDN_CUDA "Build the CUDA device (requires the CUDA toolkit)" ON)
|
|
47
|
+
else()
|
|
48
|
+
set(MITSUBA_OIDN_CUDA OFF)
|
|
49
|
+
endif()
|
|
50
|
+
|
|
51
|
+
set(OIDN_DEVICE_METAL ${MITSUBA_OIDN_METAL} CACHE BOOL "" FORCE)
|
|
52
|
+
|
|
53
|
+
# All shared libraries live side by side in the package directory
|
|
54
|
+
if (APPLE)
|
|
55
|
+
set(_rpath "@loader_path")
|
|
56
|
+
else()
|
|
57
|
+
set(_rpath "$ORIGIN")
|
|
58
|
+
endif()
|
|
59
|
+
set(OIDN_INSTALL_RPATH "${_rpath}" CACHE STRING "" FORCE)
|
|
60
|
+
|
|
61
|
+
if (MITSUBA_OIDN_CUDA)
|
|
62
|
+
find_package(CUDAToolkit 12.8 QUIET)
|
|
63
|
+
if (NOT EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/ext/oidn/external/cutlass/include/cutlass")
|
|
64
|
+
# CUTLASS is a large submodule that source distributions leave out
|
|
65
|
+
message(STATUS "mitsuba-oidn: CUTLASS not found, skipping the CUDA device")
|
|
66
|
+
set(OIDN_DEVICE_CUDA OFF CACHE BOOL "" FORCE)
|
|
67
|
+
elseif (CUDAToolkit_FOUND)
|
|
68
|
+
message(STATUS "mitsuba-oidn: building the CUDA device (CUDA ${CUDAToolkit_VERSION})")
|
|
69
|
+
set(OIDN_DEVICE_CUDA ON CACHE BOOL "" FORCE)
|
|
70
|
+
else()
|
|
71
|
+
message(STATUS "mitsuba-oidn: CUDA toolkit not found, skipping the CUDA device")
|
|
72
|
+
set(OIDN_DEVICE_CUDA OFF CACHE BOOL "" FORCE)
|
|
73
|
+
endif()
|
|
74
|
+
else()
|
|
75
|
+
set(OIDN_DEVICE_CUDA OFF CACHE BOOL "" FORCE)
|
|
76
|
+
endif()
|
|
77
|
+
|
|
78
|
+
# The weights are stored with git-lfs; a pointer file in their place would build
|
|
79
|
+
# a library with unusable filters
|
|
80
|
+
set(_weights_probe "${CMAKE_CURRENT_SOURCE_DIR}/ext/oidn/weights/rt_hdr.tza")
|
|
81
|
+
if (NOT EXISTS "${_weights_probe}")
|
|
82
|
+
message(FATAL_ERROR "OIDN weights are missing. Run 'git submodule update --init --recursive'.")
|
|
83
|
+
endif()
|
|
84
|
+
file(SIZE "${_weights_probe}" _weights_size)
|
|
85
|
+
if (_weights_size LESS 100000)
|
|
86
|
+
message(FATAL_ERROR "OIDN weights are git-lfs pointer files. Run 'git lfs pull' in ext/oidn/weights.")
|
|
87
|
+
endif()
|
|
88
|
+
|
|
89
|
+
add_subdirectory(ext/oidn)
|
|
90
|
+
|
|
91
|
+
set(MITSUBA_OIDN_LIBS OpenImageDenoise OpenImageDenoise_core OpenImageDenoise_device_cpu)
|
|
92
|
+
if (OIDN_DEVICE_METAL)
|
|
93
|
+
list(APPEND MITSUBA_OIDN_LIBS OpenImageDenoise_device_metal)
|
|
94
|
+
endif()
|
|
95
|
+
|
|
96
|
+
# ------------------------------------------------------------------------------
|
|
97
|
+
# Python extension
|
|
98
|
+
# ------------------------------------------------------------------------------
|
|
99
|
+
|
|
100
|
+
nanobind_add_module(_mitsuba_oidn_ext
|
|
101
|
+
NB_DOMAIN mitsuba_oidn
|
|
102
|
+
BACKEND_MODULE nanobind_backend
|
|
103
|
+
src/ext.cpp
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
target_link_libraries(_mitsuba_oidn_ext PRIVATE OpenImageDenoise)
|
|
107
|
+
target_compile_features(_mitsuba_oidn_ext PRIVATE cxx_std_17)
|
|
108
|
+
|
|
109
|
+
set_target_properties(_mitsuba_oidn_ext PROPERTIES INSTALL_RPATH "${_rpath}")
|
|
110
|
+
|
|
111
|
+
install(TARGETS _mitsuba_oidn_ext ${MITSUBA_OIDN_LIBS}
|
|
112
|
+
COMPONENT python
|
|
113
|
+
LIBRARY DESTINATION mitsuba_oidn
|
|
114
|
+
RUNTIME DESTINATION mitsuba_oidn
|
|
115
|
+
ARCHIVE DESTINATION mitsuba_oidn/lib
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
# Type stubs (the build-tree extension resolves the OIDN libraries via its build rpath)
|
|
119
|
+
if (NOT WIN32)
|
|
120
|
+
nanobind_add_stub(mitsuba_oidn_stub
|
|
121
|
+
MODULE _mitsuba_oidn_ext
|
|
122
|
+
OUTPUT "${CMAKE_CURRENT_BINARY_DIR}/_mitsuba_oidn_ext.pyi"
|
|
123
|
+
MARKER_FILE "${CMAKE_CURRENT_BINARY_DIR}/py.typed"
|
|
124
|
+
PYTHON_PATH $<TARGET_FILE_DIR:_mitsuba_oidn_ext>
|
|
125
|
+
DEPENDS _mitsuba_oidn_ext
|
|
126
|
+
)
|
|
127
|
+
install(FILES
|
|
128
|
+
"${CMAKE_CURRENT_BINARY_DIR}/_mitsuba_oidn_ext.pyi"
|
|
129
|
+
"${CMAKE_CURRENT_BINARY_DIR}/py.typed"
|
|
130
|
+
DESTINATION mitsuba_oidn
|
|
131
|
+
COMPONENT python
|
|
132
|
+
)
|
|
133
|
+
endif()
|
|
134
|
+
|
|
135
|
+
if (OIDN_DEVICE_CUDA)
|
|
136
|
+
# OIDN builds the CUDA device as an external project that pre-installs its module
|
|
137
|
+
# into a staging directory at build time
|
|
138
|
+
if (WIN32)
|
|
139
|
+
set(_cuda_dir "${CMAKE_INSTALL_BINDIR}")
|
|
140
|
+
else()
|
|
141
|
+
set(_cuda_dir "${CMAKE_INSTALL_LIBDIR}")
|
|
142
|
+
endif()
|
|
143
|
+
install(DIRECTORY "${OIDN_ROOT_BINARY_DIR}/devices/cuda/preinstall/${_cuda_dir}/"
|
|
144
|
+
DESTINATION mitsuba_oidn
|
|
145
|
+
COMPONENT python
|
|
146
|
+
USE_SOURCE_PERMISSIONS
|
|
147
|
+
FILES_MATCHING PATTERN "*mitsuba_oidn_device_cuda*"
|
|
148
|
+
)
|
|
149
|
+
endif()
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mitsuba-oidn
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Unofficial Python bindings for Intel Open Image Denoise (OIDN)
|
|
5
|
+
Author-Email: Wenzel Jakob <wenzel@mitsuba-renderer.org>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Topic :: Multimedia :: Graphics :: Graphics Conversion
|
|
9
|
+
Project-URL: Homepage, https://github.com/mitsuba-renderer/mitsuba-oidn
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Requires-Dist: nanobind-backend>=1.0
|
|
12
|
+
Requires-Dist: drjit>=1.5.0
|
|
13
|
+
Provides-Extra: test
|
|
14
|
+
Requires-Dist: pytest; extra == "test"
|
|
15
|
+
Requires-Dist: numpy; extra == "test"
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
|
|
18
|
+
# mitsuba-oidn
|
|
19
|
+
|
|
20
|
+
This package provides unofficial Python bindings for [Intel Open Image
|
|
21
|
+
Denoise](https://www.openimagedenoise.org) (OIDN) for use with Dr.Jit and the
|
|
22
|
+
Mitsuba renderer.
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
import mitsuba as mi
|
|
26
|
+
import mitsuba_oidn as oidn
|
|
27
|
+
|
|
28
|
+
mi.set_variant("cuda_ad_rgb")
|
|
29
|
+
scene = mi.load_dict(mi.cornell_box())
|
|
30
|
+
color = mi.render(scene, spp=16) # mi.TensorXf of shape (H, W, 3)
|
|
31
|
+
denoised = oidn.denoise(color, hdr=True) # also an mi.TensorXf, on the GPU
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
The `color` argument can be any array type that supports DLPack or the buffer
|
|
35
|
+
protocol, and the result uses the same type and device.
|
|
36
|
+
|
|
37
|
+
The package runs on Intel/ARM CPUs, CUDA, and on Apple Metal. It exchanges
|
|
38
|
+
tensors via the buffer protocol and DLPack for compatibility with Dr.Jit,
|
|
39
|
+
NumPy, PyTorch, JAX, MLX, etc., and it accesses their memory directly whenever
|
|
40
|
+
the target device can.
|
|
41
|
+
|
|
42
|
+
## Why another binding?
|
|
43
|
+
|
|
44
|
+
Several unofficial bindings of OIDN already exist (e.g.,
|
|
45
|
+
[pyoidn](https://github.com/Hyiker/pyoidn), which uses cffi to expose the C API
|
|
46
|
+
directly). This project uses [nanobind](https://github.com/wjakob/nanobind) to
|
|
47
|
+
create bindings that feel more natural in Python. They automatically commit and
|
|
48
|
+
release resources and raise errors as Python exceptions. Pixel formats and
|
|
49
|
+
dimensions are inferred from nd-array signatures.
|
|
50
|
+
|
|
51
|
+
The bindings are designed to interoperate with
|
|
52
|
+
[Dr.Jit](https://github.com/mitsuba-renderer/drjit) and [Mitsuba
|
|
53
|
+
3](https://github.com/mitsuba-renderer/mitsuba3). The copy of OIDN bundled here
|
|
54
|
+
is modified to use Dr.Jit's
|
|
55
|
+
[nanothread](https://github.com/mitsuba-renderer/nanothread) thread pool
|
|
56
|
+
instead of spinning up another redundant thread pool via oneTBB. For this
|
|
57
|
+
reason, the package depends on Dr.Jit.
|
|
58
|
+
|
|
59
|
+
## Installation
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
pip install mitsuba-oidn
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Wheels are available for Linux (x86_64, aarch64), Windows (x86_64), and
|
|
66
|
+
macOS (arm64), matching the platforms supported by Dr.Jit. The Linux x86_64
|
|
67
|
+
and Windows wheels include the CUDA device, which activates when an NVIDIA
|
|
68
|
+
driver is present. The macOS wheel includes the Metal device.
|
|
69
|
+
|
|
70
|
+
## The `denoise()` function
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
oidn.denoise(color, albedo=None, normal=None, *, hdr=False, srgb=False,
|
|
74
|
+
clean_aux=False, quality=oidn.Quality.High, input_scale=None,
|
|
75
|
+
filter="RT", device=None, output=None)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Images are arrays of shape `(H, W)` or `(H, W, C)` with `C <= 4` and dtype
|
|
79
|
+
`float32` or `float16`. A fourth channel is ignored on input. The result has
|
|
80
|
+
the same framework and lives on the same device as `color`: a NumPy array
|
|
81
|
+
yields a NumPy array, a CUDA tensor yields a CUDA tensor. Its shape is
|
|
82
|
+
`(H, W, min(C, 3))`, or `(H, W)` for two-dimensional input.
|
|
83
|
+
|
|
84
|
+
- `hdr`, `srgb`, `clean_aux`, `quality`, and `input_scale` map to the
|
|
85
|
+
parameters of the OIDN `RT` filter. Set `hdr=True` for linear radiance
|
|
86
|
+
values without an upper bound, and `clean_aux=True` when the albedo and
|
|
87
|
+
normal images are noise-free. See the [OIDN
|
|
88
|
+
documentation](https://www.openimagedenoise.org/documentation.html) for
|
|
89
|
+
details.
|
|
90
|
+
- `device` selects the device. By default, CUDA arrays use a CUDA device with
|
|
91
|
+
the matching ordinal, and host arrays use the fastest physical device in the
|
|
92
|
+
system, which can be overridden with the `OIDN_DEFAULT_DEVICE` environment
|
|
93
|
+
variable (`cpu`, `cuda`, `metal`, or a physical device ID).
|
|
94
|
+
- `output` supplies a preallocated array that is filled in place and returned.
|
|
95
|
+
With an RGBA output array, OIDN writes the RGB channels and leaves alpha
|
|
96
|
+
untouched.
|
|
97
|
+
- Filters are expensive to create, so `denoise()` caches a few of them, keyed
|
|
98
|
+
on image size, format, feature set, and parameters. Repeated calls at the
|
|
99
|
+
same resolution pay only for the actual filtering.
|
|
100
|
+
|
|
101
|
+
When the device cannot access an input array directly, for example a NumPy
|
|
102
|
+
array passed to a CUDA device, `denoise()` copies it into a device buffer.
|
|
103
|
+
Otherwise no copies are made.
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
import numpy as np
|
|
107
|
+
import torch
|
|
108
|
+
import mitsuba_oidn as oidn
|
|
109
|
+
|
|
110
|
+
# NumPy, CPU or Metal depending on the fastest available device
|
|
111
|
+
out = oidn.denoise(np.asarray(color, dtype=np.float32), hdr=True)
|
|
112
|
+
|
|
113
|
+
# PyTorch on the GPU: zero-copy in and out
|
|
114
|
+
color = torch.rand(1080, 1920, 3, device="cuda")
|
|
115
|
+
out = oidn.denoise(color, quality=oidn.Quality.Balanced)
|
|
116
|
+
assert out.device == color.device
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## The object API
|
|
120
|
+
|
|
121
|
+
The `denoise()` function covers the common case. The classes below mirror the
|
|
122
|
+
OIDN object model for applications that need control over devices, memory,
|
|
123
|
+
and filter lifetime, for instance when denoising many frames or several AOVs
|
|
124
|
+
that share auxiliary images.
|
|
125
|
+
|
|
126
|
+
### Devices
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
oidn.physical_devices() # list of PhysicalDevice: id, name, type, uuid, ...
|
|
130
|
+
|
|
131
|
+
dev = oidn.Device() # fastest physical device
|
|
132
|
+
dev = oidn.Device(oidn.DeviceType.CPU)
|
|
133
|
+
dev = oidn.Device.from_physical(id) # also from_uuid(), from_luid(), from_pci_address()
|
|
134
|
+
dev = oidn.Device.cuda(device_id=0, stream=torch.cuda.current_stream().cuda_stream)
|
|
135
|
+
dev = oidn.Device.metal(command_queue) # raw id<MTLCommandQueue> pointer
|
|
136
|
+
|
|
137
|
+
dev.num_threads = 4 # CPU only: 0 shares the Dr.Jit pool (default),
|
|
138
|
+
# a positive value creates a private pool
|
|
139
|
+
dev.verbose = 1
|
|
140
|
+
dev.type, dev.version, dev.system_memory_supported, dev.managed_memory_supported
|
|
141
|
+
dev.sync() # wait for asynchronous work
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
A device commits itself when the first buffer or filter is created. Parameters
|
|
145
|
+
such as `num_threads` must be set before that point.
|
|
146
|
+
|
|
147
|
+
### Filters
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
flt = dev.new_filter("RT") # or "RTLightmap"
|
|
151
|
+
|
|
152
|
+
flt.set_image("color", color) # arrays: layout inferred, zero-copy when possible
|
|
153
|
+
flt.set_image("albedo", albedo)
|
|
154
|
+
flt.set_image("normal", normal)
|
|
155
|
+
flt.set_image("output", output) # must be writable
|
|
156
|
+
flt.hdr = True
|
|
157
|
+
flt.clean_aux = True
|
|
158
|
+
flt.quality = oidn.Quality.High
|
|
159
|
+
flt.input_scale = 0.5 # None selects automatic scaling
|
|
160
|
+
flt.max_memory_mb = 2048
|
|
161
|
+
flt.set("cleanAux", True) # generic access by OIDN parameter name
|
|
162
|
+
flt.set_data("weights", blob) # user-trained weights (bytes or uint8 array)
|
|
163
|
+
flt.set_progress_monitor(lambda p: True) # return False to cancel
|
|
164
|
+
|
|
165
|
+
flt.execute() # commits pending changes, runs, and waits
|
|
166
|
+
flt.execute_async(); dev.sync()
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
`set_image()` accepts arrays with the layout rules of `denoise()`. Row and
|
|
170
|
+
pixel strides are passed to OIDN, so slices of larger arrays work as long as
|
|
171
|
+
the channel dimension stays contiguous. Whether an array can be bound without
|
|
172
|
+
a copy depends on the device:
|
|
173
|
+
|
|
174
|
+
| Device | Bound without copy |
|
|
175
|
+
|---|---|
|
|
176
|
+
| CPU | any host array, including CUDA pinned and managed memory |
|
|
177
|
+
| CUDA | arrays on the same CUDA device, pinned host memory, and host memory if the GPU supports pageable memory access |
|
|
178
|
+
| Metal | any host array, and views of buffers created on the device |
|
|
179
|
+
|
|
180
|
+
Use `dev.can_share(array)` to test this in advance. When binding is not
|
|
181
|
+
possible, `set_image()` raises a `TypeError` that points at the buffer API.
|
|
182
|
+
|
|
183
|
+
Filters can also take a `Buffer` with an explicit description:
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
flt.set_image("color", buf, format=oidn.Format.Float3, width=w, height=h,
|
|
187
|
+
byte_offset=0, pixel_stride=0, row_stride=0)
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
### Buffers
|
|
191
|
+
|
|
192
|
+
Buffers are memory allocations made by a device. They are the way to work
|
|
193
|
+
with memory that the host cannot address, such as dedicated GPU memory, and
|
|
194
|
+
they provide zero-copy views on unified-memory systems.
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
buf = dev.new_buffer(nbytes) # host and device accessible
|
|
198
|
+
buf = dev.new_buffer(nbytes, oidn.Storage.Device) # device memory only
|
|
199
|
+
buf = dev.new_shared_buffer(array) # wrap device-accessible memory
|
|
200
|
+
|
|
201
|
+
buf.size, buf.storage, buf.device, buf.data_ptr
|
|
202
|
+
|
|
203
|
+
view = buf.view("float32", (h, w, 3)) # DLPack and buffer-protocol object
|
|
204
|
+
img = np.from_dlpack(view) # or torch.from_dlpack(view), ...
|
|
205
|
+
|
|
206
|
+
buf.write(host_array); buf.read(host_array) # copies through the host
|
|
207
|
+
buf.write_async(src); buf.read_async(dst); dev.sync()
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
Arrays created from `buf.view()` are recognized by `set_image()` and bound
|
|
211
|
+
through the underlying buffer. On Apple silicon, rendering into such a view
|
|
212
|
+
and denoising it involves no copies at all. On a CUDA device, a device-storage
|
|
213
|
+
buffer viewed through `torch.from_dlpack()` gives a tensor that OIDN wrote
|
|
214
|
+
directly.
|
|
215
|
+
|
|
216
|
+
### Errors
|
|
217
|
+
|
|
218
|
+
All OIDN errors raise `oidn.Error`, whose `code` attribute is an
|
|
219
|
+
`oidn.ErrorCode`. Cancellation through a progress monitor raises
|
|
220
|
+
`oidn.Error` with `ErrorCode.Cancelled`. An exception raised inside the
|
|
221
|
+
progress monitor cancels the filter and propagates unchanged.
|
|
222
|
+
|
|
223
|
+
## Building from source
|
|
224
|
+
|
|
225
|
+
The build needs CMake 3.21 or newer, a C++17 compiler, and a binary release
|
|
226
|
+
of [ISPC](https://ispc.github.io/downloads.html) unpacked into `ext/ispc`,
|
|
227
|
+
so that `ext/ispc/bin/ispc` exists. Metal support requires Xcode 15 or newer.
|
|
228
|
+
CUDA support requires CUDA 12.8 or newer and is enabled automatically when the
|
|
229
|
+
toolkit is found.
|
|
230
|
+
|
|
231
|
+
```
|
|
232
|
+
git clone --recursive https://github.com/mitsuba-renderer/mitsuba-oidn
|
|
233
|
+
cd mitsuba-oidn
|
|
234
|
+
pip install nanobind==3.0.1 scikit-build-core
|
|
235
|
+
pip install --no-build-isolation -ve .
|
|
236
|
+
pytest
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
The OIDN weights are stored with git-lfs, which must be installed before
|
|
240
|
+
cloning.
|
|
241
|
+
|
|
242
|
+
## License
|
|
243
|
+
|
|
244
|
+
mitsuba-oidn is licensed under the BSD 3-Clause license. It bundles Intel Open
|
|
245
|
+
Image Denoise, which is licensed under the Apache License 2.0.
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
# mitsuba-oidn
|
|
2
|
+
|
|
3
|
+
This package provides unofficial Python bindings for [Intel Open Image
|
|
4
|
+
Denoise](https://www.openimagedenoise.org) (OIDN) for use with Dr.Jit and the
|
|
5
|
+
Mitsuba renderer.
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
import mitsuba as mi
|
|
9
|
+
import mitsuba_oidn as oidn
|
|
10
|
+
|
|
11
|
+
mi.set_variant("cuda_ad_rgb")
|
|
12
|
+
scene = mi.load_dict(mi.cornell_box())
|
|
13
|
+
color = mi.render(scene, spp=16) # mi.TensorXf of shape (H, W, 3)
|
|
14
|
+
denoised = oidn.denoise(color, hdr=True) # also an mi.TensorXf, on the GPU
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
The `color` argument can be any array type that supports DLPack or the buffer
|
|
18
|
+
protocol, and the result uses the same type and device.
|
|
19
|
+
|
|
20
|
+
The package runs on Intel/ARM CPUs, CUDA, and on Apple Metal. It exchanges
|
|
21
|
+
tensors via the buffer protocol and DLPack for compatibility with Dr.Jit,
|
|
22
|
+
NumPy, PyTorch, JAX, MLX, etc., and it accesses their memory directly whenever
|
|
23
|
+
the target device can.
|
|
24
|
+
|
|
25
|
+
## Why another binding?
|
|
26
|
+
|
|
27
|
+
Several unofficial bindings of OIDN already exist (e.g.,
|
|
28
|
+
[pyoidn](https://github.com/Hyiker/pyoidn), which uses cffi to expose the C API
|
|
29
|
+
directly). This project uses [nanobind](https://github.com/wjakob/nanobind) to
|
|
30
|
+
create bindings that feel more natural in Python. They automatically commit and
|
|
31
|
+
release resources and raise errors as Python exceptions. Pixel formats and
|
|
32
|
+
dimensions are inferred from nd-array signatures.
|
|
33
|
+
|
|
34
|
+
The bindings are designed to interoperate with
|
|
35
|
+
[Dr.Jit](https://github.com/mitsuba-renderer/drjit) and [Mitsuba
|
|
36
|
+
3](https://github.com/mitsuba-renderer/mitsuba3). The copy of OIDN bundled here
|
|
37
|
+
is modified to use Dr.Jit's
|
|
38
|
+
[nanothread](https://github.com/mitsuba-renderer/nanothread) thread pool
|
|
39
|
+
instead of spinning up another redundant thread pool via oneTBB. For this
|
|
40
|
+
reason, the package depends on Dr.Jit.
|
|
41
|
+
|
|
42
|
+
## Installation
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
pip install mitsuba-oidn
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Wheels are available for Linux (x86_64, aarch64), Windows (x86_64), and
|
|
49
|
+
macOS (arm64), matching the platforms supported by Dr.Jit. The Linux x86_64
|
|
50
|
+
and Windows wheels include the CUDA device, which activates when an NVIDIA
|
|
51
|
+
driver is present. The macOS wheel includes the Metal device.
|
|
52
|
+
|
|
53
|
+
## The `denoise()` function
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
oidn.denoise(color, albedo=None, normal=None, *, hdr=False, srgb=False,
|
|
57
|
+
clean_aux=False, quality=oidn.Quality.High, input_scale=None,
|
|
58
|
+
filter="RT", device=None, output=None)
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Images are arrays of shape `(H, W)` or `(H, W, C)` with `C <= 4` and dtype
|
|
62
|
+
`float32` or `float16`. A fourth channel is ignored on input. The result has
|
|
63
|
+
the same framework and lives on the same device as `color`: a NumPy array
|
|
64
|
+
yields a NumPy array, a CUDA tensor yields a CUDA tensor. Its shape is
|
|
65
|
+
`(H, W, min(C, 3))`, or `(H, W)` for two-dimensional input.
|
|
66
|
+
|
|
67
|
+
- `hdr`, `srgb`, `clean_aux`, `quality`, and `input_scale` map to the
|
|
68
|
+
parameters of the OIDN `RT` filter. Set `hdr=True` for linear radiance
|
|
69
|
+
values without an upper bound, and `clean_aux=True` when the albedo and
|
|
70
|
+
normal images are noise-free. See the [OIDN
|
|
71
|
+
documentation](https://www.openimagedenoise.org/documentation.html) for
|
|
72
|
+
details.
|
|
73
|
+
- `device` selects the device. By default, CUDA arrays use a CUDA device with
|
|
74
|
+
the matching ordinal, and host arrays use the fastest physical device in the
|
|
75
|
+
system, which can be overridden with the `OIDN_DEFAULT_DEVICE` environment
|
|
76
|
+
variable (`cpu`, `cuda`, `metal`, or a physical device ID).
|
|
77
|
+
- `output` supplies a preallocated array that is filled in place and returned.
|
|
78
|
+
With an RGBA output array, OIDN writes the RGB channels and leaves alpha
|
|
79
|
+
untouched.
|
|
80
|
+
- Filters are expensive to create, so `denoise()` caches a few of them, keyed
|
|
81
|
+
on image size, format, feature set, and parameters. Repeated calls at the
|
|
82
|
+
same resolution pay only for the actual filtering.
|
|
83
|
+
|
|
84
|
+
When the device cannot access an input array directly, for example a NumPy
|
|
85
|
+
array passed to a CUDA device, `denoise()` copies it into a device buffer.
|
|
86
|
+
Otherwise no copies are made.
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
import numpy as np
|
|
90
|
+
import torch
|
|
91
|
+
import mitsuba_oidn as oidn
|
|
92
|
+
|
|
93
|
+
# NumPy, CPU or Metal depending on the fastest available device
|
|
94
|
+
out = oidn.denoise(np.asarray(color, dtype=np.float32), hdr=True)
|
|
95
|
+
|
|
96
|
+
# PyTorch on the GPU: zero-copy in and out
|
|
97
|
+
color = torch.rand(1080, 1920, 3, device="cuda")
|
|
98
|
+
out = oidn.denoise(color, quality=oidn.Quality.Balanced)
|
|
99
|
+
assert out.device == color.device
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## The object API
|
|
103
|
+
|
|
104
|
+
The `denoise()` function covers the common case. The classes below mirror the
|
|
105
|
+
OIDN object model for applications that need control over devices, memory,
|
|
106
|
+
and filter lifetime, for instance when denoising many frames or several AOVs
|
|
107
|
+
that share auxiliary images.
|
|
108
|
+
|
|
109
|
+
### Devices
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
oidn.physical_devices() # list of PhysicalDevice: id, name, type, uuid, ...
|
|
113
|
+
|
|
114
|
+
dev = oidn.Device() # fastest physical device
|
|
115
|
+
dev = oidn.Device(oidn.DeviceType.CPU)
|
|
116
|
+
dev = oidn.Device.from_physical(id) # also from_uuid(), from_luid(), from_pci_address()
|
|
117
|
+
dev = oidn.Device.cuda(device_id=0, stream=torch.cuda.current_stream().cuda_stream)
|
|
118
|
+
dev = oidn.Device.metal(command_queue) # raw id<MTLCommandQueue> pointer
|
|
119
|
+
|
|
120
|
+
dev.num_threads = 4 # CPU only: 0 shares the Dr.Jit pool (default),
|
|
121
|
+
# a positive value creates a private pool
|
|
122
|
+
dev.verbose = 1
|
|
123
|
+
dev.type, dev.version, dev.system_memory_supported, dev.managed_memory_supported
|
|
124
|
+
dev.sync() # wait for asynchronous work
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
A device commits itself when the first buffer or filter is created. Parameters
|
|
128
|
+
such as `num_threads` must be set before that point.
|
|
129
|
+
|
|
130
|
+
### Filters
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
flt = dev.new_filter("RT") # or "RTLightmap"
|
|
134
|
+
|
|
135
|
+
flt.set_image("color", color) # arrays: layout inferred, zero-copy when possible
|
|
136
|
+
flt.set_image("albedo", albedo)
|
|
137
|
+
flt.set_image("normal", normal)
|
|
138
|
+
flt.set_image("output", output) # must be writable
|
|
139
|
+
flt.hdr = True
|
|
140
|
+
flt.clean_aux = True
|
|
141
|
+
flt.quality = oidn.Quality.High
|
|
142
|
+
flt.input_scale = 0.5 # None selects automatic scaling
|
|
143
|
+
flt.max_memory_mb = 2048
|
|
144
|
+
flt.set("cleanAux", True) # generic access by OIDN parameter name
|
|
145
|
+
flt.set_data("weights", blob) # user-trained weights (bytes or uint8 array)
|
|
146
|
+
flt.set_progress_monitor(lambda p: True) # return False to cancel
|
|
147
|
+
|
|
148
|
+
flt.execute() # commits pending changes, runs, and waits
|
|
149
|
+
flt.execute_async(); dev.sync()
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
`set_image()` accepts arrays with the layout rules of `denoise()`. Row and
|
|
153
|
+
pixel strides are passed to OIDN, so slices of larger arrays work as long as
|
|
154
|
+
the channel dimension stays contiguous. Whether an array can be bound without
|
|
155
|
+
a copy depends on the device:
|
|
156
|
+
|
|
157
|
+
| Device | Bound without copy |
|
|
158
|
+
|---|---|
|
|
159
|
+
| CPU | any host array, including CUDA pinned and managed memory |
|
|
160
|
+
| CUDA | arrays on the same CUDA device, pinned host memory, and host memory if the GPU supports pageable memory access |
|
|
161
|
+
| Metal | any host array, and views of buffers created on the device |
|
|
162
|
+
|
|
163
|
+
Use `dev.can_share(array)` to test this in advance. When binding is not
|
|
164
|
+
possible, `set_image()` raises a `TypeError` that points at the buffer API.
|
|
165
|
+
|
|
166
|
+
Filters can also take a `Buffer` with an explicit description:
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
flt.set_image("color", buf, format=oidn.Format.Float3, width=w, height=h,
|
|
170
|
+
byte_offset=0, pixel_stride=0, row_stride=0)
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### Buffers
|
|
174
|
+
|
|
175
|
+
Buffers are memory allocations made by a device. They are the way to work
|
|
176
|
+
with memory that the host cannot address, such as dedicated GPU memory, and
|
|
177
|
+
they provide zero-copy views on unified-memory systems.
|
|
178
|
+
|
|
179
|
+
```python
|
|
180
|
+
buf = dev.new_buffer(nbytes) # host and device accessible
|
|
181
|
+
buf = dev.new_buffer(nbytes, oidn.Storage.Device) # device memory only
|
|
182
|
+
buf = dev.new_shared_buffer(array) # wrap device-accessible memory
|
|
183
|
+
|
|
184
|
+
buf.size, buf.storage, buf.device, buf.data_ptr
|
|
185
|
+
|
|
186
|
+
view = buf.view("float32", (h, w, 3)) # DLPack and buffer-protocol object
|
|
187
|
+
img = np.from_dlpack(view) # or torch.from_dlpack(view), ...
|
|
188
|
+
|
|
189
|
+
buf.write(host_array); buf.read(host_array) # copies through the host
|
|
190
|
+
buf.write_async(src); buf.read_async(dst); dev.sync()
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
Arrays created from `buf.view()` are recognized by `set_image()` and bound
|
|
194
|
+
through the underlying buffer. On Apple silicon, rendering into such a view
|
|
195
|
+
and denoising it involves no copies at all. On a CUDA device, a device-storage
|
|
196
|
+
buffer viewed through `torch.from_dlpack()` gives a tensor that OIDN wrote
|
|
197
|
+
directly.
|
|
198
|
+
|
|
199
|
+
### Errors
|
|
200
|
+
|
|
201
|
+
All OIDN errors raise `oidn.Error`, whose `code` attribute is an
|
|
202
|
+
`oidn.ErrorCode`. Cancellation through a progress monitor raises
|
|
203
|
+
`oidn.Error` with `ErrorCode.Cancelled`. An exception raised inside the
|
|
204
|
+
progress monitor cancels the filter and propagates unchanged.
|
|
205
|
+
|
|
206
|
+
## Building from source
|
|
207
|
+
|
|
208
|
+
The build needs CMake 3.21 or newer, a C++17 compiler, and a binary release
|
|
209
|
+
of [ISPC](https://ispc.github.io/downloads.html) unpacked into `ext/ispc`,
|
|
210
|
+
so that `ext/ispc/bin/ispc` exists. Metal support requires Xcode 15 or newer.
|
|
211
|
+
CUDA support requires CUDA 12.8 or newer and is enabled automatically when the
|
|
212
|
+
toolkit is found.
|
|
213
|
+
|
|
214
|
+
```
|
|
215
|
+
git clone --recursive https://github.com/mitsuba-renderer/mitsuba-oidn
|
|
216
|
+
cd mitsuba-oidn
|
|
217
|
+
pip install nanobind==3.0.1 scikit-build-core
|
|
218
|
+
pip install --no-build-isolation -ve .
|
|
219
|
+
pytest
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
The OIDN weights are stored with git-lfs, which must be installed before
|
|
223
|
+
cloning.
|
|
224
|
+
|
|
225
|
+
## License
|
|
226
|
+
|
|
227
|
+
mitsuba-oidn is licensed under the BSD 3-Clause license. It bundles Intel Open
|
|
228
|
+
Image Denoise, which is licensed under the Apache License 2.0.
|