mitsuba-oidn 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (298) hide show
  1. mitsuba_oidn-0.1.0/.gitignore +7 -0
  2. mitsuba_oidn-0.1.0/.gitmodules +3 -0
  3. mitsuba_oidn-0.1.0/CMakeLists.txt +149 -0
  4. mitsuba_oidn-0.1.0/PKG-INFO +245 -0
  5. mitsuba_oidn-0.1.0/README.md +228 -0
  6. mitsuba_oidn-0.1.0/ext/oidn/.gitignore +97 -0
  7. mitsuba_oidn-0.1.0/ext/oidn/CHANGELOG.md +431 -0
  8. mitsuba_oidn-0.1.0/ext/oidn/CMakeLists.txt +210 -0
  9. mitsuba_oidn-0.1.0/ext/oidn/LICENSE.txt +202 -0
  10. mitsuba_oidn-0.1.0/ext/oidn/MITSUBA_CHANGES.md +20 -0
  11. mitsuba_oidn-0.1.0/ext/oidn/README.md +2270 -0
  12. mitsuba_oidn-0.1.0/ext/oidn/SECURITY.md +20 -0
  13. mitsuba_oidn-0.1.0/ext/oidn/api/CMakeLists.txt +42 -0
  14. mitsuba_oidn-0.1.0/ext/oidn/api/api.cpp +1110 -0
  15. mitsuba_oidn-0.1.0/ext/oidn/cmake/Config.cmake.in +23 -0
  16. mitsuba_oidn-0.1.0/ext/oidn/cmake/FindLevelZero.cmake +107 -0
  17. mitsuba_oidn-0.1.0/ext/oidn/cmake/FindTBB.cmake +493 -0
  18. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_bnns.cmake +14 -0
  19. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_common.cmake +65 -0
  20. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_common_external.cmake +16 -0
  21. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_ispc.cmake +242 -0
  22. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_macros.cmake +171 -0
  23. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_metal.cmake +80 -0
  24. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_package.cmake +99 -0
  25. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_platform.cmake +301 -0
  26. mitsuba_oidn-0.1.0/ext/oidn/cmake/oidn_version.cmake +10 -0
  27. mitsuba_oidn-0.1.0/ext/oidn/common/CMakeLists.txt +54 -0
  28. mitsuba_oidn-0.1.0/ext/oidn/common/common.cpp +65 -0
  29. mitsuba_oidn-0.1.0/ext/oidn/common/common.h +35 -0
  30. mitsuba_oidn-0.1.0/ext/oidn/common/export.linux.map.in +12 -0
  31. mitsuba_oidn-0.1.0/ext/oidn/common/export.macos.map.in +7 -0
  32. mitsuba_oidn-0.1.0/ext/oidn/common/half.cpp +116 -0
  33. mitsuba_oidn-0.1.0/ext/oidn/common/half.h +31 -0
  34. mitsuba_oidn-0.1.0/ext/oidn/common/oidn.rc +36 -0
  35. mitsuba_oidn-0.1.0/ext/oidn/common/oidn_utils.cpp +117 -0
  36. mitsuba_oidn-0.1.0/ext/oidn/common/oidn_utils.h +25 -0
  37. mitsuba_oidn-0.1.0/ext/oidn/common/platform.cpp +137 -0
  38. mitsuba_oidn-0.1.0/ext/oidn/common/platform.h +389 -0
  39. mitsuba_oidn-0.1.0/ext/oidn/common/timer.h +36 -0
  40. mitsuba_oidn-0.1.0/ext/oidn/core/CMakeLists.txt +126 -0
  41. mitsuba_oidn-0.1.0/ext/oidn/core/arena.cpp +103 -0
  42. mitsuba_oidn-0.1.0/ext/oidn/core/arena.h +95 -0
  43. mitsuba_oidn-0.1.0/ext/oidn/core/arena_planner.cpp +162 -0
  44. mitsuba_oidn-0.1.0/ext/oidn/core/arena_planner.h +97 -0
  45. mitsuba_oidn-0.1.0/ext/oidn/core/autoexposure.h +56 -0
  46. mitsuba_oidn-0.1.0/ext/oidn/core/buffer.cpp +202 -0
  47. mitsuba_oidn-0.1.0/ext/oidn/core/buffer.h +153 -0
  48. mitsuba_oidn-0.1.0/ext/oidn/core/color.cpp +18 -0
  49. mitsuba_oidn-0.1.0/ext/oidn/core/color.h +220 -0
  50. mitsuba_oidn-0.1.0/ext/oidn/core/concat_conv.cpp +212 -0
  51. mitsuba_oidn-0.1.0/ext/oidn/core/concat_conv.h +133 -0
  52. mitsuba_oidn-0.1.0/ext/oidn/core/context.cpp +56 -0
  53. mitsuba_oidn-0.1.0/ext/oidn/core/context.h +139 -0
  54. mitsuba_oidn-0.1.0/ext/oidn/core/conv.cpp +87 -0
  55. mitsuba_oidn-0.1.0/ext/oidn/core/conv.h +63 -0
  56. mitsuba_oidn-0.1.0/ext/oidn/core/data.h +45 -0
  57. mitsuba_oidn-0.1.0/ext/oidn/core/device.cpp +373 -0
  58. mitsuba_oidn-0.1.0/ext/oidn/core/device.h +202 -0
  59. mitsuba_oidn-0.1.0/ext/oidn/core/device_factory.h +53 -0
  60. mitsuba_oidn-0.1.0/ext/oidn/core/engine.cpp +162 -0
  61. mitsuba_oidn-0.1.0/ext/oidn/core/engine.h +149 -0
  62. mitsuba_oidn-0.1.0/ext/oidn/core/exception.cpp +17 -0
  63. mitsuba_oidn-0.1.0/ext/oidn/core/exception.h +36 -0
  64. mitsuba_oidn-0.1.0/ext/oidn/core/filter.cpp +86 -0
  65. mitsuba_oidn-0.1.0/ext/oidn/core/filter.h +53 -0
  66. mitsuba_oidn-0.1.0/ext/oidn/core/graph.cpp +591 -0
  67. mitsuba_oidn-0.1.0/ext/oidn/core/graph.h +132 -0
  68. mitsuba_oidn-0.1.0/ext/oidn/core/heap.cpp +92 -0
  69. mitsuba_oidn-0.1.0/ext/oidn/core/heap.h +69 -0
  70. mitsuba_oidn-0.1.0/ext/oidn/core/image.cpp +148 -0
  71. mitsuba_oidn-0.1.0/ext/oidn/core/image.h +121 -0
  72. mitsuba_oidn-0.1.0/ext/oidn/core/image_accessor.h +249 -0
  73. mitsuba_oidn-0.1.0/ext/oidn/core/image_copy.h +30 -0
  74. mitsuba_oidn-0.1.0/ext/oidn/core/input_process.cpp +74 -0
  75. mitsuba_oidn-0.1.0/ext/oidn/core/input_process.h +53 -0
  76. mitsuba_oidn-0.1.0/ext/oidn/core/kernel.h +464 -0
  77. mitsuba_oidn-0.1.0/ext/oidn/core/math.h +92 -0
  78. mitsuba_oidn-0.1.0/ext/oidn/core/module.cpp +199 -0
  79. mitsuba_oidn-0.1.0/ext/oidn/core/module.h +53 -0
  80. mitsuba_oidn-0.1.0/ext/oidn/core/op.cpp +24 -0
  81. mitsuba_oidn-0.1.0/ext/oidn/core/op.h +53 -0
  82. mitsuba_oidn-0.1.0/ext/oidn/core/output_process.cpp +54 -0
  83. mitsuba_oidn-0.1.0/ext/oidn/core/output_process.h +42 -0
  84. mitsuba_oidn-0.1.0/ext/oidn/core/pool.cpp +37 -0
  85. mitsuba_oidn-0.1.0/ext/oidn/core/pool.h +38 -0
  86. mitsuba_oidn-0.1.0/ext/oidn/core/progress.cpp +40 -0
  87. mitsuba_oidn-0.1.0/ext/oidn/core/progress.h +48 -0
  88. mitsuba_oidn-0.1.0/ext/oidn/core/record.h +31 -0
  89. mitsuba_oidn-0.1.0/ext/oidn/core/ref.h +161 -0
  90. mitsuba_oidn-0.1.0/ext/oidn/core/rt_filter.cpp +131 -0
  91. mitsuba_oidn-0.1.0/ext/oidn/core/rt_filter.h +25 -0
  92. mitsuba_oidn-0.1.0/ext/oidn/core/rtlightmap_filter.cpp +77 -0
  93. mitsuba_oidn-0.1.0/ext/oidn/core/rtlightmap_filter.h +25 -0
  94. mitsuba_oidn-0.1.0/ext/oidn/core/semaphore.h +21 -0
  95. mitsuba_oidn-0.1.0/ext/oidn/core/subdevice.cpp +36 -0
  96. mitsuba_oidn-0.1.0/ext/oidn/core/subdevice.h +39 -0
  97. mitsuba_oidn-0.1.0/ext/oidn/core/tensor.cpp +186 -0
  98. mitsuba_oidn-0.1.0/ext/oidn/core/tensor.h +119 -0
  99. mitsuba_oidn-0.1.0/ext/oidn/core/tensor_accessor.h +92 -0
  100. mitsuba_oidn-0.1.0/ext/oidn/core/tensor_desc.h +187 -0
  101. mitsuba_oidn-0.1.0/ext/oidn/core/tensor_layout.h +449 -0
  102. mitsuba_oidn-0.1.0/ext/oidn/core/tensor_reorder.cpp +101 -0
  103. mitsuba_oidn-0.1.0/ext/oidn/core/tensor_reorder.h +12 -0
  104. mitsuba_oidn-0.1.0/ext/oidn/core/thread.cpp +335 -0
  105. mitsuba_oidn-0.1.0/ext/oidn/core/thread.h +197 -0
  106. mitsuba_oidn-0.1.0/ext/oidn/core/tile.h +20 -0
  107. mitsuba_oidn-0.1.0/ext/oidn/core/tza.cpp +138 -0
  108. mitsuba_oidn-0.1.0/ext/oidn/core/tza.h +13 -0
  109. mitsuba_oidn-0.1.0/ext/oidn/core/unet_filter.cpp +691 -0
  110. mitsuba_oidn-0.1.0/ext/oidn/core/unet_filter.h +131 -0
  111. mitsuba_oidn-0.1.0/ext/oidn/core/upsample.cpp +37 -0
  112. mitsuba_oidn-0.1.0/ext/oidn/core/upsample.h +37 -0
  113. mitsuba_oidn-0.1.0/ext/oidn/core/vec.h +279 -0
  114. mitsuba_oidn-0.1.0/ext/oidn/core/verbose.h +47 -0
  115. mitsuba_oidn-0.1.0/ext/oidn/devices/CMakeLists.txt +190 -0
  116. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/CMakeLists.txt +184 -0
  117. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_common.cpp +65 -0
  118. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_common.h +15 -0
  119. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_conv.cpp +73 -0
  120. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_conv.h +29 -0
  121. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_engine.cpp +24 -0
  122. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_engine.h +20 -0
  123. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_pool.cpp +51 -0
  124. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/bnns/bnns_pool.h +26 -0
  125. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/color.ispc +172 -0
  126. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/color.isph +41 -0
  127. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_autoexposure.cpp +62 -0
  128. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_autoexposure.h +23 -0
  129. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_autoexposure.ispc +25 -0
  130. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_common.cpp +190 -0
  131. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_common.h +18 -0
  132. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv.cpp +103 -0
  133. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv.h +27 -0
  134. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv.ispc +237 -0
  135. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv_amx.cpp +158 -0
  136. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv_amx.h +36 -0
  137. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_conv_amx.ispc +424 -0
  138. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_device.cpp +264 -0
  139. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_device.h +69 -0
  140. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_engine.cpp +217 -0
  141. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_engine.h +74 -0
  142. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_image_copy.cpp +31 -0
  143. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_image_copy.h +23 -0
  144. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_image_copy.ispc +19 -0
  145. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process.cpp +48 -0
  146. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process.h +23 -0
  147. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process.isph +136 -0
  148. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process_f16.ispc +10 -0
  149. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_input_process_f32.ispc +10 -0
  150. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_module.cpp +24 -0
  151. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process.cpp +45 -0
  152. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process.h +23 -0
  153. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process.isph +71 -0
  154. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process_f16.ispc +10 -0
  155. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_output_process_f32.ispc +10 -0
  156. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool.cpp +47 -0
  157. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool.h +23 -0
  158. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool.isph +35 -0
  159. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool_f16.ispc +14 -0
  160. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_pool_f32.ispc +14 -0
  161. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample.cpp +83 -0
  162. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample.h +23 -0
  163. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample.isph +34 -0
  164. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample_f16.ispc +14 -0
  165. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/cpu_upsample_f32.ispc +14 -0
  166. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/image_accessor.isph +84 -0
  167. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/math.isph +107 -0
  168. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/platform.ispc +38 -0
  169. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/platform.isph +47 -0
  170. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tasking.cpp +51 -0
  171. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tasking.h +170 -0
  172. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tensor_accessor.isph +218 -0
  173. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/tile.isph +12 -0
  174. mitsuba_oidn-0.1.0/ext/oidn/devices/cpu/vec.isph +693 -0
  175. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/CMakeLists.txt +141 -0
  176. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_conv.cu +58 -0
  177. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_conv.h +13 -0
  178. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_device.cpp +278 -0
  179. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_device.h +65 -0
  180. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_engine.cu +262 -0
  181. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_engine.h +168 -0
  182. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_buffer.cpp +110 -0
  183. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_buffer.h +30 -0
  184. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_semaphore.cpp +67 -0
  185. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_external_semaphore.h +34 -0
  186. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cuda_module.cpp +46 -0
  187. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/curtn.cpp +1233 -0
  188. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/curtn.h +17 -0
  189. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cutlass_conv.h +333 -0
  190. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cutlass_conv_sm75.cu +25 -0
  191. mitsuba_oidn-0.1.0/ext/oidn/devices/cuda/cutlass_conv_sm80.cu +25 -0
  192. mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_autoexposure.h +267 -0
  193. mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_image_copy.h +73 -0
  194. mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_input_process.h +280 -0
  195. mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_output_process.h +133 -0
  196. mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_pool.h +84 -0
  197. mitsuba_oidn-0.1.0/ext/oidn/devices/gpu/gpu_upsample.h +84 -0
  198. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/CMakeLists.txt +107 -0
  199. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/ck_conv.h +71 -0
  200. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/ck_conv_dl.cpp +217 -0
  201. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/ck_conv_wmma.cpp +222 -0
  202. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_conv.cpp +60 -0
  203. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_conv.h +13 -0
  204. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_device.cpp +252 -0
  205. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_device.h +66 -0
  206. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_engine.cpp +268 -0
  207. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_engine.h +163 -0
  208. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_buffer.cpp +112 -0
  209. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_buffer.h +30 -0
  210. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_semaphore.cpp +58 -0
  211. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_external_semaphore.h +34 -0
  212. mitsuba_oidn-0.1.0/ext/oidn/devices/hip/hip_module.cpp +41 -0
  213. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/CMakeLists.txt +60 -0
  214. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_buffer.h +50 -0
  215. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_buffer.mm +245 -0
  216. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_common.h +32 -0
  217. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_common.mm +104 -0
  218. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_concat_conv.h +33 -0
  219. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_concat_conv.mm +117 -0
  220. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_conv.h +32 -0
  221. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_conv.mm +118 -0
  222. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_device.h +52 -0
  223. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_device.mm +145 -0
  224. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_engine.h +141 -0
  225. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_engine.mm +211 -0
  226. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_heap.h +35 -0
  227. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_heap.mm +76 -0
  228. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_kernels.metal +79 -0
  229. mitsuba_oidn-0.1.0/ext/oidn/devices/metal/metal_module.mm +39 -0
  230. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/CMakeLists.txt +210 -0
  231. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_common.h +392 -0
  232. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_concat_conv.h +175 -0
  233. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_conv.h +169 -0
  234. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_conv_base.h +213 -0
  235. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_device.cpp +592 -0
  236. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_device.h +101 -0
  237. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_device_table.h +101 -0
  238. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_engine.cpp +231 -0
  239. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_engine.h +174 -0
  240. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_external_buffer.cpp +82 -0
  241. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_external_buffer.h +28 -0
  242. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_input_process.h +176 -0
  243. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_module.cpp +51 -0
  244. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops.h +48 -0
  245. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xe2.cpp +9 -0
  246. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xehpc.cpp +9 -0
  247. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xehpg.cpp +9 -0
  248. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_ops_xelp.cpp +9 -0
  249. mitsuba_oidn-0.1.0/ext/oidn/devices/sycl/sycl_output_process.h +128 -0
  250. mitsuba_oidn-0.1.0/ext/oidn/external/catch.hpp +17937 -0
  251. mitsuba_oidn-0.1.0/ext/oidn/external/level_zero/ze_intel_gpu.h +610 -0
  252. mitsuba_oidn-0.1.0/ext/oidn/external/level_zero/ze_stypes.h +69 -0
  253. mitsuba_oidn-0.1.0/ext/oidn/include/OpenImageDenoise/config.h.in +82 -0
  254. mitsuba_oidn-0.1.0/ext/oidn/include/OpenImageDenoise/oidn.h +637 -0
  255. mitsuba_oidn-0.1.0/ext/oidn/include/OpenImageDenoise/oidn.hpp +1293 -0
  256. mitsuba_oidn-0.1.0/ext/oidn/scripts/blob_to_cpp.py +91 -0
  257. mitsuba_oidn-0.1.0/ext/oidn/scripts/build.py +243 -0
  258. mitsuba_oidn-0.1.0/ext/oidn/scripts/build_src.py +32 -0
  259. mitsuba_oidn-0.1.0/ext/oidn/scripts/build_weights.py +32 -0
  260. mitsuba_oidn-0.1.0/ext/oidn/scripts/common.py +69 -0
  261. mitsuba_oidn-0.1.0/ext/oidn/scripts/csan.supp.xml +10 -0
  262. mitsuba_oidn-0.1.0/ext/oidn/scripts/protex_scan.sh +59 -0
  263. mitsuba_oidn-0.1.0/ext/oidn/scripts/store-files.sh +11 -0
  264. mitsuba_oidn-0.1.0/ext/oidn/scripts/test.py +358 -0
  265. mitsuba_oidn-0.1.0/ext/oidn/scripts/valgrind.supp +120 -0
  266. mitsuba_oidn-0.1.0/ext/oidn/third-party-programs-DPCPP.txt +110 -0
  267. mitsuba_oidn-0.1.0/ext/oidn/third-party-programs-oneTBB.txt +198 -0
  268. mitsuba_oidn-0.1.0/ext/oidn/third-party-programs.txt +636 -0
  269. mitsuba_oidn-0.1.0/ext/oidn/weights/LICENSE.txt +202 -0
  270. mitsuba_oidn-0.1.0/ext/oidn/weights/README.md +7 -0
  271. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_alb.tza +0 -0
  272. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_alb_large.tza +0 -0
  273. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr.tza +0 -0
  274. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb.tza +0 -0
  275. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb_nrm.tza +0 -0
  276. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb_nrm_small.tza +0 -0
  277. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_alb_small.tza +0 -0
  278. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_calb_cnrm.tza +0 -0
  279. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_calb_cnrm_large.tza +0 -0
  280. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_calb_cnrm_small.tza +0 -0
  281. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_hdr_small.tza +0 -0
  282. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr.tza +0 -0
  283. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb.tza +0 -0
  284. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb_nrm.tza +0 -0
  285. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb_nrm_small.tza +0 -0
  286. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_alb_small.tza +0 -0
  287. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_calb_cnrm.tza +0 -0
  288. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_calb_cnrm_small.tza +0 -0
  289. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_ldr_small.tza +0 -0
  290. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_nrm.tza +0 -0
  291. mitsuba_oidn-0.1.0/ext/oidn/weights/rt_nrm_large.tza +0 -0
  292. mitsuba_oidn-0.1.0/ext/oidn/weights/rtlightmap_dir.tza +0 -0
  293. mitsuba_oidn-0.1.0/ext/oidn/weights/rtlightmap_hdr.tza +0 -0
  294. mitsuba_oidn-0.1.0/pyproject.toml +83 -0
  295. mitsuba_oidn-0.1.0/src/ext.cpp +1109 -0
  296. mitsuba_oidn-0.1.0/src/mitsuba_oidn/__init__.py +76 -0
  297. mitsuba_oidn-0.1.0/src/mitsuba_oidn/_denoise.py +211 -0
  298. mitsuba_oidn-0.1.0/tests/test_basic.py +191 -0
@@ -0,0 +1,7 @@
1
+ build/
2
+ dist/
3
+ *.egg-info/
4
+ __pycache__/
5
+ *.pyc
6
+ ext/ispc/
7
+ .pytest_cache/
@@ -0,0 +1,3 @@
1
+ [submodule "ext/oidn"]
2
+ path = ext/oidn
3
+ url = https://github.com/mitsuba-renderer/oidn.git
@@ -0,0 +1,149 @@
1
+ cmake_minimum_required(VERSION 3.21...3.31)
2
+
3
+ project(mitsuba_oidn LANGUAGES C CXX)
4
+
5
+ if (NOT SKBUILD)
6
+ message(WARNING "This CMake file is meant to be executed through scikit-build-core. "
7
+ "To build the package, run 'pip install .' or, for development, "
8
+ "'pip install --no-build-isolation -ve .'")
9
+ endif()
10
+
11
+ find_package(Python 3.10
12
+ REQUIRED COMPONENTS Interpreter Development.Module
13
+ OPTIONAL_COMPONENTS Development.SABIModule)
14
+
15
+ find_package(nanobind CONFIG REQUIRED)
16
+
17
+ # ------------------------------------------------------------------------------
18
+ # Open Image Denoise (vendored, patched to use an external thread pool)
19
+ # ------------------------------------------------------------------------------
20
+
21
+ set(OIDN_APPS OFF CACHE BOOL "" FORCE)
22
+ set(OIDN_LIBRARY_NAME "mitsuba_oidn" CACHE STRING "" FORCE)
23
+ set(OIDN_API_NAMESPACE "mitsuba_oidn" CACHE STRING "" FORCE)
24
+ set(OIDN_LIBRARY_VERSIONED OFF CACHE BOOL "" FORCE)
25
+ set(OIDN_INSTALL_DEPENDENCIES OFF CACHE BOOL "" FORCE)
26
+ set(OIDN_DEVICE_CPU ON CACHE BOOL "" FORCE)
27
+
28
+ # ISPC binary unpacked into ext/ispc (see README)
29
+ if (NOT ISPC_EXECUTABLE)
30
+ set(_ispc "${CMAKE_CURRENT_SOURCE_DIR}/ext/ispc/bin/ispc")
31
+ if (WIN32)
32
+ set(_ispc "${_ispc}.exe")
33
+ endif()
34
+ if (EXISTS "${_ispc}")
35
+ set(ISPC_EXECUTABLE "${_ispc}" CACHE FILEPATH "Path to the ISPC executable." FORCE)
36
+ endif()
37
+ endif()
38
+
39
+ if (APPLE AND CMAKE_SYSTEM_PROCESSOR MATCHES "arm64")
40
+ option(MITSUBA_OIDN_METAL "Build the Metal device" ON)
41
+ else()
42
+ set(MITSUBA_OIDN_METAL OFF)
43
+ endif()
44
+
45
+ if (NOT APPLE)
46
+ option(MITSUBA_OIDN_CUDA "Build the CUDA device (requires the CUDA toolkit)" ON)
47
+ else()
48
+ set(MITSUBA_OIDN_CUDA OFF)
49
+ endif()
50
+
51
+ set(OIDN_DEVICE_METAL ${MITSUBA_OIDN_METAL} CACHE BOOL "" FORCE)
52
+
53
+ # All shared libraries live side by side in the package directory
54
+ if (APPLE)
55
+ set(_rpath "@loader_path")
56
+ else()
57
+ set(_rpath "$ORIGIN")
58
+ endif()
59
+ set(OIDN_INSTALL_RPATH "${_rpath}" CACHE STRING "" FORCE)
60
+
61
+ if (MITSUBA_OIDN_CUDA)
62
+ find_package(CUDAToolkit 12.8 QUIET)
63
+ if (NOT EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/ext/oidn/external/cutlass/include/cutlass")
64
+ # CUTLASS is a large submodule that source distributions leave out
65
+ message(STATUS "mitsuba-oidn: CUTLASS not found, skipping the CUDA device")
66
+ set(OIDN_DEVICE_CUDA OFF CACHE BOOL "" FORCE)
67
+ elseif (CUDAToolkit_FOUND)
68
+ message(STATUS "mitsuba-oidn: building the CUDA device (CUDA ${CUDAToolkit_VERSION})")
69
+ set(OIDN_DEVICE_CUDA ON CACHE BOOL "" FORCE)
70
+ else()
71
+ message(STATUS "mitsuba-oidn: CUDA toolkit not found, skipping the CUDA device")
72
+ set(OIDN_DEVICE_CUDA OFF CACHE BOOL "" FORCE)
73
+ endif()
74
+ else()
75
+ set(OIDN_DEVICE_CUDA OFF CACHE BOOL "" FORCE)
76
+ endif()
77
+
78
+ # The weights are stored with git-lfs; a pointer file in their place would build
79
+ # a library with unusable filters
80
+ set(_weights_probe "${CMAKE_CURRENT_SOURCE_DIR}/ext/oidn/weights/rt_hdr.tza")
81
+ if (NOT EXISTS "${_weights_probe}")
82
+ message(FATAL_ERROR "OIDN weights are missing. Run 'git submodule update --init --recursive'.")
83
+ endif()
84
+ file(SIZE "${_weights_probe}" _weights_size)
85
+ if (_weights_size LESS 100000)
86
+ message(FATAL_ERROR "OIDN weights are git-lfs pointer files. Run 'git lfs pull' in ext/oidn/weights.")
87
+ endif()
88
+
89
+ add_subdirectory(ext/oidn)
90
+
91
+ set(MITSUBA_OIDN_LIBS OpenImageDenoise OpenImageDenoise_core OpenImageDenoise_device_cpu)
92
+ if (OIDN_DEVICE_METAL)
93
+ list(APPEND MITSUBA_OIDN_LIBS OpenImageDenoise_device_metal)
94
+ endif()
95
+
96
+ # ------------------------------------------------------------------------------
97
+ # Python extension
98
+ # ------------------------------------------------------------------------------
99
+
100
+ nanobind_add_module(_mitsuba_oidn_ext
101
+ NB_DOMAIN mitsuba_oidn
102
+ BACKEND_MODULE nanobind_backend
103
+ src/ext.cpp
104
+ )
105
+
106
+ target_link_libraries(_mitsuba_oidn_ext PRIVATE OpenImageDenoise)
107
+ target_compile_features(_mitsuba_oidn_ext PRIVATE cxx_std_17)
108
+
109
+ set_target_properties(_mitsuba_oidn_ext PROPERTIES INSTALL_RPATH "${_rpath}")
110
+
111
+ install(TARGETS _mitsuba_oidn_ext ${MITSUBA_OIDN_LIBS}
112
+ COMPONENT python
113
+ LIBRARY DESTINATION mitsuba_oidn
114
+ RUNTIME DESTINATION mitsuba_oidn
115
+ ARCHIVE DESTINATION mitsuba_oidn/lib
116
+ )
117
+
118
+ # Type stubs (the build-tree extension resolves the OIDN libraries via its build rpath)
119
+ if (NOT WIN32)
120
+ nanobind_add_stub(mitsuba_oidn_stub
121
+ MODULE _mitsuba_oidn_ext
122
+ OUTPUT "${CMAKE_CURRENT_BINARY_DIR}/_mitsuba_oidn_ext.pyi"
123
+ MARKER_FILE "${CMAKE_CURRENT_BINARY_DIR}/py.typed"
124
+ PYTHON_PATH $<TARGET_FILE_DIR:_mitsuba_oidn_ext>
125
+ DEPENDS _mitsuba_oidn_ext
126
+ )
127
+ install(FILES
128
+ "${CMAKE_CURRENT_BINARY_DIR}/_mitsuba_oidn_ext.pyi"
129
+ "${CMAKE_CURRENT_BINARY_DIR}/py.typed"
130
+ DESTINATION mitsuba_oidn
131
+ COMPONENT python
132
+ )
133
+ endif()
134
+
135
+ if (OIDN_DEVICE_CUDA)
136
+ # OIDN builds the CUDA device as an external project that pre-installs its module
137
+ # into a staging directory at build time
138
+ if (WIN32)
139
+ set(_cuda_dir "${CMAKE_INSTALL_BINDIR}")
140
+ else()
141
+ set(_cuda_dir "${CMAKE_INSTALL_LIBDIR}")
142
+ endif()
143
+ install(DIRECTORY "${OIDN_ROOT_BINARY_DIR}/devices/cuda/preinstall/${_cuda_dir}/"
144
+ DESTINATION mitsuba_oidn
145
+ COMPONENT python
146
+ USE_SOURCE_PERMISSIONS
147
+ FILES_MATCHING PATTERN "*mitsuba_oidn_device_cuda*"
148
+ )
149
+ endif()
@@ -0,0 +1,245 @@
1
+ Metadata-Version: 2.4
2
+ Name: mitsuba-oidn
3
+ Version: 0.1.0
4
+ Summary: Unofficial Python bindings for Intel Open Image Denoise (OIDN)
5
+ Author-Email: Wenzel Jakob <wenzel@mitsuba-renderer.org>
6
+ License-Expression: BSD-3-Clause
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Topic :: Multimedia :: Graphics :: Graphics Conversion
9
+ Project-URL: Homepage, https://github.com/mitsuba-renderer/mitsuba-oidn
10
+ Requires-Python: >=3.10
11
+ Requires-Dist: nanobind-backend>=1.0
12
+ Requires-Dist: drjit>=1.5.0
13
+ Provides-Extra: test
14
+ Requires-Dist: pytest; extra == "test"
15
+ Requires-Dist: numpy; extra == "test"
16
+ Description-Content-Type: text/markdown
17
+
18
+ # mitsuba-oidn
19
+
20
+ This package provides unofficial Python bindings for [Intel Open Image
21
+ Denoise](https://www.openimagedenoise.org) (OIDN) for use with Dr.Jit and the
22
+ Mitsuba renderer.
23
+
24
+ ```python
25
+ import mitsuba as mi
26
+ import mitsuba_oidn as oidn
27
+
28
+ mi.set_variant("cuda_ad_rgb")
29
+ scene = mi.load_dict(mi.cornell_box())
30
+ color = mi.render(scene, spp=16) # mi.TensorXf of shape (H, W, 3)
31
+ denoised = oidn.denoise(color, hdr=True) # also an mi.TensorXf, on the GPU
32
+ ```
33
+
34
+ The `color` argument can be any array type that supports DLPack or the buffer
35
+ protocol, and the result uses the same type and device.
36
+
37
+ The package runs on Intel/ARM CPUs, CUDA, and on Apple Metal. It exchanges
38
+ tensors via the buffer protocol and DLPack for compatibility with Dr.Jit,
39
+ NumPy, PyTorch, JAX, MLX, etc., and it accesses their memory directly whenever
40
+ the target device can.
41
+
42
+ ## Why another binding?
43
+
44
+ Several unofficial bindings of OIDN already exist (e.g.,
45
+ [pyoidn](https://github.com/Hyiker/pyoidn), which uses cffi to expose the C API
46
+ directly). This project uses [nanobind](https://github.com/wjakob/nanobind) to
47
+ create bindings that feel more natural in Python. They automatically commit and
48
+ release resources and raise errors as Python exceptions. Pixel formats and
49
+ dimensions are inferred from nd-array signatures.
50
+
51
+ The bindings are designed to interoperate with
52
+ [Dr.Jit](https://github.com/mitsuba-renderer/drjit) and [Mitsuba
53
+ 3](https://github.com/mitsuba-renderer/mitsuba3). The copy of OIDN bundled here
54
+ is modified to use Dr.Jit's
55
+ [nanothread](https://github.com/mitsuba-renderer/nanothread) thread pool
56
+ instead of spinning up another redundant thread pool via oneTBB. For this
57
+ reason, the package depends on Dr.Jit.
58
+
59
+ ## Installation
60
+
61
+ ```
62
+ pip install mitsuba-oidn
63
+ ```
64
+
65
+ Wheels are available for Linux (x86_64, aarch64), Windows (x86_64), and
66
+ macOS (arm64), matching the platforms supported by Dr.Jit. The Linux x86_64
67
+ and Windows wheels include the CUDA device, which activates when an NVIDIA
68
+ driver is present. The macOS wheel includes the Metal device.
69
+
70
+ ## The `denoise()` function
71
+
72
+ ```python
73
+ oidn.denoise(color, albedo=None, normal=None, *, hdr=False, srgb=False,
74
+ clean_aux=False, quality=oidn.Quality.High, input_scale=None,
75
+ filter="RT", device=None, output=None)
76
+ ```
77
+
78
+ Images are arrays of shape `(H, W)` or `(H, W, C)` with `C <= 4` and dtype
79
+ `float32` or `float16`. A fourth channel is ignored on input. The result has
80
+ the same framework and lives on the same device as `color`: a NumPy array
81
+ yields a NumPy array, a CUDA tensor yields a CUDA tensor. Its shape is
82
+ `(H, W, min(C, 3))`, or `(H, W)` for two-dimensional input.
83
+
84
+ - `hdr`, `srgb`, `clean_aux`, `quality`, and `input_scale` map to the
85
+ parameters of the OIDN `RT` filter. Set `hdr=True` for linear radiance
86
+ values without an upper bound, and `clean_aux=True` when the albedo and
87
+ normal images are noise-free. See the [OIDN
88
+ documentation](https://www.openimagedenoise.org/documentation.html) for
89
+ details.
90
+ - `device` selects the device. By default, CUDA arrays use a CUDA device with
91
+ the matching ordinal, and host arrays use the fastest physical device in the
92
+ system, which can be overridden with the `OIDN_DEFAULT_DEVICE` environment
93
+ variable (`cpu`, `cuda`, `metal`, or a physical device ID).
94
+ - `output` supplies a preallocated array that is filled in place and returned.
95
+ With an RGBA output array, OIDN writes the RGB channels and leaves alpha
96
+ untouched.
97
+ - Filters are expensive to create, so `denoise()` caches a few of them, keyed
98
+ on image size, format, feature set, and parameters. Repeated calls at the
99
+ same resolution pay only for the actual filtering.
100
+
101
+ When the device cannot access an input array directly, for example a NumPy
102
+ array passed to a CUDA device, `denoise()` copies it into a device buffer.
103
+ Otherwise no copies are made.
104
+
105
+ ```python
106
+ import numpy as np
107
+ import torch
108
+ import mitsuba_oidn as oidn
109
+
110
+ # NumPy, CPU or Metal depending on the fastest available device
111
+ out = oidn.denoise(np.asarray(color, dtype=np.float32), hdr=True)
112
+
113
+ # PyTorch on the GPU: zero-copy in and out
114
+ color = torch.rand(1080, 1920, 3, device="cuda")
115
+ out = oidn.denoise(color, quality=oidn.Quality.Balanced)
116
+ assert out.device == color.device
117
+ ```
118
+
119
+ ## The object API
120
+
121
+ The `denoise()` function covers the common case. The classes below mirror the
122
+ OIDN object model for applications that need control over devices, memory,
123
+ and filter lifetime, for instance when denoising many frames or several AOVs
124
+ that share auxiliary images.
125
+
126
+ ### Devices
127
+
128
+ ```python
129
+ oidn.physical_devices() # list of PhysicalDevice: id, name, type, uuid, ...
130
+
131
+ dev = oidn.Device() # fastest physical device
132
+ dev = oidn.Device(oidn.DeviceType.CPU)
133
+ dev = oidn.Device.from_physical(id) # also from_uuid(), from_luid(), from_pci_address()
134
+ dev = oidn.Device.cuda(device_id=0, stream=torch.cuda.current_stream().cuda_stream)
135
+ dev = oidn.Device.metal(command_queue) # raw id<MTLCommandQueue> pointer
136
+
137
+ dev.num_threads = 4 # CPU only: 0 shares the Dr.Jit pool (default),
138
+ # a positive value creates a private pool
139
+ dev.verbose = 1
140
+ dev.type, dev.version, dev.system_memory_supported, dev.managed_memory_supported
141
+ dev.sync() # wait for asynchronous work
142
+ ```
143
+
144
+ A device commits itself when the first buffer or filter is created. Parameters
145
+ such as `num_threads` must be set before that point.
146
+
147
+ ### Filters
148
+
149
+ ```python
150
+ flt = dev.new_filter("RT") # or "RTLightmap"
151
+
152
+ flt.set_image("color", color) # arrays: layout inferred, zero-copy when possible
153
+ flt.set_image("albedo", albedo)
154
+ flt.set_image("normal", normal)
155
+ flt.set_image("output", output) # must be writable
156
+ flt.hdr = True
157
+ flt.clean_aux = True
158
+ flt.quality = oidn.Quality.High
159
+ flt.input_scale = 0.5 # None selects automatic scaling
160
+ flt.max_memory_mb = 2048
161
+ flt.set("cleanAux", True) # generic access by OIDN parameter name
162
+ flt.set_data("weights", blob) # user-trained weights (bytes or uint8 array)
163
+ flt.set_progress_monitor(lambda p: True) # return False to cancel
164
+
165
+ flt.execute() # commits pending changes, runs, and waits
166
+ flt.execute_async(); dev.sync()
167
+ ```
168
+
169
+ `set_image()` accepts arrays with the layout rules of `denoise()`. Row and
170
+ pixel strides are passed to OIDN, so slices of larger arrays work as long as
171
+ the channel dimension stays contiguous. Whether an array can be bound without
172
+ a copy depends on the device:
173
+
174
+ | Device | Bound without copy |
175
+ |---|---|
176
+ | CPU | any host array, including CUDA pinned and managed memory |
177
+ | CUDA | arrays on the same CUDA device, pinned host memory, and host memory if the GPU supports pageable memory access |
178
+ | Metal | any host array, and views of buffers created on the device |
179
+
180
+ Use `dev.can_share(array)` to test this in advance. When binding is not
181
+ possible, `set_image()` raises a `TypeError` that points at the buffer API.
182
+
183
+ Filters can also take a `Buffer` with an explicit description:
184
+
185
+ ```python
186
+ flt.set_image("color", buf, format=oidn.Format.Float3, width=w, height=h,
187
+ byte_offset=0, pixel_stride=0, row_stride=0)
188
+ ```
189
+
190
+ ### Buffers
191
+
192
+ Buffers are memory allocations made by a device. They are the way to work
193
+ with memory that the host cannot address, such as dedicated GPU memory, and
194
+ they provide zero-copy views on unified-memory systems.
195
+
196
+ ```python
197
+ buf = dev.new_buffer(nbytes) # host and device accessible
198
+ buf = dev.new_buffer(nbytes, oidn.Storage.Device) # device memory only
199
+ buf = dev.new_shared_buffer(array) # wrap device-accessible memory
200
+
201
+ buf.size, buf.storage, buf.device, buf.data_ptr
202
+
203
+ view = buf.view("float32", (h, w, 3)) # DLPack and buffer-protocol object
204
+ img = np.from_dlpack(view) # or torch.from_dlpack(view), ...
205
+
206
+ buf.write(host_array); buf.read(host_array) # copies through the host
207
+ buf.write_async(src); buf.read_async(dst); dev.sync()
208
+ ```
209
+
210
+ Arrays created from `buf.view()` are recognized by `set_image()` and bound
211
+ through the underlying buffer. On Apple silicon, rendering into such a view
212
+ and denoising it involves no copies at all. On a CUDA device, a device-storage
213
+ buffer viewed through `torch.from_dlpack()` gives a tensor that OIDN wrote
214
+ directly.
215
+
216
+ ### Errors
217
+
218
+ All OIDN errors raise `oidn.Error`, whose `code` attribute is an
219
+ `oidn.ErrorCode`. Cancellation through a progress monitor raises
220
+ `oidn.Error` with `ErrorCode.Cancelled`. An exception raised inside the
221
+ progress monitor cancels the filter and propagates unchanged.
222
+
223
+ ## Building from source
224
+
225
+ The build needs CMake 3.21 or newer, a C++17 compiler, and a binary release
226
+ of [ISPC](https://ispc.github.io/downloads.html) unpacked into `ext/ispc`,
227
+ so that `ext/ispc/bin/ispc` exists. Metal support requires Xcode 15 or newer.
228
+ CUDA support requires CUDA 12.8 or newer and is enabled automatically when the
229
+ toolkit is found.
230
+
231
+ ```
232
+ git clone --recursive https://github.com/mitsuba-renderer/mitsuba-oidn
233
+ cd mitsuba-oidn
234
+ pip install nanobind==3.0.1 scikit-build-core
235
+ pip install --no-build-isolation -ve .
236
+ pytest
237
+ ```
238
+
239
+ The OIDN weights are stored with git-lfs, which must be installed before
240
+ cloning.
241
+
242
+ ## License
243
+
244
+ mitsuba-oidn is licensed under the BSD 3-Clause license. It bundles Intel Open
245
+ Image Denoise, which is licensed under the Apache License 2.0.
@@ -0,0 +1,228 @@
1
+ # mitsuba-oidn
2
+
3
+ This package provides unofficial Python bindings for [Intel Open Image
4
+ Denoise](https://www.openimagedenoise.org) (OIDN) for use with Dr.Jit and the
5
+ Mitsuba renderer.
6
+
7
+ ```python
8
+ import mitsuba as mi
9
+ import mitsuba_oidn as oidn
10
+
11
+ mi.set_variant("cuda_ad_rgb")
12
+ scene = mi.load_dict(mi.cornell_box())
13
+ color = mi.render(scene, spp=16) # mi.TensorXf of shape (H, W, 3)
14
+ denoised = oidn.denoise(color, hdr=True) # also an mi.TensorXf, on the GPU
15
+ ```
16
+
17
+ The `color` argument can be any array type that supports DLPack or the buffer
18
+ protocol, and the result uses the same type and device.
19
+
20
+ The package runs on Intel/ARM CPUs, CUDA, and on Apple Metal. It exchanges
21
+ tensors via the buffer protocol and DLPack for compatibility with Dr.Jit,
22
+ NumPy, PyTorch, JAX, MLX, etc., and it accesses their memory directly whenever
23
+ the target device can.
24
+
25
+ ## Why another binding?
26
+
27
+ Several unofficial bindings of OIDN already exist (e.g.,
28
+ [pyoidn](https://github.com/Hyiker/pyoidn), which uses cffi to expose the C API
29
+ directly). This project uses [nanobind](https://github.com/wjakob/nanobind) to
30
+ create bindings that feel more natural in Python. They automatically commit and
31
+ release resources and raise errors as Python exceptions. Pixel formats and
32
+ dimensions are inferred from nd-array signatures.
33
+
34
+ The bindings are designed to interoperate with
35
+ [Dr.Jit](https://github.com/mitsuba-renderer/drjit) and [Mitsuba
36
+ 3](https://github.com/mitsuba-renderer/mitsuba3). The copy of OIDN bundled here
37
+ is modified to use Dr.Jit's
38
+ [nanothread](https://github.com/mitsuba-renderer/nanothread) thread pool
39
+ instead of spinning up another redundant thread pool via oneTBB. For this
40
+ reason, the package depends on Dr.Jit.
41
+
42
+ ## Installation
43
+
44
+ ```
45
+ pip install mitsuba-oidn
46
+ ```
47
+
48
+ Wheels are available for Linux (x86_64, aarch64), Windows (x86_64), and
49
+ macOS (arm64), matching the platforms supported by Dr.Jit. The Linux x86_64
50
+ and Windows wheels include the CUDA device, which activates when an NVIDIA
51
+ driver is present. The macOS wheel includes the Metal device.
52
+
53
+ ## The `denoise()` function
54
+
55
+ ```python
56
+ oidn.denoise(color, albedo=None, normal=None, *, hdr=False, srgb=False,
57
+ clean_aux=False, quality=oidn.Quality.High, input_scale=None,
58
+ filter="RT", device=None, output=None)
59
+ ```
60
+
61
+ Images are arrays of shape `(H, W)` or `(H, W, C)` with `C <= 4` and dtype
62
+ `float32` or `float16`. A fourth channel is ignored on input. The result has
63
+ the same framework and lives on the same device as `color`: a NumPy array
64
+ yields a NumPy array, a CUDA tensor yields a CUDA tensor. Its shape is
65
+ `(H, W, min(C, 3))`, or `(H, W)` for two-dimensional input.
66
+
67
+ - `hdr`, `srgb`, `clean_aux`, `quality`, and `input_scale` map to the
68
+ parameters of the OIDN `RT` filter. Set `hdr=True` for linear radiance
69
+ values without an upper bound, and `clean_aux=True` when the albedo and
70
+ normal images are noise-free. See the [OIDN
71
+ documentation](https://www.openimagedenoise.org/documentation.html) for
72
+ details.
73
+ - `device` selects the device. By default, CUDA arrays use a CUDA device with
74
+ the matching ordinal, and host arrays use the fastest physical device in the
75
+ system, which can be overridden with the `OIDN_DEFAULT_DEVICE` environment
76
+ variable (`cpu`, `cuda`, `metal`, or a physical device ID).
77
+ - `output` supplies a preallocated array that is filled in place and returned.
78
+ With an RGBA output array, OIDN writes the RGB channels and leaves alpha
79
+ untouched.
80
+ - Filters are expensive to create, so `denoise()` caches a few of them, keyed
81
+ on image size, format, feature set, and parameters. Repeated calls at the
82
+ same resolution pay only for the actual filtering.
83
+
84
+ When the device cannot access an input array directly, for example a NumPy
85
+ array passed to a CUDA device, `denoise()` copies it into a device buffer.
86
+ Otherwise no copies are made.
87
+
88
+ ```python
89
+ import numpy as np
90
+ import torch
91
+ import mitsuba_oidn as oidn
92
+
93
+ # NumPy, CPU or Metal depending on the fastest available device
94
+ out = oidn.denoise(np.asarray(color, dtype=np.float32), hdr=True)
95
+
96
+ # PyTorch on the GPU: zero-copy in and out
97
+ color = torch.rand(1080, 1920, 3, device="cuda")
98
+ out = oidn.denoise(color, quality=oidn.Quality.Balanced)
99
+ assert out.device == color.device
100
+ ```
101
+
102
+ ## The object API
103
+
104
+ The `denoise()` function covers the common case. The classes below mirror the
105
+ OIDN object model for applications that need control over devices, memory,
106
+ and filter lifetime, for instance when denoising many frames or several AOVs
107
+ that share auxiliary images.
108
+
109
+ ### Devices
110
+
111
+ ```python
112
+ oidn.physical_devices() # list of PhysicalDevice: id, name, type, uuid, ...
113
+
114
+ dev = oidn.Device() # fastest physical device
115
+ dev = oidn.Device(oidn.DeviceType.CPU)
116
+ dev = oidn.Device.from_physical(id) # also from_uuid(), from_luid(), from_pci_address()
117
+ dev = oidn.Device.cuda(device_id=0, stream=torch.cuda.current_stream().cuda_stream)
118
+ dev = oidn.Device.metal(command_queue) # raw id<MTLCommandQueue> pointer
119
+
120
+ dev.num_threads = 4 # CPU only: 0 shares the Dr.Jit pool (default),
121
+ # a positive value creates a private pool
122
+ dev.verbose = 1
123
+ dev.type, dev.version, dev.system_memory_supported, dev.managed_memory_supported
124
+ dev.sync() # wait for asynchronous work
125
+ ```
126
+
127
+ A device commits itself when the first buffer or filter is created. Parameters
128
+ such as `num_threads` must be set before that point.
129
+
130
+ ### Filters
131
+
132
+ ```python
133
+ flt = dev.new_filter("RT") # or "RTLightmap"
134
+
135
+ flt.set_image("color", color) # arrays: layout inferred, zero-copy when possible
136
+ flt.set_image("albedo", albedo)
137
+ flt.set_image("normal", normal)
138
+ flt.set_image("output", output) # must be writable
139
+ flt.hdr = True
140
+ flt.clean_aux = True
141
+ flt.quality = oidn.Quality.High
142
+ flt.input_scale = 0.5 # None selects automatic scaling
143
+ flt.max_memory_mb = 2048
144
+ flt.set("cleanAux", True) # generic access by OIDN parameter name
145
+ flt.set_data("weights", blob) # user-trained weights (bytes or uint8 array)
146
+ flt.set_progress_monitor(lambda p: True) # return False to cancel
147
+
148
+ flt.execute() # commits pending changes, runs, and waits
149
+ flt.execute_async(); dev.sync()
150
+ ```
151
+
152
+ `set_image()` accepts arrays with the layout rules of `denoise()`. Row and
153
+ pixel strides are passed to OIDN, so slices of larger arrays work as long as
154
+ the channel dimension stays contiguous. Whether an array can be bound without
155
+ a copy depends on the device:
156
+
157
+ | Device | Bound without copy |
158
+ |---|---|
159
+ | CPU | any host array, including CUDA pinned and managed memory |
160
+ | CUDA | arrays on the same CUDA device, pinned host memory, and host memory if the GPU supports pageable memory access |
161
+ | Metal | any host array, and views of buffers created on the device |
162
+
163
+ Use `dev.can_share(array)` to test this in advance. When binding is not
164
+ possible, `set_image()` raises a `TypeError` that points at the buffer API.
165
+
166
+ Filters can also take a `Buffer` with an explicit description:
167
+
168
+ ```python
169
+ flt.set_image("color", buf, format=oidn.Format.Float3, width=w, height=h,
170
+ byte_offset=0, pixel_stride=0, row_stride=0)
171
+ ```
172
+
173
+ ### Buffers
174
+
175
+ Buffers are memory allocations made by a device. They are the way to work
176
+ with memory that the host cannot address, such as dedicated GPU memory, and
177
+ they provide zero-copy views on unified-memory systems.
178
+
179
+ ```python
180
+ buf = dev.new_buffer(nbytes) # host and device accessible
181
+ buf = dev.new_buffer(nbytes, oidn.Storage.Device) # device memory only
182
+ buf = dev.new_shared_buffer(array) # wrap device-accessible memory
183
+
184
+ buf.size, buf.storage, buf.device, buf.data_ptr
185
+
186
+ view = buf.view("float32", (h, w, 3)) # DLPack and buffer-protocol object
187
+ img = np.from_dlpack(view) # or torch.from_dlpack(view), ...
188
+
189
+ buf.write(host_array); buf.read(host_array) # copies through the host
190
+ buf.write_async(src); buf.read_async(dst); dev.sync()
191
+ ```
192
+
193
+ Arrays created from `buf.view()` are recognized by `set_image()` and bound
194
+ through the underlying buffer. On Apple silicon, rendering into such a view
195
+ and denoising it involves no copies at all. On a CUDA device, a device-storage
196
+ buffer viewed through `torch.from_dlpack()` gives a tensor that OIDN wrote
197
+ directly.
198
+
199
+ ### Errors
200
+
201
+ All OIDN errors raise `oidn.Error`, whose `code` attribute is an
202
+ `oidn.ErrorCode`. Cancellation through a progress monitor raises
203
+ `oidn.Error` with `ErrorCode.Cancelled`. An exception raised inside the
204
+ progress monitor cancels the filter and propagates unchanged.
205
+
206
+ ## Building from source
207
+
208
+ The build needs CMake 3.21 or newer, a C++17 compiler, and a binary release
209
+ of [ISPC](https://ispc.github.io/downloads.html) unpacked into `ext/ispc`,
210
+ so that `ext/ispc/bin/ispc` exists. Metal support requires Xcode 15 or newer.
211
+ CUDA support requires CUDA 12.8 or newer and is enabled automatically when the
212
+ toolkit is found.
213
+
214
+ ```
215
+ git clone --recursive https://github.com/mitsuba-renderer/mitsuba-oidn
216
+ cd mitsuba-oidn
217
+ pip install nanobind==3.0.1 scikit-build-core
218
+ pip install --no-build-isolation -ve .
219
+ pytest
220
+ ```
221
+
222
+ The OIDN weights are stored with git-lfs, which must be installed before
223
+ cloning.
224
+
225
+ ## License
226
+
227
+ mitsuba-oidn is licensed under the BSD 3-Clause license. It bundles Intel Open
228
+ Image Denoise, which is licensed under the Apache License 2.0.