oidn-web 0.3.5 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/CHANGELOG.md +49 -0
  2. package/README.md +140 -4
  3. package/benchmarks/compare.mjs +651 -0
  4. package/benchmarks/leak.mjs +255 -0
  5. package/benchmarks/results/before-spatial.json +391 -0
  6. package/benchmarks/results/before-spatial.md +47 -0
  7. package/benchmarks/results/int8-scan.json +2007 -0
  8. package/benchmarks/results/int8-scan.md +160 -0
  9. package/benchmarks/results/int8-w8a8-scan.json +2007 -0
  10. package/benchmarks/results/int8-w8a8-scan.md +160 -0
  11. package/benchmarks/results/int8-weight-channel.json +1413 -0
  12. package/benchmarks/results/int8-weight-channel.md +118 -0
  13. package/benchmarks/results/int8-weight-only.json +1437 -0
  14. package/benchmarks/results/int8-weight-only.md +118 -0
  15. package/benchmarks/results/kernel-webnn-final.json +1115 -0
  16. package/benchmarks/results/kernel-webnn-final.md +104 -0
  17. package/benchmarks/results/latest-optimized.json +375 -0
  18. package/benchmarks/results/latest-optimized.md +47 -0
  19. package/benchmarks/results/latest.json +391 -0
  20. package/benchmarks/results/latest.md +47 -0
  21. package/benchmarks/results/profile-baseline.json +331 -0
  22. package/benchmarks/results/profile-baseline.md +12 -0
  23. package/benchmarks/results/profile-conv2x.json +331 -0
  24. package/benchmarks/results/profile-conv2x.md +12 -0
  25. package/benchmarks/results/profile-fast-init.json +385 -0
  26. package/benchmarks/results/profile-fast-init.md +47 -0
  27. package/benchmarks/results/profile-fp16-fma.json +369 -0
  28. package/benchmarks/results/profile-fp16-fma.md +47 -0
  29. package/benchmarks/results/profile-fp16-tiled-decoder.json +369 -0
  30. package/benchmarks/results/profile-fp16-tiled-decoder.md +47 -0
  31. package/benchmarks/results/profile-fp16-tiled-encoder.json +369 -0
  32. package/benchmarks/results/profile-fp16-tiled-encoder.md +47 -0
  33. package/benchmarks/results/profile-fp16-unfused-pool.json +385 -0
  34. package/benchmarks/results/profile-fp16-unfused-pool.md +47 -0
  35. package/benchmarks/results/profile-input-major.json +347 -0
  36. package/benchmarks/results/profile-input-major.md +12 -0
  37. package/benchmarks/results/profile-k16.json +347 -0
  38. package/benchmarks/results/profile-k16.md +12 -0
  39. package/benchmarks/results/profile-k4.json +347 -0
  40. package/benchmarks/results/profile-k4.md +12 -0
  41. package/benchmarks/results/profile-pool-reuse.json +331 -0
  42. package/benchmarks/results/profile-pool-reuse.md +12 -0
  43. package/benchmarks/results/profile-precompiled.json +385 -0
  44. package/benchmarks/results/profile-precompiled.md +47 -0
  45. package/benchmarks/results/profile-static-channels.json +385 -0
  46. package/benchmarks/results/profile-static-channels.md +47 -0
  47. package/benchmarks/results/profile-static-io.json +385 -0
  48. package/benchmarks/results/profile-static-io.md +47 -0
  49. package/benchmarks/results/profile-tiled-conv.json +331 -0
  50. package/benchmarks/results/profile-tiled-conv.md +12 -0
  51. package/benchmarks/results/profile-tiled-decoder.json +347 -0
  52. package/benchmarks/results/profile-tiled-decoder.md +12 -0
  53. package/benchmarks/results/profile-tiled-matmul.json +331 -0
  54. package/benchmarks/results/profile-tiled-matmul.md +12 -0
  55. package/benchmarks/results/profile-unfused-decoder.json +379 -0
  56. package/benchmarks/results/profile-unfused-decoder.md +12 -0
  57. package/benchmarks/results/profile-unfused-pool.json +347 -0
  58. package/benchmarks/results/profile-unfused-pool.md +12 -0
  59. package/benchmarks/results/spatial-auto.json +575 -0
  60. package/benchmarks/results/spatial-auto.md +61 -0
  61. package/benchmarks/results/subgroup-smoke.json +1094 -0
  62. package/benchmarks/results/subgroup-smoke.md +104 -0
  63. package/benchmarks/results/webnn-smoke.json +739 -0
  64. package/benchmarks/results/webnn-smoke.md +76 -0
  65. package/dist/oidn.js +4166 -22580
  66. package/dist/oidn.umd.cjs +776 -5799
  67. package/lib/UNet.d.ts +66 -15
  68. package/lib/UNet.js +162 -257
  69. package/lib/UNet.js.map +1 -1
  70. package/lib/WGPUComputePass.d.ts +1 -1
  71. package/lib/WGPUComputePass.js +6 -4
  72. package/lib/WGPUComputePass.js.map +1 -1
  73. package/lib/backend.d.ts +8 -4
  74. package/lib/backend.js +36 -44
  75. package/lib/backend.js.map +1 -1
  76. package/lib/graphOptimizer.d.ts +54 -0
  77. package/lib/graphOptimizer.js +216 -0
  78. package/lib/graphOptimizer.js.map +1 -0
  79. package/lib/main.d.ts +33 -10
  80. package/lib/main.js +5 -0
  81. package/lib/main.js.map +1 -1
  82. package/lib/modelSpec.d.ts +80 -0
  83. package/lib/modelSpec.js +270 -0
  84. package/lib/modelSpec.js.map +1 -0
  85. package/lib/nativeUNet.d.ts +67 -0
  86. package/lib/nativeUNet.js +1735 -0
  87. package/lib/nativeUNet.js.map +1 -0
  88. package/lib/process.js +3 -0
  89. package/lib/process.js.map +1 -1
  90. package/lib/resourceTracker.d.ts +26 -0
  91. package/lib/resourceTracker.js +65 -0
  92. package/lib/resourceTracker.js.map +1 -0
  93. package/lib/tileScheduler.d.ts +33 -0
  94. package/lib/tileScheduler.js +86 -0
  95. package/lib/tileScheduler.js.map +1 -0
  96. package/lib/webnnUNet.d.ts +52 -0
  97. package/lib/webnnUNet.js +535 -0
  98. package/lib/webnnUNet.js.map +1 -0
  99. package/package.json +9 -5
  100. package/scripts/inspect-model.mjs +64 -0
  101. package/src/UNet.ts +236 -339
  102. package/src/WGPUComputePass.ts +6 -4
  103. package/src/backend.ts +42 -55
  104. package/src/graphOptimizer.ts +301 -0
  105. package/src/main.ts +71 -11
  106. package/src/modelSpec.ts +414 -0
  107. package/src/nativeUNet.ts +2256 -0
  108. package/src/process.ts +3 -1
  109. package/src/resourceTracker.ts +94 -0
  110. package/src/tileScheduler.ts +138 -0
  111. package/src/webnnUNet.ts +812 -0
  112. package/tests/modelSpec.test.mjs +128 -0
  113. package/tests/resourceLifecycle.test.mjs +383 -0
  114. package/tests/tileScheduler.test.mjs +90 -0
  115. package/lib/helper.d.ts +0 -4
  116. package/lib/helper.js +0 -33
  117. package/lib/helper.js.map +0 -1
  118. package/lib/kernels.d.ts +0 -1
  119. package/lib/kernels.js +0 -26
  120. package/lib/kernels.js.map +0 -1
  121. package/src/helper.ts +0 -43
  122. package/src/kernels.ts +0 -31
package/CHANGELOG.md ADDED
@@ -0,0 +1,49 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ ## [0.4.0] - 2026-08-20
6
+
7
+ oidn-web 0.4.0 replaces the TensorFlow.js inference stack with a purpose-built, model-driven WebGPU runtime. Existing `initUNetFromURL` and `initUNetFromBuffer` integrations remain supported while gaining native FP16, adaptive scheduling, runtime diagnostics, and stronger model validation.
8
+
9
+ ### Highlights
10
+
11
+ - Removed TensorFlow.js and its WebGPU backend from the runtime dependencies.
12
+ - Added a custom WGSL U-Net executor with automatic FP16 selection and a deterministic FP32 fallback.
13
+ - Added built-in topology descriptors for the current OIDN small and large RT models, including clean auxiliary variants.
14
+ - Reduced the reference Vite ESM bundle from approximately 168.8 KB to 31.1 KB gzip compared with 0.3.5, a reduction of about 82%.
15
+ - Reached a 27.6 ms median for a 512 x 512 Clean Aux Large inference on the tested Apple GPU, versus 35.0 ms for the TensorFlow.js baseline (1.27x faster). Performance depends on the browser, GPU, model, and tile size.
16
+
17
+ ### Added
18
+
19
+ - Model-driven graph planning and lifetime-aware activation-buffer reuse.
20
+ - Versioned `UNetModelSpec` descriptors, topology detection, and validation of tensor names, layouts, data types, byte lengths, kernel shapes, bias shapes, and graph channel flow.
21
+ - `modelSpec` initialization option for future or custom OIDN model topologies.
22
+ - `npm run model:inspect` for model hashes, tensor signatures, and descriptor compatibility checks.
23
+ - Dynamic tile sizing with configurable minimum, initial, and maximum tile sizes and a target GPU time.
24
+ - GPU queue backpressure between tiles to keep cancellation and rendering interaction responsive.
25
+ - Runtime diagnostics through `getRuntimeInfo()`, including the selected engine, precision, model, kernel capabilities, tile state, and resource statistics.
26
+ - Per-layer GPU profiling through `profileNextExecution()` and `getLastExecutionProfile()` when `timestamp-query` is enabled.
27
+ - Cross-version browser benchmarks with queue-complete timings, sampled output validation, TFJS baseline comparison, and per-layer GPU profiles.
28
+ - Explicit, idempotent resource disposal and resource lifecycle accounting.
29
+ - Automated model, graph, scheduler, resource leak, allocation rollback, and late-async-cleanup tests.
30
+
31
+ ### Changed
32
+
33
+ - U-Net inference now runs directly in WGSL. The stable `auto` engine selects the native WGSL backend and no longer initializes TensorFlow.js.
34
+ - FP16-capable devices retain TZA half-float weights and use short FP16 FMA accumulation groups folded into FP32 accumulators. The final output remains FP32.
35
+ - Devices without `shader-f16` automatically use native FP32 inference.
36
+ - Convolution tensors and activations use a blocked four-channel layout with channel-specialized shaders.
37
+ - Decoder `upsample + concat + conv` patterns are fused, and all network passes for a tile are encoded into one command buffer.
38
+ - Shape-independent pipelines are compiled asynchronously before initialization resolves to avoid first-denoise shader compilation stalls.
39
+ - `maxTileSize` is now a hard upper bound for the adaptive tile controller.
40
+ - A shared `GPUDevice` uses FP16 only when `shader-f16` was requested during device creation; WebGPU features cannot be enabled afterward.
41
+
42
+ ### Migration notes
43
+
44
+ - No changes are required for basic `initUNetFromURL` or `initUNetFromBuffer` usage.
45
+ - Call `dispose()` when a U-Net instance is no longer needed.
46
+ - When supplying an existing `GPUDevice`, request `shader-f16` before creating the device if FP16 inference is desired.
47
+ - Set `dynamicTile: false` to restore fixed-size tiling.
48
+
49
+ [0.4.0]: https://github.com/pissang/oidn-web/compare/v0.3.5...v0.4.0
package/README.md CHANGED
@@ -9,9 +9,22 @@ It's used in the [Vector to 3D](https://www.figma.com/community/plugin/126460021
9
9
  | :----------------------------------------------------------------------------------------------: | :--------------------------------------------------------------------------------: | :--------------------------------------------------------------------------------------: |
10
10
  | ![](https://github.com/pissang/oidn-web/blob/main/examples/test/ground-truth.png 'Ground Truth') | ![](https://github.com/pissang/oidn-web/blob/main/examples/test/noisy.png 'Noisy') | ![](https://github.com/pissang/oidn-web/blob/main/examples/test/denoised.png 'Denoised') |
11
11
 
12
- ## How it Works.
12
+ ## How it works
13
13
 
14
- It uses [tfjs](https://github.com/tensorflow/tfjs) to build the UNet model used by the OIDN. Then use the model to do prediction with a WebGPU backend from the image data.
14
+ The OIDN U-Net runs directly on WebGPU with model-driven WGSL compute
15
+ pipelines. Convolution activations use a blocked
16
+ four-channel layout, decoder `upsample + concat + conv` patterns are fused,
17
+ and all network dispatches for a tile are submitted in one command buffer.
18
+ FP32 convolutions use channel-specialized implicit-GEMM tiles; FP16 uses
19
+ channel-specialized vector FMA and separate max-pool passes, selected from the
20
+ same model descriptor.
21
+
22
+ TZA half-float weights stay half-float when the device enables `shader-f16`.
23
+ FP16 products are accumulated in short half-precision groups and periodically
24
+ folded into FP32 accumulators; the final output is FP32. Devices without
25
+ `shader-f16` automatically use the native FP32 path. Shape-independent compute
26
+ pipelines compile asynchronously before initialization resolves, so first-use
27
+ shader compilation does not interrupt an interactive denoise.
15
28
 
16
29
  ## How to Use
17
30
 
@@ -42,7 +55,7 @@ initUNetFromURL('./weights/rt_ldr.tza').then((unet) => {
42
55
  const abortDenoising = unet.tileExecute({
43
56
  // The color input for LDR image is 4 channels.
44
57
  // In the format of Uint8ClampedArray or Uint8Array.
45
- color: { data: noisyImageData, width, height },
58
+ color: noisyImageData,
46
59
  done(denoised) {
47
60
  console.log('Finished');
48
61
  },
@@ -111,7 +124,7 @@ If you already have a WebGPU path tracer. You can integrate the oidn-web into yo
111
124
  initUNetFromURL(
112
125
  './weights/rt_hdr_alb_nrm.tza',
113
126
  {
114
- // Share GPUDevice and GPUAdapterInfo to the TFJS WebGPU backend
127
+ // Share GPUDevice and GPUAdapterInfo with the native WGSL runtime.
115
128
  device,
116
129
  adapterInfo
117
130
  },
@@ -153,6 +166,129 @@ initUNetFromURL('./weights/rt_hdr_alb_nrm_small.tza', ...);
153
166
 
154
167
  Other combinations can be found in the [oidn-weights](https://github.com/RenderKit/oidn-weights)
155
168
 
169
+ ### FP16 and runtime information
170
+
171
+ Standalone initialization requests `shader-f16` when the adapter supports it.
172
+ When sharing a device, optional features must be requested when that device is
173
+ created; WebGPU features cannot be enabled afterward.
174
+
175
+ ```ts
176
+ const requiredFeatures = adapter.features.has('shader-f16')
177
+ ? ['shader-f16']
178
+ : [];
179
+ const device = await adapter.requestDevice({ requiredFeatures });
180
+
181
+ const unet = await initUNetFromURL(
182
+ modelUrl,
183
+ { device, adapterInfo },
184
+ {
185
+ aux: true,
186
+ hdr: true,
187
+ precision: 'auto' // 'fp16' enforces support; 'fp32' is deterministic fallback
188
+ }
189
+ );
190
+
191
+ console.log(unet.getRuntimeInfo());
192
+ // { gpuEngine: 'wgsl', precision: 'fp16', model: 'oidn-unet-large-v1', ... }
193
+ ```
194
+
195
+ For one-shot native GPU timings, request a profile immediately before an
196
+ execution. This is available when the shared device enabled `timestamp-query`:
197
+
198
+ ```ts
199
+ if (unet.profileNextExecution()) {
200
+ unet.tileExecute({
201
+ color,
202
+ albedo,
203
+ normal,
204
+ done: async () => {
205
+ console.table((await unet.getLastExecutionProfile()).layers);
206
+ }
207
+ });
208
+ }
209
+ ```
210
+
211
+ ### Updating to a new OIDN model
212
+
213
+ TZA stores tensors but not the executable graph. The runtime therefore keeps
214
+ the graph in a versioned `UNetModelSpec`, separate from shader and precision
215
+ code. Built-in descriptors cover the current OIDN small and large RT U-Nets.
216
+ At load time the descriptor is detected from the complete tensor-name set, and
217
+ tensor layout, dtype, byte length, kernel shape, bias shape, and graph channel
218
+ flow are validated before GPU resources are created.
219
+
220
+ If an OIDN update keeps one of these topologies and tensor names, changed
221
+ channel widths are handled automatically. If it adds or renames nodes, add a
222
+ new descriptor (or pass `modelSpec`) and its validation fixture. Existing graph
223
+ fusion rules apply to the new descriptor without changes to WGSL kernels.
224
+
225
+ Use the inspection command to get a stable SHA-256, full tensor signature, and
226
+ descriptor compatibility result for an upstream weight file:
227
+
228
+ ```shell
229
+ npm run model:inspect -- weights/rt_hdr_alb_nrm.tza
230
+ ```
231
+
232
+ ```ts
233
+ const unet = await initUNetFromURL(newModelUrl, backend, {
234
+ aux: true,
235
+ hdr: true,
236
+ modelSpec: newOidnModelSpec
237
+ });
238
+ ```
239
+
240
+ ### GPU backpressure and dynamic tiles
241
+
242
+ `tileExecute` waits for the submitted GPU work of a tile before scheduling the
243
+ next tile. This keeps at most one OIDN tile in flight, which makes cancellation
244
+ responsive instead of leaving queued denoising work ahead of interactive
245
+ rendering.
246
+
247
+ Tile sizing is adaptive by default. `maxTileSize` is a hard upper bound; the
248
+ completed GPU time of a tiled execution adjusts the tile size used by the next
249
+ execution. Single-tile images do not affect the estimate. The default range
250
+ starts at 384 pixels, does not go below 256, and targets about 16 ms of GPU work
251
+ per tile.
252
+
253
+ ```ts
254
+ initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', backend, {
255
+ aux: true,
256
+ hdr: true,
257
+ maxTileSize: 512,
258
+ dynamicTile: {
259
+ minTileSize: 256,
260
+ initialTileSize: 384,
261
+ targetTileTimeMs: 16
262
+ }
263
+ });
264
+
265
+ // Restore fixed-size behavior when deterministic tiling is preferred.
266
+ initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', backend, {
267
+ aux: true,
268
+ hdr: true,
269
+ maxTileSize: 512,
270
+ dynamicTile: false
271
+ });
272
+ ```
273
+
274
+ ### Benchmark the native runtime against TFJS
275
+
276
+ The browser benchmark automatically finds the nearest ancestor whose package
277
+ still depends on TensorFlow.js, builds that commit in a temporary worktree, and
278
+ compares it with the current WGSL FP32 and FP16 runtimes. Each measured run
279
+ waits for the WebGPU queue to finish, so the result includes execution rather
280
+ than only JavaScript command submission. It also samples output against the
281
+ TFJS FP32 result and, when timestamp queries are supported, reports the five
282
+ most expensive native network nodes.
283
+
284
+ ```shell
285
+ npm run benchmark -- --width 512 --height 512 --tile-size 512 --runs 5
286
+ ```
287
+
288
+ Results are printed as a table and written to
289
+ `benchmarks/results/latest.{json,md}`. Use `--baseline <commit>` to pin an
290
+ explicit historical version or `--chrome <path>` to select a browser.
291
+
156
292
  ## Credits
157
293
 
158
294
  Huge thanks to Max Liani for his series: https://maxliani.wordpress.com/2023/03/17/dnnd-1-a-deep-neural-network-dive/. My work is mostly inspired by it.