oidn-web 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/CHANGELOG.md +45 -0
  2. package/README.md +83 -19
  3. package/dist/oidn.js +3239 -2642
  4. package/dist/oidn.umd.cjs +512 -296
  5. package/lib/UNet.d.ts +54 -20
  6. package/lib/UNet.js +194 -118
  7. package/lib/UNet.js.map +1 -1
  8. package/lib/backend.d.ts +1 -8
  9. package/lib/backend.js +1 -9
  10. package/lib/backend.js.map +1 -1
  11. package/lib/finalRgbShader.d.ts +13 -0
  12. package/lib/finalRgbShader.js +160 -0
  13. package/lib/finalRgbShader.js.map +1 -0
  14. package/lib/graphOptimizer.js +1 -2
  15. package/lib/graphOptimizer.js.map +1 -1
  16. package/lib/hdrTransfer.d.ts +14 -0
  17. package/lib/hdrTransfer.js +61 -0
  18. package/lib/hdrTransfer.js.map +1 -0
  19. package/lib/main.d.ts +16 -7
  20. package/lib/main.js +4 -5
  21. package/lib/main.js.map +1 -1
  22. package/lib/nativeUNet.d.ts +39 -3
  23. package/lib/nativeUNet.js +449 -120
  24. package/lib/nativeUNet.js.map +1 -1
  25. package/lib/process.d.ts +5 -11
  26. package/lib/process.js +35 -49
  27. package/lib/process.js.map +1 -1
  28. package/lib/tileScheduler.d.ts +32 -4
  29. package/lib/tileScheduler.js +133 -20
  30. package/lib/tileScheduler.js.map +1 -1
  31. package/package.json +9 -2
  32. package/src/UNet.ts +287 -158
  33. package/src/backend.ts +1 -14
  34. package/src/finalRgbShader.ts +186 -0
  35. package/src/graphOptimizer.ts +1 -2
  36. package/src/hdrTransfer.ts +88 -0
  37. package/src/main.ts +28 -13
  38. package/src/nativeUNet.ts +515 -116
  39. package/src/process.ts +43 -70
  40. package/src/tileScheduler.ts +216 -24
  41. package/benchmarks/compare.mjs +0 -651
  42. package/benchmarks/leak.mjs +0 -255
  43. package/benchmarks/results/before-spatial.json +0 -391
  44. package/benchmarks/results/before-spatial.md +0 -47
  45. package/benchmarks/results/int8-scan.json +0 -2007
  46. package/benchmarks/results/int8-scan.md +0 -160
  47. package/benchmarks/results/int8-w8a8-scan.json +0 -2007
  48. package/benchmarks/results/int8-w8a8-scan.md +0 -160
  49. package/benchmarks/results/int8-weight-channel.json +0 -1413
  50. package/benchmarks/results/int8-weight-channel.md +0 -118
  51. package/benchmarks/results/int8-weight-only.json +0 -1437
  52. package/benchmarks/results/int8-weight-only.md +0 -118
  53. package/benchmarks/results/kernel-webnn-final.json +0 -1115
  54. package/benchmarks/results/kernel-webnn-final.md +0 -104
  55. package/benchmarks/results/latest-optimized.json +0 -375
  56. package/benchmarks/results/latest-optimized.md +0 -47
  57. package/benchmarks/results/latest.json +0 -391
  58. package/benchmarks/results/latest.md +0 -47
  59. package/benchmarks/results/profile-baseline.json +0 -331
  60. package/benchmarks/results/profile-baseline.md +0 -12
  61. package/benchmarks/results/profile-conv2x.json +0 -331
  62. package/benchmarks/results/profile-conv2x.md +0 -12
  63. package/benchmarks/results/profile-fast-init.json +0 -385
  64. package/benchmarks/results/profile-fast-init.md +0 -47
  65. package/benchmarks/results/profile-fp16-fma.json +0 -369
  66. package/benchmarks/results/profile-fp16-fma.md +0 -47
  67. package/benchmarks/results/profile-fp16-tiled-decoder.json +0 -369
  68. package/benchmarks/results/profile-fp16-tiled-decoder.md +0 -47
  69. package/benchmarks/results/profile-fp16-tiled-encoder.json +0 -369
  70. package/benchmarks/results/profile-fp16-tiled-encoder.md +0 -47
  71. package/benchmarks/results/profile-fp16-unfused-pool.json +0 -385
  72. package/benchmarks/results/profile-fp16-unfused-pool.md +0 -47
  73. package/benchmarks/results/profile-input-major.json +0 -347
  74. package/benchmarks/results/profile-input-major.md +0 -12
  75. package/benchmarks/results/profile-k16.json +0 -347
  76. package/benchmarks/results/profile-k16.md +0 -12
  77. package/benchmarks/results/profile-k4.json +0 -347
  78. package/benchmarks/results/profile-k4.md +0 -12
  79. package/benchmarks/results/profile-pool-reuse.json +0 -331
  80. package/benchmarks/results/profile-pool-reuse.md +0 -12
  81. package/benchmarks/results/profile-precompiled.json +0 -385
  82. package/benchmarks/results/profile-precompiled.md +0 -47
  83. package/benchmarks/results/profile-static-channels.json +0 -385
  84. package/benchmarks/results/profile-static-channels.md +0 -47
  85. package/benchmarks/results/profile-static-io.json +0 -385
  86. package/benchmarks/results/profile-static-io.md +0 -47
  87. package/benchmarks/results/profile-tiled-conv.json +0 -331
  88. package/benchmarks/results/profile-tiled-conv.md +0 -12
  89. package/benchmarks/results/profile-tiled-decoder.json +0 -347
  90. package/benchmarks/results/profile-tiled-decoder.md +0 -12
  91. package/benchmarks/results/profile-tiled-matmul.json +0 -331
  92. package/benchmarks/results/profile-tiled-matmul.md +0 -12
  93. package/benchmarks/results/profile-unfused-decoder.json +0 -379
  94. package/benchmarks/results/profile-unfused-decoder.md +0 -12
  95. package/benchmarks/results/profile-unfused-pool.json +0 -347
  96. package/benchmarks/results/profile-unfused-pool.md +0 -12
  97. package/benchmarks/results/spatial-auto.json +0 -575
  98. package/benchmarks/results/spatial-auto.md +0 -61
  99. package/benchmarks/results/subgroup-smoke.json +0 -1094
  100. package/benchmarks/results/subgroup-smoke.md +0 -104
  101. package/benchmarks/results/webnn-smoke.json +0 -739
  102. package/benchmarks/results/webnn-smoke.md +0 -76
  103. package/scripts/inspect-model.mjs +0 -64
  104. package/tests/modelSpec.test.mjs +0 -128
  105. package/tests/resourceLifecycle.test.mjs +0 -383
  106. package/tests/tileScheduler.test.mjs +0 -90
package/CHANGELOG.md CHANGED
@@ -2,6 +2,50 @@
2
2
 
3
3
  All notable changes to this project are documented in this file.
4
4
 
5
+ ## [0.5.0] - 2026-10-09
6
+
7
+ oidn-web 0.5.0 simplifies device setup, makes tiled execution always settle, and switches FP16 convolutions to implicit GEMM. It contains breaking changes for TypeScript callers that pass `adapterInfo`, for code that constructs `UNet` directly, and for code that relies on square tiles or per-frame tile pacing.
8
+
9
+ ### Breaking changes
10
+
11
+ - `initUNetFromURL` and `initUNetFromBuffer` take `{ device }` as the second argument. `adapterInfo` was only needed by the removed TensorFlow.js backend and is no longer accepted by the type. Extra properties are ignored at runtime, but TypeScript rejects an object literal that still contains `adapterInfo`.
12
+ - `new UNet(tensors, device, options)` takes a `GPUDevice` instead of a `{ device, adapterInfo }` object.
13
+ - `tileExecute` now continues on the event loop between tiles by default instead of waiting for `requestAnimationFrame`. Pass `scheduling: 'animation-frame'` to keep pacing tiles to display frames.
14
+ - Tiles are balanced rectangles with overlap only on edges shared with another tile, instead of fixed-size squares. Tile count, position, and size reported to `progress` differ from 0.4.0. The default `dynamicTile.initialTileSize` is now 432.
15
+ - `UNetExecutionStats` no longer has `tileWidth` and `tileHeight`. Use `tileColumns`, `tileRows`, `tileOverlap`, `inputPixelCount`, and `inputShapeCount` instead.
16
+ - FP16 convolutions use implicit GEMM by default. Output differs slightly from 0.4.0 because accumulation order changed. Set `kernel: 'direct'` to reproduce the 0.4.0 FP16 path.
17
+
18
+ ### Added
19
+
20
+ - `hdrTransfer: 'log'` for RTLightmap HDR models, alongside the default PU transfer.
21
+ - `error` callback on `tileExecute` for asynchronous failures, WebGPU device loss, and exceptions thrown by `progress` or `done`. `progress` and `done` may return promises.
22
+ - `scheduling: 'event-loop' | 'animation-frame'`, `tileOverlap`, and `wholeImage` options on `tileExecute`.
23
+ - `prepareForImage(width, height)` to create per-shape GPU resources before the first denoise.
24
+ - `planTileGrid` and its `TilePlan`, `PlannedTile`, and `TileRect` types.
25
+ - Experimental `gemm` tuning options. These are intended for benchmarking and are not covered by semver.
26
+ - `hdrTransfer` and `activeExecutionCount` in `getRuntimeInfo()`, and the active GEMM configuration under `kernel.gemm`.
27
+
28
+ ### Changed
29
+
30
+ - Implicit GEMM tiles are selected by output alignment and GPU limits, with optimized addressing, weight layout, register tiles, shared-memory layout, and pooling access.
31
+ - The final RGB convolution uses an adaptive shared-memory cache.
32
+ - Adaptive tile sizing uses the smoothed P75 tile GPU time, excludes the cold first tile, ignores cancelled and single-tile work, and buckets input shapes to at most two sizes.
33
+ - `animation-frame` scheduling falls back to a 100 ms timer so execution still completes in hidden tabs.
34
+
35
+ ### Fixed
36
+
37
+ - Tiled execution always settles with `done` or `error`, or stops silently after abort, including when `requestAnimationFrame` never fires or the device is lost.
38
+ - Completed executions release their device-loss listeners.
39
+ - Edge tiles whose size is not a multiple of 16 replicate edge pixels into the padded model input.
40
+ - The npm package no longer includes benchmark results, tests, and scripts.
41
+
42
+ ### Migration notes
43
+
44
+ - Replace `{ device, adapterInfo }` with `{ device }`.
45
+ - Replace `new UNet(tensors, { device, adapterInfo }, options)` with `new UNet(tensors, device, options)`.
46
+ - Interactive renderers that share the GPU with OIDN should pass `scheduling: 'animation-frame'`.
47
+ - Pass an `error` callback to handle failures. Without one, failures are logged to the console.
48
+
5
49
  ## [0.4.0] - 2026-08-20
6
50
 
7
51
  oidn-web 0.4.0 replaces the TensorFlow.js inference stack with a purpose-built, model-driven WebGPU runtime. Existing `initUNetFromURL` and `initUNetFromBuffer` integrations remain supported while gaining native FP16, adaptive scheduling, runtime diagnostics, and stronger model validation.
@@ -46,4 +90,5 @@ oidn-web 0.4.0 replaces the TensorFlow.js inference stack with a purpose-built,
46
90
  - When supplying an existing `GPUDevice`, request `shader-f16` before creating the device if FP16 inference is desired.
47
91
  - Set `dynamicTile: false` to restore fixed-size tiling.
48
92
 
93
+ [0.5.0]: https://github.com/pissang/oidn-web/compare/v0.4.0...v0.5.0
49
94
  [0.4.0]: https://github.com/pissang/oidn-web/compare/v0.3.5...v0.4.0
package/README.md CHANGED
@@ -15,9 +15,9 @@ The OIDN U-Net runs directly on WebGPU with model-driven WGSL compute
15
15
  pipelines. Convolution activations use a blocked
16
16
  four-channel layout, decoder `upsample + concat + conv` patterns are fused,
17
17
  and all network dispatches for a tile are submitted in one command buffer.
18
- FP32 convolutions use channel-specialized implicit-GEMM tiles; FP16 uses
19
- channel-specialized vector FMA and separate max-pool passes, selected from the
20
- same model descriptor.
18
+ FP16 and FP32 convolutions use channel-specialized implicit-GEMM tiles by
19
+ default, with a direct convolution for the final output layer and separate
20
+ max-pool passes.
21
21
 
22
22
  TZA half-float weights stay half-float when the device enables `shader-f16`.
23
23
  FP16 products are accumulated in short half-precision groups and periodically
@@ -50,8 +50,8 @@ initUNetFromURL('./weights/rt_ldr.tza').then((unet) => {
50
50
  .getContext('2d')
51
51
  .getImageData(0, 0, width, height);
52
52
 
53
- // Tile execute the denoising.
54
- // If the resolution is high. It will split the input into tiles and execute one tile per frame.
53
+ // Tile execute the denoising. High resolutions use balanced rectangular
54
+ // tiles with overlap only at boundaries shared by another tile.
55
55
  const abortDenoising = unet.tileExecute({
56
56
  // The color input for LDR image is 4 channels.
57
57
  // In the format of Uint8ClampedArray or Uint8Array.
@@ -89,6 +89,22 @@ initUNetFromURL('./weights/rt_hdr.tza', undefined, {
89
89
  });
90
90
  ```
91
91
 
92
+ HDR transfer defaults to the PU curve used by the regular RT models. Models
93
+ trained for the RTLightmap filter use the logarithmic curve from upstream OIDN;
94
+ select it explicitly when loading such weights:
95
+
96
+ ```ts
97
+ const lightmap = await initUNetFromURL('./weights/rtlightmap_hdr.tza', undefined, {
98
+ hdr: true,
99
+ hdrTransfer: 'log'
100
+ });
101
+ ```
102
+
103
+ The `log` transfer maps `y` to `log(1 + y) / log(65505)` and reverses this
104
+ before writing HDR output. For GPU-buffer inputs, callers may continue to
105
+ pre-scale the complete image once and leave the runtime's `inputScale` at its
106
+ existing default of `1`.
107
+
92
108
  ### Use auxiliary images
93
109
 
94
110
  ```ts
@@ -124,9 +140,8 @@ If you already have a WebGPU path tracer. You can integrate the oidn-web into yo
124
140
  initUNetFromURL(
125
141
  './weights/rt_hdr_alb_nrm.tza',
126
142
  {
127
- // Share GPUDevice and GPUAdapterInfo with the native WGSL runtime.
128
- device,
129
- adapterInfo
143
+ // Share the GPUDevice with the native WGSL runtime.
144
+ device
130
145
  },
131
146
  {
132
147
  aux: true,
@@ -180,7 +195,7 @@ const device = await adapter.requestDevice({ requiredFeatures });
180
195
 
181
196
  const unet = await initUNetFromURL(
182
197
  modelUrl,
183
- { device, adapterInfo },
198
+ { device },
184
199
  {
185
200
  aux: true,
186
201
  hdr: true,
@@ -230,40 +245,75 @@ npm run model:inspect -- weights/rt_hdr_alb_nrm.tza
230
245
  ```
231
246
 
232
247
  ```ts
233
- const unet = await initUNetFromURL(newModelUrl, backend, {
248
+ const unet = await initUNetFromURL(newModelUrl, undefined, {
234
249
  aux: true,
235
250
  hdr: true,
236
251
  modelSpec: newOidnModelSpec
237
252
  });
238
253
  ```
239
254
 
240
- ### GPU backpressure and dynamic tiles
255
+ ### GPU backpressure, scheduling, and dynamic tiles
241
256
 
242
257
  `tileExecute` waits for the submitted GPU work of a tile before scheduling the
243
258
  next tile. This keeps at most one OIDN tile in flight, which makes cancellation
244
259
  responsive instead of leaving queued denoising work ahead of interactive
245
260
  rendering.
246
261
 
247
- Tile sizing is adaptive by default. `maxTileSize` is a hard upper bound; the
248
- completed GPU time of a tiled execution adjusts the tile size used by the next
249
- execution. Single-tile images do not affect the estimate. The default range
250
- starts at 384 pixels, does not go below 256, and targets about 16 ms of GPU work
251
- per tile.
262
+ Between tiles, JavaScript yields according to `scheduling`:
263
+
264
+ - `'event-loop'` (default) continues on the next macrotask. Execution finishes
265
+ as soon as the GPU allows and also completes in background tabs.
266
+ - `'animation-frame'` waits for the next display frame, so at most one tile
267
+ runs per frame. Use it when OIDN shares the GPU with an interactive renderer
268
+ and frame rate matters more than denoise latency. If no frame arrives within
269
+ 100 ms (for example in a hidden tab), the next tile runs anyway.
270
+
271
+ Asynchronous failures, including WebGPU device loss and exceptions thrown by
272
+ `progress` or `done`, are reported to the optional `error` callback. Unless the
273
+ returned abort function is called first, every execution ends by calling `done`
274
+ or `error`; without an `error` callback, failures are logged to the console.
275
+
276
+ Tile sizing is adaptive by default. `maxTileSize` is a hard upper bound. Each
277
+ execution partitions the image into balanced rectangular output regions and
278
+ adds model context only on edges shared with another tile. Input shapes are
279
+ bucketed to at most two sizes so the native execution cache remains stable.
280
+ The smoothed P75 GPU time of completed tiled executions adjusts the maximum tile
281
+ size used by the next execution. The potentially cold first tile is excluded,
282
+ cancelled and single-tile work is ignored, and a layout is held for at least two
283
+ complete executions. The default range starts at 432 pixels, does not go below
284
+ 256, changes in 16-pixel steps, and targets about 16 ms of GPU work per tile.
252
285
 
253
286
  ```ts
254
- initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', backend, {
287
+ const unet = await initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', undefined, {
255
288
  aux: true,
256
289
  hdr: true,
257
290
  maxTileSize: 512,
258
291
  dynamicTile: {
259
292
  minTileSize: 256,
260
- initialTileSize: 384,
293
+ initialTileSize: 432,
261
294
  targetTileTimeMs: 16
262
295
  }
263
296
  });
264
297
 
298
+ // An interactive renderer can pace tiles to display frames. A smaller halo
299
+ // lowers per-tile cost; the default overlap is half of the model receptive
300
+ // field rounded up to 16 pixels.
301
+ const abortDenoising = unet.tileExecute({
302
+ color,
303
+ albedo,
304
+ normal,
305
+ tileOverlap: 80,
306
+ scheduling: 'animation-frame',
307
+ done(denoised) {
308
+ // ...
309
+ },
310
+ error(reason) {
311
+ console.error(reason);
312
+ }
313
+ });
314
+
265
315
  // Restore fixed-size behavior when deterministic tiling is preferred.
266
- initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', backend, {
316
+ initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', undefined, {
267
317
  aux: true,
268
318
  hdr: true,
269
319
  maxTileSize: 512,
@@ -271,6 +321,20 @@ initUNetFromURL('./weights/rt_hdr_alb_nrm.tza', backend, {
271
321
  });
272
322
  ```
273
323
 
324
+ ### Warm up and whole-image execution
325
+
326
+ The native runtime creates per-shape GPU buffers and pipelines the first time
327
+ it sees a tile input shape. Call `prepareForImage` with the expected image size while a loading
328
+ state is still visible, so the first `tileExecute` does not stall:
329
+
330
+ ```ts
331
+ await unet.prepareForImage(width, height);
332
+ ```
333
+
334
+ Pass `wholeImage: true` to `tileExecute` (and to `prepareForImage`) to run the
335
+ complete image as one tile and skip tiling overhead. It ignores `maxTileSize`,
336
+ so the image must fit the device's buffer and dispatch limits.
337
+
274
338
  ### Benchmark the native runtime against TFJS
275
339
 
276
340
  The browser benchmark automatically finds the nearest ancestor whose package