@aardworx/wombat.rendering 0.9.32 → 0.18.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/dist/runtime/derivedModes/kernelCodegen.d.ts +90 -0
  2. package/dist/runtime/derivedModes/kernelCodegen.d.ts.map +1 -0
  3. package/dist/runtime/derivedModes/kernelCodegen.js +623 -0
  4. package/dist/runtime/derivedModes/kernelCodegen.js.map +1 -0
  5. package/dist/runtime/derivedModes/modeKeyCpu.d.ts.map +1 -1
  6. package/dist/runtime/derivedModes/modeKeyCpu.js +23 -67
  7. package/dist/runtime/derivedModes/modeKeyCpu.js.map +1 -1
  8. package/dist/runtime/derivedModes/partitionDispatcher.d.ts +79 -0
  9. package/dist/runtime/derivedModes/partitionDispatcher.d.ts.map +1 -0
  10. package/dist/runtime/derivedModes/partitionDispatcher.js +303 -0
  11. package/dist/runtime/derivedModes/partitionDispatcher.js.map +1 -0
  12. package/dist/runtime/derivedModes/partitionKernel.d.ts +12 -0
  13. package/dist/runtime/derivedModes/partitionKernel.d.ts.map +1 -0
  14. package/dist/runtime/derivedModes/partitionKernel.js +127 -0
  15. package/dist/runtime/derivedModes/partitionKernel.js.map +1 -0
  16. package/dist/runtime/derivedModes/partitionKernelLayout.d.ts +4 -0
  17. package/dist/runtime/derivedModes/partitionKernelLayout.d.ts.map +1 -0
  18. package/dist/runtime/derivedModes/partitionKernelLayout.js +24 -0
  19. package/dist/runtime/derivedModes/partitionKernelLayout.js.map +1 -0
  20. package/dist/runtime/derivedModes/rule.d.ts +38 -86
  21. package/dist/runtime/derivedModes/rule.d.ts.map +1 -1
  22. package/dist/runtime/derivedModes/rule.js +34 -74
  23. package/dist/runtime/derivedModes/rule.js.map +1 -1
  24. package/dist/runtime/derivedUniforms/dispatch.d.ts +10 -8
  25. package/dist/runtime/derivedUniforms/dispatch.d.ts.map +1 -1
  26. package/dist/runtime/derivedUniforms/dispatch.js +31 -18
  27. package/dist/runtime/derivedUniforms/dispatch.js.map +1 -1
  28. package/dist/runtime/derivedUniforms/marker.d.ts +90 -11
  29. package/dist/runtime/derivedUniforms/marker.d.ts.map +1 -1
  30. package/dist/runtime/derivedUniforms/marker.js +161 -15
  31. package/dist/runtime/derivedUniforms/marker.js.map +1 -1
  32. package/dist/runtime/derivedUniforms/sceneIntegration.d.ts +25 -6
  33. package/dist/runtime/derivedUniforms/sceneIntegration.d.ts.map +1 -1
  34. package/dist/runtime/derivedUniforms/sceneIntegration.js +61 -20
  35. package/dist/runtime/derivedUniforms/sceneIntegration.js.map +1 -1
  36. package/dist/runtime/heapEligibility.d.ts.map +1 -1
  37. package/dist/runtime/heapEligibility.js +22 -2
  38. package/dist/runtime/heapEligibility.js.map +1 -1
  39. package/dist/runtime/heapScene/chunkedArena.d.ts +76 -0
  40. package/dist/runtime/heapScene/chunkedArena.d.ts.map +1 -0
  41. package/dist/runtime/heapScene/chunkedArena.js +223 -0
  42. package/dist/runtime/heapScene/chunkedArena.js.map +1 -0
  43. package/dist/runtime/heapScene/freelist.d.ts +50 -0
  44. package/dist/runtime/heapScene/freelist.d.ts.map +1 -0
  45. package/dist/runtime/heapScene/freelist.js +160 -0
  46. package/dist/runtime/heapScene/freelist.js.map +1 -0
  47. package/dist/runtime/heapScene/growBuffer.d.ts +23 -2
  48. package/dist/runtime/heapScene/growBuffer.d.ts.map +1 -1
  49. package/dist/runtime/heapScene/growBuffer.js +35 -3
  50. package/dist/runtime/heapScene/growBuffer.js.map +1 -1
  51. package/dist/runtime/heapScene/pools.d.ts +65 -42
  52. package/dist/runtime/heapScene/pools.d.ts.map +1 -1
  53. package/dist/runtime/heapScene/pools.js +206 -134
  54. package/dist/runtime/heapScene/pools.js.map +1 -1
  55. package/dist/runtime/heapScene.d.ts +12 -0
  56. package/dist/runtime/heapScene.d.ts.map +1 -1
  57. package/dist/runtime/heapScene.js +949 -167
  58. package/dist/runtime/heapScene.js.map +1 -1
  59. package/dist/runtime/hybridScene.d.ts +7 -0
  60. package/dist/runtime/hybridScene.d.ts.map +1 -1
  61. package/dist/runtime/hybridScene.js +1 -0
  62. package/dist/runtime/hybridScene.js.map +1 -1
  63. package/dist/runtime/index.d.ts +4 -3
  64. package/dist/runtime/index.d.ts.map +1 -1
  65. package/dist/runtime/index.js +4 -3
  66. package/dist/runtime/index.js.map +1 -1
  67. package/dist/runtime/renderTask.d.ts +5 -0
  68. package/dist/runtime/renderTask.d.ts.map +1 -1
  69. package/dist/runtime/renderTask.js +1 -0
  70. package/dist/runtime/renderTask.js.map +1 -1
  71. package/dist/runtime/runtime.d.ts +6 -0
  72. package/dist/runtime/runtime.d.ts.map +1 -1
  73. package/dist/runtime/runtime.js +1 -0
  74. package/dist/runtime/runtime.js.map +1 -1
  75. package/package.json +1 -1
  76. package/src/runtime/derivedModes/kernelCodegen.ts +673 -0
  77. package/src/runtime/derivedModes/modeKeyCpu.ts +22 -60
  78. package/src/runtime/derivedModes/partitionDispatcher.ts +349 -0
  79. package/src/runtime/derivedModes/partitionKernelLayout.ts +25 -0
  80. package/src/runtime/derivedModes/rule.ts +66 -145
  81. package/src/runtime/derivedUniforms/dispatch.ts +48 -21
  82. package/src/runtime/derivedUniforms/marker.ts +210 -16
  83. package/src/runtime/derivedUniforms/sceneIntegration.ts +67 -20
  84. package/src/runtime/heapEligibility.ts +22 -1
  85. package/src/runtime/heapScene/chunkedArena.ts +238 -0
  86. package/src/runtime/heapScene/freelist.ts +161 -0
  87. package/src/runtime/heapScene/growBuffer.ts +39 -2
  88. package/src/runtime/heapScene/pools.ts +231 -143
  89. package/src/runtime/heapScene.ts +1134 -161
  90. package/src/runtime/hybridScene.ts +8 -0
  91. package/src/runtime/index.ts +12 -9
  92. package/src/runtime/renderTask.ts +6 -0
  93. package/src/runtime/runtime.ts +7 -0
  94. package/src/runtime/derivedModes/gpuDispatcher.ts +0 -226
  95. package/src/runtime/derivedModes/gpuKernel.ts +0 -98
@@ -85,6 +85,13 @@ export interface CompileHybridSceneOptions {
85
85
  * §7 derived-uniforms opt-in — pass-through to `BuildHeapSceneOptions`.
86
86
  */
87
87
  readonly enableDerivedUniforms?: boolean;
88
+ /**
89
+ * §3 — per-arena-chunk byte cap. Pass-through to
90
+ * `BuildHeapSceneOptions.maxChunkBytes`. Demo / test value
91
+ * (e.g. 4 MB) exercises multi-chunk routing without needing
92
+ * massive workloads.
93
+ */
94
+ readonly maxChunkBytes?: number;
88
95
  }
89
96
 
90
97
  export interface HybridScene {
@@ -226,6 +233,7 @@ export function compileHybridScene(
226
233
  atlasPool,
227
234
  ...(opts.enableFamilyMerge === true ? { enableFamilyMerge: true } : {}),
228
235
  ...(opts.enableDerivedUniforms === false ? { enableDerivedUniforms: false } : {}),
236
+ ...(opts.maxChunkBytes !== undefined ? { maxChunkBytes: opts.maxChunkBytes } : {}),
229
237
  });
230
238
 
231
239
  // ─── Legacy subset → RenderTree → ScenePass ──────────────────────
@@ -130,18 +130,21 @@ export {
130
130
  export {
131
131
  derivedMode,
132
132
  isDerivedModeRule,
133
- flipCull,
134
- gpuFlipCullByDeterminant,
135
133
  type DerivedModeRule,
136
- type DerivedModeBuilder,
137
- type GpuRuleSpec,
138
134
  type ModeAxis,
139
135
  type ModeValue,
140
136
  } from "./derivedModes/rule.js";
141
137
 
142
- export { GpuDerivedModesScene } from "./derivedModes/gpuDispatcher.js";
143
138
  export {
144
- GPU_FLIP_CULL_BY_DET_WGSL,
145
- CULL_TO_U32,
146
- U32_TO_CULL,
147
- } from "./derivedModes/gpuKernel.js";
139
+ PARTITION_RECORD_PREFIX_U32,
140
+ partitionRecordU32,
141
+ partitionRecordBytes,
142
+ } from "./derivedModes/partitionKernelLayout.js";
143
+ export { GpuPartitionScene, type PartitionSceneSpec } from "./derivedModes/partitionDispatcher.js";
144
+ export {
145
+ emitPartitionKernel,
146
+ substituteReadInput, substituteReadInputInStmt,
147
+ collectUniformReadsInStmt,
148
+ rewriteOutputsToSlotIndices,
149
+ type RuleCodegenSpec, type KernelCodegenSpec,
150
+ } from "./derivedModes/kernelCodegen.js";
@@ -54,6 +54,11 @@ export interface RuntimeContext {
54
54
  * compiled against this context.
55
55
  */
56
56
  readonly enableDerivedUniforms?: boolean;
57
+ /**
58
+ * §3 — per-arena-chunk byte cap, propagated to every HybridScene
59
+ * compiled against this context. Default unset → adapter limit.
60
+ */
61
+ readonly maxChunkBytes?: number;
57
62
  }
58
63
 
59
64
  class RenderTask implements IRenderTask {
@@ -256,6 +261,7 @@ class RenderTask implements IRenderTask {
256
261
  ...(this.ctx.heapEnabled !== undefined ? { heapEnabled: this.ctx.heapEnabled } : {}),
257
262
  ...(this.ctx.enableFamilyMerge === true ? { enableFamilyMerge: true } : {}),
258
263
  ...(this.ctx.enableDerivedUniforms === false ? { enableDerivedUniforms: false } : {}),
264
+ ...(this.ctx.maxChunkBytes !== undefined ? { maxChunkBytes: this.ctx.maxChunkBytes } : {}),
259
265
  });
260
266
  this._scenes.set(cmd, s);
261
267
  }
@@ -41,6 +41,12 @@ export interface RuntimeOptions {
41
41
  * ProjTrafo as `aval<Trafo3d>` for any derived they consume.
42
42
  */
43
43
  readonly enableDerivedUniforms?: boolean;
44
+ /**
45
+ * §3 — per-arena-chunk byte cap for the heap path. Threaded into
46
+ * `HybridSceneOptions.maxChunkBytes`. Pass a small value (e.g. 4 MB)
47
+ * from a demo / test to exercise multi-chunk routing.
48
+ */
49
+ readonly maxChunkBytes?: number;
44
50
  }
45
51
 
46
52
  /**
@@ -77,6 +83,7 @@ export class Runtime {
77
83
  ...(opts.heapEnabled !== undefined ? { heapEnabled: opts.heapEnabled } : {}),
78
84
  ...(opts.enableFamilyMerge === true ? { enableFamilyMerge: true } : {}),
79
85
  ...(opts.enableDerivedUniforms === false ? { enableDerivedUniforms: false } : {}),
86
+ ...(opts.maxChunkBytes !== undefined ? { maxChunkBytes: opts.maxChunkBytes } : {}),
80
87
  };
81
88
  // `device.lost` is a real-WebGPU promise; mock devices may not
82
89
  // expose it. Treat as "never lost" in that case.
@@ -1,226 +0,0 @@
1
- // Scene-level dispatcher for the v1 GPU-evaluated derived-mode rule
2
- // (flip-cull-by-determinant; see ./gpuKernel.ts).
3
- //
4
- // Lifecycle
5
- // =========
6
- //
7
- // - Construct once per heap scene with the device.
8
- // - registerRo(drawId, modelRef, declaredCull): register an RO that
9
- // has the GPU flip-cull rule. modelRef = byte offset into the
10
- // arena where the RO's ModelTrafo mat4 lives.
11
- // - deregisterRo(drawId): unregister on removeDraw.
12
- // - markDirty(): caller hints that an input changed; the next
13
- // `dispatch()` will run the kernel.
14
- // - dispatch(arenaBuf, enc): if dirty, encode the compute pass into
15
- // `enc`. Always copies the output buffer into a staging buffer for
16
- // CPU readback.
17
- // - finish(): mapAsync the staging buffer and resolve with the
18
- // per-RO output array. The scene calls this AFTER submit and
19
- // before deciding rebuckets. The promise resolves on the next
20
- // microtask after the GPU completes the copy.
21
- //
22
- // The buffers grow lazily (pow2) as ROs are added.
23
-
24
- import { GPU_FLIP_CULL_BY_DET_WGSL, CULL_TO_U32 } from "./gpuKernel.js";
25
- import type { CullMode } from "../pipelineCache/index.js";
26
- import type { aval } from "@aardworx/wombat.adaptive";
27
- import { AdaptiveToken } from "@aardworx/wombat.adaptive";
28
-
29
- const NO_RULE = 0xFFFFFFFF;
30
-
31
- interface RoEntry {
32
- /** Byte offset of the RO's ModelTrafo mat4 in the arena. */
33
- modelRef: number;
34
- /** SG-declared cull mode — either a plain value or an aval whose
35
- * current value the dispatcher reads each frame. */
36
- declared: CullMode | aval<CullMode>;
37
- }
38
-
39
- const WG_SIZE = 64;
40
- const POW2 = (n: number): number => { let p = 1; while (p < n) p <<= 1; return Math.max(64, p); };
41
-
42
- export class GpuDerivedModesScene {
43
- readonly device: GPUDevice;
44
-
45
- private readonly entries: (RoEntry | undefined)[] = [];
46
- private liveCount = 0;
47
- private dirty = false;
48
-
49
- // GPU resources — lazily allocated on first registerRo.
50
- private inputBuf: GPUBuffer | null = null;
51
- private outputBuf: GPUBuffer | null = null;
52
- private stagingBuf: GPUBuffer | null = null;
53
- private paramsBuf: GPUBuffer | null = null;
54
- private pipeline: GPUComputePipeline | null = null;
55
- private bindGroup: GPUBindGroup | null = null;
56
- private layout: GPUBindGroupLayout | null = null;
57
- /** Capacity in number of u32 entries (= numROs the buffers can hold). */
58
- private capacity = 0;
59
- /** Last-known arena buffer, used to detect bind-group invalidation. */
60
- private arenaBuf: GPUBuffer | null = null;
61
- /** True while the staging buffer's mapAsync is in flight. We must
62
- * not enqueue a new copyBufferToBuffer into it until it's been
63
- * unmapped (mapAsync resolved + unmap called). */
64
- private readbackInFlight = false;
65
-
66
- /** Per-RO last-known output. CPU mirror, populated by finish(). */
67
- readonly lastOutput: Uint32Array;
68
-
69
- constructor(device: GPUDevice) {
70
- this.device = device;
71
- this.lastOutput = new Uint32Array(0);
72
- }
73
-
74
- get registered(): number { return this.liveCount; }
75
-
76
- registerRo(
77
- drawId: number,
78
- modelRef: number,
79
- declared: CullMode | aval<CullMode>,
80
- ): void {
81
- this.entries[drawId] = { modelRef, declared };
82
- this.liveCount++;
83
- this.dirty = true;
84
- }
85
-
86
- deregisterRo(drawId: number): void {
87
- if (this.entries[drawId] === undefined) return;
88
- this.entries[drawId] = undefined;
89
- this.liveCount--;
90
- this.dirty = true;
91
- }
92
-
93
- markDirty(): void { this.dirty = true; }
94
- consumeDirty(): boolean { const was = this.dirty; this.dirty = false; return was; }
95
-
96
- private ensureResources(arenaBuf: GPUBuffer, numROs: number): void {
97
- const needCap = POW2(Math.max(1, numROs));
98
- if (this.outputBuf === null || needCap > this.capacity) {
99
- this.inputBuf ?.destroy();
100
- this.outputBuf ?.destroy();
101
- this.stagingBuf?.destroy();
102
- const bytes = needCap * 4;
103
- this.inputBuf = this.device.createBuffer({ size: bytes, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, label: "gpuDerivedModes/in" });
104
- this.outputBuf = this.device.createBuffer({ size: bytes, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC | GPUBufferUsage.COPY_DST, label: "gpuDerivedModes/out" });
105
- this.stagingBuf = this.device.createBuffer({ size: bytes, usage: GPUBufferUsage.MAP_READ | GPUBufferUsage.COPY_DST, label: "gpuDerivedModes/staging" });
106
- (this as { lastOutput: Uint32Array }).lastOutput = new Uint32Array(needCap);
107
- this.capacity = needCap;
108
- this.bindGroup = null;
109
- }
110
- if (this.paramsBuf === null) {
111
- this.paramsBuf = this.device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST, label: "gpuDerivedModes/params" });
112
- }
113
- if (this.layout === null) {
114
- this.layout = this.device.createBindGroupLayout({
115
- label: "gpuDerivedModes/bgl",
116
- entries: [
117
- { binding: 0, visibility: GPUShaderStage.COMPUTE, buffer: { type: "read-only-storage" } },
118
- { binding: 1, visibility: GPUShaderStage.COMPUTE, buffer: { type: "read-only-storage" } },
119
- { binding: 2, visibility: GPUShaderStage.COMPUTE, buffer: { type: "storage" } },
120
- { binding: 3, visibility: GPUShaderStage.COMPUTE, buffer: { type: "uniform" } },
121
- ],
122
- });
123
- }
124
- if (this.pipeline === null) {
125
- const module = this.device.createShaderModule({ code: GPU_FLIP_CULL_BY_DET_WGSL, label: "gpuDerivedModes/module" });
126
- this.pipeline = this.device.createComputePipeline({
127
- label: "gpuDerivedModes/pipeline",
128
- layout: this.device.createPipelineLayout({ bindGroupLayouts: [this.layout] }),
129
- compute: { module, entryPoint: "evaluate" },
130
- });
131
- }
132
- if (this.bindGroup === null || this.arenaBuf !== arenaBuf) {
133
- this.bindGroup = this.device.createBindGroup({
134
- label: "gpuDerivedModes/bindGroup",
135
- layout: this.layout,
136
- entries: [
137
- { binding: 0, resource: { buffer: arenaBuf } },
138
- { binding: 1, resource: { buffer: this.inputBuf! } },
139
- { binding: 2, resource: { buffer: this.outputBuf! } },
140
- { binding: 3, resource: { buffer: this.paramsBuf! } },
141
- ],
142
- });
143
- this.arenaBuf = arenaBuf;
144
- }
145
- }
146
-
147
- /** Upload current per-RO input refs + params; called when dirty. */
148
- private uploadInputs(numROs: number): void {
149
- const refs = new Uint32Array(this.capacity).fill(NO_RULE);
150
- let declaredU32: number = CULL_TO_U32.back;
151
- for (let i = 0; i < numROs; i++) {
152
- const e = this.entries[i];
153
- if (e !== undefined) {
154
- refs[i] = e.modelRef;
155
- // Resolve the declared each dispatch. v1 supports one shared
156
- // declared per kernel (single uniform in the WGSL). Last
157
- // registered entry's declared wins. For the typical SG shape
158
- // where a single `<Sg CullMode={cv}>` wraps every leaf, every
159
- // entry's declared is the same aval — so picking the last is
160
- // equivalent to picking any. If the user genuinely binds
161
- // distinct declareds across the scene, an axis-domain
162
- // analyzer would be the correct generalization.
163
- const d = e.declared;
164
- const v = typeof d === "string"
165
- ? d
166
- : (d as aval<CullMode>).getValue(AdaptiveToken.top);
167
- declaredU32 = CULL_TO_U32[v];
168
- }
169
- }
170
- this.device.queue.writeBuffer(this.inputBuf!, 0, refs.buffer, refs.byteOffset, refs.byteLength);
171
- const params = new Uint32Array([numROs, declaredU32, 0, 0]);
172
- this.device.queue.writeBuffer(this.paramsBuf!, 0, params.buffer, params.byteOffset, params.byteLength);
173
- }
174
-
175
- /**
176
- * Encode the compute dispatch + the staging copy into `enc`. Caller
177
- * is responsible for `device.queue.submit([enc.finish()])`. Returns
178
- * the number of ROs the kernel was dispatched against.
179
- */
180
- dispatch(arenaBuf: GPUBuffer, numROs: number, enc: GPUCommandEncoder): number {
181
- this.ensureResources(arenaBuf, numROs);
182
- // If the previous frame's readback hasn't resolved yet, we can
183
- // still dispatch the compute (writes to outputBuf), but we must
184
- // skip the staging copy — otherwise WebGPU rejects the submit
185
- // with "buffer used in submit while mapped". Cheap to skip: the
186
- // next dispatch will re-copy when staging is unmapped.
187
- this.uploadInputs(numROs);
188
- const pass = enc.beginComputePass({ label: "gpuDerivedModes/pass" });
189
- pass.setPipeline(this.pipeline!);
190
- pass.setBindGroup(0, this.bindGroup!);
191
- pass.dispatchWorkgroups(Math.ceil(numROs / WG_SIZE));
192
- pass.end();
193
- if (!this.readbackInFlight) {
194
- const bytes = numROs * 4;
195
- enc.copyBufferToBuffer(this.outputBuf!, 0, this.stagingBuf!, 0, bytes);
196
- this.readbackInFlight = true;
197
- }
198
- return numROs;
199
- }
200
-
201
- /**
202
- * mapAsync the staging buffer + copy values into `lastOutput`.
203
- * Resolves on the GPU finishing the copy. Caller diffs lastOutput
204
- * against its own cached values to find changed ROs.
205
- */
206
- async finish(numROs: number): Promise<void> {
207
- if (this.stagingBuf === null || numROs === 0) return;
208
- if (!this.readbackInFlight) return; // dispatch skipped the copy; nothing to read
209
- const bytes = numROs * 4;
210
- try {
211
- await this.stagingBuf.mapAsync(GPUMapMode.READ, 0, bytes);
212
- const view = new Uint32Array(this.stagingBuf.getMappedRange(0, bytes).slice(0));
213
- this.stagingBuf.unmap();
214
- (this as { lastOutput: Uint32Array }).lastOutput.set(view, 0);
215
- } finally {
216
- this.readbackInFlight = false;
217
- }
218
- }
219
-
220
- dispose(): void {
221
- this.inputBuf ?.destroy();
222
- this.outputBuf ?.destroy();
223
- this.stagingBuf?.destroy();
224
- this.paramsBuf ?.destroy();
225
- }
226
- }
@@ -1,98 +0,0 @@
1
- // GPU-evaluated derived-mode rules — minimum viable kernel.
2
- //
3
- // CPU eval (Task 2 Phase 1-3, 0.9.17) runs rule closures in JS. For
4
- // rules whose inputs live GPU-side (per-RO trafos, etc.) JS evaluation
5
- // fans out CPU work that's wasted: the trafo bytes are right there in
6
- // the arena. This module runs the rule on the GPU instead, in one
7
- // dispatch over all dirty ROs per frame.
8
- //
9
- // v1 scope is deliberately narrow: ONE hard-coded WGSL kernel for
10
- // "flip cullMode by sign of det(upper3x3 of an arena mat4)". Picked
11
- // because it's the motivating case from docs/derived-modes.md and
12
- // proves the GPU-eval path end-to-end. Generalizes via a traced IR
13
- // builder in v2 (see derived-mode-rules-plan.md Phase 5).
14
- //
15
- // Architecture
16
- // ============
17
- //
18
- // - One **module-wide** `GpuDerivedModesScene` per heap scene. Holds:
19
- // * `roOutputBuf` — `array<u32>` sized to scene.numROs.
20
- // Each entry = the rule's modeKey contribution.
21
- // * `roInputBuf` — `array<u32>` (one ref-into-arena per RO)
22
- // telling the kernel where to read the input.
23
- // * `stagingBuf` — MAP_READ + COPY_DST mirror of roOutputBuf,
24
- // for the per-frame readback.
25
- // - Dispatch each frame in `update()`, ONLY when something dirty.
26
- // - `mapAsync(staging)` resolves on a microtask; the CPU diffs the
27
- // buffer against `roLastKey[]`. Any change → mark drawId dirty;
28
- // the existing rebucket flow does the rest.
29
- //
30
- // The dirty gate keeps steady-state cost ≈ 0 — when no input aval
31
- // marks AND no RO add/remove, we skip the dispatch + readback.
32
-
33
- export const GPU_FLIP_CULL_BY_DET_WGSL = /* wgsl */ `
34
- // Per-RO record. Index i corresponds to drawId i in the scene's
35
- // flat RO table.
36
- struct GpuRuleParams {
37
- numROs: u32,
38
- declared: u32, // 0 = none, 1 = front, 2 = back (must match CULL_MODE[] order in bitfield.ts)
39
- _pad0: u32,
40
- _pad1: u32,
41
- };
42
-
43
- @group(0) @binding(0) var<storage, read> arena: array<u32>;
44
- // One per-RO entry: byte offset within the arena of the ROs ModelTrafo
45
- // allocation header. Data is at offset + 16 (ALLOC_HEADER_PAD_TO).
46
- // 0xFFFFFFFFu means "this RO does NOT have a flip-cull-by-det rule".
47
- @group(0) @binding(1) var<storage, read> roInputRefs: array<u32>;
48
- @group(0) @binding(2) var<storage, read_write> roOutputs: array<u32>;
49
- @group(0) @binding(3) var<uniform> uIn: GpuRuleParams;
50
-
51
- // Read a row-major mat4 from the arena at byte offset \`refBytes + 16\`
52
- // (header is 16 bytes wide due to ALLOC_HEADER_PAD_TO). Returns the
53
- // 3×3 upper-left in row-major order: [m00,m01,m02, m10,m11,m12, m20,m21,m22].
54
- fn loadUpper3x3(refBytes: u32) -> array<f32, 9> {
55
- let baseU32 = (refBytes + 16u) >> 2u;
56
- // The arena is a Float32 storage of mat4 in row-major, 16 floats.
57
- // Read 9 of them.
58
- return array<f32, 9>(
59
- bitcast<f32>(arena[baseU32 + 0u]), bitcast<f32>(arena[baseU32 + 1u]), bitcast<f32>(arena[baseU32 + 2u]),
60
- bitcast<f32>(arena[baseU32 + 4u]), bitcast<f32>(arena[baseU32 + 5u]), bitcast<f32>(arena[baseU32 + 6u]),
61
- bitcast<f32>(arena[baseU32 + 8u]), bitcast<f32>(arena[baseU32 + 9u]), bitcast<f32>(arena[baseU32 + 10u]),
62
- );
63
- }
64
-
65
- fn det3x3(m: array<f32, 9>) -> f32 {
66
- return m[0] * (m[4] * m[8] - m[5] * m[7])
67
- - m[1] * (m[3] * m[8] - m[5] * m[6])
68
- + m[2] * (m[3] * m[7] - m[4] * m[6]);
69
- }
70
-
71
- // Mirror of bitfield.ts CULL_MODE[] order: 0=none, 1=front, 2=back.
72
- fn flipCull(c: u32) -> u32 {
73
- if (c == 1u) { return 2u; } // front -> back
74
- if (c == 2u) { return 1u; } // back -> front
75
- return c; // none
76
- }
77
-
78
- @compute @workgroup_size(64)
79
- fn evaluate(@builtin(global_invocation_id) gid: vec3<u32>) {
80
- let i = gid.x;
81
- if (i >= uIn.numROs) { return; }
82
- let refByte = roInputRefs[i];
83
- if (refByte == 0xFFFFFFFFu) {
84
- // No rule on this RO; leave existing output untouched. Allows a
85
- // single kernel dispatch to skip non-participating ROs.
86
- return;
87
- }
88
- let m = loadUpper3x3(refByte);
89
- let d = det3x3(m);
90
- let declared = uIn.declared;
91
- let outVal = select(declared, flipCull(declared), d < 0.0);
92
- roOutputs[i] = outVal;
93
- }
94
- `;
95
-
96
- /** Map between bitfield.ts's CULL_MODE order and the kernel's u32 enum. */
97
- export const CULL_TO_U32 = { "none": 0, "front": 1, "back": 2 } as const;
98
- export const U32_TO_CULL = ["none", "front", "back"] as const;