@aardworx/wombat.rendering 0.19.8 → 0.19.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aardworx/wombat.rendering",
3
- "version": "0.19.8",
3
+ "version": "0.19.9",
4
4
  "description": "WebGPU rendering layer for the Wombat TypeScript stack — RenderObject/RenderTask/runtime + window glue, port of Aardvark.Rendering's lower layers on top of WebGPU and wombat.shader.",
5
5
  "license": "MIT",
6
6
  "author": "krauthaufen",
@@ -27,13 +27,14 @@
27
27
  // `ro.samplers.count <= 1` are classifier invariants.
28
28
 
29
29
  import { type aval, type AdaptiveToken } from "@aardworx/wombat.adaptive";
30
- import type { IBuffer, HostBufferSource } from "../core/buffer.js";
31
- import type { BufferView } from "../core/bufferView.js";
30
+ import { IBuffer, type HostBufferSource } from "../core/buffer.js";
31
+ import { BufferView } from "../core/bufferView.js";
32
32
  import { ITexture, type HostTextureSource } from "../core/texture.js";
33
33
  import { ISampler } from "../core/sampler.js";
34
34
  import type { RenderObject } from "../core/renderObject.js";
35
35
  import { asAttributeProvider, asUniformProvider } from "../core/provider.js";
36
36
  import type { HeapDrawSpec, HeapTextureSet } from "./heapScene.js";
37
+ import { isBufferView } from "./heapScene/pools.js";
37
38
  import { compileHeapEffect } from "./heapEffect.js";
38
39
  import { AtlasPool } from "./textureAtlas/atlasPool.js";
39
40
 
@@ -50,13 +51,29 @@ function asUint32(data: HostBufferSource): Uint32Array {
50
51
  return new Uint32Array(data);
51
52
  }
52
53
 
54
+ // Widen 16-bit indices to u32 (the heap's indexStorage is u32). Copy, not a
55
+ // reinterpret view — each u16 index becomes a u32 element.
56
+ function widenU16(data: HostBufferSource): Uint32Array {
57
+ const u16 = data instanceof Uint16Array
58
+ ? data
59
+ : ArrayBuffer.isView(data)
60
+ ? new Uint16Array(data.buffer, data.byteOffset, data.byteLength / 2)
61
+ : new Uint16Array(data);
62
+ return Uint32Array.from(u16);
63
+ }
64
+
53
65
  // IndexPool keys on aval identity → ROs sharing the same
54
66
  // `ro.indices.buffer` must produce the SAME downstream
55
67
  // `aval<Uint32Array>`. A fresh `.map(…)` per RO would defeat the
56
68
  // pool's identity-based sharing and balloon GPU memory (e.g. 11K
57
69
  // spheres each allocating their own copy of the 6144-index buffer
58
70
  // → 270 MB).
59
- const indicesAvalCache = new WeakMap<aval<IBuffer>, aval<Uint32Array>>();
71
+ // Keyed by (buffer aval, baseVertex). baseVertex is baked into the index
72
+ // VALUES (index[i] + baseVertex), so a buffer drawn with two different
73
+ // baseVertex offsets needs two distinct transcoded arrays — but the common
74
+ // case (baseVertex 0, or per-primitive index buffers that aren't shared)
75
+ // keeps one entry per buffer and preserves the pool's identity sharing.
76
+ const indicesAvalCache = new WeakMap<aval<IBuffer>, Map<number, aval<Uint32Array>>>();
60
77
 
61
78
  // Per-aval shared "current pool ref" cell. ROs that share an
62
79
  // `aval<ITexture>` (and therefore one AtlasPool entry) all read
@@ -71,20 +88,79 @@ function currentRefCellFor(av: aval<ITexture>, initial: number): { ref: number }
71
88
  }
72
89
  return c;
73
90
  }
74
- function indicesAvalFor(ibAval: aval<IBuffer>): aval<Uint32Array> {
75
- let av = indicesAvalCache.get(ibAval);
91
+ function indicesAvalFor(ibAval: aval<IBuffer>, indexFormat: GPUIndexFormat, baseVertex: number): aval<Uint32Array> {
92
+ let byBase = indicesAvalCache.get(ibAval);
93
+ if (byBase === undefined) {
94
+ byBase = new Map();
95
+ indicesAvalCache.set(ibAval, byBase);
96
+ }
97
+ let av = byBase.get(baseVertex);
76
98
  if (av === undefined) {
77
99
  av = ibAval.map((ib: IBuffer) => {
78
100
  if (ib.kind !== "host") {
79
101
  throw new Error("heapAdapter: index buffer flipped to native GPUBuffer; classifier should have caught this");
80
102
  }
81
- return asUint32(ib.data);
103
+ const u32 = indexFormat === "uint16" ? widenU16(ib.data) : asUint32(ib.data);
104
+ if (baseVertex === 0) return u32;
105
+ // Fold baseVertex into the index values so the megacall decode
106
+ // (vid = indexStorage[...]) lands on the right vertex with no
107
+ // extra record field. Copy (don't mutate the shared view).
108
+ const out = new Uint32Array(u32.length);
109
+ for (let i = 0; i < u32.length; i++) out[i] = u32[i]! + baseVertex;
110
+ return out;
82
111
  });
83
- indicesAvalCache.set(ibAval, av);
112
+ byBase.set(baseVertex, av);
84
113
  }
85
114
  return av;
86
115
  }
87
116
 
117
+ // Byte view over a host buffer source (for the de-interleave gather).
118
+ function bytesOf(data: HostBufferSource): Uint8Array {
119
+ if (data instanceof Uint8Array) return data;
120
+ if (ArrayBuffer.isView(data)) return new Uint8Array(data.buffer, data.byteOffset, data.byteLength);
121
+ return new Uint8Array(data);
122
+ }
123
+
124
+ // De-interleaved (tight, offset-0) views, keyed by (source buffer aval,
125
+ // offset, stride) so ROs sharing one interleaved buffer + the same attribute
126
+ // share a single gathered allocation.
127
+ const deinterleaveCache = new WeakMap<aval<IBuffer>, Map<string, aval<IBuffer>>>();
128
+
129
+ /**
130
+ * Make a tight, offset-0 BufferView the arena can ingest. Tight/broadcast
131
+ * views pass through unchanged; interleaved or offset views are gathered
132
+ * into a fresh packed host buffer (every `stride` bytes from `offset`,
133
+ * `byteSize` each) wrapped as a new whole-buffer view.
134
+ */
135
+ function tightenBufferView(bv: BufferView): BufferView {
136
+ if (bv.singleValue !== undefined) return bv; // broadcast → lowered to a uniform
137
+ const byteSize = bv.elementType.byteSize;
138
+ const offset = bv.offset ?? 0;
139
+ const stride = (bv.stride !== undefined && bv.stride > 0) ? bv.stride : byteSize;
140
+ if (offset === 0 && stride === byteSize) return bv; // already tight
141
+ let byKey = deinterleaveCache.get(bv.buffer);
142
+ if (byKey === undefined) { byKey = new Map(); deinterleaveCache.set(bv.buffer, byKey); }
143
+ const key = `${offset}:${stride}:${byteSize}`;
144
+ let tight = byKey.get(key);
145
+ if (tight === undefined) {
146
+ tight = bv.buffer.map((ib: IBuffer): IBuffer => {
147
+ if (ib.kind !== "host") {
148
+ throw new Error("heapAdapter: interleaved attribute buffer flipped to native GPUBuffer; classifier should have caught this");
149
+ }
150
+ const src = bytesOf(ib.data);
151
+ const count = Math.max(0, Math.floor((src.byteLength - offset - byteSize) / stride) + 1);
152
+ const out = new Uint8Array(count * byteSize);
153
+ for (let i = 0; i < count; i++) {
154
+ const s = offset + i * stride;
155
+ out.set(src.subarray(s, s + byteSize), i * byteSize);
156
+ }
157
+ return IBuffer.fromHost(out);
158
+ });
159
+ byKey.set(key, tight);
160
+ }
161
+ return BufferView.ofBuffer(tight, bv.elementType);
162
+ }
163
+
88
164
  /**
89
165
  * Pull `(format, width, height, mipLevelCount)` out of an ITexture so
90
166
  * we can run Tier-S eligibility against it. Both variants expose this:
@@ -197,7 +273,7 @@ export function renderObjectToHeapSpec(
197
273
  const inputs: { [name: string]: aval<unknown> | unknown } = {};
198
274
  for (const a of schema.attributes) {
199
275
  const bv = vAttr.tryGet(a.name);
200
- if (bv !== undefined) inputs[a.name] = bv;
276
+ if (bv !== undefined) inputs[a.name] = isBufferView(bv) ? tightenBufferView(bv) : bv;
201
277
  }
202
278
  const pullUniform = (name: string): void => {
203
279
  if (Object.prototype.hasOwnProperty.call(inputs, name)) return;
@@ -222,17 +298,25 @@ export function renderObjectToHeapSpec(
222
298
  instanceAttributes = {};
223
299
  for (const n of names) {
224
300
  const bv = iAttr.tryGet(n);
225
- if (bv !== undefined) instanceAttributes[n] = bv;
301
+ if (bv !== undefined) instanceAttributes[n] = isBufferView(bv) ? tightenBufferView(bv) : bv;
226
302
  }
227
303
  }
228
304
  }
229
305
 
306
+ // DrawCall: read once (snapshot — like instanceCount/vertexCount).
307
+ // Slice offsets fold in here: firstIndex → record indexStart,
308
+ // indexCount → record slice length, baseVertex → baked into the
309
+ // transcoded index values (so it needs no record field).
310
+ const dc = ro.drawCall.getValue(token);
311
+ const baseVertex = dc.kind === "indexed" ? dc.baseVertex : 0;
312
+
230
313
  // 2. Indices: BufferView → aval<Uint32Array>. Map the underlying
231
- // IBuffer aval; the heap path's IndexPool will key on this aval.
314
+ // IBuffer aval; the heap path's IndexPool will key on this aval
315
+ // (and on baseVertex, since it's baked into the values).
232
316
  // Non-indexed ROs (no `ro.indices`) carry no index aval — the spec
233
317
  // then sets `vertexCount` and the megacall decodes the vertex directly.
234
318
  const indices: aval<Uint32Array> | undefined =
235
- ro.indices !== undefined ? indicesAvalFor(ro.indices.buffer) : undefined;
319
+ ro.indices !== undefined ? indicesAvalFor(ro.indices.buffer, ro.indices.elementType.indexFormat ?? "uint32", baseVertex) : undefined;
236
320
 
237
321
  // 3. Texture/sampler: single-pair v1. The Sg compile layer binds
238
322
  // each texture aval under both `name` and `${name}_view` so the
@@ -355,19 +439,20 @@ export function renderObjectToHeapSpec(
355
439
  );
356
440
  }
357
441
 
358
- // 4. DrawCall: read once. Classifier validates fields are heap-
359
- // compatible (indexed, instanceCount=1, zero offsets).
360
- const dc = ro.drawCall.getValue(token);
442
+ // 4. DrawCall geometry. Classifier validates fields are heap-
443
+ // compatible (instanceCount≥1, firstInstance=0, non-indexed firstVertex=0).
361
444
  if (dc.kind === "indexed" && indices === undefined) {
362
445
  throw new Error("heapAdapter: indexed drawCall without indices");
363
446
  }
364
447
  if (dc.kind === "non-indexed" && indices !== undefined) {
365
448
  throw new Error("heapAdapter: non-indexed drawCall with an index buffer");
366
449
  }
367
- // Indexed → carry the index aval; non-indexed → carry the vertex count and
368
- // the megacall uses the local vertex index directly (no index lookup).
450
+ // Indexed → carry the index aval + the firstIndex/indexCount slice (folded
451
+ // into the record's indexStart/indexCount; baseVertex is already baked into
452
+ // the index values). Non-indexed → carry the vertex count; the megacall uses
453
+ // the local vertex index directly (no index lookup).
369
454
  const geom = dc.kind === "indexed"
370
- ? { indices: indices! }
455
+ ? { indices: indices!, firstIndex: dc.firstIndex, indexCount: dc.indexCount }
371
456
  : { vertexCount: dc.vertexCount };
372
457
 
373
458
  return {
@@ -185,27 +185,17 @@ export function isHeapEligible(ro: RenderObject): aval<boolean> {
185
185
  return AVal.constant(false);
186
186
  }
187
187
 
188
- // 2. BufferView stride wedge only tight per-vertex/per-instance
189
- // layouts (and broadcasts via `singleValue` / stride 0) are
190
- // ingested. The shader-side cyclic addressing (`vid % length`,
191
- // `iidx % length`) makes broadcasts a degenerate length-1 case,
192
- // so we don't reject them here but interleaved strides remain
193
- // unsupported. Per-instance attributes follow the same rule.
194
- let strideBlocked = false;
195
- const checkStride = (v: BufferView): void => {
196
- const stride = v.stride ?? v.elementType.byteSize;
197
- const isBroadcast = v.singleValue !== undefined || stride === 0;
198
- if (!isBroadcast && stride !== v.elementType.byteSize) strideBlocked = true;
199
- };
200
- for (const v of attrViews(asAttributeProvider(ro.vertexAttributes))) checkStride(v);
201
- if (ro.instanceAttributes !== undefined) {
202
- for (const v of attrViews(asAttributeProvider(ro.instanceAttributes))) checkStride(v);
203
- }
204
- if (strideBlocked) return AVal.constant(false);
188
+ // 2. Interleaved / offset attributes are handled by de-interleaving at
189
+ // ingest (heapAdapter.tightenBufferView gathers a tight, offset-0
190
+ // copy keyed by (buffer, offset, stride) so sharing is preserved), so
191
+ // no stride/offset wedge remains. Broadcasts (singleValue / stride 0)
192
+ // still pass through as a degenerate length-1 case.
205
193
 
206
- // 3. Index-format wedge — heap stores indices as u32.
207
- if (ro.indices !== undefined && ro.indices.elementType.indexFormat !== "uint32") {
208
- return AVal.constant(false);
194
+ // 3. Index-format wedge — heap stores indices as u32; u16 is widened
195
+ // to u32 at ingest (heapAdapter.widenU16). Anything else is rejected.
196
+ if (ro.indices !== undefined) {
197
+ const fmt = ro.indices.elementType.indexFormat;
198
+ if (fmt !== "uint32" && fmt !== "uint16") return AVal.constant(false);
209
199
  }
210
200
 
211
201
  const buffers = bufferAvals(ro);
@@ -233,27 +223,19 @@ export function isHeapEligible(ro: RenderObject): aval<boolean> {
233
223
  if (dc.firstInstance !== 0) return false;
234
224
  if (dc.kind === "indexed") {
235
225
  if (ro.indices === undefined) return false;
236
- if (dc.baseVertex !== 0) return false;
237
- if (dc.firstIndex !== 0) return false;
238
- // Whole-buffer wedge: the heap ingests the ENTIRE index + vertex
239
- // buffers for an RO (IndexPool/AttributeArena copy `arr.length`,
240
- // not the drawCall slice). An RO that consumes only a PREFIX of a
241
- // shared index buffer (firstIndex=0 but indexCount < bufferCount
242
- // e.g. one glyph in a multi-glyph text run whose glyphs share a
243
- // single atlas index buffer) would therefore drag the rest of the
244
- // buffer into its draw: the trailing indices reference vertices
245
- // past this RO's range, painting stray triangles. Until the heap
246
- // honours the drawCall slice, only ROs that own their whole index
247
- // buffer are eligible.
248
- const ib = ro.indices.buffer.getValue(token);
249
- if (ib.kind === "host") {
250
- const idxBufCount = ib.sizeBytes >>> 2; // u32 indices
251
- if (dc.indexCount !== idxBufCount) return false;
252
- }
226
+ // Offsets and shared sub-buffer slices are all eligible now: the heap
227
+ // still ingests the WHOLE index + vertex buffers (one shared arena
228
+ // allocation per buffer aval sharing is the heap's whole point), but
229
+ // the record now honours the drawCall slice. firstIndex folds into the
230
+ // record's indexStart, indexCount becomes the slice length (so a single
231
+ // glyph of a multi-glyph run reads only its run of indices), and
232
+ // baseVertex is baked into the transcoded index values at ingest
233
+ // (heapAdapter.indicesAvalFor). No drawCall wedge remains here.
253
234
  } else {
254
235
  // Non-indexed: the megacall uses the local vertex index directly, so a
255
- // prefix read is safe (vid = 0..vertexCount-1) no whole-buffer wedge.
256
- // firstVertex must be 0 (the heap maps vid attribute[vid] from base 0).
236
+ // prefix read is safe (vid = 0..vertexCount-1). firstVertex must be 0 —
237
+ // a non-indexed vertex offset would need a per-record field (the index
238
+ // bake-in trick doesn't apply with no index buffer), so it's deferred.
257
239
  if (ro.indices !== undefined) return false;
258
240
  if (dc.firstVertex !== 0) return false;
259
241
  }
@@ -465,6 +465,17 @@ export interface HeapDrawSpec {
465
465
  readonly indices?: aval<Uint32Array> | Uint32Array;
466
466
  /** Vertex count for a non-indexed draw (when `indices` is omitted). */
467
467
  readonly vertexCount?: number;
468
+ /**
469
+ * Optional index slice into a (possibly shared) index buffer. The heap
470
+ * still allocates/uploads the WHOLE `indices` buffer once per aval (so N
471
+ * draws sharing it cost one allocation), but this draw emits only
472
+ * `[firstIndex, firstIndex+indexCount)`. Folds into the record's
473
+ * indexStart (`alloc.firstIndex + firstIndex`) and indexCount. Omit for a
474
+ * whole-buffer draw (defaults: firstIndex 0, indexCount = buffer length).
475
+ * (baseVertex is handled upstream by baking it into the index values.)
476
+ */
477
+ readonly firstIndex?: number;
478
+ readonly indexCount?: number;
468
479
  /**
469
480
  * Optional texture set. When present, adds a `texture_2d<f32>` at
470
481
  * binding 4 and a sampler at binding 5; the FS must declare them.
@@ -3025,8 +3036,11 @@ export function buildHeapScene(
3025
3036
  ? indexPool.acquire(device, arena.indices, indicesAval!, bucket.chunkIdx, readPlain(spec.indices!) as Uint32Array)
3026
3037
  : undefined;
3027
3038
  // Emit count per instance + the firstIndex field written to the record.
3028
- const emitCount = hasIndices ? idxAlloc!.count : (spec.vertexCount ?? 0);
3029
- const firstIndexField = hasIndices ? idxAlloc!.firstIndex : HEAP_NONINDEXED;
3039
+ // Slice: emit only [firstIndex, firstIndex+indexCount) of the (shared,
3040
+ // whole-buffer) index allocation. firstIndex folds into indexStart;
3041
+ // indexCount defaults to the whole buffer.
3042
+ const emitCount = hasIndices ? (spec.indexCount ?? idxAlloc!.count) : (spec.vertexCount ?? 0);
3043
+ const firstIndexField = hasIndices ? (idxAlloc!.firstIndex + (spec.firstIndex ?? 0)) : HEAP_NONINDEXED;
3030
3044
 
3031
3045
  const localSlot = bucket.drawHeap.alloc();
3032
3046
  // Per-RO instancing: read `spec.instanceCount` (defaults to 1).