@driftengine/texture 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/LICENSE +202 -0
  2. package/NOTICE +29 -0
  3. package/README.md +106 -0
  4. package/dist/decodeCpu.d.ts +59 -0
  5. package/dist/decodeCpu.js +234 -0
  6. package/dist/decodeGraph.d.ts +105 -0
  7. package/dist/decodeGraph.js +180 -0
  8. package/dist/half.d.ts +24 -0
  9. package/dist/half.js +86 -0
  10. package/dist/index.d.ts +66 -0
  11. package/dist/index.js +55 -0
  12. package/dist/inference.d.ts +53 -0
  13. package/dist/inference.js +243 -0
  14. package/dist/materialArray.d.ts +38 -0
  15. package/dist/materialArray.js +40 -0
  16. package/dist/mipNdf.d.ts +29 -0
  17. package/dist/mipNdf.js +53 -0
  18. package/dist/overlay/journal.d.ts +78 -0
  19. package/dist/overlay/journal.js +171 -0
  20. package/dist/overlay/sparse.d.ts +68 -0
  21. package/dist/overlay/sparse.js +212 -0
  22. package/dist/progressive.d.ts +30 -0
  23. package/dist/progressive.js +56 -0
  24. package/dist/residency/pageCache.d.ts +103 -0
  25. package/dist/residency/pageCache.js +184 -0
  26. package/dist/residency/predict.d.ts +55 -0
  27. package/dist/residency/predict.js +51 -0
  28. package/dist/residency/predictor.d.ts +16 -0
  29. package/dist/residency/predictor.js +44 -0
  30. package/dist/residency/queue.d.ts +26 -0
  31. package/dist/residency/queue.js +52 -0
  32. package/dist/residency/stream.d.ts +66 -0
  33. package/dist/residency/stream.js +142 -0
  34. package/dist/residency/table.d.ts +36 -0
  35. package/dist/residency/table.js +72 -0
  36. package/dist/residency/viewTiles.d.ts +108 -0
  37. package/dist/residency/viewTiles.js +419 -0
  38. package/dist/semantics.d.ts +52 -0
  39. package/dist/semantics.js +76 -0
  40. package/dist/tensor/architecture.d.ts +53 -0
  41. package/dist/tensor/architecture.js +96 -0
  42. package/dist/tensor/attention.d.ts +5 -0
  43. package/dist/tensor/attention.js +62 -0
  44. package/dist/tensor/denseOperators.d.ts +2 -0
  45. package/dist/tensor/denseOperators.js +136 -0
  46. package/dist/tensor/graph.d.ts +83 -0
  47. package/dist/tensor/graph.js +175 -0
  48. package/dist/tensor/linear.d.ts +49 -0
  49. package/dist/tensor/linear.js +136 -0
  50. package/dist/tensor/operatorKit.d.ts +27 -0
  51. package/dist/tensor/operatorKit.js +45 -0
  52. package/dist/tensor/operators.d.ts +3 -0
  53. package/dist/tensor/operators.js +24 -0
  54. package/dist/tensor/resize.d.ts +6 -0
  55. package/dist/tensor/resize.js +107 -0
  56. package/dist/tensor/reuse.d.ts +33 -0
  57. package/dist/tensor/reuse.js +59 -0
  58. package/dist/tensor/shapeOperators.d.ts +3 -0
  59. package/dist/tensor/shapeOperators.js +173 -0
  60. package/dist/tensor/spatial.d.ts +34 -0
  61. package/dist/tensor/spatial.js +131 -0
  62. package/dist/tensor/spatialOperators.d.ts +2 -0
  63. package/dist/tensor/spatialOperators.js +138 -0
  64. package/dist/tileHash.d.ts +29 -0
  65. package/dist/tileHash.js +50 -0
  66. package/dist/timeNodes.d.ts +26 -0
  67. package/dist/timeNodes.js +48 -0
  68. package/package.json +59 -0
  69. package/src/decodeCpu.ts +308 -0
  70. package/src/decodeGraph.ts +214 -0
  71. package/src/half.ts +86 -0
  72. package/src/index.ts +175 -0
  73. package/src/inference.ts +278 -0
  74. package/src/materialArray.ts +67 -0
  75. package/src/mipNdf.ts +63 -0
  76. package/src/overlay/journal.ts +218 -0
  77. package/src/overlay/sparse.ts +275 -0
  78. package/src/progressive.ts +60 -0
  79. package/src/residency/pageCache.ts +233 -0
  80. package/src/residency/predict.ts +74 -0
  81. package/src/residency/predictor.ts +62 -0
  82. package/src/residency/queue.ts +78 -0
  83. package/src/residency/stream.ts +194 -0
  84. package/src/residency/table.ts +89 -0
  85. package/src/residency/viewTiles.ts +553 -0
  86. package/src/semantics.ts +114 -0
  87. package/src/tensor/architecture.ts +140 -0
  88. package/src/tensor/attention.ts +75 -0
  89. package/src/tensor/denseOperators.ts +153 -0
  90. package/src/tensor/graph.ts +244 -0
  91. package/src/tensor/linear.ts +153 -0
  92. package/src/tensor/operatorKit.ts +76 -0
  93. package/src/tensor/operators.ts +28 -0
  94. package/src/tensor/resize.ts +140 -0
  95. package/src/tensor/shapeOperators.ts +173 -0
  96. package/src/tensor/spatial.ts +178 -0
  97. package/src/tensor/spatialOperators.ts +182 -0
  98. package/src/tileHash.ts +60 -0
  99. package/src/timeNodes.ts +60 -0
package/src/index.ts ADDED
@@ -0,0 +1,175 @@
1
+ /*! DriftEngine | Copyright 2026 Drift Technologies | Apache-2.0 | https://github.com/drftrun/driftengine */
2
+ /**
3
+ * DriftTexture: a texture as a compiled, sampled field rather than an image.
4
+ *
5
+ * Every engine in the world treats a texture as a bag of texels. This one treats it as a small
6
+ * program sampled at `(uv, t, params)` — one object per *material* rather than per channel,
7
+ * carrying every channel, every level and its own decoder.
8
+ *
9
+ * Three consequences fall out of that shape and none of them is available to an image:
10
+ *
11
+ * - **Channels compress together**, because they are correlated and compressing them apart throws
12
+ * that correlation away.
13
+ * - **Time is an argument, not a clock**, so an animated texture sampled at simulation time is
14
+ * byte-exact under replay — which no other engine's animated texture can claim.
15
+ * - **The decoder is data walked by one shader**, never a permutation. `ARCHITECTURE.md` measured
16
+ * what the other way costs at 196,910 gzipped bytes for a single flag.
17
+ *
18
+ * What ships here is the format, the reference decode, the material binding, predictive
19
+ * residency and the writable overlay. The device interpreter the GPU-driven pipeline samples with
20
+ * is `@driftengine/core`'s, checked against `decodeCpu` by `scripts/gpu-parity.mjs`.
21
+ */
22
+ export {
23
+ CHANNEL_SEMANTICS,
24
+ isColour,
25
+ linearToSrgb,
26
+ needsVarianceMips,
27
+ normaliseSample,
28
+ semanticAt,
29
+ semanticIndex,
30
+ srgbToLinear,
31
+ } from './semantics.ts';
32
+ export type { ChannelSemantic, ChannelSpec } from './semantics.ts';
33
+ export { reduceNormalMip, toksvigRoughness } from './mipNdf.ts';
34
+ export { createTileIndex, hashTile, internTile, tileSlotCount } from './tileHash.ts';
35
+ export type { TileIndex } from './tileHash.ts';
36
+ export {
37
+ activationBound,
38
+ evalNetwork,
39
+ evalNetworkHalf,
40
+ halfPrecisionErrorBound,
41
+ networkWeightCount,
42
+ } from './inference.ts';
43
+ export { HALF_MAX, fromHalfBits, halfWeights, roundHalf, toHalfBits } from './half.ts';
44
+ /*
45
+ * The operators a small transformer is built from, as references the device kernels are held to:
46
+ * the second half of the one inference runtime, beside the perceptron's evaluation above.
47
+ */
48
+ export { addBias, erf, gelu, layerNorm, matmul, softmax } from './tensor/linear.ts';
49
+ export { attention } from './tensor/attention.ts';
50
+ export { resize } from './tensor/resize.ts';
51
+ export { conv2d, convTranspose2d, maxPool2d, patchEmbed } from './tensor/spatial.ts';
52
+ export {
53
+ createGraphEvaluator,
54
+ graphForDevice,
55
+ graphFromStored,
56
+ graphShapes,
57
+ validateGraph,
58
+ } from './tensor/graph.ts';
59
+ export { CONSTANT_PREFIX, graphFromWeights } from './tensor/architecture.ts';
60
+ export type { Architecture, GraphBuilder, WeightSource, Weights } from './tensor/architecture.ts';
61
+ export type {
62
+ GraphEvaluator,
63
+ GraphNode,
64
+ GraphTensor,
65
+ GraphValue,
66
+ NetworkGraph,
67
+ StoredGraph,
68
+ } from './tensor/graph.ts';
69
+ export { OPERATORS } from './tensor/operators.ts';
70
+ export type { AttributeValue, Attributes, Operator } from './tensor/operators.ts';
71
+ export type { NetworkShape } from './inference.ts';
72
+ export { flipbookFrame, latentLerpWeights } from './timeNodes.ts';
73
+ export type { LerpWeights } from './timeNodes.ts';
74
+ export {
75
+ ADDRESS_MODE,
76
+ ADDRESS_MODE_COUNT,
77
+ DECODE_OP,
78
+ MAX_REGISTERS,
79
+ NODE_STRIDE,
80
+ addDecodeNode,
81
+ createDecodeGraph,
82
+ decodeDecodeGraph,
83
+ encodeDecodeGraph,
84
+ graphRegisterCount,
85
+ nodeA,
86
+ nodeB,
87
+ nodeOp,
88
+ nodeOut,
89
+ validateDecodeGraph,
90
+ } from './decodeGraph.ts';
91
+ export type { DecodeGraph, DecodeOp } from './decodeGraph.ts';
92
+ export { REMAP_SEMANTICS, createDecodeRegisters, decodeCpu } from './decodeCpu.ts';
93
+ export type { DecodeResources, LatentImage, LatentLevel } from './decodeCpu.ts';
94
+ export { progressiveOrder, usableAt } from './progressive.ts';
95
+ export { arrayDescriptor, assignLayer, createMaterialArray, layerOf } from './materialArray.ts';
96
+ export type { MaterialArray } from './materialArray.ts';
97
+
98
+ /*
99
+ * Predictive residency: the half that needs the simulation's save and restore rather than the
100
+ * renderer. See `residency/predict.ts` for why looking ahead is available here and nowhere else.
101
+ */
102
+ export { predictViews } from './residency/predict.ts';
103
+ export type { SimulationHandle } from './residency/predict.ts';
104
+ export {
105
+ TILE_ABSENT,
106
+ TILE_REQUESTED,
107
+ TILE_RESIDENT,
108
+ createResidencyTable,
109
+ evict,
110
+ leastRecentlyUsed,
111
+ markRequested,
112
+ markResident,
113
+ residentCount,
114
+ slotFor,
115
+ tileState,
116
+ touchTile,
117
+ } from './residency/table.ts';
118
+ export type { ResidencyTable } from './residency/table.ts';
119
+ export { createPrefetchQueue, enqueue, queueSize, takeBatch } from './residency/queue.ts';
120
+ export type { PrefetchQueue } from './residency/queue.ts';
121
+ export {
122
+ acquirePage,
123
+ beginCacheFrame,
124
+ createPageCache,
125
+ decodeMode,
126
+ ensurePageDecoded,
127
+ occupiedPages,
128
+ pageDecoded,
129
+ pageSlot,
130
+ releasePage,
131
+ setDecodeMode,
132
+ takeEvicted,
133
+ } from './residency/pageCache.ts';
134
+ export type { DecodeMode, PageCache, PageCacheOptions, PageDecode } from './residency/pageCache.ts';
135
+ export {
136
+ clearMissing,
137
+ createStreamer,
138
+ forgetMissing,
139
+ pumpStreamer,
140
+ streamerInFlight,
141
+ } from './residency/stream.ts';
142
+ export type { Streamer, StreamerOptions, TileSource } from './residency/stream.ts';
143
+ export {
144
+ ADDRESS_CLAMP,
145
+ ADDRESS_WRAP,
146
+ OVERLAY_CHANNELS,
147
+ compositeOverlay,
148
+ createOverlay,
149
+ overlayTileCount,
150
+ overlayTiles,
151
+ sampleOverlay,
152
+ writeOverlay,
153
+ writtenMaskAt,
154
+ } from './overlay/sparse.ts';
155
+ export type { Overlay, OverlayOptions, OverlayTile } from './overlay/sparse.ts';
156
+ export {
157
+ JOURNAL_ENTRY_BYTES,
158
+ JOURNAL_MAGIC,
159
+ JOURNAL_VERSION,
160
+ applyOverlayJournal,
161
+ clearOverlay,
162
+ createOverlayJournal,
163
+ decodeOverlayJournal,
164
+ emptyLike,
165
+ encodeOverlayJournal,
166
+ journalLength,
167
+ recordOverlayWrite,
168
+ rewindOverlay,
169
+ truncateJournalFrom,
170
+ } from './overlay/journal.ts';
171
+ export type { OverlayJournal } from './overlay/journal.ts';
172
+ export { runPrediction } from './residency/predictor.ts';
173
+ export { LEVEL_MARGIN, latentTileGrid, tilesForView } from './residency/viewTiles.ts';
174
+ export type { InstanceTileInfo, MaterialTileGrid } from './residency/viewTiles.ts';
175
+ export type { Predictor } from './residency/predictor.ts';
@@ -0,0 +1,278 @@
1
+ /**
2
+ * A small network evaluator, and the only one this engine has.
3
+ *
4
+ * **Three things use it**: a DriftTexture's decode, reconstruction's refinement tier, and capture's
5
+ * reconstruction. That is the argument for building it carefully and once — a second inference path
6
+ * is a second set of numerical conventions to keep in step, and the first symptom of them drifting
7
+ * is a picture that is slightly wrong everywhere.
8
+ *
9
+ * **The constraint that shapes it: WebGPU exposes no hardware matrix units.** The desktop graphics
10
+ * interfaces reach them through cooperative vectors and the published neural-texture work depends
11
+ * on that. So the network here has to be small enough that ordinary vector arithmetic suffices,
12
+ * and its size is decided by measurement against a frame budget rather than by copying an
13
+ * architecture that assumed hardware this platform does not have.
14
+ *
15
+ * **The activation applies to hidden layers and not to the output.** Applying it to the output
16
+ * clamps every channel into the activation's range, which produces a washed-out picture that a
17
+ * training loss will not reveal — the network simply learns around it and infers badly.
18
+ */
19
+ import { fromHalfBits, roundHalf } from './half.ts';
20
+
21
+ export interface NetworkShape {
22
+ readonly inputs: number;
23
+ readonly hidden: readonly number[];
24
+ readonly outputs: number;
25
+ }
26
+
27
+ /** How many weights and biases a shape needs, laid out layer by layer. */
28
+ export function networkWeightCount(shape: NetworkShape): number {
29
+ let total = 0;
30
+ let previous = shape.inputs;
31
+ for (const width of shape.hidden) {
32
+ total += previous * width + width;
33
+ previous = width;
34
+ }
35
+ return total + previous * shape.outputs + shape.outputs;
36
+ }
37
+
38
+ /** Rectified linear, which is what the generated shader will use too. */
39
+ function activate(value: number): number {
40
+ return value > 0 ? value : 0;
41
+ }
42
+
43
+ /**
44
+ * Evaluate the network, writing `shape.outputs` values into `out`.
45
+ *
46
+ * Weights are read in layer order: for each layer, the weight matrix row-major by output, then the
47
+ * biases. `scratch` must hold at least twice the widest layer and is reused across calls so this
48
+ * allocates nothing.
49
+ */
50
+ export function evalNetwork(
51
+ shape: NetworkShape,
52
+ weights: Float32Array,
53
+ input: Float32Array,
54
+ out: Float32Array,
55
+ scratch: Float32Array,
56
+ ): void {
57
+ const widest = Math.max(shape.inputs, shape.outputs, ...shape.hidden, 1);
58
+ let current = scratch.subarray(0, widest);
59
+ let next = scratch.subarray(widest, widest * 2);
60
+ for (let i = 0; i < shape.inputs; i += 1) current[i] = input[i] as number;
61
+
62
+ let at = 0;
63
+ let previous = shape.inputs;
64
+ for (const width of shape.hidden) {
65
+ for (let o = 0; o < width; o += 1) {
66
+ let sum = 0;
67
+ for (let i = 0; i < previous; i += 1) {
68
+ sum += (current[i] as number) * (weights[at + o * previous + i] as number);
69
+ }
70
+ next[o] = activate(sum + (weights[at + width * previous + o] as number));
71
+ }
72
+ at += previous * width + width;
73
+ previous = width;
74
+ const swap = current;
75
+ current = next;
76
+ next = swap;
77
+ }
78
+
79
+ for (let o = 0; o < shape.outputs; o += 1) {
80
+ let sum = 0;
81
+ for (let i = 0; i < previous; i += 1) {
82
+ sum += (current[i] as number) * (weights[at + o * previous + i] as number);
83
+ }
84
+ /* No activation here. See the header. */
85
+ out[o] = sum + (weights[at + shape.outputs * previous + o] as number);
86
+ }
87
+ }
88
+
89
+ /*
90
+ * The most one rounding can move a value of this magnitude: half a unit in the last place, which is
91
+ * 2^-11 of the value for a normal half-precision number and 2^-25 below them. Single precision's
92
+ * own rounding, which the reference and the device do too, is the same argument at 2^-24.
93
+ */
94
+ function halfRounding(magnitude: number): number {
95
+ return Math.max(magnitude * 2 ** -11, 2 ** -25);
96
+ }
97
+
98
+ function singleRounding(magnitude: number): number {
99
+ return Math.max(magnitude * 2 ** -24, 2 ** -150);
100
+ }
101
+
102
+ /**
103
+ * How far the half-precision evaluation of this network, at this input, can sit from the
104
+ * single-precision one.
105
+ *
106
+ * **A bound per network rather than one tolerance for all**, because a sampled tolerance is a claim
107
+ * about the corpus that sampled it — 2,400 networks put the worst relative error at 0.00213, the
108
+ * device's own corpus reached 0.0039, and 40,000 reached 0.0066. This carries the error instead:
109
+ * the weights' and the input's own rounding exactly, then half a unit in the last place for every
110
+ * product, every partial sum and the bias, through each layer — a rectifier moves no error, since
111
+ * it is one-Lipschitz. **It holds whichever form a device computes**: every operation rounded, or
112
+ * each multiply-add contracted into a single rounding, which WGSL permits and this project's
113
+ * development machine does. It assumes no value overflows, which `activationBound` is for.
114
+ */
115
+ export function halfPrecisionErrorBound(
116
+ shape: NetworkShape,
117
+ weights: Float32Array,
118
+ input: Float32Array,
119
+ ): number {
120
+ let value = Array.from({ length: shape.inputs }, (_, i) => input[i] as number);
121
+ let error = value.map((x) => Math.abs(roundHalf(x) - x) + singleRounding(Math.abs(x)));
122
+ let worst = 0;
123
+ let at = 0;
124
+ let previous = shape.inputs;
125
+ const layers = shape.hidden.length + 1;
126
+ for (let layer = 0; layer < layers; layer += 1) {
127
+ const width = layer < shape.hidden.length ? (shape.hidden[layer] as number) : shape.outputs;
128
+ const hidden = layer < shape.hidden.length;
129
+ const nextValue: number[] = [];
130
+ const nextError: number[] = [];
131
+ for (let o = 0; o < width; o += 1) {
132
+ let exact = 0;
133
+ let carried = 0;
134
+ let magnitude = 0;
135
+ let rounding = 0;
136
+ for (let i = 0; i < previous; i += 1) {
137
+ const weight = weights[at + o * previous + i] as number;
138
+ const drift = Math.abs(roundHalf(weight) - weight);
139
+ const x = value[i] as number;
140
+ const e = error[i] as number;
141
+ exact += x * weight;
142
+ /* |xh·wh − x·w| ≤ |xh − x|·|wh| + |x|·|wh − w|. */
143
+ carried += e * (Math.abs(weight) + drift) + Math.abs(x) * drift;
144
+ const product = (Math.abs(x) + e) * (Math.abs(weight) + drift);
145
+ magnitude += product + rounding;
146
+ rounding +=
147
+ halfRounding(product) +
148
+ halfRounding(magnitude) +
149
+ singleRounding(product) +
150
+ singleRounding(magnitude);
151
+ }
152
+ const bias = weights[at + width * previous + o] as number;
153
+ const biasDrift = Math.abs(roundHalf(bias) - bias);
154
+ exact += bias;
155
+ carried += biasDrift;
156
+ magnitude += Math.abs(bias) + biasDrift + rounding;
157
+ rounding += halfRounding(magnitude) + singleRounding(magnitude);
158
+ const total = carried + rounding;
159
+ if (hidden) {
160
+ nextValue.push(Math.max(0, exact));
161
+ nextError.push(total);
162
+ } else {
163
+ worst = Math.max(worst, total);
164
+ }
165
+ }
166
+ at += previous * width + width;
167
+ previous = width;
168
+ value = nextValue;
169
+ error = nextError;
170
+ }
171
+ return worst;
172
+ }
173
+
174
+ /**
175
+ * Evaluate the network as a device with `enable f16` would, writing `shape.outputs` values into
176
+ * `out`.
177
+ *
178
+ * **Every multiply and every add is rounded**, because that is what a sixteen-bit adder does, and
179
+ * the arithmetic is in doubles so the rounding is the only approximation. The order is the
180
+ * device's: the weighted inputs summed in index order, then the bias, then the rectifier —
181
+ * `render/shaders/network.wgsl.ts` in `@driftengine/core` is written to match it.
182
+ *
183
+ * `weights` are half-precision bits, as `halfWeights` produces and an `NNET` chunk stores. Inputs
184
+ * are rounded to half precision on the way in. `scratch` holds at least twice the widest layer.
185
+ */
186
+ export function evalNetworkHalf(
187
+ shape: NetworkShape,
188
+ weights: Uint16Array,
189
+ input: Float32Array,
190
+ out: Float32Array,
191
+ scratch: Float64Array,
192
+ contract = false,
193
+ ): void {
194
+ const widest = Math.max(shape.inputs, shape.outputs, ...shape.hidden, 1);
195
+ let current = scratch.subarray(0, widest);
196
+ let next = scratch.subarray(widest, widest * 2);
197
+ for (let i = 0; i < shape.inputs; i += 1) current[i] = roundHalf(input[i] as number);
198
+
199
+ let at = 0;
200
+ let previous = shape.inputs;
201
+ const layers = shape.hidden.length + 1;
202
+ for (let layer = 0; layer < layers; layer += 1) {
203
+ const width = layer < shape.hidden.length ? (shape.hidden[layer] as number) : shape.outputs;
204
+ const hidden = layer < shape.hidden.length;
205
+ for (let o = 0; o < width; o += 1) {
206
+ let sum = 0;
207
+ for (let i = 0; i < previous; i += 1) {
208
+ const product =
209
+ (current[i] as number) * fromHalfBits(weights[at + o * previous + i] as number);
210
+ sum = contract ? roundHalf(sum + product) : roundHalf(sum + roundHalf(product));
211
+ }
212
+ sum = roundHalf(sum + fromHalfBits(weights[at + width * previous + o] as number));
213
+ const value = hidden ? (sum > 0 ? sum : 0) : sum;
214
+ if (hidden) next[o] = value;
215
+ else out[o] = value;
216
+ }
217
+ at += previous * width + width;
218
+ previous = width;
219
+ const swap = current;
220
+ current = next;
221
+ next = swap;
222
+ }
223
+ }
224
+
225
+ /**
226
+ * The largest magnitude any product, partial sum or output of the network can reach over the given
227
+ * input box.
228
+ *
229
+ * **What a consumer asks before it chooses half precision.** Past 65,504 the format has no finite
230
+ * value, and a partial sum can pass that on its way to a total that does not — so this bounds each
231
+ * neuron by its bias plus the sum of every weight times the largest magnitude its input can take,
232
+ * which covers every prefix of the sum at once. Intervals are carried through the layers, and a
233
+ * rectifier clips them, which is what keeps the bound from growing without cause.
234
+ */
235
+ export function activationBound(
236
+ shape: NetworkShape,
237
+ weights: Float32Array,
238
+ low: Float32Array,
239
+ high: Float32Array,
240
+ ): number {
241
+ let lower = Array.from({ length: shape.inputs }, (_, i) => low[i] as number);
242
+ let upper = Array.from({ length: shape.inputs }, (_, i) => high[i] as number);
243
+ let bound = 0;
244
+ for (let i = 0; i < shape.inputs; i += 1) {
245
+ bound = Math.max(bound, Math.abs(lower[i] as number), Math.abs(upper[i] as number));
246
+ }
247
+ let at = 0;
248
+ let previous = shape.inputs;
249
+ const layers = shape.hidden.length + 1;
250
+ for (let layer = 0; layer < layers; layer += 1) {
251
+ const width = layer < shape.hidden.length ? (shape.hidden[layer] as number) : shape.outputs;
252
+ const hidden = layer < shape.hidden.length;
253
+ const nextLower: number[] = [];
254
+ const nextUpper: number[] = [];
255
+ for (let o = 0; o < width; o += 1) {
256
+ const bias = weights[at + width * previous + o] as number;
257
+ let magnitude = Math.abs(bias);
258
+ let min = bias;
259
+ let max = bias;
260
+ for (let i = 0; i < previous; i += 1) {
261
+ const weight = weights[at + o * previous + i] as number;
262
+ const a = weight * (lower[i] as number);
263
+ const b = weight * (upper[i] as number);
264
+ min += Math.min(a, b);
265
+ max += Math.max(a, b);
266
+ magnitude += Math.max(Math.abs(a), Math.abs(b));
267
+ }
268
+ bound = Math.max(bound, magnitude);
269
+ nextLower.push(hidden ? Math.max(0, min) : min);
270
+ nextUpper.push(hidden ? Math.max(0, max) : max);
271
+ }
272
+ at += previous * width + width;
273
+ previous = width;
274
+ lower = nextLower;
275
+ upper = nextUpper;
276
+ }
277
+ return bound;
278
+ }
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Materials as slices of a few texture arrays, which is what WebGPU can bind.
3
+ *
4
+ * **This is the binding design, and it is why this package lands in the same wave as the
5
+ * GPU-driven pipeline rather than later.** WebGPU has no bindless resource access, which is the
6
+ * usual way a GPU-driven renderer reaches a thousand materials from one indirect draw. A thousand
7
+ * materials as six thousand independent textures is unreachable. A thousand materials as a
8
+ * thousand latent slices in a handful of arrays, indexed by a material identifier read from a
9
+ * storage buffer, is ordinary.
10
+ *
11
+ * **Identical content shares a layer.** The tile index already hashes by content, so two materials
12
+ * whose latents are the same occupy one layer — which is the deduplication paying for itself a
13
+ * second time, in binding slots rather than in bytes.
14
+ */
15
+ export interface MaterialArray {
16
+ /** Layer per material identifier. */
17
+ layerOf: Map<number, number>;
18
+ /** Which content hash each layer holds, so identical latents share one. */
19
+ layerByHash: Map<string, number>;
20
+ layerCapacity: number;
21
+ tileSize: number;
22
+ used: number;
23
+ }
24
+
25
+ export function createMaterialArray(layerCapacity: number, tileSize: number): MaterialArray {
26
+ return {
27
+ layerOf: new Map(),
28
+ layerByHash: new Map(),
29
+ layerCapacity,
30
+ tileSize,
31
+ used: 0,
32
+ };
33
+ }
34
+
35
+ /**
36
+ * Give this material a layer. Returns it, or -1 when the array is full.
37
+ *
38
+ * **Full reports failure rather than evicting.** An eviction here would silently repoint a
39
+ * material that is still being drawn, which is a texture swap nobody asked for in the middle of a
40
+ * frame; a caller that runs out needs to know.
41
+ */
42
+ export function assignLayer(array: MaterialArray, materialId: number, contentHash: string): number {
43
+ const existing = array.layerOf.get(materialId);
44
+ if (existing !== undefined) return existing;
45
+
46
+ const shared = array.layerByHash.get(contentHash);
47
+ if (shared !== undefined) {
48
+ array.layerOf.set(materialId, shared);
49
+ return shared;
50
+ }
51
+
52
+ if (array.used >= array.layerCapacity) return -1;
53
+ const layer = array.used;
54
+ array.used += 1;
55
+ array.layerByHash.set(contentHash, layer);
56
+ array.layerOf.set(materialId, layer);
57
+ return layer;
58
+ }
59
+
60
+ export function layerOf(array: MaterialArray, materialId: number): number {
61
+ return array.layerOf.get(materialId) ?? -1;
62
+ }
63
+
64
+ /** What a bind group needs to know: how many layers and how large each is. */
65
+ export function arrayDescriptor(array: MaterialArray): { layers: number; size: number } {
66
+ return { layers: array.used, size: array.tileSize };
67
+ }
package/src/mipNdf.ts ADDED
@@ -0,0 +1,63 @@
1
+ /**
2
+ * Mip levels for a normal map that get rougher instead of sparkling.
3
+ *
4
+ * **Averaging four normals and renormalising throws away how much they disagreed — and that
5
+ * disagreement *is* roughness at the smaller scale.** What comes back is a distant surface whose
6
+ * highlight flickers as the camera moves, which reads as an aliasing problem in the renderer and
7
+ * is a mip chain problem in the asset.
8
+ *
9
+ * Every engine can do this and almost none does it by default, because it needs the mip chain to
10
+ * write roughness as well as normals — which needs the two to live in one object. In a
11
+ * DriftTexture they do, which is why this is the default here rather than an option.
12
+ *
13
+ * **The reduction adds variance and never removes it.** A surface already rough at the finer level
14
+ * cannot become smoother by being viewed from further away.
15
+ */
16
+
17
+ /**
18
+ * How much roughness a shortened average normal implies.
19
+ *
20
+ * The averaged normal's length falls as the normals it averaged disagree, so `1 - length` is a
21
+ * direct measure of the variance lost to averaging. Added in variance rather than in roughness,
22
+ * because variances of independent sources add and roughnesses do not.
23
+ */
24
+ export function toksvigRoughness(normalLength: number, roughness: number): number {
25
+ const length = Math.min(1, Math.max(0, normalLength));
26
+ const added = 1 - length;
27
+ return Math.sqrt(Math.min(1, roughness * roughness + added));
28
+ }
29
+
30
+ /**
31
+ * Reduce a block of normals to one, writing both the direction and the roughness it implies.
32
+ *
33
+ * `src` holds `width * height` normals as xyz triples, already unbiased into -1..1.
34
+ */
35
+ export function reduceNormalMip(
36
+ src: Float32Array,
37
+ width: number,
38
+ height: number,
39
+ outNormal: Float32Array,
40
+ outRoughness: Float32Array,
41
+ baseRoughness = 0,
42
+ ): void {
43
+ let x = 0;
44
+ let y = 0;
45
+ let z = 0;
46
+ const count = width * height;
47
+ for (let i = 0; i < count; i += 1) {
48
+ x += src[i * 3] as number;
49
+ y += src[i * 3 + 1] as number;
50
+ z += src[i * 3 + 2] as number;
51
+ }
52
+ x /= count;
53
+ y /= count;
54
+ z /= count;
55
+
56
+ /* The length before normalising is the whole signal. Normalising first discards it. */
57
+ const length = Math.hypot(x, y, z);
58
+ const k = length > 0 ? 1 / length : 0;
59
+ outNormal[0] = x * k;
60
+ outNormal[1] = y * k;
61
+ outNormal[2] = z * k;
62
+ outRoughness[0] = toksvigRoughness(length, baseRoughness);
63
+ }