@driftengine/texture 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +29 -0
- package/README.md +106 -0
- package/dist/decodeCpu.d.ts +59 -0
- package/dist/decodeCpu.js +234 -0
- package/dist/decodeGraph.d.ts +105 -0
- package/dist/decodeGraph.js +180 -0
- package/dist/half.d.ts +24 -0
- package/dist/half.js +86 -0
- package/dist/index.d.ts +66 -0
- package/dist/index.js +55 -0
- package/dist/inference.d.ts +53 -0
- package/dist/inference.js +243 -0
- package/dist/materialArray.d.ts +38 -0
- package/dist/materialArray.js +40 -0
- package/dist/mipNdf.d.ts +29 -0
- package/dist/mipNdf.js +53 -0
- package/dist/overlay/journal.d.ts +78 -0
- package/dist/overlay/journal.js +171 -0
- package/dist/overlay/sparse.d.ts +68 -0
- package/dist/overlay/sparse.js +212 -0
- package/dist/progressive.d.ts +30 -0
- package/dist/progressive.js +56 -0
- package/dist/residency/pageCache.d.ts +103 -0
- package/dist/residency/pageCache.js +184 -0
- package/dist/residency/predict.d.ts +55 -0
- package/dist/residency/predict.js +51 -0
- package/dist/residency/predictor.d.ts +16 -0
- package/dist/residency/predictor.js +44 -0
- package/dist/residency/queue.d.ts +26 -0
- package/dist/residency/queue.js +52 -0
- package/dist/residency/stream.d.ts +66 -0
- package/dist/residency/stream.js +142 -0
- package/dist/residency/table.d.ts +36 -0
- package/dist/residency/table.js +72 -0
- package/dist/residency/viewTiles.d.ts +108 -0
- package/dist/residency/viewTiles.js +419 -0
- package/dist/semantics.d.ts +52 -0
- package/dist/semantics.js +76 -0
- package/dist/tensor/architecture.d.ts +53 -0
- package/dist/tensor/architecture.js +96 -0
- package/dist/tensor/attention.d.ts +5 -0
- package/dist/tensor/attention.js +62 -0
- package/dist/tensor/denseOperators.d.ts +2 -0
- package/dist/tensor/denseOperators.js +136 -0
- package/dist/tensor/graph.d.ts +83 -0
- package/dist/tensor/graph.js +175 -0
- package/dist/tensor/linear.d.ts +49 -0
- package/dist/tensor/linear.js +136 -0
- package/dist/tensor/operatorKit.d.ts +27 -0
- package/dist/tensor/operatorKit.js +45 -0
- package/dist/tensor/operators.d.ts +3 -0
- package/dist/tensor/operators.js +24 -0
- package/dist/tensor/resize.d.ts +6 -0
- package/dist/tensor/resize.js +107 -0
- package/dist/tensor/reuse.d.ts +33 -0
- package/dist/tensor/reuse.js +59 -0
- package/dist/tensor/shapeOperators.d.ts +3 -0
- package/dist/tensor/shapeOperators.js +173 -0
- package/dist/tensor/spatial.d.ts +34 -0
- package/dist/tensor/spatial.js +131 -0
- package/dist/tensor/spatialOperators.d.ts +2 -0
- package/dist/tensor/spatialOperators.js +138 -0
- package/dist/tileHash.d.ts +29 -0
- package/dist/tileHash.js +50 -0
- package/dist/timeNodes.d.ts +26 -0
- package/dist/timeNodes.js +48 -0
- package/package.json +59 -0
- package/src/decodeCpu.ts +308 -0
- package/src/decodeGraph.ts +214 -0
- package/src/half.ts +86 -0
- package/src/index.ts +175 -0
- package/src/inference.ts +278 -0
- package/src/materialArray.ts +67 -0
- package/src/mipNdf.ts +63 -0
- package/src/overlay/journal.ts +218 -0
- package/src/overlay/sparse.ts +275 -0
- package/src/progressive.ts +60 -0
- package/src/residency/pageCache.ts +233 -0
- package/src/residency/predict.ts +74 -0
- package/src/residency/predictor.ts +62 -0
- package/src/residency/queue.ts +78 -0
- package/src/residency/stream.ts +194 -0
- package/src/residency/table.ts +89 -0
- package/src/residency/viewTiles.ts +553 -0
- package/src/semantics.ts +114 -0
- package/src/tensor/architecture.ts +140 -0
- package/src/tensor/attention.ts +75 -0
- package/src/tensor/denseOperators.ts +153 -0
- package/src/tensor/graph.ts +244 -0
- package/src/tensor/linear.ts +153 -0
- package/src/tensor/operatorKit.ts +76 -0
- package/src/tensor/operators.ts +28 -0
- package/src/tensor/resize.ts +140 -0
- package/src/tensor/shapeOperators.ts +173 -0
- package/src/tensor/spatial.ts +178 -0
- package/src/tensor/spatialOperators.ts +182 -0
- package/src/tileHash.ts +60 -0
- package/src/timeNodes.ts +60 -0
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A small network evaluator, and the only one this engine has.
|
|
3
|
+
*
|
|
4
|
+
* **Three things use it**: a DriftTexture's decode, reconstruction's refinement tier, and capture's
|
|
5
|
+
* reconstruction. That is the argument for building it carefully and once — a second inference path
|
|
6
|
+
* is a second set of numerical conventions to keep in step, and the first symptom of them drifting
|
|
7
|
+
* is a picture that is slightly wrong everywhere.
|
|
8
|
+
*
|
|
9
|
+
* **The constraint that shapes it: WebGPU exposes no hardware matrix units.** The desktop graphics
|
|
10
|
+
* interfaces reach them through cooperative vectors and the published neural-texture work depends
|
|
11
|
+
* on that. So the network here has to be small enough that ordinary vector arithmetic suffices,
|
|
12
|
+
* and its size is decided by measurement against a frame budget rather than by copying an
|
|
13
|
+
* architecture that assumed hardware this platform does not have.
|
|
14
|
+
*
|
|
15
|
+
* **The activation applies to hidden layers and not to the output.** Applying it to the output
|
|
16
|
+
* clamps every channel into the activation's range, which produces a washed-out picture that a
|
|
17
|
+
* training loss will not reveal — the network simply learns around it and infers badly.
|
|
18
|
+
*/
|
|
19
|
+
import { fromHalfBits, roundHalf } from './half.js';
|
|
20
|
+
/** How many weights and biases a shape needs, laid out layer by layer. */
|
|
21
|
+
export function networkWeightCount(shape) {
|
|
22
|
+
let total = 0;
|
|
23
|
+
let previous = shape.inputs;
|
|
24
|
+
for (const width of shape.hidden) {
|
|
25
|
+
total += previous * width + width;
|
|
26
|
+
previous = width;
|
|
27
|
+
}
|
|
28
|
+
return total + previous * shape.outputs + shape.outputs;
|
|
29
|
+
}
|
|
30
|
+
/** Rectified linear, which is what the generated shader will use too. */
|
|
31
|
+
function activate(value) {
|
|
32
|
+
return value > 0 ? value : 0;
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* Evaluate the network, writing `shape.outputs` values into `out`.
|
|
36
|
+
*
|
|
37
|
+
* Weights are read in layer order: for each layer, the weight matrix row-major by output, then the
|
|
38
|
+
* biases. `scratch` must hold at least twice the widest layer and is reused across calls so this
|
|
39
|
+
* allocates nothing.
|
|
40
|
+
*/
|
|
41
|
+
export function evalNetwork(shape, weights, input, out, scratch) {
|
|
42
|
+
const widest = Math.max(shape.inputs, shape.outputs, ...shape.hidden, 1);
|
|
43
|
+
let current = scratch.subarray(0, widest);
|
|
44
|
+
let next = scratch.subarray(widest, widest * 2);
|
|
45
|
+
for (let i = 0; i < shape.inputs; i += 1)
|
|
46
|
+
current[i] = input[i];
|
|
47
|
+
let at = 0;
|
|
48
|
+
let previous = shape.inputs;
|
|
49
|
+
for (const width of shape.hidden) {
|
|
50
|
+
for (let o = 0; o < width; o += 1) {
|
|
51
|
+
let sum = 0;
|
|
52
|
+
for (let i = 0; i < previous; i += 1) {
|
|
53
|
+
sum += current[i] * weights[at + o * previous + i];
|
|
54
|
+
}
|
|
55
|
+
next[o] = activate(sum + weights[at + width * previous + o]);
|
|
56
|
+
}
|
|
57
|
+
at += previous * width + width;
|
|
58
|
+
previous = width;
|
|
59
|
+
const swap = current;
|
|
60
|
+
current = next;
|
|
61
|
+
next = swap;
|
|
62
|
+
}
|
|
63
|
+
for (let o = 0; o < shape.outputs; o += 1) {
|
|
64
|
+
let sum = 0;
|
|
65
|
+
for (let i = 0; i < previous; i += 1) {
|
|
66
|
+
sum += current[i] * weights[at + o * previous + i];
|
|
67
|
+
}
|
|
68
|
+
/* No activation here. See the header. */
|
|
69
|
+
out[o] = sum + weights[at + shape.outputs * previous + o];
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
/*
|
|
73
|
+
* The most one rounding can move a value of this magnitude: half a unit in the last place, which is
|
|
74
|
+
* 2^-11 of the value for a normal half-precision number and 2^-25 below them. Single precision's
|
|
75
|
+
* own rounding, which the reference and the device do too, is the same argument at 2^-24.
|
|
76
|
+
*/
|
|
77
|
+
function halfRounding(magnitude) {
|
|
78
|
+
return Math.max(magnitude * 2 ** -11, 2 ** -25);
|
|
79
|
+
}
|
|
80
|
+
function singleRounding(magnitude) {
|
|
81
|
+
return Math.max(magnitude * 2 ** -24, 2 ** -150);
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* How far the half-precision evaluation of this network, at this input, can sit from the
|
|
85
|
+
* single-precision one.
|
|
86
|
+
*
|
|
87
|
+
* **A bound per network rather than one tolerance for all**, because a sampled tolerance is a claim
|
|
88
|
+
* about the corpus that sampled it — 2,400 networks put the worst relative error at 0.00213, the
|
|
89
|
+
* device's own corpus reached 0.0039, and 40,000 reached 0.0066. This carries the error instead:
|
|
90
|
+
* the weights' and the input's own rounding exactly, then half a unit in the last place for every
|
|
91
|
+
* product, every partial sum and the bias, through each layer — a rectifier moves no error, since
|
|
92
|
+
* it is one-Lipschitz. **It holds whichever form a device computes**: every operation rounded, or
|
|
93
|
+
* each multiply-add contracted into a single rounding, which WGSL permits and this project's
|
|
94
|
+
* development machine does. It assumes no value overflows, which `activationBound` is for.
|
|
95
|
+
*/
|
|
96
|
+
export function halfPrecisionErrorBound(shape, weights, input) {
|
|
97
|
+
let value = Array.from({ length: shape.inputs }, (_, i) => input[i]);
|
|
98
|
+
let error = value.map((x) => Math.abs(roundHalf(x) - x) + singleRounding(Math.abs(x)));
|
|
99
|
+
let worst = 0;
|
|
100
|
+
let at = 0;
|
|
101
|
+
let previous = shape.inputs;
|
|
102
|
+
const layers = shape.hidden.length + 1;
|
|
103
|
+
for (let layer = 0; layer < layers; layer += 1) {
|
|
104
|
+
const width = layer < shape.hidden.length ? shape.hidden[layer] : shape.outputs;
|
|
105
|
+
const hidden = layer < shape.hidden.length;
|
|
106
|
+
const nextValue = [];
|
|
107
|
+
const nextError = [];
|
|
108
|
+
for (let o = 0; o < width; o += 1) {
|
|
109
|
+
let exact = 0;
|
|
110
|
+
let carried = 0;
|
|
111
|
+
let magnitude = 0;
|
|
112
|
+
let rounding = 0;
|
|
113
|
+
for (let i = 0; i < previous; i += 1) {
|
|
114
|
+
const weight = weights[at + o * previous + i];
|
|
115
|
+
const drift = Math.abs(roundHalf(weight) - weight);
|
|
116
|
+
const x = value[i];
|
|
117
|
+
const e = error[i];
|
|
118
|
+
exact += x * weight;
|
|
119
|
+
/* |xh·wh − x·w| ≤ |xh − x|·|wh| + |x|·|wh − w|. */
|
|
120
|
+
carried += e * (Math.abs(weight) + drift) + Math.abs(x) * drift;
|
|
121
|
+
const product = (Math.abs(x) + e) * (Math.abs(weight) + drift);
|
|
122
|
+
magnitude += product + rounding;
|
|
123
|
+
rounding +=
|
|
124
|
+
halfRounding(product) +
|
|
125
|
+
halfRounding(magnitude) +
|
|
126
|
+
singleRounding(product) +
|
|
127
|
+
singleRounding(magnitude);
|
|
128
|
+
}
|
|
129
|
+
const bias = weights[at + width * previous + o];
|
|
130
|
+
const biasDrift = Math.abs(roundHalf(bias) - bias);
|
|
131
|
+
exact += bias;
|
|
132
|
+
carried += biasDrift;
|
|
133
|
+
magnitude += Math.abs(bias) + biasDrift + rounding;
|
|
134
|
+
rounding += halfRounding(magnitude) + singleRounding(magnitude);
|
|
135
|
+
const total = carried + rounding;
|
|
136
|
+
if (hidden) {
|
|
137
|
+
nextValue.push(Math.max(0, exact));
|
|
138
|
+
nextError.push(total);
|
|
139
|
+
}
|
|
140
|
+
else {
|
|
141
|
+
worst = Math.max(worst, total);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
at += previous * width + width;
|
|
145
|
+
previous = width;
|
|
146
|
+
value = nextValue;
|
|
147
|
+
error = nextError;
|
|
148
|
+
}
|
|
149
|
+
return worst;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Evaluate the network as a device with `enable f16` would, writing `shape.outputs` values into
|
|
153
|
+
* `out`.
|
|
154
|
+
*
|
|
155
|
+
* **Every multiply and every add is rounded**, because that is what a sixteen-bit adder does, and
|
|
156
|
+
* the arithmetic is in doubles so the rounding is the only approximation. The order is the
|
|
157
|
+
* device's: the weighted inputs summed in index order, then the bias, then the rectifier —
|
|
158
|
+
* `render/shaders/network.wgsl.ts` in `@driftengine/core` is written to match it.
|
|
159
|
+
*
|
|
160
|
+
* `weights` are half-precision bits, as `halfWeights` produces and an `NNET` chunk stores. Inputs
|
|
161
|
+
* are rounded to half precision on the way in. `scratch` holds at least twice the widest layer.
|
|
162
|
+
*/
|
|
163
|
+
export function evalNetworkHalf(shape, weights, input, out, scratch, contract = false) {
|
|
164
|
+
const widest = Math.max(shape.inputs, shape.outputs, ...shape.hidden, 1);
|
|
165
|
+
let current = scratch.subarray(0, widest);
|
|
166
|
+
let next = scratch.subarray(widest, widest * 2);
|
|
167
|
+
for (let i = 0; i < shape.inputs; i += 1)
|
|
168
|
+
current[i] = roundHalf(input[i]);
|
|
169
|
+
let at = 0;
|
|
170
|
+
let previous = shape.inputs;
|
|
171
|
+
const layers = shape.hidden.length + 1;
|
|
172
|
+
for (let layer = 0; layer < layers; layer += 1) {
|
|
173
|
+
const width = layer < shape.hidden.length ? shape.hidden[layer] : shape.outputs;
|
|
174
|
+
const hidden = layer < shape.hidden.length;
|
|
175
|
+
for (let o = 0; o < width; o += 1) {
|
|
176
|
+
let sum = 0;
|
|
177
|
+
for (let i = 0; i < previous; i += 1) {
|
|
178
|
+
const product = current[i] * fromHalfBits(weights[at + o * previous + i]);
|
|
179
|
+
sum = contract ? roundHalf(sum + product) : roundHalf(sum + roundHalf(product));
|
|
180
|
+
}
|
|
181
|
+
sum = roundHalf(sum + fromHalfBits(weights[at + width * previous + o]));
|
|
182
|
+
const value = hidden ? (sum > 0 ? sum : 0) : sum;
|
|
183
|
+
if (hidden)
|
|
184
|
+
next[o] = value;
|
|
185
|
+
else
|
|
186
|
+
out[o] = value;
|
|
187
|
+
}
|
|
188
|
+
at += previous * width + width;
|
|
189
|
+
previous = width;
|
|
190
|
+
const swap = current;
|
|
191
|
+
current = next;
|
|
192
|
+
next = swap;
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
/**
|
|
196
|
+
* The largest magnitude any product, partial sum or output of the network can reach over the given
|
|
197
|
+
* input box.
|
|
198
|
+
*
|
|
199
|
+
* **What a consumer asks before it chooses half precision.** Past 65,504 the format has no finite
|
|
200
|
+
* value, and a partial sum can pass that on its way to a total that does not — so this bounds each
|
|
201
|
+
* neuron by its bias plus the sum of every weight times the largest magnitude its input can take,
|
|
202
|
+
* which covers every prefix of the sum at once. Intervals are carried through the layers, and a
|
|
203
|
+
* rectifier clips them, which is what keeps the bound from growing without cause.
|
|
204
|
+
*/
|
|
205
|
+
export function activationBound(shape, weights, low, high) {
|
|
206
|
+
let lower = Array.from({ length: shape.inputs }, (_, i) => low[i]);
|
|
207
|
+
let upper = Array.from({ length: shape.inputs }, (_, i) => high[i]);
|
|
208
|
+
let bound = 0;
|
|
209
|
+
for (let i = 0; i < shape.inputs; i += 1) {
|
|
210
|
+
bound = Math.max(bound, Math.abs(lower[i]), Math.abs(upper[i]));
|
|
211
|
+
}
|
|
212
|
+
let at = 0;
|
|
213
|
+
let previous = shape.inputs;
|
|
214
|
+
const layers = shape.hidden.length + 1;
|
|
215
|
+
for (let layer = 0; layer < layers; layer += 1) {
|
|
216
|
+
const width = layer < shape.hidden.length ? shape.hidden[layer] : shape.outputs;
|
|
217
|
+
const hidden = layer < shape.hidden.length;
|
|
218
|
+
const nextLower = [];
|
|
219
|
+
const nextUpper = [];
|
|
220
|
+
for (let o = 0; o < width; o += 1) {
|
|
221
|
+
const bias = weights[at + width * previous + o];
|
|
222
|
+
let magnitude = Math.abs(bias);
|
|
223
|
+
let min = bias;
|
|
224
|
+
let max = bias;
|
|
225
|
+
for (let i = 0; i < previous; i += 1) {
|
|
226
|
+
const weight = weights[at + o * previous + i];
|
|
227
|
+
const a = weight * lower[i];
|
|
228
|
+
const b = weight * upper[i];
|
|
229
|
+
min += Math.min(a, b);
|
|
230
|
+
max += Math.max(a, b);
|
|
231
|
+
magnitude += Math.max(Math.abs(a), Math.abs(b));
|
|
232
|
+
}
|
|
233
|
+
bound = Math.max(bound, magnitude);
|
|
234
|
+
nextLower.push(hidden ? Math.max(0, min) : min);
|
|
235
|
+
nextUpper.push(hidden ? Math.max(0, max) : max);
|
|
236
|
+
}
|
|
237
|
+
at += previous * width + width;
|
|
238
|
+
previous = width;
|
|
239
|
+
lower = nextLower;
|
|
240
|
+
upper = nextUpper;
|
|
241
|
+
}
|
|
242
|
+
return bound;
|
|
243
|
+
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Materials as slices of a few texture arrays, which is what WebGPU can bind.
|
|
3
|
+
*
|
|
4
|
+
* **This is the binding design, and it is why this package lands in the same wave as the
|
|
5
|
+
* GPU-driven pipeline rather than later.** WebGPU has no bindless resource access, which is the
|
|
6
|
+
* usual way a GPU-driven renderer reaches a thousand materials from one indirect draw. A thousand
|
|
7
|
+
* materials as six thousand independent textures is unreachable. A thousand materials as a
|
|
8
|
+
* thousand latent slices in a handful of arrays, indexed by a material identifier read from a
|
|
9
|
+
* storage buffer, is ordinary.
|
|
10
|
+
*
|
|
11
|
+
* **Identical content shares a layer.** The tile index already hashes by content, so two materials
|
|
12
|
+
* whose latents are the same occupy one layer — which is the deduplication paying for itself a
|
|
13
|
+
* second time, in binding slots rather than in bytes.
|
|
14
|
+
*/
|
|
15
|
+
export interface MaterialArray {
|
|
16
|
+
/** Layer per material identifier. */
|
|
17
|
+
layerOf: Map<number, number>;
|
|
18
|
+
/** Which content hash each layer holds, so identical latents share one. */
|
|
19
|
+
layerByHash: Map<string, number>;
|
|
20
|
+
layerCapacity: number;
|
|
21
|
+
tileSize: number;
|
|
22
|
+
used: number;
|
|
23
|
+
}
|
|
24
|
+
export declare function createMaterialArray(layerCapacity: number, tileSize: number): MaterialArray;
|
|
25
|
+
/**
|
|
26
|
+
* Give this material a layer. Returns it, or -1 when the array is full.
|
|
27
|
+
*
|
|
28
|
+
* **Full reports failure rather than evicting.** An eviction here would silently repoint a
|
|
29
|
+
* material that is still being drawn, which is a texture swap nobody asked for in the middle of a
|
|
30
|
+
* frame; a caller that runs out needs to know.
|
|
31
|
+
*/
|
|
32
|
+
export declare function assignLayer(array: MaterialArray, materialId: number, contentHash: string): number;
|
|
33
|
+
export declare function layerOf(array: MaterialArray, materialId: number): number;
|
|
34
|
+
/** What a bind group needs to know: how many layers and how large each is. */
|
|
35
|
+
export declare function arrayDescriptor(array: MaterialArray): {
|
|
36
|
+
layers: number;
|
|
37
|
+
size: number;
|
|
38
|
+
};
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
export function createMaterialArray(layerCapacity, tileSize) {
|
|
2
|
+
return {
|
|
3
|
+
layerOf: new Map(),
|
|
4
|
+
layerByHash: new Map(),
|
|
5
|
+
layerCapacity,
|
|
6
|
+
tileSize,
|
|
7
|
+
used: 0,
|
|
8
|
+
};
|
|
9
|
+
}
|
|
10
|
+
/**
|
|
11
|
+
* Give this material a layer. Returns it, or -1 when the array is full.
|
|
12
|
+
*
|
|
13
|
+
* **Full reports failure rather than evicting.** An eviction here would silently repoint a
|
|
14
|
+
* material that is still being drawn, which is a texture swap nobody asked for in the middle of a
|
|
15
|
+
* frame; a caller that runs out needs to know.
|
|
16
|
+
*/
|
|
17
|
+
export function assignLayer(array, materialId, contentHash) {
|
|
18
|
+
const existing = array.layerOf.get(materialId);
|
|
19
|
+
if (existing !== undefined)
|
|
20
|
+
return existing;
|
|
21
|
+
const shared = array.layerByHash.get(contentHash);
|
|
22
|
+
if (shared !== undefined) {
|
|
23
|
+
array.layerOf.set(materialId, shared);
|
|
24
|
+
return shared;
|
|
25
|
+
}
|
|
26
|
+
if (array.used >= array.layerCapacity)
|
|
27
|
+
return -1;
|
|
28
|
+
const layer = array.used;
|
|
29
|
+
array.used += 1;
|
|
30
|
+
array.layerByHash.set(contentHash, layer);
|
|
31
|
+
array.layerOf.set(materialId, layer);
|
|
32
|
+
return layer;
|
|
33
|
+
}
|
|
34
|
+
export function layerOf(array, materialId) {
|
|
35
|
+
return array.layerOf.get(materialId) ?? -1;
|
|
36
|
+
}
|
|
37
|
+
/** What a bind group needs to know: how many layers and how large each is. */
|
|
38
|
+
export function arrayDescriptor(array) {
|
|
39
|
+
return { layers: array.used, size: array.tileSize };
|
|
40
|
+
}
|
package/dist/mipNdf.d.ts
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mip levels for a normal map that get rougher instead of sparkling.
|
|
3
|
+
*
|
|
4
|
+
* **Averaging four normals and renormalising throws away how much they disagreed — and that
|
|
5
|
+
* disagreement *is* roughness at the smaller scale.** What comes back is a distant surface whose
|
|
6
|
+
* highlight flickers as the camera moves, which reads as an aliasing problem in the renderer and
|
|
7
|
+
* is a mip chain problem in the asset.
|
|
8
|
+
*
|
|
9
|
+
* Every engine can do this and almost none does it by default, because it needs the mip chain to
|
|
10
|
+
* write roughness as well as normals — which needs the two to live in one object. In a
|
|
11
|
+
* DriftTexture they do, which is why this is the default here rather than an option.
|
|
12
|
+
*
|
|
13
|
+
* **The reduction adds variance and never removes it.** A surface already rough at the finer level
|
|
14
|
+
* cannot become smoother by being viewed from further away.
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* How much roughness a shortened average normal implies.
|
|
18
|
+
*
|
|
19
|
+
* The averaged normal's length falls as the normals it averaged disagree, so `1 - length` is a
|
|
20
|
+
* direct measure of the variance lost to averaging. Added in variance rather than in roughness,
|
|
21
|
+
* because variances of independent sources add and roughnesses do not.
|
|
22
|
+
*/
|
|
23
|
+
export declare function toksvigRoughness(normalLength: number, roughness: number): number;
|
|
24
|
+
/**
|
|
25
|
+
* Reduce a block of normals to one, writing both the direction and the roughness it implies.
|
|
26
|
+
*
|
|
27
|
+
* `src` holds `width * height` normals as xyz triples, already unbiased into -1..1.
|
|
28
|
+
*/
|
|
29
|
+
export declare function reduceNormalMip(src: Float32Array, width: number, height: number, outNormal: Float32Array, outRoughness: Float32Array, baseRoughness?: number): void;
|
package/dist/mipNdf.js
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mip levels for a normal map that get rougher instead of sparkling.
|
|
3
|
+
*
|
|
4
|
+
* **Averaging four normals and renormalising throws away how much they disagreed — and that
|
|
5
|
+
* disagreement *is* roughness at the smaller scale.** What comes back is a distant surface whose
|
|
6
|
+
* highlight flickers as the camera moves, which reads as an aliasing problem in the renderer and
|
|
7
|
+
* is a mip chain problem in the asset.
|
|
8
|
+
*
|
|
9
|
+
* Every engine can do this and almost none does it by default, because it needs the mip chain to
|
|
10
|
+
* write roughness as well as normals — which needs the two to live in one object. In a
|
|
11
|
+
* DriftTexture they do, which is why this is the default here rather than an option.
|
|
12
|
+
*
|
|
13
|
+
* **The reduction adds variance and never removes it.** A surface already rough at the finer level
|
|
14
|
+
* cannot become smoother by being viewed from further away.
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* How much roughness a shortened average normal implies.
|
|
18
|
+
*
|
|
19
|
+
* The averaged normal's length falls as the normals it averaged disagree, so `1 - length` is a
|
|
20
|
+
* direct measure of the variance lost to averaging. Added in variance rather than in roughness,
|
|
21
|
+
* because variances of independent sources add and roughnesses do not.
|
|
22
|
+
*/
|
|
23
|
+
export function toksvigRoughness(normalLength, roughness) {
|
|
24
|
+
const length = Math.min(1, Math.max(0, normalLength));
|
|
25
|
+
const added = 1 - length;
|
|
26
|
+
return Math.sqrt(Math.min(1, roughness * roughness + added));
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* Reduce a block of normals to one, writing both the direction and the roughness it implies.
|
|
30
|
+
*
|
|
31
|
+
* `src` holds `width * height` normals as xyz triples, already unbiased into -1..1.
|
|
32
|
+
*/
|
|
33
|
+
export function reduceNormalMip(src, width, height, outNormal, outRoughness, baseRoughness = 0) {
|
|
34
|
+
let x = 0;
|
|
35
|
+
let y = 0;
|
|
36
|
+
let z = 0;
|
|
37
|
+
const count = width * height;
|
|
38
|
+
for (let i = 0; i < count; i += 1) {
|
|
39
|
+
x += src[i * 3];
|
|
40
|
+
y += src[i * 3 + 1];
|
|
41
|
+
z += src[i * 3 + 2];
|
|
42
|
+
}
|
|
43
|
+
x /= count;
|
|
44
|
+
y /= count;
|
|
45
|
+
z /= count;
|
|
46
|
+
/* The length before normalising is the whole signal. Normalising first discards it. */
|
|
47
|
+
const length = Math.hypot(x, y, z);
|
|
48
|
+
const k = length > 0 ? 1 / length : 0;
|
|
49
|
+
outNormal[0] = x * k;
|
|
50
|
+
outNormal[1] = y * k;
|
|
51
|
+
outNormal[2] = z * k;
|
|
52
|
+
outRoughness[0] = toksvigRoughness(length, baseRoughness);
|
|
53
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A scorch mark is in the input log, so a replay burns the same wall.
|
|
3
|
+
*
|
|
4
|
+
* **Runtime-mutable textures normally break replay.** The marks are not part of the simulation and
|
|
5
|
+
* nothing records them, so a recording plays back a world where the walls are clean — and the
|
|
6
|
+
* failure has the worst possible shape: nothing goes wrong at the time, everything looks right, and
|
|
7
|
+
* the recording is wrong forever. Anybody who later debugs from it is debugging a session that did
|
|
8
|
+
* not happen.
|
|
9
|
+
*
|
|
10
|
+
* Here a write is recorded in the same log the input goes into, so replaying a session reproduces
|
|
11
|
+
* the exact mark in the exact place — and the same mechanism means a rollback un-draws what the
|
|
12
|
+
* rolled-back frames drew, because rolling back an overlay *is* rebuilding it from the journal up
|
|
13
|
+
* to the frame you rolled back to. An overlay cannot be snapshotted the way the world is: it is
|
|
14
|
+
* sparse and unbounded, and a snapshot per frame of something that grows is the cost the sparseness
|
|
15
|
+
* was for.
|
|
16
|
+
*
|
|
17
|
+
* **Applying does not clear, and rewinding does.** A replay that always cleared could not be used
|
|
18
|
+
* to catch up frame by frame, which is what a live session does. `rewindOverlay` is the one that
|
|
19
|
+
* puts the world back, and it says so in its name.
|
|
20
|
+
*
|
|
21
|
+
* **The journal is truncated on rollback, and forgetting that is the bug.** Re-simulating after a
|
|
22
|
+
* rollback records its writes again; the frames that did not happen must not still be in the log,
|
|
23
|
+
* or a replay from the start draws both the mark that happened and the one that was undone.
|
|
24
|
+
*/
|
|
25
|
+
import { type Overlay } from './sparse.ts';
|
|
26
|
+
/** Bytes an entry takes on the wire: two `u32` and four `f32`. */
|
|
27
|
+
export declare const JOURNAL_ENTRY_BYTES = 24;
|
|
28
|
+
/** `"DOVJ"`, little-endian, as every other record in this package spells its magic. */
|
|
29
|
+
export declare const JOURNAL_MAGIC = 1247170372;
|
|
30
|
+
export declare const JOURNAL_VERSION = 1;
|
|
31
|
+
export interface OverlayJournal {
|
|
32
|
+
/** One frame number an entry. */
|
|
33
|
+
frames: Int32Array;
|
|
34
|
+
/** One channel index an entry. */
|
|
35
|
+
channels: Int32Array;
|
|
36
|
+
/** Four values an entry: u, v, radius, value. */
|
|
37
|
+
values: Float32Array;
|
|
38
|
+
count: number;
|
|
39
|
+
}
|
|
40
|
+
export declare function createOverlayJournal(capacity?: number): OverlayJournal;
|
|
41
|
+
/**
|
|
42
|
+
* Record a write. Stored as `f32`, which is what makes the encoding round-trip exactly.
|
|
43
|
+
*
|
|
44
|
+
* A journal of `f64` written out as `f32` reproduces a *nearly* identical mark, and "nearly" in a
|
|
45
|
+
* replay is a divergence that appears at the worst moment — the fingerprint that no longer matches
|
|
46
|
+
* a recording, for a reason nobody would look for in a texture.
|
|
47
|
+
*/
|
|
48
|
+
export declare function recordOverlayWrite(journal: OverlayJournal, frame: number, u: number, v: number, radius: number, channel: number, value: number): void;
|
|
49
|
+
export declare function journalLength(journal: OverlayJournal): number;
|
|
50
|
+
/**
|
|
51
|
+
* Forget every entry from `frame` onward. What a rollback owes the journal.
|
|
52
|
+
*
|
|
53
|
+
* Entries are recorded in frame order, so this is a truncation rather than a filter — and where
|
|
54
|
+
* they are not, the scan below still removes exactly the right ones.
|
|
55
|
+
*/
|
|
56
|
+
export declare function truncateJournalFrom(journal: OverlayJournal, frame: number): number;
|
|
57
|
+
/** Apply every entry up to and including `upToFrame`, onto whatever the overlay already holds. */
|
|
58
|
+
export declare function applyOverlayJournal(overlay: Overlay, journal: OverlayJournal, upToFrame: number): number;
|
|
59
|
+
/** Throw away every written tile. The overlay keeps its shape and holds nothing. */
|
|
60
|
+
export declare function clearOverlay(overlay: Overlay): void;
|
|
61
|
+
/**
|
|
62
|
+
* Put the overlay back to how it was at `upToFrame`: clear, then replay.
|
|
63
|
+
*
|
|
64
|
+
* Rebuilt rather than undone, because a write is not invertible — two marks on one texel leave no
|
|
65
|
+
* record of what was underneath, and an overlay that tried to undo would need a history per texel,
|
|
66
|
+
* which is the dense layer the whole design exists to avoid.
|
|
67
|
+
*/
|
|
68
|
+
export declare function rewindOverlay(overlay: Overlay, journal: OverlayJournal, upToFrame: number): number;
|
|
69
|
+
/** An overlay with the same shape as `like` and nothing written. What a replay starts from. */
|
|
70
|
+
export declare function emptyLike(like: Overlay): Overlay;
|
|
71
|
+
export declare function encodeOverlayJournal(journal: OverlayJournal): Uint8Array;
|
|
72
|
+
/**
|
|
73
|
+
* Read a journal back. Null where the bytes are not one.
|
|
74
|
+
*
|
|
75
|
+
* Null rather than a partial journal: a replay against half a log is a session that diverges partway
|
|
76
|
+
* through for no visible reason, which is worse than one that refuses to start.
|
|
77
|
+
*/
|
|
78
|
+
export declare function decodeOverlayJournal(bytes: Uint8Array): OverlayJournal | null;
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A scorch mark is in the input log, so a replay burns the same wall.
|
|
3
|
+
*
|
|
4
|
+
* **Runtime-mutable textures normally break replay.** The marks are not part of the simulation and
|
|
5
|
+
* nothing records them, so a recording plays back a world where the walls are clean — and the
|
|
6
|
+
* failure has the worst possible shape: nothing goes wrong at the time, everything looks right, and
|
|
7
|
+
* the recording is wrong forever. Anybody who later debugs from it is debugging a session that did
|
|
8
|
+
* not happen.
|
|
9
|
+
*
|
|
10
|
+
* Here a write is recorded in the same log the input goes into, so replaying a session reproduces
|
|
11
|
+
* the exact mark in the exact place — and the same mechanism means a rollback un-draws what the
|
|
12
|
+
* rolled-back frames drew, because rolling back an overlay *is* rebuilding it from the journal up
|
|
13
|
+
* to the frame you rolled back to. An overlay cannot be snapshotted the way the world is: it is
|
|
14
|
+
* sparse and unbounded, and a snapshot per frame of something that grows is the cost the sparseness
|
|
15
|
+
* was for.
|
|
16
|
+
*
|
|
17
|
+
* **Applying does not clear, and rewinding does.** A replay that always cleared could not be used
|
|
18
|
+
* to catch up frame by frame, which is what a live session does. `rewindOverlay` is the one that
|
|
19
|
+
* puts the world back, and it says so in its name.
|
|
20
|
+
*
|
|
21
|
+
* **The journal is truncated on rollback, and forgetting that is the bug.** Re-simulating after a
|
|
22
|
+
* rollback records its writes again; the frames that did not happen must not still be in the log,
|
|
23
|
+
* or a replay from the start draws both the mark that happened and the one that was undone.
|
|
24
|
+
*/
|
|
25
|
+
import { createOverlay, writeOverlay } from './sparse.js';
|
|
26
|
+
/** Bytes an entry takes on the wire: two `u32` and four `f32`. */
|
|
27
|
+
export const JOURNAL_ENTRY_BYTES = 24;
|
|
28
|
+
/** `"DOVJ"`, little-endian, as every other record in this package spells its magic. */
|
|
29
|
+
export const JOURNAL_MAGIC = 0x4a564f44;
|
|
30
|
+
export const JOURNAL_VERSION = 1;
|
|
31
|
+
export function createOverlayJournal(capacity = 256) {
|
|
32
|
+
const size = Math.max(1, Math.floor(capacity));
|
|
33
|
+
return {
|
|
34
|
+
frames: new Int32Array(size),
|
|
35
|
+
channels: new Int32Array(size),
|
|
36
|
+
values: new Float32Array(size * 4),
|
|
37
|
+
count: 0,
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
function grow(journal) {
|
|
41
|
+
if (journal.count < journal.frames.length)
|
|
42
|
+
return;
|
|
43
|
+
const size = journal.frames.length * 2;
|
|
44
|
+
const frames = new Int32Array(size);
|
|
45
|
+
const channels = new Int32Array(size);
|
|
46
|
+
const values = new Float32Array(size * 4);
|
|
47
|
+
frames.set(journal.frames);
|
|
48
|
+
channels.set(journal.channels);
|
|
49
|
+
values.set(journal.values);
|
|
50
|
+
journal.frames = frames;
|
|
51
|
+
journal.channels = channels;
|
|
52
|
+
journal.values = values;
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Record a write. Stored as `f32`, which is what makes the encoding round-trip exactly.
|
|
56
|
+
*
|
|
57
|
+
* A journal of `f64` written out as `f32` reproduces a *nearly* identical mark, and "nearly" in a
|
|
58
|
+
* replay is a divergence that appears at the worst moment — the fingerprint that no longer matches
|
|
59
|
+
* a recording, for a reason nobody would look for in a texture.
|
|
60
|
+
*/
|
|
61
|
+
export function recordOverlayWrite(journal, frame, u, v, radius, channel, value) {
|
|
62
|
+
grow(journal);
|
|
63
|
+
const at = journal.count;
|
|
64
|
+
journal.frames[at] = frame;
|
|
65
|
+
journal.channels[at] = channel;
|
|
66
|
+
journal.values[at * 4] = u;
|
|
67
|
+
journal.values[at * 4 + 1] = v;
|
|
68
|
+
journal.values[at * 4 + 2] = radius;
|
|
69
|
+
journal.values[at * 4 + 3] = value;
|
|
70
|
+
journal.count += 1;
|
|
71
|
+
}
|
|
72
|
+
export function journalLength(journal) {
|
|
73
|
+
return journal.count;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Forget every entry from `frame` onward. What a rollback owes the journal.
|
|
77
|
+
*
|
|
78
|
+
* Entries are recorded in frame order, so this is a truncation rather than a filter — and where
|
|
79
|
+
* they are not, the scan below still removes exactly the right ones.
|
|
80
|
+
*/
|
|
81
|
+
export function truncateJournalFrom(journal, frame) {
|
|
82
|
+
let kept = 0;
|
|
83
|
+
for (let at = 0; at < journal.count; at += 1) {
|
|
84
|
+
if (journal.frames[at] >= frame)
|
|
85
|
+
continue;
|
|
86
|
+
if (kept !== at) {
|
|
87
|
+
journal.frames[kept] = journal.frames[at];
|
|
88
|
+
journal.channels[kept] = journal.channels[at];
|
|
89
|
+
for (let c = 0; c < 4; c += 1) {
|
|
90
|
+
journal.values[kept * 4 + c] = journal.values[at * 4 + c];
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
kept += 1;
|
|
94
|
+
}
|
|
95
|
+
const dropped = journal.count - kept;
|
|
96
|
+
journal.count = kept;
|
|
97
|
+
return dropped;
|
|
98
|
+
}
|
|
99
|
+
/** Apply every entry up to and including `upToFrame`, onto whatever the overlay already holds. */
|
|
100
|
+
export function applyOverlayJournal(overlay, journal, upToFrame) {
|
|
101
|
+
let applied = 0;
|
|
102
|
+
for (let at = 0; at < journal.count; at += 1) {
|
|
103
|
+
if (journal.frames[at] > upToFrame)
|
|
104
|
+
continue;
|
|
105
|
+
writeOverlay(overlay, journal.values[at * 4], journal.values[at * 4 + 1], journal.values[at * 4 + 2], journal.channels[at], journal.values[at * 4 + 3]);
|
|
106
|
+
applied += 1;
|
|
107
|
+
}
|
|
108
|
+
return applied;
|
|
109
|
+
}
|
|
110
|
+
/** Throw away every written tile. The overlay keeps its shape and holds nothing. */
|
|
111
|
+
export function clearOverlay(overlay) {
|
|
112
|
+
overlay.tiles.clear();
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* Put the overlay back to how it was at `upToFrame`: clear, then replay.
|
|
116
|
+
*
|
|
117
|
+
* Rebuilt rather than undone, because a write is not invertible — two marks on one texel leave no
|
|
118
|
+
* record of what was underneath, and an overlay that tried to undo would need a history per texel,
|
|
119
|
+
* which is the dense layer the whole design exists to avoid.
|
|
120
|
+
*/
|
|
121
|
+
export function rewindOverlay(overlay, journal, upToFrame) {
|
|
122
|
+
clearOverlay(overlay);
|
|
123
|
+
return applyOverlayJournal(overlay, journal, upToFrame);
|
|
124
|
+
}
|
|
125
|
+
/** An overlay with the same shape as `like` and nothing written. What a replay starts from. */
|
|
126
|
+
export function emptyLike(like) {
|
|
127
|
+
return createOverlay(like.tileSize, {
|
|
128
|
+
tilesAcross: like.tilesAcross,
|
|
129
|
+
addressMode: like.addressMode,
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
export function encodeOverlayJournal(journal) {
|
|
133
|
+
const bytes = new Uint8Array(12 + journal.count * JOURNAL_ENTRY_BYTES);
|
|
134
|
+
const view = new DataView(bytes.buffer);
|
|
135
|
+
view.setUint32(0, JOURNAL_MAGIC, true);
|
|
136
|
+
view.setUint32(4, JOURNAL_VERSION, true);
|
|
137
|
+
view.setUint32(8, journal.count, true);
|
|
138
|
+
for (let at = 0; at < journal.count; at += 1) {
|
|
139
|
+
const base = 12 + at * JOURNAL_ENTRY_BYTES;
|
|
140
|
+
view.setInt32(base, journal.frames[at], true);
|
|
141
|
+
view.setInt32(base + 4, journal.channels[at], true);
|
|
142
|
+
for (let c = 0; c < 4; c += 1) {
|
|
143
|
+
view.setFloat32(base + 8 + c * 4, journal.values[at * 4 + c], true);
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
return bytes;
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* Read a journal back. Null where the bytes are not one.
|
|
150
|
+
*
|
|
151
|
+
* Null rather than a partial journal: a replay against half a log is a session that diverges partway
|
|
152
|
+
* through for no visible reason, which is worse than one that refuses to start.
|
|
153
|
+
*/
|
|
154
|
+
export function decodeOverlayJournal(bytes) {
|
|
155
|
+
if (bytes.byteLength < 12)
|
|
156
|
+
return null;
|
|
157
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
158
|
+
if (view.getUint32(0, true) !== JOURNAL_MAGIC)
|
|
159
|
+
return null;
|
|
160
|
+
if (view.getUint32(4, true) !== JOURNAL_VERSION)
|
|
161
|
+
return null;
|
|
162
|
+
const count = view.getUint32(8, true);
|
|
163
|
+
if (bytes.byteLength !== 12 + count * JOURNAL_ENTRY_BYTES)
|
|
164
|
+
return null;
|
|
165
|
+
const journal = createOverlayJournal(Math.max(1, count));
|
|
166
|
+
for (let at = 0; at < count; at += 1) {
|
|
167
|
+
const base = 12 + at * JOURNAL_ENTRY_BYTES;
|
|
168
|
+
recordOverlayWrite(journal, view.getInt32(base, true), view.getFloat32(base + 8, true), view.getFloat32(base + 12, true), view.getFloat32(base + 16, true), view.getInt32(base + 4, true), view.getFloat32(base + 20, true));
|
|
169
|
+
}
|
|
170
|
+
return journal;
|
|
171
|
+
}
|