@driftengine/texture 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +29 -0
- package/README.md +106 -0
- package/dist/decodeCpu.d.ts +59 -0
- package/dist/decodeCpu.js +234 -0
- package/dist/decodeGraph.d.ts +105 -0
- package/dist/decodeGraph.js +180 -0
- package/dist/half.d.ts +24 -0
- package/dist/half.js +86 -0
- package/dist/index.d.ts +66 -0
- package/dist/index.js +55 -0
- package/dist/inference.d.ts +53 -0
- package/dist/inference.js +243 -0
- package/dist/materialArray.d.ts +38 -0
- package/dist/materialArray.js +40 -0
- package/dist/mipNdf.d.ts +29 -0
- package/dist/mipNdf.js +53 -0
- package/dist/overlay/journal.d.ts +78 -0
- package/dist/overlay/journal.js +171 -0
- package/dist/overlay/sparse.d.ts +68 -0
- package/dist/overlay/sparse.js +212 -0
- package/dist/progressive.d.ts +30 -0
- package/dist/progressive.js +56 -0
- package/dist/residency/pageCache.d.ts +103 -0
- package/dist/residency/pageCache.js +184 -0
- package/dist/residency/predict.d.ts +55 -0
- package/dist/residency/predict.js +51 -0
- package/dist/residency/predictor.d.ts +16 -0
- package/dist/residency/predictor.js +44 -0
- package/dist/residency/queue.d.ts +26 -0
- package/dist/residency/queue.js +52 -0
- package/dist/residency/stream.d.ts +66 -0
- package/dist/residency/stream.js +142 -0
- package/dist/residency/table.d.ts +36 -0
- package/dist/residency/table.js +72 -0
- package/dist/residency/viewTiles.d.ts +108 -0
- package/dist/residency/viewTiles.js +419 -0
- package/dist/semantics.d.ts +52 -0
- package/dist/semantics.js +76 -0
- package/dist/tensor/architecture.d.ts +53 -0
- package/dist/tensor/architecture.js +96 -0
- package/dist/tensor/attention.d.ts +5 -0
- package/dist/tensor/attention.js +62 -0
- package/dist/tensor/denseOperators.d.ts +2 -0
- package/dist/tensor/denseOperators.js +136 -0
- package/dist/tensor/graph.d.ts +83 -0
- package/dist/tensor/graph.js +175 -0
- package/dist/tensor/linear.d.ts +49 -0
- package/dist/tensor/linear.js +136 -0
- package/dist/tensor/operatorKit.d.ts +27 -0
- package/dist/tensor/operatorKit.js +45 -0
- package/dist/tensor/operators.d.ts +3 -0
- package/dist/tensor/operators.js +24 -0
- package/dist/tensor/resize.d.ts +6 -0
- package/dist/tensor/resize.js +107 -0
- package/dist/tensor/reuse.d.ts +33 -0
- package/dist/tensor/reuse.js +59 -0
- package/dist/tensor/shapeOperators.d.ts +3 -0
- package/dist/tensor/shapeOperators.js +173 -0
- package/dist/tensor/spatial.d.ts +34 -0
- package/dist/tensor/spatial.js +131 -0
- package/dist/tensor/spatialOperators.d.ts +2 -0
- package/dist/tensor/spatialOperators.js +138 -0
- package/dist/tileHash.d.ts +29 -0
- package/dist/tileHash.js +50 -0
- package/dist/timeNodes.d.ts +26 -0
- package/dist/timeNodes.js +48 -0
- package/package.json +59 -0
- package/src/decodeCpu.ts +308 -0
- package/src/decodeGraph.ts +214 -0
- package/src/half.ts +86 -0
- package/src/index.ts +175 -0
- package/src/inference.ts +278 -0
- package/src/materialArray.ts +67 -0
- package/src/mipNdf.ts +63 -0
- package/src/overlay/journal.ts +218 -0
- package/src/overlay/sparse.ts +275 -0
- package/src/progressive.ts +60 -0
- package/src/residency/pageCache.ts +233 -0
- package/src/residency/predict.ts +74 -0
- package/src/residency/predictor.ts +62 -0
- package/src/residency/queue.ts +78 -0
- package/src/residency/stream.ts +194 -0
- package/src/residency/table.ts +89 -0
- package/src/residency/viewTiles.ts +553 -0
- package/src/semantics.ts +114 -0
- package/src/tensor/architecture.ts +140 -0
- package/src/tensor/attention.ts +75 -0
- package/src/tensor/denseOperators.ts +153 -0
- package/src/tensor/graph.ts +244 -0
- package/src/tensor/linear.ts +153 -0
- package/src/tensor/operatorKit.ts +76 -0
- package/src/tensor/operators.ts +28 -0
- package/src/tensor/resize.ts +140 -0
- package/src/tensor/shapeOperators.ts +173 -0
- package/src/tensor/spatial.ts +178 -0
- package/src/tensor/spatialOperators.ts +182 -0
- package/src/tileHash.ts +60 -0
- package/src/timeNodes.ts +60 -0
package/src/decodeCpu.ts
ADDED
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The reference decoder: what the shader has to agree with.
|
|
3
|
+
*
|
|
4
|
+
* **A second implementation is not duplication here, it is the specification.** A shader cannot be
|
|
5
|
+
* unit-tested at speed, and a decoder that is subtly wrong is a picture that is subtly wrong
|
|
6
|
+
* everywhere — the hardest defect in this whole design to notice and the easiest to ship. So the
|
|
7
|
+
* arithmetic is written once in TypeScript, where it can be asserted in milliseconds, and the
|
|
8
|
+
* device copy is checked against it over generated input.
|
|
9
|
+
*
|
|
10
|
+
* The repository already takes this position: `splats` authors its shader once and generates the
|
|
11
|
+
* second backend from it. This is the same discipline where the second copy cannot be generated.
|
|
12
|
+
*
|
|
13
|
+
* **Deterministic by construction.** No clock, no random, and `t` is the caller's. Decoding the
|
|
14
|
+
* same inputs twice is bit-identical, which is what the fingerprint test rests on.
|
|
15
|
+
*/
|
|
16
|
+
import {
|
|
17
|
+
ADDRESS_MODE,
|
|
18
|
+
DECODE_OP,
|
|
19
|
+
MAX_REGISTERS,
|
|
20
|
+
nodeA,
|
|
21
|
+
nodeB,
|
|
22
|
+
nodeOp,
|
|
23
|
+
nodeOut,
|
|
24
|
+
} from './decodeGraph.ts';
|
|
25
|
+
import type { DecodeGraph } from './decodeGraph.ts';
|
|
26
|
+
import { evalNetwork } from './inference.ts';
|
|
27
|
+
import type { NetworkShape } from './inference.ts';
|
|
28
|
+
import { CHANNEL_SEMANTICS, normaliseSample } from './semantics.ts';
|
|
29
|
+
import type { ChannelSemantic } from './semantics.ts';
|
|
30
|
+
import { flipbookFrame } from './timeNodes.ts';
|
|
31
|
+
|
|
32
|
+
/** One level of a latent's chain. Components are the image's own. */
|
|
33
|
+
export interface LatentLevel {
|
|
34
|
+
data: Float32Array;
|
|
35
|
+
width: number;
|
|
36
|
+
height: number;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export interface LatentImage {
|
|
40
|
+
data: Float32Array;
|
|
41
|
+
width: number;
|
|
42
|
+
height: number;
|
|
43
|
+
/** Components per texel, up to four. */
|
|
44
|
+
channels: number;
|
|
45
|
+
/**
|
|
46
|
+
* Levels 1 onward, each half the one before. `mips[k]` is level `k + 1`.
|
|
47
|
+
*
|
|
48
|
+
* Absent is a single level, which is every image written before 2026-09-17 — and at `lod` 0 the
|
|
49
|
+
* chain is never read, so those callers decode exactly what they did.
|
|
50
|
+
*/
|
|
51
|
+
mips?: readonly LatentLevel[];
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface DecodeResources {
|
|
55
|
+
latents: readonly LatentImage[];
|
|
56
|
+
blocks: readonly LatentImage[];
|
|
57
|
+
networks: readonly { shape: NetworkShape; weights: Float32Array }[];
|
|
58
|
+
/**
|
|
59
|
+
* Four floats per slot, for `CONSTANT`.
|
|
60
|
+
*
|
|
61
|
+
* Optional, so every caller written before Wave 4C still type-checks and still runs — a graph
|
|
62
|
+
* with no constant in it never reads this.
|
|
63
|
+
*/
|
|
64
|
+
constants?: Float32Array;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* The order `REMAP_CHANNEL` packs a semantic in, which is `CHANNEL_SEMANTICS` and not a copy of it.
|
|
69
|
+
*
|
|
70
|
+
* **It was a copy until 2026-09-20**, written out again here — the same eleven names in the same
|
|
71
|
+
* order in two files, with nothing holding them together. That is a list a file's channels are
|
|
72
|
+
* *numbered* by: the two drifting apart means every material written by one and read by the other
|
|
73
|
+
* decodes its channels as something else, and a normal map read as roughness inverts every
|
|
74
|
+
* highlight in a scene. The name is kept because `terrainTexture.ts`, the WGSL interpreter's test
|
|
75
|
+
* and this file all say `REMAP_SEMANTICS` when they mean the packing.
|
|
76
|
+
*/
|
|
77
|
+
export const REMAP_SEMANTICS: readonly ChannelSemantic[] = CHANNEL_SEMANTICS;
|
|
78
|
+
|
|
79
|
+
/** Lattice addressing only: where a coordinate lands before it is scaled across the lattice. */
|
|
80
|
+
function address(value: number, mode: number): number {
|
|
81
|
+
if (mode === ADDRESS_MODE.LATTICE_WRAP) return value - Math.floor(value);
|
|
82
|
+
return Math.min(1, Math.max(0, value));
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** A texel index inside the image: wrapped for centre wrap, clamped for everything else. */
|
|
86
|
+
function texelIndex(value: number, size: number, mode: number): number {
|
|
87
|
+
if (mode === ADDRESS_MODE.CENTRE_WRAP) return ((value % size) + size) % size;
|
|
88
|
+
return Math.min(size - 1, Math.max(0, value));
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Bilinear on one level, with the graph's address mode applied first.
|
|
93
|
+
*
|
|
94
|
+
* **For the lattice modes this is byte-identical to what it replaced**: `su` is inside
|
|
95
|
+
* `[0, width - 1]`, so clamping `floor(su)` and `floor(su) + 1` is exactly the `min` it was.
|
|
96
|
+
*/
|
|
97
|
+
function sampleLevel(
|
|
98
|
+
level: LatentLevel,
|
|
99
|
+
channels: number,
|
|
100
|
+
u: number,
|
|
101
|
+
v: number,
|
|
102
|
+
mode: number,
|
|
103
|
+
out: Float32Array,
|
|
104
|
+
): void {
|
|
105
|
+
const centre = mode === ADDRESS_MODE.CENTRE_CLAMP || mode === ADDRESS_MODE.CENTRE_WRAP;
|
|
106
|
+
const su = centre ? u * level.width - 0.5 : address(u, mode) * (level.width - 1);
|
|
107
|
+
const sv = centre ? v * level.height - 0.5 : address(v, mode) * (level.height - 1);
|
|
108
|
+
const baseX = Math.floor(su);
|
|
109
|
+
const baseY = Math.floor(sv);
|
|
110
|
+
const fx = su - baseX;
|
|
111
|
+
const fy = sv - baseY;
|
|
112
|
+
const x0 = texelIndex(baseX, level.width, mode);
|
|
113
|
+
const x1 = texelIndex(baseX + 1, level.width, mode);
|
|
114
|
+
const y0 = texelIndex(baseY, level.height, mode);
|
|
115
|
+
const y1 = texelIndex(baseY + 1, level.height, mode);
|
|
116
|
+
for (let c = 0; c < 4; c += 1) {
|
|
117
|
+
if (c >= channels) {
|
|
118
|
+
out[c] = c === 3 ? 1 : 0;
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
const at = (x: number, y: number) => level.data[(y * level.width + x) * channels + c] as number;
|
|
122
|
+
const top = at(x0, y0) * (1 - fx) + at(x1, y0) * fx;
|
|
123
|
+
const bottom = at(x0, y1) * (1 - fx) + at(x1, y1) * fx;
|
|
124
|
+
out[c] = top * (1 - fy) + bottom * fy;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
const SECOND = new Float32Array(4);
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Trilinear: bilinear on the two levels either side of `lod`, mixed by its fraction.
|
|
132
|
+
*
|
|
133
|
+
* **At a whole level only that level is read**, so an integer `lod` — and 0 above all — costs and
|
|
134
|
+
* returns exactly one bilinear sample.
|
|
135
|
+
*/
|
|
136
|
+
function sampleImage(
|
|
137
|
+
image: LatentImage,
|
|
138
|
+
u: number,
|
|
139
|
+
v: number,
|
|
140
|
+
mode: number,
|
|
141
|
+
lod: number,
|
|
142
|
+
out: Float32Array,
|
|
143
|
+
): void {
|
|
144
|
+
const top = image.mips?.length ?? 0;
|
|
145
|
+
const level = lod > 0 ? Math.min(lod, top) : 0;
|
|
146
|
+
const lower = Math.floor(level);
|
|
147
|
+
const blend = level - lower;
|
|
148
|
+
const levelOf = (k: number): LatentLevel =>
|
|
149
|
+
k === 0 ? image : (image.mips?.[k - 1] as LatentLevel);
|
|
150
|
+
sampleLevel(levelOf(lower), image.channels, u, v, mode, out);
|
|
151
|
+
if (blend === 0) return;
|
|
152
|
+
sampleLevel(levelOf(Math.min(lower + 1, top)), image.channels, u, v, mode, SECOND);
|
|
153
|
+
for (let c = 0; c < 4; c += 1) {
|
|
154
|
+
out[c] = (out[c] as number) * (1 - blend) + (SECOND[c] as number) * blend;
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/** A deterministic integer hash, so the procedural node reproduces across runs and platforms. */
|
|
159
|
+
function hash2(x: number, y: number, seed: number): number {
|
|
160
|
+
let h =
|
|
161
|
+
Math.imul(x | 0, 0x27d4eb2d) ^ Math.imul(y | 0, 0x165667b1) ^ Math.imul(seed | 0, 0x9e3779b1);
|
|
162
|
+
h = Math.imul(h ^ (h >>> 15), 0x2c1b3c6d);
|
|
163
|
+
h = Math.imul(h ^ (h >>> 12), 0x297a2d39);
|
|
164
|
+
return ((h ^ (h >>> 15)) >>> 0) / 0xffffffff;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function valueNoise(u: number, v: number, seed: number): number {
|
|
168
|
+
const x0 = Math.floor(u);
|
|
169
|
+
const y0 = Math.floor(v);
|
|
170
|
+
const fx = u - x0;
|
|
171
|
+
const fy = v - y0;
|
|
172
|
+
const sx = fx * fx * (3 - 2 * fx);
|
|
173
|
+
const sy = fy * fy * (3 - 2 * fy);
|
|
174
|
+
const a = hash2(x0, y0, seed);
|
|
175
|
+
const b = hash2(x0 + 1, y0, seed);
|
|
176
|
+
const c = hash2(x0, y0 + 1, seed);
|
|
177
|
+
const d = hash2(x0 + 1, y0 + 1, seed);
|
|
178
|
+
return (a * (1 - sx) + b * sx) * (1 - sy) + (c * (1 - sx) + d * sx) * sy;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
const SCRATCH = new Float32Array(256);
|
|
182
|
+
const SAMPLE = new Float32Array(4);
|
|
183
|
+
const NET_IN = new Float32Array(64);
|
|
184
|
+
const NET_OUT = new Float32Array(64);
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Run the graph at one texture coordinate and one time, writing four components into `out`.
|
|
188
|
+
*
|
|
189
|
+
* `registers` is the caller's, `MAX_REGISTERS * 4` floats, reused so this allocates nothing.
|
|
190
|
+
* `lod` picks the level a latent is sampled at; 0, and anything that is not a positive number, is
|
|
191
|
+
* level 0.
|
|
192
|
+
*/
|
|
193
|
+
export function decodeCpu(
|
|
194
|
+
graph: DecodeGraph,
|
|
195
|
+
resources: DecodeResources,
|
|
196
|
+
u: number,
|
|
197
|
+
v: number,
|
|
198
|
+
t: number,
|
|
199
|
+
out: Float32Array,
|
|
200
|
+
registers: Float32Array,
|
|
201
|
+
lod = 0,
|
|
202
|
+
): void {
|
|
203
|
+
for (let i = 0; i < graph.count; i += 1) {
|
|
204
|
+
const op = nodeOp(graph, i);
|
|
205
|
+
const a = nodeA(graph, i);
|
|
206
|
+
const b = nodeB(graph, i);
|
|
207
|
+
const dst = nodeOut(graph, i) * 4;
|
|
208
|
+
|
|
209
|
+
switch (op) {
|
|
210
|
+
case DECODE_OP.CONSTANT: {
|
|
211
|
+
const constants = resources.constants;
|
|
212
|
+
const base = a * 4;
|
|
213
|
+
for (let c = 0; c < 4; c += 1) registers[dst + c] = constants?.[base + c] ?? 0;
|
|
214
|
+
break;
|
|
215
|
+
}
|
|
216
|
+
case DECODE_OP.SAMPLE_LATENT: {
|
|
217
|
+
const image = resources.latents[a];
|
|
218
|
+
if (image === undefined) break;
|
|
219
|
+
sampleImage(image, u, v, graph.addressMode, lod, SAMPLE);
|
|
220
|
+
registers.set(SAMPLE, dst);
|
|
221
|
+
break;
|
|
222
|
+
}
|
|
223
|
+
case DECODE_OP.SAMPLE_BLOCK: {
|
|
224
|
+
const image = resources.blocks[a];
|
|
225
|
+
if (image === undefined) break;
|
|
226
|
+
sampleImage(image, u, v, graph.addressMode, lod, SAMPLE);
|
|
227
|
+
registers.set(SAMPLE, dst);
|
|
228
|
+
break;
|
|
229
|
+
}
|
|
230
|
+
case DECODE_OP.PROCEDURAL_FBM: {
|
|
231
|
+
let sum = 0;
|
|
232
|
+
let amplitude = 0.5;
|
|
233
|
+
let frequency = 1;
|
|
234
|
+
const octaves = Math.max(1, b);
|
|
235
|
+
for (let o = 0; o < octaves; o += 1) {
|
|
236
|
+
sum += valueNoise(u * frequency * 8, v * frequency * 8, a) * amplitude;
|
|
237
|
+
amplitude *= 0.5;
|
|
238
|
+
frequency *= 2;
|
|
239
|
+
}
|
|
240
|
+
registers[dst] = sum;
|
|
241
|
+
registers[dst + 1] = sum;
|
|
242
|
+
registers[dst + 2] = sum;
|
|
243
|
+
registers[dst + 3] = 1;
|
|
244
|
+
break;
|
|
245
|
+
}
|
|
246
|
+
case DECODE_OP.FLIPBOOK_INDEX: {
|
|
247
|
+
registers[dst] = flipbookFrame(t, a, b, true);
|
|
248
|
+
registers[dst + 1] = 0;
|
|
249
|
+
registers[dst + 2] = 0;
|
|
250
|
+
registers[dst + 3] = 1;
|
|
251
|
+
break;
|
|
252
|
+
}
|
|
253
|
+
case DECODE_OP.LATENT_LERP: {
|
|
254
|
+
const mix = t - Math.floor(t);
|
|
255
|
+
const src0 = a * 4;
|
|
256
|
+
const src1 = b * 4;
|
|
257
|
+
for (let c = 0; c < 4; c += 1) {
|
|
258
|
+
registers[dst + c] =
|
|
259
|
+
(registers[src0 + c] as number) * (1 - mix) + (registers[src1 + c] as number) * mix;
|
|
260
|
+
}
|
|
261
|
+
break;
|
|
262
|
+
}
|
|
263
|
+
case DECODE_OP.REMAP_CHANNEL: {
|
|
264
|
+
const src = a * 4;
|
|
265
|
+
for (let c = 0; c < 4; c += 1) registers[dst + c] = registers[src + c] as number;
|
|
266
|
+
const semantic = REMAP_SEMANTICS[b >>> 4] ?? 'mask-linear';
|
|
267
|
+
const component = b & 0xf;
|
|
268
|
+
SAMPLE.set(registers.subarray(dst, dst + 4));
|
|
269
|
+
normaliseSample(SAMPLE, { semantic, component }, registers[src + component] as number);
|
|
270
|
+
registers.set(SAMPLE, dst);
|
|
271
|
+
break;
|
|
272
|
+
}
|
|
273
|
+
case DECODE_OP.COMPOSITE: {
|
|
274
|
+
const src0 = a * 4;
|
|
275
|
+
const src1 = b * 4;
|
|
276
|
+
const alpha = registers[src0 + 3] as number;
|
|
277
|
+
for (let c = 0; c < 3; c += 1) {
|
|
278
|
+
registers[dst + c] =
|
|
279
|
+
(registers[src0 + c] as number) * alpha + (registers[src1 + c] as number) * (1 - alpha);
|
|
280
|
+
}
|
|
281
|
+
registers[dst + 3] = alpha + (registers[src1 + 3] as number) * (1 - alpha);
|
|
282
|
+
break;
|
|
283
|
+
}
|
|
284
|
+
case DECODE_OP.EVAL_NETWORK: {
|
|
285
|
+
const net = resources.networks[b];
|
|
286
|
+
if (net === undefined) break;
|
|
287
|
+
const src = a * 4;
|
|
288
|
+
for (let c = 0; c < net.shape.inputs; c += 1) {
|
|
289
|
+
NET_IN[c] = c < 4 ? (registers[src + c] as number) : 0;
|
|
290
|
+
}
|
|
291
|
+
evalNetwork(net.shape, net.weights, NET_IN, NET_OUT, SCRATCH);
|
|
292
|
+
for (let c = 0; c < 4; c += 1) {
|
|
293
|
+
registers[dst + c] = c < net.shape.outputs ? (NET_OUT[c] as number) : 0;
|
|
294
|
+
}
|
|
295
|
+
break;
|
|
296
|
+
}
|
|
297
|
+
default:
|
|
298
|
+
break;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
const result = graph.result * 4;
|
|
302
|
+
for (let c = 0; c < 4; c += 1) out[c] = registers[result + c] as number;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/** A registers array of the right size, for a caller that holds one across calls. */
|
|
306
|
+
export function createDecodeRegisters(): Float32Array {
|
|
307
|
+
return new Float32Array(MAX_REGISTERS * 4);
|
|
308
|
+
}
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A texture's decoder, as data.
|
|
3
|
+
*
|
|
4
|
+
* **This is the file where the expensive mistake would be made, so the reasoning is here.** The
|
|
5
|
+
* decode program is uploaded as a uniform buffer and walked by one shader. There is no code
|
|
6
|
+
* generation, no `#define`, no shader per material and no variant — because `ARCHITECTURE.md` has
|
|
7
|
+
* already measured what the other way costs, with the number: a fifth permutation flag took the
|
|
8
|
+
* generated WGSL corpus from 914 KB to 1,906 KB and cost **196,910 gzipped bytes on every
|
|
9
|
+
* consumer**, including those who never enabled it, because deflate's window is 32 KB and
|
|
10
|
+
* near-identical permutations do not deduplicate. A decode graph as permutations would be that
|
|
11
|
+
* mistake with a far larger exponent — per material rather than per feature.
|
|
12
|
+
*
|
|
13
|
+
* As data it costs bytes, linearly, and a new operation is one more case in one switch.
|
|
14
|
+
*
|
|
15
|
+
* **A node is four words: op, a, b, out.** Which of `a` and `b` name registers rather than
|
|
16
|
+
* immediates is a property of the operation, held in `OP_ARGS` so that the validator and both
|
|
17
|
+
* interpreters read one table rather than three copies of a convention.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
export const DECODE_OP = {
|
|
21
|
+
/** a = latent slot. Samples it at (u, v). */
|
|
22
|
+
SAMPLE_LATENT: 0,
|
|
23
|
+
/** a = register holding the input vector, b = network slot. */
|
|
24
|
+
EVAL_NETWORK: 1,
|
|
25
|
+
/** a = block-compressed slot. The passthrough for content a network does not help. */
|
|
26
|
+
SAMPLE_BLOCK: 2,
|
|
27
|
+
/** a = seed, b = octaves. */
|
|
28
|
+
PROCEDURAL_FBM: 3,
|
|
29
|
+
/** a = frame count, b = frames per second. Writes the frame index into x. */
|
|
30
|
+
FLIPBOOK_INDEX: 4,
|
|
31
|
+
/** a, b = registers. Mixes them by the fractional part of the sample time. */
|
|
32
|
+
LATENT_LERP: 5,
|
|
33
|
+
/** a = register, b = packed channel spec. Applies the declared convention. */
|
|
34
|
+
REMAP_CHANNEL: 6,
|
|
35
|
+
/** a, b = registers. `a` over `b`. */
|
|
36
|
+
COMPOSITE: 7,
|
|
37
|
+
/**
|
|
38
|
+
* a = constant slot. Writes that four-component value.
|
|
39
|
+
*
|
|
40
|
+
* **Added for Wave 4C, and the reason is the rule rather than the node.** A material graph
|
|
41
|
+
* compiles to this vocabulary and not to a shader, so a node the vocabulary cannot express grows
|
|
42
|
+
* the vocabulary — which costs bytes linearly, because an operation is data. A colour picker is
|
|
43
|
+
* the first node anybody puts in a material graph and there was nothing here that could hold one.
|
|
44
|
+
*/
|
|
45
|
+
CONSTANT: 8,
|
|
46
|
+
} as const;
|
|
47
|
+
|
|
48
|
+
export type DecodeOp = (typeof DECODE_OP)[keyof typeof DECODE_OP];
|
|
49
|
+
|
|
50
|
+
/** Whether each operation's `a` and `b` name registers. One table, three readers. */
|
|
51
|
+
const OP_ARGS: Readonly<Record<number, { a: boolean; b: boolean }>> = {
|
|
52
|
+
[DECODE_OP.SAMPLE_LATENT]: { a: false, b: false },
|
|
53
|
+
[DECODE_OP.EVAL_NETWORK]: { a: true, b: false },
|
|
54
|
+
[DECODE_OP.SAMPLE_BLOCK]: { a: false, b: false },
|
|
55
|
+
[DECODE_OP.PROCEDURAL_FBM]: { a: false, b: false },
|
|
56
|
+
[DECODE_OP.FLIPBOOK_INDEX]: { a: false, b: false },
|
|
57
|
+
[DECODE_OP.LATENT_LERP]: { a: true, b: true },
|
|
58
|
+
[DECODE_OP.REMAP_CHANNEL]: { a: true, b: false },
|
|
59
|
+
[DECODE_OP.COMPOSITE]: { a: true, b: true },
|
|
60
|
+
[DECODE_OP.CONSTANT]: { a: false, b: false },
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Where a coordinate lands, and what happens outside the unit square.
|
|
65
|
+
*
|
|
66
|
+
* **Two conventions, because two kinds of data need them.** A *lattice* puts `u = 0` on the centre
|
|
67
|
+
* of the first texel and `u = 1` on the centre of the last, which is what a height field sampled at
|
|
68
|
+
* its own vertices wants — `terrainTexture.ts` reads `x / (width - 1)` and lands on texel `x`
|
|
69
|
+
* exactly. A *centre* mode puts `u = 0` on the first texel's edge, which is what every GPU sampler
|
|
70
|
+
* does and what a surface texture tiled across a mesh wants. Added 2026-09-17 so the device
|
|
71
|
+
* interpreter has a reference to agree with; modes 0 and 1 did not move.
|
|
72
|
+
*
|
|
73
|
+
* Lattice wrap has a seam a tiling texture shows: `u = 0.999` reads the last texel and `u = 1`
|
|
74
|
+
* the first. Centre wrap blends across it, as a repeating sampler does.
|
|
75
|
+
*/
|
|
76
|
+
export const ADDRESS_MODE = {
|
|
77
|
+
LATTICE_CLAMP: 0,
|
|
78
|
+
LATTICE_WRAP: 1,
|
|
79
|
+
CENTRE_CLAMP: 2,
|
|
80
|
+
CENTRE_WRAP: 3,
|
|
81
|
+
} as const;
|
|
82
|
+
|
|
83
|
+
/** How many address modes there are. A graph asking for this many or more is refused. */
|
|
84
|
+
export const ADDRESS_MODE_COUNT = 4;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Registers the interpreter has, each a four-component vector.
|
|
88
|
+
*
|
|
89
|
+
* A budget rather than a limit discovered on one device: a graph that needs more is refused at
|
|
90
|
+
* encode time, where a person can see it, rather than compiling to a shader that fails on the
|
|
91
|
+
* hardware with the smallest uniform space.
|
|
92
|
+
*/
|
|
93
|
+
export const MAX_REGISTERS = 16;
|
|
94
|
+
|
|
95
|
+
/** Words per node: op, a, b, out. */
|
|
96
|
+
export const NODE_STRIDE = 4;
|
|
97
|
+
|
|
98
|
+
export interface DecodeGraph {
|
|
99
|
+
/** Four words per node. */
|
|
100
|
+
nodes: Uint32Array;
|
|
101
|
+
count: number;
|
|
102
|
+
/** Which register holds the finished channel vector. */
|
|
103
|
+
result: number;
|
|
104
|
+
/** One of `ADDRESS_MODE`. */
|
|
105
|
+
addressMode: number;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
export function createDecodeGraph(capacity: number): DecodeGraph {
|
|
109
|
+
return { nodes: new Uint32Array(capacity * NODE_STRIDE), count: 0, result: 0, addressMode: 0 };
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
export function addDecodeNode(
|
|
113
|
+
graph: DecodeGraph,
|
|
114
|
+
op: number,
|
|
115
|
+
a: number,
|
|
116
|
+
b: number,
|
|
117
|
+
out: number,
|
|
118
|
+
): number {
|
|
119
|
+
const at = graph.count * NODE_STRIDE;
|
|
120
|
+
graph.nodes[at] = op;
|
|
121
|
+
graph.nodes[at + 1] = a;
|
|
122
|
+
graph.nodes[at + 2] = b;
|
|
123
|
+
graph.nodes[at + 3] = out;
|
|
124
|
+
graph.count += 1;
|
|
125
|
+
return graph.count - 1;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
export function nodeOp(graph: DecodeGraph, index: number): number {
|
|
129
|
+
return graph.nodes[index * NODE_STRIDE] as number;
|
|
130
|
+
}
|
|
131
|
+
export function nodeA(graph: DecodeGraph, index: number): number {
|
|
132
|
+
return graph.nodes[index * NODE_STRIDE + 1] as number;
|
|
133
|
+
}
|
|
134
|
+
export function nodeB(graph: DecodeGraph, index: number): number {
|
|
135
|
+
return graph.nodes[index * NODE_STRIDE + 2] as number;
|
|
136
|
+
}
|
|
137
|
+
export function nodeOut(graph: DecodeGraph, index: number): number {
|
|
138
|
+
return graph.nodes[index * NODE_STRIDE + 3] as number;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/** The highest register the graph writes, plus one. */
|
|
142
|
+
export function graphRegisterCount(graph: DecodeGraph): number {
|
|
143
|
+
let highest = graph.result;
|
|
144
|
+
for (let i = 0; i < graph.count; i += 1) highest = Math.max(highest, nodeOut(graph, i));
|
|
145
|
+
return highest + 1;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Whether this graph is one an interpreter can run.
|
|
150
|
+
*
|
|
151
|
+
* Returns a message rather than throwing, matching `validateGraph` in the frame package and for
|
|
152
|
+
* the same reason: whether a malformed graph is an assertion or a skipped material is the
|
|
153
|
+
* caller's decision.
|
|
154
|
+
*/
|
|
155
|
+
export function validateDecodeGraph(graph: DecodeGraph): string | null {
|
|
156
|
+
if (
|
|
157
|
+
!Number.isInteger(graph.addressMode) ||
|
|
158
|
+
graph.addressMode < 0 ||
|
|
159
|
+
graph.addressMode >= ADDRESS_MODE_COUNT
|
|
160
|
+
) {
|
|
161
|
+
return `address mode ${graph.addressMode} is not one of the ${ADDRESS_MODE_COUNT} the interpreter knows`;
|
|
162
|
+
}
|
|
163
|
+
const written = new Set<number>();
|
|
164
|
+
for (let i = 0; i < graph.count; i += 1) {
|
|
165
|
+
const op = nodeOp(graph, i);
|
|
166
|
+
const args = OP_ARGS[op];
|
|
167
|
+
if (args === undefined) return `node ${i} has unknown opcode ${op}`;
|
|
168
|
+
|
|
169
|
+
const out = nodeOut(graph, i);
|
|
170
|
+
if (out >= MAX_REGISTERS) {
|
|
171
|
+
return `node ${i} writes register ${out}, past the ${MAX_REGISTERS} the interpreter has`;
|
|
172
|
+
}
|
|
173
|
+
if (args.a && !written.has(nodeA(graph, i))) {
|
|
174
|
+
return `node ${i} reads register ${nodeA(graph, i)}, which nothing wrote`;
|
|
175
|
+
}
|
|
176
|
+
if (args.b && !written.has(nodeB(graph, i))) {
|
|
177
|
+
return `node ${i} reads register ${nodeB(graph, i)}, which nothing wrote`;
|
|
178
|
+
}
|
|
179
|
+
written.add(out);
|
|
180
|
+
}
|
|
181
|
+
if (!written.has(graph.result)) {
|
|
182
|
+
return `the result register ${graph.result} is never written`;
|
|
183
|
+
}
|
|
184
|
+
if (graphRegisterCount(graph) > MAX_REGISTERS) {
|
|
185
|
+
return `the graph needs ${graphRegisterCount(graph)} registers, past the ${MAX_REGISTERS} available`;
|
|
186
|
+
}
|
|
187
|
+
return null;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/** `count`, `result`, `addressMode`, then the nodes. */
|
|
191
|
+
export function encodeDecodeGraph(graph: DecodeGraph): Uint8Array {
|
|
192
|
+
const bytes = new Uint8Array(12 + graph.count * NODE_STRIDE * 4);
|
|
193
|
+
const view = new DataView(bytes.buffer);
|
|
194
|
+
view.setUint32(0, graph.count, true);
|
|
195
|
+
view.setUint32(4, graph.result, true);
|
|
196
|
+
view.setUint32(8, graph.addressMode, true);
|
|
197
|
+
new Uint32Array(bytes.buffer, 12, graph.count * NODE_STRIDE).set(
|
|
198
|
+
graph.nodes.subarray(0, graph.count * NODE_STRIDE),
|
|
199
|
+
);
|
|
200
|
+
return bytes;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
export function decodeDecodeGraph(bytes: Uint8Array): DecodeGraph {
|
|
204
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
205
|
+
const count = view.getUint32(0, true);
|
|
206
|
+
const graph = createDecodeGraph(Math.max(1, count));
|
|
207
|
+
graph.count = count;
|
|
208
|
+
graph.result = view.getUint32(4, true);
|
|
209
|
+
graph.addressMode = view.getUint32(8, true);
|
|
210
|
+
for (let i = 0; i < count * NODE_STRIDE; i += 1) {
|
|
211
|
+
graph.nodes[i] = view.getUint32(12 + i * 4, true);
|
|
212
|
+
}
|
|
213
|
+
return graph;
|
|
214
|
+
}
|
package/src/half.ts
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* IEEE 754 binary16 — half precision — on a platform that has no type for it.
|
|
3
|
+
*
|
|
4
|
+
* **The half-precision network path needs a reference, and the reference needs this.** A shader
|
|
5
|
+
* declared `enable f16` computes in sixteen bits, and the only way to say what it should have
|
|
6
|
+
* produced is to round the same arithmetic to the same sixteen bits here. Node 22 has no
|
|
7
|
+
* `Float16Array` and no `Math.f16round`, so the rounding is written out: one sign bit, five exponent
|
|
8
|
+
* bits biased by fifteen, ten mantissa bits, ties to even, overflow to infinity.
|
|
9
|
+
*
|
|
10
|
+
* **What the format cannot hold is the reason the path is optional.** The largest finite value is
|
|
11
|
+
* 65,504 and the smallest normal one is 2^-14, so a network whose activations leave that range is
|
|
12
|
+
* wrong in half precision in a way no tolerance covers — `activationBound` in `inference.ts` is what
|
|
13
|
+
* a consumer asks before choosing it.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
/** The largest finite half-precision value. */
|
|
17
|
+
export const HALF_MAX = 65504;
|
|
18
|
+
|
|
19
|
+
/* One scratch view, so reading a double's exponent allocates nothing. */
|
|
20
|
+
const BITS = new DataView(new ArrayBuffer(8));
|
|
21
|
+
|
|
22
|
+
/*
|
|
23
|
+
* Round to the nearest integer, a tie to the even one. `Math.round` rounds a tie up, which is not
|
|
24
|
+
* what any floating-point format does and would make every tie in the tests an error of one unit.
|
|
25
|
+
*/
|
|
26
|
+
function roundEven(value: number): number {
|
|
27
|
+
const floor = Math.floor(value);
|
|
28
|
+
const fraction = value - floor;
|
|
29
|
+
if (fraction > 0.5) return floor + 1;
|
|
30
|
+
if (fraction < 0.5) return floor;
|
|
31
|
+
return floor % 2 === 0 ? floor : floor + 1;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** The sixteen bits a value rounds to, as an unsigned integer. */
|
|
35
|
+
export function toHalfBits(value: number): number {
|
|
36
|
+
if (Number.isNaN(value)) return 0x7e00;
|
|
37
|
+
const sign = value < 0 || Object.is(value, -0) ? 0x8000 : 0;
|
|
38
|
+
const magnitude = Math.abs(value);
|
|
39
|
+
if (magnitude < 2 ** -14) {
|
|
40
|
+
/*
|
|
41
|
+
* Subnormal: the value in units of 2^-24, which is the mantissa directly. A result of 1,024 is
|
|
42
|
+
* the smallest normal value, and its bits are exactly 1,024 — the carry is free.
|
|
43
|
+
*/
|
|
44
|
+
return sign | roundEven(magnitude * 2 ** 24);
|
|
45
|
+
}
|
|
46
|
+
/*
|
|
47
|
+
* The exponent from the double's own bits rather than from `Math.log2`, which the language
|
|
48
|
+
* leaves approximate — exact here, because every magnitude this far down the function is a normal
|
|
49
|
+
* double.
|
|
50
|
+
*/
|
|
51
|
+
BITS.setFloat64(0, magnitude);
|
|
52
|
+
let exponent = ((BITS.getUint16(0) >> 4) & 0x7ff) - 1023;
|
|
53
|
+
let mantissa = roundEven((magnitude / 2 ** exponent - 1) * 1024);
|
|
54
|
+
if (mantissa === 1024) {
|
|
55
|
+
mantissa = 0;
|
|
56
|
+
exponent += 1;
|
|
57
|
+
}
|
|
58
|
+
/*
|
|
59
|
+
* Overflow is decided here and only here. 65,520 is halfway between the largest finite value and
|
|
60
|
+
* 2^16, its mantissa rounds up to the carry above, and the exponent it lands on is 16.
|
|
61
|
+
*/
|
|
62
|
+
if (exponent > 15) return sign | 0x7c00;
|
|
63
|
+
return sign | ((exponent + 15) << 10) | mantissa;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** The value sixteen bits hold. */
|
|
67
|
+
export function fromHalfBits(bits: number): number {
|
|
68
|
+
const sign = (bits & 0x8000) !== 0 ? -1 : 1;
|
|
69
|
+
const exponent = (bits >> 10) & 0x1f;
|
|
70
|
+
const mantissa = bits & 0x03ff;
|
|
71
|
+
if (exponent === 0) return sign * mantissa * 2 ** -24;
|
|
72
|
+
if (exponent === 0x1f) return mantissa === 0 ? sign * Infinity : Number.NaN;
|
|
73
|
+
return sign * (1 + mantissa / 1024) * 2 ** (exponent - 15);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** The nearest half-precision value, as a number. */
|
|
77
|
+
export function roundHalf(value: number): number {
|
|
78
|
+
return fromHalfBits(toHalfBits(value));
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** A whole weight array as half-precision bits, each weight rounded on its own. */
|
|
82
|
+
export function halfWeights(weights: Float32Array): Uint16Array {
|
|
83
|
+
const out = new Uint16Array(weights.length);
|
|
84
|
+
for (let i = 0; i < weights.length; i += 1) out[i] = toHalfBits(weights[i] as number);
|
|
85
|
+
return out;
|
|
86
|
+
}
|