@driftengine/texture 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/LICENSE +202 -0
  2. package/NOTICE +29 -0
  3. package/README.md +106 -0
  4. package/dist/decodeCpu.d.ts +59 -0
  5. package/dist/decodeCpu.js +234 -0
  6. package/dist/decodeGraph.d.ts +105 -0
  7. package/dist/decodeGraph.js +180 -0
  8. package/dist/half.d.ts +24 -0
  9. package/dist/half.js +86 -0
  10. package/dist/index.d.ts +66 -0
  11. package/dist/index.js +55 -0
  12. package/dist/inference.d.ts +53 -0
  13. package/dist/inference.js +243 -0
  14. package/dist/materialArray.d.ts +38 -0
  15. package/dist/materialArray.js +40 -0
  16. package/dist/mipNdf.d.ts +29 -0
  17. package/dist/mipNdf.js +53 -0
  18. package/dist/overlay/journal.d.ts +78 -0
  19. package/dist/overlay/journal.js +171 -0
  20. package/dist/overlay/sparse.d.ts +68 -0
  21. package/dist/overlay/sparse.js +212 -0
  22. package/dist/progressive.d.ts +30 -0
  23. package/dist/progressive.js +56 -0
  24. package/dist/residency/pageCache.d.ts +103 -0
  25. package/dist/residency/pageCache.js +184 -0
  26. package/dist/residency/predict.d.ts +55 -0
  27. package/dist/residency/predict.js +51 -0
  28. package/dist/residency/predictor.d.ts +16 -0
  29. package/dist/residency/predictor.js +44 -0
  30. package/dist/residency/queue.d.ts +26 -0
  31. package/dist/residency/queue.js +52 -0
  32. package/dist/residency/stream.d.ts +66 -0
  33. package/dist/residency/stream.js +142 -0
  34. package/dist/residency/table.d.ts +36 -0
  35. package/dist/residency/table.js +72 -0
  36. package/dist/residency/viewTiles.d.ts +108 -0
  37. package/dist/residency/viewTiles.js +419 -0
  38. package/dist/semantics.d.ts +52 -0
  39. package/dist/semantics.js +76 -0
  40. package/dist/tensor/architecture.d.ts +53 -0
  41. package/dist/tensor/architecture.js +96 -0
  42. package/dist/tensor/attention.d.ts +5 -0
  43. package/dist/tensor/attention.js +62 -0
  44. package/dist/tensor/denseOperators.d.ts +2 -0
  45. package/dist/tensor/denseOperators.js +136 -0
  46. package/dist/tensor/graph.d.ts +83 -0
  47. package/dist/tensor/graph.js +175 -0
  48. package/dist/tensor/linear.d.ts +49 -0
  49. package/dist/tensor/linear.js +136 -0
  50. package/dist/tensor/operatorKit.d.ts +27 -0
  51. package/dist/tensor/operatorKit.js +45 -0
  52. package/dist/tensor/operators.d.ts +3 -0
  53. package/dist/tensor/operators.js +24 -0
  54. package/dist/tensor/resize.d.ts +6 -0
  55. package/dist/tensor/resize.js +107 -0
  56. package/dist/tensor/reuse.d.ts +33 -0
  57. package/dist/tensor/reuse.js +59 -0
  58. package/dist/tensor/shapeOperators.d.ts +3 -0
  59. package/dist/tensor/shapeOperators.js +173 -0
  60. package/dist/tensor/spatial.d.ts +34 -0
  61. package/dist/tensor/spatial.js +131 -0
  62. package/dist/tensor/spatialOperators.d.ts +2 -0
  63. package/dist/tensor/spatialOperators.js +138 -0
  64. package/dist/tileHash.d.ts +29 -0
  65. package/dist/tileHash.js +50 -0
  66. package/dist/timeNodes.d.ts +26 -0
  67. package/dist/timeNodes.js +48 -0
  68. package/package.json +59 -0
  69. package/src/decodeCpu.ts +308 -0
  70. package/src/decodeGraph.ts +214 -0
  71. package/src/half.ts +86 -0
  72. package/src/index.ts +175 -0
  73. package/src/inference.ts +278 -0
  74. package/src/materialArray.ts +67 -0
  75. package/src/mipNdf.ts +63 -0
  76. package/src/overlay/journal.ts +218 -0
  77. package/src/overlay/sparse.ts +275 -0
  78. package/src/progressive.ts +60 -0
  79. package/src/residency/pageCache.ts +233 -0
  80. package/src/residency/predict.ts +74 -0
  81. package/src/residency/predictor.ts +62 -0
  82. package/src/residency/queue.ts +78 -0
  83. package/src/residency/stream.ts +194 -0
  84. package/src/residency/table.ts +89 -0
  85. package/src/residency/viewTiles.ts +553 -0
  86. package/src/semantics.ts +114 -0
  87. package/src/tensor/architecture.ts +140 -0
  88. package/src/tensor/attention.ts +75 -0
  89. package/src/tensor/denseOperators.ts +153 -0
  90. package/src/tensor/graph.ts +244 -0
  91. package/src/tensor/linear.ts +153 -0
  92. package/src/tensor/operatorKit.ts +76 -0
  93. package/src/tensor/operators.ts +28 -0
  94. package/src/tensor/resize.ts +140 -0
  95. package/src/tensor/shapeOperators.ts +173 -0
  96. package/src/tensor/spatial.ts +178 -0
  97. package/src/tensor/spatialOperators.ts +182 -0
  98. package/src/tileHash.ts +60 -0
  99. package/src/timeNodes.ts +60 -0
@@ -0,0 +1,105 @@
1
+ /**
2
+ * A texture's decoder, as data.
3
+ *
4
+ * **This is the file where the expensive mistake would be made, so the reasoning is here.** The
5
+ * decode program is uploaded as a uniform buffer and walked by one shader. There is no code
6
+ * generation, no `#define`, no shader per material and no variant — because `ARCHITECTURE.md` has
7
+ * already measured what the other way costs, with the number: a fifth permutation flag took the
8
+ * generated WGSL corpus from 914 KB to 1,906 KB and cost **196,910 gzipped bytes on every
9
+ * consumer**, including those who never enabled it, because deflate's window is 32 KB and
10
+ * near-identical permutations do not deduplicate. A decode graph as permutations would be that
11
+ * mistake with a far larger exponent — per material rather than per feature.
12
+ *
13
+ * As data it costs bytes, linearly, and a new operation is one more case in one switch.
14
+ *
15
+ * **A node is four words: op, a, b, out.** Which of `a` and `b` name registers rather than
16
+ * immediates is a property of the operation, held in `OP_ARGS` so that the validator and both
17
+ * interpreters read one table rather than three copies of a convention.
18
+ */
19
+ export declare const DECODE_OP: {
20
+ /** a = latent slot. Samples it at (u, v). */
21
+ readonly SAMPLE_LATENT: 0;
22
+ /** a = register holding the input vector, b = network slot. */
23
+ readonly EVAL_NETWORK: 1;
24
+ /** a = block-compressed slot. The passthrough for content a network does not help. */
25
+ readonly SAMPLE_BLOCK: 2;
26
+ /** a = seed, b = octaves. */
27
+ readonly PROCEDURAL_FBM: 3;
28
+ /** a = frame count, b = frames per second. Writes the frame index into x. */
29
+ readonly FLIPBOOK_INDEX: 4;
30
+ /** a, b = registers. Mixes them by the fractional part of the sample time. */
31
+ readonly LATENT_LERP: 5;
32
+ /** a = register, b = packed channel spec. Applies the declared convention. */
33
+ readonly REMAP_CHANNEL: 6;
34
+ /** a, b = registers. `a` over `b`. */
35
+ readonly COMPOSITE: 7;
36
+ /**
37
+ * a = constant slot. Writes that four-component value.
38
+ *
39
+ * **Added for Wave 4C, and the reason is the rule rather than the node.** A material graph
40
+ * compiles to this vocabulary and not to a shader, so a node the vocabulary cannot express grows
41
+ * the vocabulary — which costs bytes linearly, because an operation is data. A colour picker is
42
+ * the first node anybody puts in a material graph and there was nothing here that could hold one.
43
+ */
44
+ readonly CONSTANT: 8;
45
+ };
46
+ export type DecodeOp = (typeof DECODE_OP)[keyof typeof DECODE_OP];
47
+ /**
48
+ * Where a coordinate lands, and what happens outside the unit square.
49
+ *
50
+ * **Two conventions, because two kinds of data need them.** A *lattice* puts `u = 0` on the centre
51
+ * of the first texel and `u = 1` on the centre of the last, which is what a height field sampled at
52
+ * its own vertices wants — `terrainTexture.ts` reads `x / (width - 1)` and lands on texel `x`
53
+ * exactly. A *centre* mode puts `u = 0` on the first texel's edge, which is what every GPU sampler
54
+ * does and what a surface texture tiled across a mesh wants. Added 2026-09-17 so the device
55
+ * interpreter has a reference to agree with; modes 0 and 1 did not move.
56
+ *
57
+ * Lattice wrap has a seam a tiling texture shows: `u = 0.999` reads the last texel and `u = 1`
58
+ * the first. Centre wrap blends across it, as a repeating sampler does.
59
+ */
60
+ export declare const ADDRESS_MODE: {
61
+ readonly LATTICE_CLAMP: 0;
62
+ readonly LATTICE_WRAP: 1;
63
+ readonly CENTRE_CLAMP: 2;
64
+ readonly CENTRE_WRAP: 3;
65
+ };
66
+ /** How many address modes there are. A graph asking for this many or more is refused. */
67
+ export declare const ADDRESS_MODE_COUNT = 4;
68
+ /**
69
+ * Registers the interpreter has, each a four-component vector.
70
+ *
71
+ * A budget rather than a limit discovered on one device: a graph that needs more is refused at
72
+ * encode time, where a person can see it, rather than compiling to a shader that fails on the
73
+ * hardware with the smallest uniform space.
74
+ */
75
+ export declare const MAX_REGISTERS = 16;
76
+ /** Words per node: op, a, b, out. */
77
+ export declare const NODE_STRIDE = 4;
78
+ export interface DecodeGraph {
79
+ /** Four words per node. */
80
+ nodes: Uint32Array;
81
+ count: number;
82
+ /** Which register holds the finished channel vector. */
83
+ result: number;
84
+ /** One of `ADDRESS_MODE`. */
85
+ addressMode: number;
86
+ }
87
+ export declare function createDecodeGraph(capacity: number): DecodeGraph;
88
+ export declare function addDecodeNode(graph: DecodeGraph, op: number, a: number, b: number, out: number): number;
89
+ export declare function nodeOp(graph: DecodeGraph, index: number): number;
90
+ export declare function nodeA(graph: DecodeGraph, index: number): number;
91
+ export declare function nodeB(graph: DecodeGraph, index: number): number;
92
+ export declare function nodeOut(graph: DecodeGraph, index: number): number;
93
+ /** The highest register the graph writes, plus one. */
94
+ export declare function graphRegisterCount(graph: DecodeGraph): number;
95
+ /**
96
+ * Whether this graph is one an interpreter can run.
97
+ *
98
+ * Returns a message rather than throwing, matching `validateGraph` in the frame package and for
99
+ * the same reason: whether a malformed graph is an assertion or a skipped material is the
100
+ * caller's decision.
101
+ */
102
+ export declare function validateDecodeGraph(graph: DecodeGraph): string | null;
103
+ /** `count`, `result`, `addressMode`, then the nodes. */
104
+ export declare function encodeDecodeGraph(graph: DecodeGraph): Uint8Array;
105
+ export declare function decodeDecodeGraph(bytes: Uint8Array): DecodeGraph;
@@ -0,0 +1,180 @@
1
+ /**
2
+ * A texture's decoder, as data.
3
+ *
4
+ * **This is the file where the expensive mistake would be made, so the reasoning is here.** The
5
+ * decode program is uploaded as a uniform buffer and walked by one shader. There is no code
6
+ * generation, no `#define`, no shader per material and no variant — because `ARCHITECTURE.md` has
7
+ * already measured what the other way costs, with the number: a fifth permutation flag took the
8
+ * generated WGSL corpus from 914 KB to 1,906 KB and cost **196,910 gzipped bytes on every
9
+ * consumer**, including those who never enabled it, because deflate's window is 32 KB and
10
+ * near-identical permutations do not deduplicate. A decode graph as permutations would be that
11
+ * mistake with a far larger exponent — per material rather than per feature.
12
+ *
13
+ * As data it costs bytes, linearly, and a new operation is one more case in one switch.
14
+ *
15
+ * **A node is four words: op, a, b, out.** Which of `a` and `b` name registers rather than
16
+ * immediates is a property of the operation, held in `OP_ARGS` so that the validator and both
17
+ * interpreters read one table rather than three copies of a convention.
18
+ */
19
+ export const DECODE_OP = {
20
+ /** a = latent slot. Samples it at (u, v). */
21
+ SAMPLE_LATENT: 0,
22
+ /** a = register holding the input vector, b = network slot. */
23
+ EVAL_NETWORK: 1,
24
+ /** a = block-compressed slot. The passthrough for content a network does not help. */
25
+ SAMPLE_BLOCK: 2,
26
+ /** a = seed, b = octaves. */
27
+ PROCEDURAL_FBM: 3,
28
+ /** a = frame count, b = frames per second. Writes the frame index into x. */
29
+ FLIPBOOK_INDEX: 4,
30
+ /** a, b = registers. Mixes them by the fractional part of the sample time. */
31
+ LATENT_LERP: 5,
32
+ /** a = register, b = packed channel spec. Applies the declared convention. */
33
+ REMAP_CHANNEL: 6,
34
+ /** a, b = registers. `a` over `b`. */
35
+ COMPOSITE: 7,
36
+ /**
37
+ * a = constant slot. Writes that four-component value.
38
+ *
39
+ * **Added for Wave 4C, and the reason is the rule rather than the node.** A material graph
40
+ * compiles to this vocabulary and not to a shader, so a node the vocabulary cannot express grows
41
+ * the vocabulary — which costs bytes linearly, because an operation is data. A colour picker is
42
+ * the first node anybody puts in a material graph and there was nothing here that could hold one.
43
+ */
44
+ CONSTANT: 8,
45
+ };
46
+ /** Whether each operation's `a` and `b` name registers. One table, three readers. */
47
+ const OP_ARGS = {
48
+ [DECODE_OP.SAMPLE_LATENT]: { a: false, b: false },
49
+ [DECODE_OP.EVAL_NETWORK]: { a: true, b: false },
50
+ [DECODE_OP.SAMPLE_BLOCK]: { a: false, b: false },
51
+ [DECODE_OP.PROCEDURAL_FBM]: { a: false, b: false },
52
+ [DECODE_OP.FLIPBOOK_INDEX]: { a: false, b: false },
53
+ [DECODE_OP.LATENT_LERP]: { a: true, b: true },
54
+ [DECODE_OP.REMAP_CHANNEL]: { a: true, b: false },
55
+ [DECODE_OP.COMPOSITE]: { a: true, b: true },
56
+ [DECODE_OP.CONSTANT]: { a: false, b: false },
57
+ };
58
+ /**
59
+ * Where a coordinate lands, and what happens outside the unit square.
60
+ *
61
+ * **Two conventions, because two kinds of data need them.** A *lattice* puts `u = 0` on the centre
62
+ * of the first texel and `u = 1` on the centre of the last, which is what a height field sampled at
63
+ * its own vertices wants — `terrainTexture.ts` reads `x / (width - 1)` and lands on texel `x`
64
+ * exactly. A *centre* mode puts `u = 0` on the first texel's edge, which is what every GPU sampler
65
+ * does and what a surface texture tiled across a mesh wants. Added 2026-09-17 so the device
66
+ * interpreter has a reference to agree with; modes 0 and 1 did not move.
67
+ *
68
+ * Lattice wrap has a seam a tiling texture shows: `u = 0.999` reads the last texel and `u = 1`
69
+ * the first. Centre wrap blends across it, as a repeating sampler does.
70
+ */
71
+ export const ADDRESS_MODE = {
72
+ LATTICE_CLAMP: 0,
73
+ LATTICE_WRAP: 1,
74
+ CENTRE_CLAMP: 2,
75
+ CENTRE_WRAP: 3,
76
+ };
77
+ /** How many address modes there are. A graph asking for this many or more is refused. */
78
+ export const ADDRESS_MODE_COUNT = 4;
79
+ /**
80
+ * Registers the interpreter has, each a four-component vector.
81
+ *
82
+ * A budget rather than a limit discovered on one device: a graph that needs more is refused at
83
+ * encode time, where a person can see it, rather than compiling to a shader that fails on the
84
+ * hardware with the smallest uniform space.
85
+ */
86
+ export const MAX_REGISTERS = 16;
87
+ /** Words per node: op, a, b, out. */
88
+ export const NODE_STRIDE = 4;
89
+ export function createDecodeGraph(capacity) {
90
+ return { nodes: new Uint32Array(capacity * NODE_STRIDE), count: 0, result: 0, addressMode: 0 };
91
+ }
92
+ export function addDecodeNode(graph, op, a, b, out) {
93
+ const at = graph.count * NODE_STRIDE;
94
+ graph.nodes[at] = op;
95
+ graph.nodes[at + 1] = a;
96
+ graph.nodes[at + 2] = b;
97
+ graph.nodes[at + 3] = out;
98
+ graph.count += 1;
99
+ return graph.count - 1;
100
+ }
101
+ export function nodeOp(graph, index) {
102
+ return graph.nodes[index * NODE_STRIDE];
103
+ }
104
+ export function nodeA(graph, index) {
105
+ return graph.nodes[index * NODE_STRIDE + 1];
106
+ }
107
+ export function nodeB(graph, index) {
108
+ return graph.nodes[index * NODE_STRIDE + 2];
109
+ }
110
+ export function nodeOut(graph, index) {
111
+ return graph.nodes[index * NODE_STRIDE + 3];
112
+ }
113
+ /** The highest register the graph writes, plus one. */
114
+ export function graphRegisterCount(graph) {
115
+ let highest = graph.result;
116
+ for (let i = 0; i < graph.count; i += 1)
117
+ highest = Math.max(highest, nodeOut(graph, i));
118
+ return highest + 1;
119
+ }
120
+ /**
121
+ * Whether this graph is one an interpreter can run.
122
+ *
123
+ * Returns a message rather than throwing, matching `validateGraph` in the frame package and for
124
+ * the same reason: whether a malformed graph is an assertion or a skipped material is the
125
+ * caller's decision.
126
+ */
127
+ export function validateDecodeGraph(graph) {
128
+ if (!Number.isInteger(graph.addressMode) ||
129
+ graph.addressMode < 0 ||
130
+ graph.addressMode >= ADDRESS_MODE_COUNT) {
131
+ return `address mode ${graph.addressMode} is not one of the ${ADDRESS_MODE_COUNT} the interpreter knows`;
132
+ }
133
+ const written = new Set();
134
+ for (let i = 0; i < graph.count; i += 1) {
135
+ const op = nodeOp(graph, i);
136
+ const args = OP_ARGS[op];
137
+ if (args === undefined)
138
+ return `node ${i} has unknown opcode ${op}`;
139
+ const out = nodeOut(graph, i);
140
+ if (out >= MAX_REGISTERS) {
141
+ return `node ${i} writes register ${out}, past the ${MAX_REGISTERS} the interpreter has`;
142
+ }
143
+ if (args.a && !written.has(nodeA(graph, i))) {
144
+ return `node ${i} reads register ${nodeA(graph, i)}, which nothing wrote`;
145
+ }
146
+ if (args.b && !written.has(nodeB(graph, i))) {
147
+ return `node ${i} reads register ${nodeB(graph, i)}, which nothing wrote`;
148
+ }
149
+ written.add(out);
150
+ }
151
+ if (!written.has(graph.result)) {
152
+ return `the result register ${graph.result} is never written`;
153
+ }
154
+ if (graphRegisterCount(graph) > MAX_REGISTERS) {
155
+ return `the graph needs ${graphRegisterCount(graph)} registers, past the ${MAX_REGISTERS} available`;
156
+ }
157
+ return null;
158
+ }
159
+ /** `count`, `result`, `addressMode`, then the nodes. */
160
+ export function encodeDecodeGraph(graph) {
161
+ const bytes = new Uint8Array(12 + graph.count * NODE_STRIDE * 4);
162
+ const view = new DataView(bytes.buffer);
163
+ view.setUint32(0, graph.count, true);
164
+ view.setUint32(4, graph.result, true);
165
+ view.setUint32(8, graph.addressMode, true);
166
+ new Uint32Array(bytes.buffer, 12, graph.count * NODE_STRIDE).set(graph.nodes.subarray(0, graph.count * NODE_STRIDE));
167
+ return bytes;
168
+ }
169
+ export function decodeDecodeGraph(bytes) {
170
+ const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
171
+ const count = view.getUint32(0, true);
172
+ const graph = createDecodeGraph(Math.max(1, count));
173
+ graph.count = count;
174
+ graph.result = view.getUint32(4, true);
175
+ graph.addressMode = view.getUint32(8, true);
176
+ for (let i = 0; i < count * NODE_STRIDE; i += 1) {
177
+ graph.nodes[i] = view.getUint32(12 + i * 4, true);
178
+ }
179
+ return graph;
180
+ }
package/dist/half.d.ts ADDED
@@ -0,0 +1,24 @@
1
+ /**
2
+ * IEEE 754 binary16 — half precision — on a platform that has no type for it.
3
+ *
4
+ * **The half-precision network path needs a reference, and the reference needs this.** A shader
5
+ * declared `enable f16` computes in sixteen bits, and the only way to say what it should have
6
+ * produced is to round the same arithmetic to the same sixteen bits here. Node 22 has no
7
+ * `Float16Array` and no `Math.f16round`, so the rounding is written out: one sign bit, five exponent
8
+ * bits biased by fifteen, ten mantissa bits, ties to even, overflow to infinity.
9
+ *
10
+ * **What the format cannot hold is the reason the path is optional.** The largest finite value is
11
+ * 65,504 and the smallest normal one is 2^-14, so a network whose activations leave that range is
12
+ * wrong in half precision in a way no tolerance covers — `activationBound` in `inference.ts` is what
13
+ * a consumer asks before choosing it.
14
+ */
15
+ /** The largest finite half-precision value. */
16
+ export declare const HALF_MAX = 65504;
17
+ /** The sixteen bits a value rounds to, as an unsigned integer. */
18
+ export declare function toHalfBits(value: number): number;
19
+ /** The value sixteen bits hold. */
20
+ export declare function fromHalfBits(bits: number): number;
21
+ /** The nearest half-precision value, as a number. */
22
+ export declare function roundHalf(value: number): number;
23
+ /** A whole weight array as half-precision bits, each weight rounded on its own. */
24
+ export declare function halfWeights(weights: Float32Array): Uint16Array;
package/dist/half.js ADDED
@@ -0,0 +1,86 @@
1
+ /**
2
+ * IEEE 754 binary16 — half precision — on a platform that has no type for it.
3
+ *
4
+ * **The half-precision network path needs a reference, and the reference needs this.** A shader
5
+ * declared `enable f16` computes in sixteen bits, and the only way to say what it should have
6
+ * produced is to round the same arithmetic to the same sixteen bits here. Node 22 has no
7
+ * `Float16Array` and no `Math.f16round`, so the rounding is written out: one sign bit, five exponent
8
+ * bits biased by fifteen, ten mantissa bits, ties to even, overflow to infinity.
9
+ *
10
+ * **What the format cannot hold is the reason the path is optional.** The largest finite value is
11
+ * 65,504 and the smallest normal one is 2^-14, so a network whose activations leave that range is
12
+ * wrong in half precision in a way no tolerance covers — `activationBound` in `inference.ts` is what
13
+ * a consumer asks before choosing it.
14
+ */
15
+ /** The largest finite half-precision value. */
16
+ export const HALF_MAX = 65504;
17
+ /* One scratch view, so reading a double's exponent allocates nothing. */
18
+ const BITS = new DataView(new ArrayBuffer(8));
19
+ /*
20
+ * Round to the nearest integer, a tie to the even one. `Math.round` rounds a tie up, which is not
21
+ * what any floating-point format does and would make every tie in the tests an error of one unit.
22
+ */
23
+ function roundEven(value) {
24
+ const floor = Math.floor(value);
25
+ const fraction = value - floor;
26
+ if (fraction > 0.5)
27
+ return floor + 1;
28
+ if (fraction < 0.5)
29
+ return floor;
30
+ return floor % 2 === 0 ? floor : floor + 1;
31
+ }
32
+ /** The sixteen bits a value rounds to, as an unsigned integer. */
33
+ export function toHalfBits(value) {
34
+ if (Number.isNaN(value))
35
+ return 0x7e00;
36
+ const sign = value < 0 || Object.is(value, -0) ? 0x8000 : 0;
37
+ const magnitude = Math.abs(value);
38
+ if (magnitude < 2 ** -14) {
39
+ /*
40
+ * Subnormal: the value in units of 2^-24, which is the mantissa directly. A result of 1,024 is
41
+ * the smallest normal value, and its bits are exactly 1,024 — the carry is free.
42
+ */
43
+ return sign | roundEven(magnitude * 2 ** 24);
44
+ }
45
+ /*
46
+ * The exponent from the double's own bits rather than from `Math.log2`, which the language
47
+ * leaves approximate — exact here, because every magnitude this far down the function is a normal
48
+ * double.
49
+ */
50
+ BITS.setFloat64(0, magnitude);
51
+ let exponent = ((BITS.getUint16(0) >> 4) & 0x7ff) - 1023;
52
+ let mantissa = roundEven((magnitude / 2 ** exponent - 1) * 1024);
53
+ if (mantissa === 1024) {
54
+ mantissa = 0;
55
+ exponent += 1;
56
+ }
57
+ /*
58
+ * Overflow is decided here and only here. 65,520 is halfway between the largest finite value and
59
+ * 2^16, its mantissa rounds up to the carry above, and the exponent it lands on is 16.
60
+ */
61
+ if (exponent > 15)
62
+ return sign | 0x7c00;
63
+ return sign | ((exponent + 15) << 10) | mantissa;
64
+ }
65
+ /** The value sixteen bits hold. */
66
+ export function fromHalfBits(bits) {
67
+ const sign = (bits & 0x8000) !== 0 ? -1 : 1;
68
+ const exponent = (bits >> 10) & 0x1f;
69
+ const mantissa = bits & 0x03ff;
70
+ if (exponent === 0)
71
+ return sign * mantissa * 2 ** -24;
72
+ if (exponent === 0x1f)
73
+ return mantissa === 0 ? sign * Infinity : Number.NaN;
74
+ return sign * (1 + mantissa / 1024) * 2 ** (exponent - 15);
75
+ }
76
+ /** The nearest half-precision value, as a number. */
77
+ export function roundHalf(value) {
78
+ return fromHalfBits(toHalfBits(value));
79
+ }
80
+ /** A whole weight array as half-precision bits, each weight rounded on its own. */
81
+ export function halfWeights(weights) {
82
+ const out = new Uint16Array(weights.length);
83
+ for (let i = 0; i < weights.length; i += 1)
84
+ out[i] = toHalfBits(weights[i]);
85
+ return out;
86
+ }
@@ -0,0 +1,66 @@
1
+ /*! DriftEngine | Copyright 2026 Drift Technologies | Apache-2.0 | https://github.com/drftrun/driftengine */
2
+ /**
3
+ * DriftTexture: a texture as a compiled, sampled field rather than an image.
4
+ *
5
+ * Every engine in the world treats a texture as a bag of texels. This one treats it as a small
6
+ * program sampled at `(uv, t, params)` — one object per *material* rather than per channel,
7
+ * carrying every channel, every level and its own decoder.
8
+ *
9
+ * Three consequences fall out of that shape and none of them is available to an image:
10
+ *
11
+ * - **Channels compress together**, because they are correlated and compressing them apart throws
12
+ * that correlation away.
13
+ * - **Time is an argument, not a clock**, so an animated texture sampled at simulation time is
14
+ * byte-exact under replay — which no other engine's animated texture can claim.
15
+ * - **The decoder is data walked by one shader**, never a permutation. `ARCHITECTURE.md` measured
16
+ * what the other way costs at 196,910 gzipped bytes for a single flag.
17
+ *
18
+ * What ships here is the format, the reference decode, the material binding, predictive
19
+ * residency and the writable overlay. The device interpreter the GPU-driven pipeline samples with
20
+ * is `@driftengine/core`'s, checked against `decodeCpu` by `scripts/gpu-parity.mjs`.
21
+ */
22
+ export { CHANNEL_SEMANTICS, isColour, linearToSrgb, needsVarianceMips, normaliseSample, semanticAt, semanticIndex, srgbToLinear, } from './semantics.ts';
23
+ export type { ChannelSemantic, ChannelSpec } from './semantics.ts';
24
+ export { reduceNormalMip, toksvigRoughness } from './mipNdf.ts';
25
+ export { createTileIndex, hashTile, internTile, tileSlotCount } from './tileHash.ts';
26
+ export type { TileIndex } from './tileHash.ts';
27
+ export { activationBound, evalNetwork, evalNetworkHalf, halfPrecisionErrorBound, networkWeightCount, } from './inference.ts';
28
+ export { HALF_MAX, fromHalfBits, halfWeights, roundHalf, toHalfBits } from './half.ts';
29
+ export { addBias, erf, gelu, layerNorm, matmul, softmax } from './tensor/linear.ts';
30
+ export { attention } from './tensor/attention.ts';
31
+ export { resize } from './tensor/resize.ts';
32
+ export { conv2d, convTranspose2d, maxPool2d, patchEmbed } from './tensor/spatial.ts';
33
+ export { createGraphEvaluator, graphForDevice, graphFromStored, graphShapes, validateGraph, } from './tensor/graph.ts';
34
+ export { CONSTANT_PREFIX, graphFromWeights } from './tensor/architecture.ts';
35
+ export type { Architecture, GraphBuilder, WeightSource, Weights } from './tensor/architecture.ts';
36
+ export type { GraphEvaluator, GraphNode, GraphTensor, GraphValue, NetworkGraph, StoredGraph, } from './tensor/graph.ts';
37
+ export { OPERATORS } from './tensor/operators.ts';
38
+ export type { AttributeValue, Attributes, Operator } from './tensor/operators.ts';
39
+ export type { NetworkShape } from './inference.ts';
40
+ export { flipbookFrame, latentLerpWeights } from './timeNodes.ts';
41
+ export type { LerpWeights } from './timeNodes.ts';
42
+ export { ADDRESS_MODE, ADDRESS_MODE_COUNT, DECODE_OP, MAX_REGISTERS, NODE_STRIDE, addDecodeNode, createDecodeGraph, decodeDecodeGraph, encodeDecodeGraph, graphRegisterCount, nodeA, nodeB, nodeOp, nodeOut, validateDecodeGraph, } from './decodeGraph.ts';
43
+ export type { DecodeGraph, DecodeOp } from './decodeGraph.ts';
44
+ export { REMAP_SEMANTICS, createDecodeRegisters, decodeCpu } from './decodeCpu.ts';
45
+ export type { DecodeResources, LatentImage, LatentLevel } from './decodeCpu.ts';
46
+ export { progressiveOrder, usableAt } from './progressive.ts';
47
+ export { arrayDescriptor, assignLayer, createMaterialArray, layerOf } from './materialArray.ts';
48
+ export type { MaterialArray } from './materialArray.ts';
49
+ export { predictViews } from './residency/predict.ts';
50
+ export type { SimulationHandle } from './residency/predict.ts';
51
+ export { TILE_ABSENT, TILE_REQUESTED, TILE_RESIDENT, createResidencyTable, evict, leastRecentlyUsed, markRequested, markResident, residentCount, slotFor, tileState, touchTile, } from './residency/table.ts';
52
+ export type { ResidencyTable } from './residency/table.ts';
53
+ export { createPrefetchQueue, enqueue, queueSize, takeBatch } from './residency/queue.ts';
54
+ export type { PrefetchQueue } from './residency/queue.ts';
55
+ export { acquirePage, beginCacheFrame, createPageCache, decodeMode, ensurePageDecoded, occupiedPages, pageDecoded, pageSlot, releasePage, setDecodeMode, takeEvicted, } from './residency/pageCache.ts';
56
+ export type { DecodeMode, PageCache, PageCacheOptions, PageDecode } from './residency/pageCache.ts';
57
+ export { clearMissing, createStreamer, forgetMissing, pumpStreamer, streamerInFlight, } from './residency/stream.ts';
58
+ export type { Streamer, StreamerOptions, TileSource } from './residency/stream.ts';
59
+ export { ADDRESS_CLAMP, ADDRESS_WRAP, OVERLAY_CHANNELS, compositeOverlay, createOverlay, overlayTileCount, overlayTiles, sampleOverlay, writeOverlay, writtenMaskAt, } from './overlay/sparse.ts';
60
+ export type { Overlay, OverlayOptions, OverlayTile } from './overlay/sparse.ts';
61
+ export { JOURNAL_ENTRY_BYTES, JOURNAL_MAGIC, JOURNAL_VERSION, applyOverlayJournal, clearOverlay, createOverlayJournal, decodeOverlayJournal, emptyLike, encodeOverlayJournal, journalLength, recordOverlayWrite, rewindOverlay, truncateJournalFrom, } from './overlay/journal.ts';
62
+ export type { OverlayJournal } from './overlay/journal.ts';
63
+ export { runPrediction } from './residency/predictor.ts';
64
+ export { LEVEL_MARGIN, latentTileGrid, tilesForView } from './residency/viewTiles.ts';
65
+ export type { InstanceTileInfo, MaterialTileGrid } from './residency/viewTiles.ts';
66
+ export type { Predictor } from './residency/predictor.ts';
package/dist/index.js ADDED
@@ -0,0 +1,55 @@
1
+ /*! DriftEngine | Copyright 2026 Drift Technologies | Apache-2.0 | https://github.com/drftrun/driftengine */
2
+ /**
3
+ * DriftTexture: a texture as a compiled, sampled field rather than an image.
4
+ *
5
+ * Every engine in the world treats a texture as a bag of texels. This one treats it as a small
6
+ * program sampled at `(uv, t, params)` — one object per *material* rather than per channel,
7
+ * carrying every channel, every level and its own decoder.
8
+ *
9
+ * Three consequences fall out of that shape and none of them is available to an image:
10
+ *
11
+ * - **Channels compress together**, because they are correlated and compressing them apart throws
12
+ * that correlation away.
13
+ * - **Time is an argument, not a clock**, so an animated texture sampled at simulation time is
14
+ * byte-exact under replay — which no other engine's animated texture can claim.
15
+ * - **The decoder is data walked by one shader**, never a permutation. `ARCHITECTURE.md` measured
16
+ * what the other way costs at 196,910 gzipped bytes for a single flag.
17
+ *
18
+ * What ships here is the format, the reference decode, the material binding, predictive
19
+ * residency and the writable overlay. The device interpreter the GPU-driven pipeline samples with
20
+ * is `@driftengine/core`'s, checked against `decodeCpu` by `scripts/gpu-parity.mjs`.
21
+ */
22
+ export { CHANNEL_SEMANTICS, isColour, linearToSrgb, needsVarianceMips, normaliseSample, semanticAt, semanticIndex, srgbToLinear, } from './semantics.js';
23
+ export { reduceNormalMip, toksvigRoughness } from './mipNdf.js';
24
+ export { createTileIndex, hashTile, internTile, tileSlotCount } from './tileHash.js';
25
+ export { activationBound, evalNetwork, evalNetworkHalf, halfPrecisionErrorBound, networkWeightCount, } from './inference.js';
26
+ export { HALF_MAX, fromHalfBits, halfWeights, roundHalf, toHalfBits } from './half.js';
27
+ /*
28
+ * The operators a small transformer is built from, as references the device kernels are held to:
29
+ * the second half of the one inference runtime, beside the perceptron's evaluation above.
30
+ */
31
+ export { addBias, erf, gelu, layerNorm, matmul, softmax } from './tensor/linear.js';
32
+ export { attention } from './tensor/attention.js';
33
+ export { resize } from './tensor/resize.js';
34
+ export { conv2d, convTranspose2d, maxPool2d, patchEmbed } from './tensor/spatial.js';
35
+ export { createGraphEvaluator, graphForDevice, graphFromStored, graphShapes, validateGraph, } from './tensor/graph.js';
36
+ export { CONSTANT_PREFIX, graphFromWeights } from './tensor/architecture.js';
37
+ export { OPERATORS } from './tensor/operators.js';
38
+ export { flipbookFrame, latentLerpWeights } from './timeNodes.js';
39
+ export { ADDRESS_MODE, ADDRESS_MODE_COUNT, DECODE_OP, MAX_REGISTERS, NODE_STRIDE, addDecodeNode, createDecodeGraph, decodeDecodeGraph, encodeDecodeGraph, graphRegisterCount, nodeA, nodeB, nodeOp, nodeOut, validateDecodeGraph, } from './decodeGraph.js';
40
+ export { REMAP_SEMANTICS, createDecodeRegisters, decodeCpu } from './decodeCpu.js';
41
+ export { progressiveOrder, usableAt } from './progressive.js';
42
+ export { arrayDescriptor, assignLayer, createMaterialArray, layerOf } from './materialArray.js';
43
+ /*
44
+ * Predictive residency: the half that needs the simulation's save and restore rather than the
45
+ * renderer. See `residency/predict.ts` for why looking ahead is available here and nowhere else.
46
+ */
47
+ export { predictViews } from './residency/predict.js';
48
+ export { TILE_ABSENT, TILE_REQUESTED, TILE_RESIDENT, createResidencyTable, evict, leastRecentlyUsed, markRequested, markResident, residentCount, slotFor, tileState, touchTile, } from './residency/table.js';
49
+ export { createPrefetchQueue, enqueue, queueSize, takeBatch } from './residency/queue.js';
50
+ export { acquirePage, beginCacheFrame, createPageCache, decodeMode, ensurePageDecoded, occupiedPages, pageDecoded, pageSlot, releasePage, setDecodeMode, takeEvicted, } from './residency/pageCache.js';
51
+ export { clearMissing, createStreamer, forgetMissing, pumpStreamer, streamerInFlight, } from './residency/stream.js';
52
+ export { ADDRESS_CLAMP, ADDRESS_WRAP, OVERLAY_CHANNELS, compositeOverlay, createOverlay, overlayTileCount, overlayTiles, sampleOverlay, writeOverlay, writtenMaskAt, } from './overlay/sparse.js';
53
+ export { JOURNAL_ENTRY_BYTES, JOURNAL_MAGIC, JOURNAL_VERSION, applyOverlayJournal, clearOverlay, createOverlayJournal, decodeOverlayJournal, emptyLike, encodeOverlayJournal, journalLength, recordOverlayWrite, rewindOverlay, truncateJournalFrom, } from './overlay/journal.js';
54
+ export { runPrediction } from './residency/predictor.js';
55
+ export { LEVEL_MARGIN, latentTileGrid, tilesForView } from './residency/viewTiles.js';
@@ -0,0 +1,53 @@
1
+ export interface NetworkShape {
2
+ readonly inputs: number;
3
+ readonly hidden: readonly number[];
4
+ readonly outputs: number;
5
+ }
6
+ /** How many weights and biases a shape needs, laid out layer by layer. */
7
+ export declare function networkWeightCount(shape: NetworkShape): number;
8
+ /**
9
+ * Evaluate the network, writing `shape.outputs` values into `out`.
10
+ *
11
+ * Weights are read in layer order: for each layer, the weight matrix row-major by output, then the
12
+ * biases. `scratch` must hold at least twice the widest layer and is reused across calls so this
13
+ * allocates nothing.
14
+ */
15
+ export declare function evalNetwork(shape: NetworkShape, weights: Float32Array, input: Float32Array, out: Float32Array, scratch: Float32Array): void;
16
+ /**
17
+ * How far the half-precision evaluation of this network, at this input, can sit from the
18
+ * single-precision one.
19
+ *
20
+ * **A bound per network rather than one tolerance for all**, because a sampled tolerance is a claim
21
+ * about the corpus that sampled it — 2,400 networks put the worst relative error at 0.00213, the
22
+ * device's own corpus reached 0.0039, and 40,000 reached 0.0066. This carries the error instead:
23
+ * the weights' and the input's own rounding exactly, then half a unit in the last place for every
24
+ * product, every partial sum and the bias, through each layer — a rectifier moves no error, since
25
+ * it is one-Lipschitz. **It holds whichever form a device computes**: every operation rounded, or
26
+ * each multiply-add contracted into a single rounding, which WGSL permits and this project's
27
+ * development machine does. It assumes no value overflows, which `activationBound` is for.
28
+ */
29
+ export declare function halfPrecisionErrorBound(shape: NetworkShape, weights: Float32Array, input: Float32Array): number;
30
+ /**
31
+ * Evaluate the network as a device with `enable f16` would, writing `shape.outputs` values into
32
+ * `out`.
33
+ *
34
+ * **Every multiply and every add is rounded**, because that is what a sixteen-bit adder does, and
35
+ * the arithmetic is in doubles so the rounding is the only approximation. The order is the
36
+ * device's: the weighted inputs summed in index order, then the bias, then the rectifier —
37
+ * `render/shaders/network.wgsl.ts` in `@driftengine/core` is written to match it.
38
+ *
39
+ * `weights` are half-precision bits, as `halfWeights` produces and an `NNET` chunk stores. Inputs
40
+ * are rounded to half precision on the way in. `scratch` holds at least twice the widest layer.
41
+ */
42
+ export declare function evalNetworkHalf(shape: NetworkShape, weights: Uint16Array, input: Float32Array, out: Float32Array, scratch: Float64Array, contract?: boolean): void;
43
+ /**
44
+ * The largest magnitude any product, partial sum or output of the network can reach over the given
45
+ * input box.
46
+ *
47
+ * **What a consumer asks before it chooses half precision.** Past 65,504 the format has no finite
48
+ * value, and a partial sum can pass that on its way to a total that does not — so this bounds each
49
+ * neuron by its bias plus the sum of every weight times the largest magnitude its input can take,
50
+ * which covers every prefix of the sum at once. Intervals are carried through the layers, and a
51
+ * rectifier clips them, which is what keeps the bound from growing without cause.
52
+ */
53
+ export declare function activationBound(shape: NetworkShape, weights: Float32Array, low: Float32Array, high: Float32Array): number;