@driftengine/texture 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/LICENSE +202 -0
  2. package/NOTICE +29 -0
  3. package/README.md +106 -0
  4. package/dist/decodeCpu.d.ts +59 -0
  5. package/dist/decodeCpu.js +234 -0
  6. package/dist/decodeGraph.d.ts +105 -0
  7. package/dist/decodeGraph.js +180 -0
  8. package/dist/half.d.ts +24 -0
  9. package/dist/half.js +86 -0
  10. package/dist/index.d.ts +66 -0
  11. package/dist/index.js +55 -0
  12. package/dist/inference.d.ts +53 -0
  13. package/dist/inference.js +243 -0
  14. package/dist/materialArray.d.ts +38 -0
  15. package/dist/materialArray.js +40 -0
  16. package/dist/mipNdf.d.ts +29 -0
  17. package/dist/mipNdf.js +53 -0
  18. package/dist/overlay/journal.d.ts +78 -0
  19. package/dist/overlay/journal.js +171 -0
  20. package/dist/overlay/sparse.d.ts +68 -0
  21. package/dist/overlay/sparse.js +212 -0
  22. package/dist/progressive.d.ts +30 -0
  23. package/dist/progressive.js +56 -0
  24. package/dist/residency/pageCache.d.ts +103 -0
  25. package/dist/residency/pageCache.js +184 -0
  26. package/dist/residency/predict.d.ts +55 -0
  27. package/dist/residency/predict.js +51 -0
  28. package/dist/residency/predictor.d.ts +16 -0
  29. package/dist/residency/predictor.js +44 -0
  30. package/dist/residency/queue.d.ts +26 -0
  31. package/dist/residency/queue.js +52 -0
  32. package/dist/residency/stream.d.ts +66 -0
  33. package/dist/residency/stream.js +142 -0
  34. package/dist/residency/table.d.ts +36 -0
  35. package/dist/residency/table.js +72 -0
  36. package/dist/residency/viewTiles.d.ts +108 -0
  37. package/dist/residency/viewTiles.js +419 -0
  38. package/dist/semantics.d.ts +52 -0
  39. package/dist/semantics.js +76 -0
  40. package/dist/tensor/architecture.d.ts +53 -0
  41. package/dist/tensor/architecture.js +96 -0
  42. package/dist/tensor/attention.d.ts +5 -0
  43. package/dist/tensor/attention.js +62 -0
  44. package/dist/tensor/denseOperators.d.ts +2 -0
  45. package/dist/tensor/denseOperators.js +136 -0
  46. package/dist/tensor/graph.d.ts +83 -0
  47. package/dist/tensor/graph.js +175 -0
  48. package/dist/tensor/linear.d.ts +49 -0
  49. package/dist/tensor/linear.js +136 -0
  50. package/dist/tensor/operatorKit.d.ts +27 -0
  51. package/dist/tensor/operatorKit.js +45 -0
  52. package/dist/tensor/operators.d.ts +3 -0
  53. package/dist/tensor/operators.js +24 -0
  54. package/dist/tensor/resize.d.ts +6 -0
  55. package/dist/tensor/resize.js +107 -0
  56. package/dist/tensor/reuse.d.ts +33 -0
  57. package/dist/tensor/reuse.js +59 -0
  58. package/dist/tensor/shapeOperators.d.ts +3 -0
  59. package/dist/tensor/shapeOperators.js +173 -0
  60. package/dist/tensor/spatial.d.ts +34 -0
  61. package/dist/tensor/spatial.js +131 -0
  62. package/dist/tensor/spatialOperators.d.ts +2 -0
  63. package/dist/tensor/spatialOperators.js +138 -0
  64. package/dist/tileHash.d.ts +29 -0
  65. package/dist/tileHash.js +50 -0
  66. package/dist/timeNodes.d.ts +26 -0
  67. package/dist/timeNodes.js +48 -0
  68. package/package.json +59 -0
  69. package/src/decodeCpu.ts +308 -0
  70. package/src/decodeGraph.ts +214 -0
  71. package/src/half.ts +86 -0
  72. package/src/index.ts +175 -0
  73. package/src/inference.ts +278 -0
  74. package/src/materialArray.ts +67 -0
  75. package/src/mipNdf.ts +63 -0
  76. package/src/overlay/journal.ts +218 -0
  77. package/src/overlay/sparse.ts +275 -0
  78. package/src/progressive.ts +60 -0
  79. package/src/residency/pageCache.ts +233 -0
  80. package/src/residency/predict.ts +74 -0
  81. package/src/residency/predictor.ts +62 -0
  82. package/src/residency/queue.ts +78 -0
  83. package/src/residency/stream.ts +194 -0
  84. package/src/residency/table.ts +89 -0
  85. package/src/residency/viewTiles.ts +553 -0
  86. package/src/semantics.ts +114 -0
  87. package/src/tensor/architecture.ts +140 -0
  88. package/src/tensor/attention.ts +75 -0
  89. package/src/tensor/denseOperators.ts +153 -0
  90. package/src/tensor/graph.ts +244 -0
  91. package/src/tensor/linear.ts +153 -0
  92. package/src/tensor/operatorKit.ts +76 -0
  93. package/src/tensor/operators.ts +28 -0
  94. package/src/tensor/resize.ts +140 -0
  95. package/src/tensor/shapeOperators.ts +173 -0
  96. package/src/tensor/spatial.ts +178 -0
  97. package/src/tensor/spatialOperators.ts +182 -0
  98. package/src/tileHash.ts +60 -0
  99. package/src/timeNodes.ts +60 -0
@@ -0,0 +1,244 @@
1
+ /**
2
+ * A network as a graph of the runtime's operators, checked before it runs and evaluated on the CPU.
3
+ *
4
+ * **Validation says what is wrong in words, and names what it is about**: an operator the runtime
5
+ * lacks, a value read before anything writes it, a shape that does not fit the node that reads it,
6
+ * an output nothing produces. A graph is checked when it is loaded — a model is refused then, with
7
+ * the reason, rather than at its first frame with a picture of zeros.
8
+ *
9
+ * **The evaluator is the reference**, which the device runner is held to: it sizes every value and
10
+ * its scratch once, when it is created, and a run writes into those buffers and allocates nothing.
11
+ * What it gives up is memory: every intermediate value keeps its own buffer, which is the plainest
12
+ * thing to compare against and the device runner's job to improve on.
13
+ */
14
+ import { fromHalfBits } from '../half.ts';
15
+ import { OPERATORS, type Attributes } from './operators.ts';
16
+ import { planReuse } from '@driftengine/core';
17
+
18
+ export interface GraphValue {
19
+ readonly name: string;
20
+ readonly shape: readonly number[];
21
+ }
22
+
23
+ export interface GraphNode {
24
+ readonly op: string;
25
+ readonly inputs: readonly string[];
26
+ readonly output: string;
27
+ readonly attributes: Attributes;
28
+ }
29
+
30
+ export interface GraphTensor {
31
+ readonly shape: readonly number[];
32
+ readonly data: Float32Array;
33
+ }
34
+
35
+ export interface NetworkGraph {
36
+ readonly inputs: readonly GraphValue[];
37
+ readonly outputs: readonly string[];
38
+ readonly nodes: readonly GraphNode[];
39
+ readonly tensors: ReadonlyMap<string, GraphTensor>;
40
+ }
41
+
42
+ const size = (shape: readonly number[]): number => shape.reduce((total, d) => total * d, 1);
43
+
44
+ /** Every value's shape, or the first reason the graph cannot run. */
45
+ function inferShapes(graph: NetworkGraph): Map<string, readonly number[]> | string {
46
+ const shapes = new Map<string, readonly number[]>();
47
+ for (const input of graph.inputs) shapes.set(input.name, input.shape);
48
+ for (const [name, tensor] of graph.tensors) {
49
+ if (tensor.data.length !== size(tensor.shape)) {
50
+ return `tensor "${name}" holds ${tensor.data.length} values and its shape [${tensor.shape.join(', ')}] needs ${size(tensor.shape)}`;
51
+ }
52
+ shapes.set(name, tensor.shape);
53
+ }
54
+ for (const node of graph.nodes) {
55
+ const operator = OPERATORS.get(node.op);
56
+ if (operator === undefined) {
57
+ return `the runtime has no operator "${node.op}" (writing "${node.output}")`;
58
+ }
59
+ const [fewest, most] = operator.arity;
60
+ if (node.inputs.length < fewest || node.inputs.length > most) {
61
+ return `${node.op} writing "${node.output}" takes ${fewest} to ${most} inputs and was given ${node.inputs.length}`;
62
+ }
63
+ const inputShapes: (readonly number[])[] = [];
64
+ for (const input of node.inputs) {
65
+ const shape = shapes.get(input);
66
+ if (shape === undefined)
67
+ return `${node.op} writing "${node.output}" reads "${input}", which nothing writes`;
68
+ inputShapes.push(shape);
69
+ }
70
+ if (shapes.has(node.output)) return `"${node.output}" is written twice`;
71
+ const ranks = operator.ranks ?? [];
72
+ for (let i = 0; i < inputShapes.length && i < ranks.length; i += 1) {
73
+ const rank = (inputShapes[i] as readonly number[]).length;
74
+ if (rank !== ranks[i]) {
75
+ return `${node.op} writing "${node.output}": "${node.inputs[i]}" has rank ${rank} and this input takes rank ${ranks[i]}`;
76
+ }
77
+ }
78
+ const shape = operator.shape(inputShapes, node.attributes);
79
+ if (typeof shape === 'string') return `${node.op} writing "${node.output}": ${shape}`;
80
+ shapes.set(node.output, shape);
81
+ }
82
+ for (const output of graph.outputs) {
83
+ if (!shapes.has(output)) return `the output "${output}" is written by nothing`;
84
+ }
85
+ return shapes;
86
+ }
87
+
88
+ /**
89
+ * A graph as a file stores it: its tensors in either precision, half as the bits a device uploads.
90
+ * `@driftengine/drft`'s `DrftGraph` is one, structurally, which is how the two packages meet
91
+ * without either importing the other.
92
+ */
93
+ export interface StoredGraph {
94
+ readonly inputs: readonly GraphValue[];
95
+ readonly outputs: readonly string[];
96
+ readonly nodes: readonly GraphNode[];
97
+ readonly tensors: readonly {
98
+ readonly name: string;
99
+ readonly shape: readonly number[];
100
+ readonly data: Float32Array | Uint16Array;
101
+ }[];
102
+ }
103
+
104
+ /**
105
+ * A stored graph as one the runtime runs, half-precision tensors decoded to single; validated, and
106
+ * refused naming what is wrong — an operator the runtime lacks is a fact about the model, and this
107
+ * is when a caller learns it.
108
+ */
109
+ export function graphFromStored(stored: StoredGraph): NetworkGraph {
110
+ const tensors = new Map<string, GraphTensor>();
111
+ for (const tensor of stored.tensors) {
112
+ const data =
113
+ tensor.data instanceof Uint16Array
114
+ ? Float32Array.from(tensor.data, (bits) => fromHalfBits(bits))
115
+ : tensor.data;
116
+ tensors.set(tensor.name, { shape: tensor.shape, data });
117
+ }
118
+ const graph: NetworkGraph = {
119
+ inputs: stored.inputs,
120
+ outputs: stored.outputs,
121
+ nodes: stored.nodes,
122
+ tensors,
123
+ };
124
+ const problem = validateGraph(graph);
125
+ if (problem !== null) throw new Error(`network graph: ${problem}`);
126
+ return graph;
127
+ }
128
+
129
+ /** Every value's shape — inputs, tensors and intermediates — refused, with the reason, if invalid. */
130
+ export function graphShapes(graph: NetworkGraph): ReadonlyMap<string, readonly number[]> {
131
+ const shapes = inferShapes(graph);
132
+ if (typeof shapes === 'string') throw new Error(`network graph: ${shapes}`);
133
+ return shapes;
134
+ }
135
+
136
+ /**
137
+ * A graph as `@driftengine/core`'s runner takes it: its tensors listed by name, and every value's
138
+ * shape as inferred here — **nothing evaluated and nothing sized**, where asking an evaluator for its
139
+ * shapes would allocate every intermediate value first. Structural, because core does not import
140
+ * this package; the shapes are this module's, so there is one copy of every operator's rule.
141
+ */
142
+ export function graphForDevice(graph: NetworkGraph): {
143
+ readonly inputs: readonly GraphValue[];
144
+ readonly outputs: readonly string[];
145
+ readonly nodes: readonly GraphNode[];
146
+ readonly tensors: readonly {
147
+ readonly name: string;
148
+ readonly shape: readonly number[];
149
+ readonly data: Float32Array;
150
+ }[];
151
+ readonly shapes: ReadonlyMap<string, readonly number[]>;
152
+ } {
153
+ return {
154
+ inputs: graph.inputs,
155
+ outputs: graph.outputs,
156
+ nodes: graph.nodes,
157
+ tensors: [...graph.tensors].map(([name, tensor]) => ({ name, ...tensor })),
158
+ shapes: graphShapes(graph),
159
+ };
160
+ }
161
+
162
+ /** Null when the graph can run; otherwise the reason, naming what it is about. */
163
+ export function validateGraph(graph: NetworkGraph): string | null {
164
+ const shapes = inferShapes(graph);
165
+ return typeof shapes === 'string' ? shapes : null;
166
+ }
167
+
168
+ export interface GraphEvaluator {
169
+ /** Every value's shape, inputs, tensors and intermediates alike. */
170
+ readonly shapes: ReadonlyMap<string, readonly number[]>;
171
+ /**
172
+ * Evaluate with these inputs. The returned arrays are the evaluator's own and are overwritten by
173
+ * the next run.
174
+ */
175
+ run(inputs: ReadonlyMap<string, Float32Array>): Map<string, Float32Array>;
176
+ }
177
+
178
+ /**
179
+ * Validate, size every value once, and return something that evaluates without allocating.
180
+ *
181
+ * **Values that cannot be alive at once share a buffer**, by `planReuse` — the rule the device's
182
+ * runner plans by too. Weights and inputs stay in the caller's own arrays and are planned by
183
+ * neither. What it buys is the difference between holding every value of a graph and holding the
184
+ * few alive at any moment: OWLv2's image graph at 960² is 3,647 MB of values and 196 MB in ten buffers,
185
+ * which is the difference between a reference that runs on an ordinary machine and one that is
186
+ * killed for its memory.
187
+ */
188
+ export function createGraphEvaluator(graph: NetworkGraph): GraphEvaluator {
189
+ const shapes = inferShapes(graph);
190
+ if (typeof shapes === 'string') throw new Error(`network graph: ${shapes}`);
191
+ const buffers = new Map<string, Float32Array>();
192
+ for (const [name, tensor] of graph.tensors) buffers.set(name, tensor.data);
193
+ const plan = planReuse({ nodes: graph.nodes, held: [], outputs: graph.outputs }, (name) =>
194
+ size(shapes.get(name) as readonly number[]),
195
+ );
196
+ const slots = plan.sizes.map((values) => new Float32Array(values));
197
+ /* A buffer is as large as the largest value it holds, so each value takes a view of its own. */
198
+ for (const [name, slot] of plan.slotOf) {
199
+ const room = size(shapes.get(name) as readonly number[]);
200
+ const held = slots[slot] as Float32Array;
201
+ buffers.set(name, held.length === room ? held : held.subarray(0, room));
202
+ }
203
+ let scratchSize = 0;
204
+ for (const node of graph.nodes) {
205
+ const operator = OPERATORS.get(node.op);
206
+ const inputShapes = node.inputs.map((input) => shapes.get(input) as readonly number[]);
207
+ scratchSize = Math.max(scratchSize, operator?.scratch?.(inputShapes, node.attributes) ?? 0);
208
+ }
209
+ const scratch = new Float32Array(scratchSize);
210
+ const outputs = new Map<string, Float32Array>();
211
+
212
+ return {
213
+ shapes,
214
+ run(inputs) {
215
+ for (const input of graph.inputs) {
216
+ const data = inputs.get(input.name);
217
+ if (data === undefined)
218
+ throw new Error(`network graph: no value for the input "${input.name}"`);
219
+ if (data.length !== size(input.shape)) {
220
+ throw new Error(
221
+ `network graph: "${input.name}" holds ${data.length} values and needs ${size(input.shape)}`,
222
+ );
223
+ }
224
+ buffers.set(input.name, data);
225
+ }
226
+ for (const node of graph.nodes) {
227
+ const operator = OPERATORS.get(node.op);
228
+ if (operator === undefined) continue;
229
+ const values = node.inputs.map((input) => buffers.get(input) as Float32Array);
230
+ const inputShapes = node.inputs.map((input) => shapes.get(input) as readonly number[]);
231
+ operator.evaluate(
232
+ values,
233
+ inputShapes,
234
+ node.attributes,
235
+ buffers.get(node.output) as Float32Array,
236
+ scratch,
237
+ );
238
+ }
239
+ outputs.clear();
240
+ for (const output of graph.outputs) outputs.set(output, buffers.get(output) as Float32Array);
241
+ return outputs;
242
+ },
243
+ };
244
+ }
@@ -0,0 +1,153 @@
1
+ /**
2
+ * The dense operators a transformer is built from: matrix multiply, bias, layer norm, GELU, softmax.
3
+ *
4
+ * **The references the device kernels are held to**, which is why they are written for exactness
5
+ * rather than speed: every sum accumulates in double precision and is rounded once, on the way into
6
+ * the caller's single-precision output. A kernel that disagrees with one of these is wrong, not
7
+ * this.
8
+ *
9
+ * **Contiguous row-major arrays with their dimensions, and nothing that allocates.** A strided view
10
+ * that permutes without copying would make every operator handle strides, and every device kernel
11
+ * would materialise the permutation anyway; so a permutation is a copy into a buffer the caller
12
+ * owns, and the operators stay one loop each.
13
+ *
14
+ * **The exact GELU, `x·Φ(x)`, and never the tanh approximation.** The two differ by 4e-4 at 3, and
15
+ * a port of a network trained with one and evaluated with the other is a different network. `erf`
16
+ * is here for that reason, accurate to double precision.
17
+ */
18
+
19
+ /**
20
+ * `out[m×n] = a[m×k] · b`, where `b` is `k×n`, or `n×k` read as its transpose when `transposeB`.
21
+ * The offsets are where each matrix starts in its array, which is how one head of many is taken.
22
+ */
23
+ export function matmul(
24
+ out: Float32Array,
25
+ a: Float32Array,
26
+ b: Float32Array,
27
+ m: number,
28
+ k: number,
29
+ n: number,
30
+ transposeB = false,
31
+ outAt = 0,
32
+ aAt = 0,
33
+ bAt = 0,
34
+ ): void {
35
+ for (let row = 0; row < m; row += 1) {
36
+ const aRow = aAt + row * k;
37
+ for (let col = 0; col < n; col += 1) {
38
+ let sum = 0;
39
+ if (transposeB) {
40
+ const bRow = bAt + col * k;
41
+ for (let i = 0; i < k; i += 1) sum += (a[aRow + i] as number) * (b[bRow + i] as number);
42
+ } else {
43
+ for (let i = 0; i < k; i += 1) {
44
+ sum += (a[aRow + i] as number) * (b[bAt + i * n + col] as number);
45
+ }
46
+ }
47
+ out[outAt + row * n + col] = sum;
48
+ }
49
+ }
50
+ }
51
+
52
+ /** Add `bias[cols]` to every one of `rows` rows of `x`, in place. */
53
+ export function addBias(x: Float32Array, rows: number, cols: number, bias: Float32Array): void {
54
+ for (let row = 0; row < rows; row += 1) {
55
+ for (let col = 0; col < cols; col += 1) {
56
+ x[row * cols + col] = (x[row * cols + col] as number) + (bias[col] as number);
57
+ }
58
+ }
59
+ }
60
+
61
+ /**
62
+ * Each row normalised to zero mean and unit variance, then scaled by `gamma` and shifted by `beta`.
63
+ *
64
+ * **The population variance, divided by the row's length and not one less**, which is what the
65
+ * upstream frameworks compute and therefore what their trained weights expect.
66
+ */
67
+ export function layerNorm(
68
+ x: Float32Array,
69
+ rows: number,
70
+ cols: number,
71
+ gamma: Float32Array,
72
+ beta: Float32Array,
73
+ epsilon: number,
74
+ out: Float32Array,
75
+ ): void {
76
+ for (let row = 0; row < rows; row += 1) {
77
+ const at = row * cols;
78
+ let mean = 0;
79
+ for (let col = 0; col < cols; col += 1) mean += x[at + col] as number;
80
+ mean /= cols;
81
+ let variance = 0;
82
+ for (let col = 0; col < cols; col += 1) variance += ((x[at + col] as number) - mean) ** 2;
83
+ variance /= cols;
84
+ const scale = 1 / Math.sqrt(variance + epsilon);
85
+ for (let col = 0; col < cols; col += 1) {
86
+ const normal = variance + epsilon === 0 ? 0 : ((x[at + col] as number) - mean) * scale;
87
+ out[at + col] = normal * (gamma[col] as number) + (beta[col] as number);
88
+ }
89
+ }
90
+ }
91
+
92
+ const TWO_OVER_ROOT_PI = 2 / Math.sqrt(Math.PI);
93
+
94
+ /**
95
+ * The error function, to double precision.
96
+ *
97
+ * A Maclaurin series below 3, where its alternating terms peak near 170 at `x = 3` and cost about
98
+ * three digits to cancellation — still 1e-13; a continued fraction for the complement above, which
99
+ * converges fastest exactly where the series is worst; and ±1 past 6, where the complement is below
100
+ * a double's last place.
101
+ */
102
+ export function erf(x: number): number {
103
+ const sign = x < 0 ? -1 : 1;
104
+ const a = Math.abs(x);
105
+ if (a >= 6) return sign;
106
+ if (a < 3) {
107
+ let term = a;
108
+ let sum = a;
109
+ const square = a * a;
110
+ for (let n = 1; n < 100; n += 1) {
111
+ term *= -square / n;
112
+ const next = term / (2 * n + 1);
113
+ sum += next;
114
+ if (Math.abs(next) < 1e-17 * Math.abs(sum)) break;
115
+ }
116
+ return sign * TWO_OVER_ROOT_PI * sum;
117
+ }
118
+ /* erfc(a) = exp(−a²)/√π · 1/(a + ½/(a + 1/(a + 3/2/(a + ...)))), evaluated from the tail up. */
119
+ let fraction = 0;
120
+ for (let n = 60; n >= 1; n -= 1) fraction = n / 2 / (a + fraction);
121
+ const complement = Math.exp(-a * a) / Math.sqrt(Math.PI) / (a + fraction);
122
+ return sign * (1 - complement);
123
+ }
124
+
125
+ /** `x·Φ(x)`, the exact GELU, from `x` into `out`. */
126
+ export function gelu(x: Float32Array, out: Float32Array): void {
127
+ for (let i = 0; i < x.length; i += 1) {
128
+ const value = x[i] as number;
129
+ out[i] = 0.5 * value * (1 + erf(value / Math.SQRT2));
130
+ }
131
+ }
132
+
133
+ /** The logistic, `1/(1 + e^−x)`, from `x` into `out`: 1 far to the right, and 0 far to the left. */
134
+ export function sigmoid(x: Float32Array, out: Float32Array): void {
135
+ for (let i = 0; i < x.length; i += 1) out[i] = 1 / (1 + Math.exp(-(x[i] as number)));
136
+ }
137
+
138
+ /**
139
+ * Each row of `x` to a distribution, stably: the row's largest value is subtracted before the
140
+ * exponential, so logits near a thousand give their logistic rather than `Infinity / Infinity`.
141
+ */
142
+ export function softmax(x: Float32Array, rows: number, cols: number, out: Float32Array): void {
143
+ for (let row = 0; row < rows; row += 1) {
144
+ const at = row * cols;
145
+ let largest = -Infinity;
146
+ for (let col = 0; col < cols; col += 1) largest = Math.max(largest, x[at + col] as number);
147
+ let total = 0;
148
+ for (let col = 0; col < cols; col += 1) total += Math.exp((x[at + col] as number) - largest);
149
+ for (let col = 0; col < cols; col += 1) {
150
+ out[at + col] = Math.exp((x[at + col] as number) - largest) / total;
151
+ }
152
+ }
153
+ }
@@ -0,0 +1,76 @@
1
+ /**
2
+ * What every operator table shares: the shape of an operator, and the few helpers they all read
3
+ * their attributes and shapes with.
4
+ */
5
+ export type AttributeValue = number | string | boolean | readonly number[];
6
+ export type Attributes = Readonly<Record<string, AttributeValue>>;
7
+ export type Shape = readonly number[];
8
+
9
+ export interface Operator {
10
+ /** How many inputs it takes, fewest and most. */
11
+ readonly arity: readonly [number, number];
12
+ /**
13
+ * The rank each input must have, in order, where it matters. **Checked before `shape` runs**,
14
+ * because an operator reading an image as `[channels, height, width]` reads a value with one more
15
+ * axis as its first three and validates it — and then declares an output it fills only part of.
16
+ */
17
+ readonly ranks?: readonly number[];
18
+ shape(inputs: readonly Shape[], attributes: Attributes): number[] | string;
19
+ /** Scratch values it needs beyond its output, if any. */
20
+ scratch?(inputs: readonly Shape[], attributes: Attributes): number;
21
+ evaluate(
22
+ inputs: readonly Float32Array[],
23
+ shapes: readonly Shape[],
24
+ attributes: Attributes,
25
+ out: Float32Array,
26
+ scratch: Float32Array,
27
+ ): void;
28
+ }
29
+
30
+ export const num = (attributes: Attributes, name: string, fallback?: number): number => {
31
+ const value = attributes[name];
32
+ if (typeof value === 'number') return value;
33
+ if (fallback !== undefined) return fallback;
34
+ throw new RangeError(`attribute "${name}" must be a number`);
35
+ };
36
+
37
+ export const list = (attributes: Attributes, name: string): readonly number[] => {
38
+ const value = attributes[name];
39
+ return Array.isArray(value) ? (value as readonly number[]) : [];
40
+ };
41
+
42
+ export const same = (a: Shape, b: Shape): boolean =>
43
+ a.length === b.length && a.every((d, i) => d === b[i]);
44
+
45
+ export const elementwise = (apply: (a: number, b: number) => number): Operator => ({
46
+ arity: [2, 2],
47
+ shape: ([a, b]) => {
48
+ const x = a as Shape;
49
+ const y = b as Shape;
50
+ if (same(x, y) || (y.length === 1 && y[0] === x[x.length - 1])) return [...x];
51
+ return `[${y.join(', ')}] is neither [${x.join(', ')}] nor its last dimension`;
52
+ },
53
+ evaluate: ([a, b], _shapes, _attributes, out) => {
54
+ const left = a as Float32Array;
55
+ const right = b as Float32Array;
56
+ const n = right.length;
57
+ for (let i = 0; i < left.length; i += 1) {
58
+ out[i] = apply(left[i] as number, right[i % n] as number);
59
+ }
60
+ },
61
+ });
62
+
63
+ /* Strides of a row-major shape. */
64
+ export function strides(shape: Shape): number[] {
65
+ const out = new Array<number>(shape.length).fill(1);
66
+ for (let d = shape.length - 2; d >= 0; d -= 1) {
67
+ out[d] = (out[d + 1] as number) * (shape[d + 1] as number);
68
+ }
69
+ return out;
70
+ }
71
+
72
+ export const product = (shape: Shape, from = 0, to = shape.length): number => {
73
+ let total = 1;
74
+ for (let d = from; d < to; d += 1) total *= shape[d] as number;
75
+ return total;
76
+ };
@@ -0,0 +1,28 @@
1
+ /**
2
+ * The operators a graph may name: what each one's output shape is, and how it is evaluated.
3
+ *
4
+ * **One table, and a graph naming anything not in it is refused at validation, by name** — the
5
+ * runtime lacking an operator is a fact about the model, and a caller learns it when the model is
6
+ * loaded rather than at its first frame. Adding an operator is adding a row here, with its reference
7
+ * in `linear.ts`, `attention.ts` or `spatial.ts` and its device kernel in core.
8
+ *
9
+ * **`linear` is the upstream layer, `x·Wᵀ + b`, with its weight in the upstream layout `[out][in]`,
10
+ * and the bias added in double precision before the one rounding** — which is what makes a
11
+ * perceptron written as a graph agree with `evalNetwork` bit for bit, and is the evidence there is
12
+ * one runtime rather than two.
13
+ *
14
+ * Shapes are checked before anything runs; a shape function returns the output shape, or a sentence
15
+ * saying why there is none.
16
+ */
17
+ import { DENSE_OPERATORS } from './denseOperators.ts';
18
+ import type { Operator } from './operatorKit.ts';
19
+ import { SHAPE_OPERATORS } from './shapeOperators.ts';
20
+ import { SPATIAL_OPERATORS } from './spatialOperators.ts';
21
+
22
+ export type { AttributeValue, Attributes, Operator } from './operatorKit.ts';
23
+
24
+ export const OPERATORS: ReadonlyMap<string, Operator> = new Map<string, Operator>([
25
+ ...DENSE_OPERATORS,
26
+ ...SPATIAL_OPERATORS,
27
+ ...SHAPE_OPERATORS,
28
+ ]);
@@ -0,0 +1,140 @@
1
+ /**
2
+ * Resizing a channel-major image, `[channels][height][width]`, by the upstream's conventions.
3
+ *
4
+ * **They are followed exactly, because they are the bug every port has.** Without aligned corners
5
+ * a destination pixel reads source `(i + ½)·scale − ½`, and for bilinear a negative source is
6
+ * clamped to zero; with aligned corners it reads `i·(in − 1)/(out − 1)`, and an output of one reads
7
+ * source zero. Bicubic uses the cubic convolution kernel with `a = −0.75`, not the −0.5 of the
8
+ * textbooks, and clamps its four taps to the border rather than its coordinate. Nearest reads
9
+ * `⌊i·scale⌋`, PyTorch's legacy `nearest` rather than `nearest-exact`.
10
+ */
11
+ const CUBIC_A = -0.75;
12
+
13
+ /* The cubic convolution kernel's two pieces, for |x| ≤ 1 and 1 < |x| < 2. */
14
+ function cubicNear(x: number): number {
15
+ return ((CUBIC_A + 2) * x - (CUBIC_A + 3)) * x * x + 1;
16
+ }
17
+
18
+ function cubicFar(x: number): number {
19
+ return ((CUBIC_A * x - 5 * CUBIC_A) * x + 8 * CUBIC_A) * x - 4 * CUBIC_A;
20
+ }
21
+
22
+ /*
23
+ * Where destination index `i` reads from, in the upstream's convention. Without aligned corners a
24
+ * destination pixel covers `input / output` source pixels — unless a step is given, which is what
25
+ * PyTorch's interpolate uses when it is handed a scale factor: the factor's inverse, not the ratio
26
+ * of the sizes it rounded to.
27
+ */
28
+ function sourceOf(
29
+ i: number,
30
+ input: number,
31
+ output: number,
32
+ align: boolean,
33
+ cubic: boolean,
34
+ step?: number,
35
+ ): number {
36
+ if (align) return output <= 1 ? 0 : (i * (input - 1)) / (output - 1);
37
+ const source = step === undefined ? ((i + 0.5) * input) / output - 0.5 : (i + 0.5) * step - 0.5;
38
+ return !cubic && source < 0 ? 0 : source;
39
+ }
40
+
41
+ function clampIndex(i: number, size: number): number {
42
+ return i < 0 ? 0 : i >= size ? size - 1 : i;
43
+ }
44
+
45
+ /**
46
+ * `out[channels][outH][outW]` resized from `input[channels][inH][inW]`. `step`, without aligned
47
+ * corners, is the source pixels a destination pixel covers, vertically and then horizontally, where
48
+ * it is not the ratio of the sizes.
49
+ */
50
+ export function resize(
51
+ out: Float32Array,
52
+ input: Float32Array,
53
+ channels: number,
54
+ inH: number,
55
+ inW: number,
56
+ outH: number,
57
+ outW: number,
58
+ mode: 'bilinear' | 'bicubic' | 'nearest',
59
+ alignCorners: boolean,
60
+ step?: readonly [number, number],
61
+ ): void {
62
+ if (mode === 'nearest') {
63
+ nearest(out, input, channels, inH, inW, outH, outW, step);
64
+ return;
65
+ }
66
+ const cubic = mode === 'bicubic';
67
+ const wx = [0, 0, 0, 0];
68
+ const wy = [0, 0, 0, 0];
69
+ for (let c = 0; c < channels; c += 1) {
70
+ const plane = c * inH * inW;
71
+ for (let y = 0; y < outH; y += 1) {
72
+ const sy = sourceOf(y, inH, outH, alignCorners, cubic, step?.[0]);
73
+ const y0 = Math.floor(sy);
74
+ const ty = sy - y0;
75
+ for (let x = 0; x < outW; x += 1) {
76
+ const sx = sourceOf(x, inW, outW, alignCorners, cubic, step?.[1]);
77
+ const x0 = Math.floor(sx);
78
+ const tx = sx - x0;
79
+ let value = 0;
80
+ if (cubic) {
81
+ wx[0] = cubicFar(tx + 1);
82
+ wx[1] = cubicNear(tx);
83
+ wx[2] = cubicNear(1 - tx);
84
+ wx[3] = cubicFar(2 - tx);
85
+ wy[0] = cubicFar(ty + 1);
86
+ wy[1] = cubicNear(ty);
87
+ wy[2] = cubicNear(1 - ty);
88
+ wy[3] = cubicFar(2 - ty);
89
+ for (let j = 0; j < 4; j += 1) {
90
+ const row = plane + clampIndex(y0 - 1 + j, inH) * inW;
91
+ let across = 0;
92
+ for (let i = 0; i < 4; i += 1) {
93
+ across += (wx[i] as number) * (input[row + clampIndex(x0 - 1 + i, inW)] as number);
94
+ }
95
+ value += (wy[j] as number) * across;
96
+ }
97
+ } else {
98
+ const x1 = Math.min(x0 + 1, inW - 1);
99
+ const y1 = Math.min(y0 + 1, inH - 1);
100
+ const top =
101
+ (input[plane + y0 * inW + x0] as number) * (1 - tx) +
102
+ (input[plane + y0 * inW + x1] as number) * tx;
103
+ const bottom =
104
+ (input[plane + y1 * inW + x0] as number) * (1 - tx) +
105
+ (input[plane + y1 * inW + x1] as number) * tx;
106
+ value = top * (1 - ty) + bottom * ty;
107
+ }
108
+ out[(c * outH + y) * outW + x] = value;
109
+ }
110
+ }
111
+ }
112
+ }
113
+
114
+ /*
115
+ * PyTorch's `nearest`: destination `i` reads source `⌊i·scale⌋`, clamped to the last, the product
116
+ * taken in single precision as its kernel takes it — `scale` the step where the node gives one, as
117
+ * a resize by a scale factor does, and the sizes' ratio otherwise.
118
+ */
119
+ function nearest(
120
+ out: Float32Array,
121
+ input: Float32Array,
122
+ channels: number,
123
+ inH: number,
124
+ inW: number,
125
+ outH: number,
126
+ outW: number,
127
+ step?: readonly [number, number],
128
+ ): void {
129
+ const scaleY = Math.fround(step?.[0] ?? inH / outH);
130
+ const scaleX = Math.fround(step?.[1] ?? inW / outW);
131
+ for (let c = 0; c < channels; c += 1) {
132
+ for (let y = 0; y < outH; y += 1) {
133
+ const sy = Math.min(Math.floor(Math.fround(y * scaleY)), inH - 1);
134
+ for (let x = 0; x < outW; x += 1) {
135
+ const sx = Math.min(Math.floor(Math.fround(x * scaleX)), inW - 1);
136
+ out[(c * outH + y) * outW + x] = input[(c * inH + sy) * inW + sx] as number;
137
+ }
138
+ }
139
+ }
140
+ }