@driftengine/texture 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +29 -0
- package/README.md +106 -0
- package/dist/decodeCpu.d.ts +59 -0
- package/dist/decodeCpu.js +234 -0
- package/dist/decodeGraph.d.ts +105 -0
- package/dist/decodeGraph.js +180 -0
- package/dist/half.d.ts +24 -0
- package/dist/half.js +86 -0
- package/dist/index.d.ts +66 -0
- package/dist/index.js +55 -0
- package/dist/inference.d.ts +53 -0
- package/dist/inference.js +243 -0
- package/dist/materialArray.d.ts +38 -0
- package/dist/materialArray.js +40 -0
- package/dist/mipNdf.d.ts +29 -0
- package/dist/mipNdf.js +53 -0
- package/dist/overlay/journal.d.ts +78 -0
- package/dist/overlay/journal.js +171 -0
- package/dist/overlay/sparse.d.ts +68 -0
- package/dist/overlay/sparse.js +212 -0
- package/dist/progressive.d.ts +30 -0
- package/dist/progressive.js +56 -0
- package/dist/residency/pageCache.d.ts +103 -0
- package/dist/residency/pageCache.js +184 -0
- package/dist/residency/predict.d.ts +55 -0
- package/dist/residency/predict.js +51 -0
- package/dist/residency/predictor.d.ts +16 -0
- package/dist/residency/predictor.js +44 -0
- package/dist/residency/queue.d.ts +26 -0
- package/dist/residency/queue.js +52 -0
- package/dist/residency/stream.d.ts +66 -0
- package/dist/residency/stream.js +142 -0
- package/dist/residency/table.d.ts +36 -0
- package/dist/residency/table.js +72 -0
- package/dist/residency/viewTiles.d.ts +108 -0
- package/dist/residency/viewTiles.js +419 -0
- package/dist/semantics.d.ts +52 -0
- package/dist/semantics.js +76 -0
- package/dist/tensor/architecture.d.ts +53 -0
- package/dist/tensor/architecture.js +96 -0
- package/dist/tensor/attention.d.ts +5 -0
- package/dist/tensor/attention.js +62 -0
- package/dist/tensor/denseOperators.d.ts +2 -0
- package/dist/tensor/denseOperators.js +136 -0
- package/dist/tensor/graph.d.ts +83 -0
- package/dist/tensor/graph.js +175 -0
- package/dist/tensor/linear.d.ts +49 -0
- package/dist/tensor/linear.js +136 -0
- package/dist/tensor/operatorKit.d.ts +27 -0
- package/dist/tensor/operatorKit.js +45 -0
- package/dist/tensor/operators.d.ts +3 -0
- package/dist/tensor/operators.js +24 -0
- package/dist/tensor/resize.d.ts +6 -0
- package/dist/tensor/resize.js +107 -0
- package/dist/tensor/reuse.d.ts +33 -0
- package/dist/tensor/reuse.js +59 -0
- package/dist/tensor/shapeOperators.d.ts +3 -0
- package/dist/tensor/shapeOperators.js +173 -0
- package/dist/tensor/spatial.d.ts +34 -0
- package/dist/tensor/spatial.js +131 -0
- package/dist/tensor/spatialOperators.d.ts +2 -0
- package/dist/tensor/spatialOperators.js +138 -0
- package/dist/tileHash.d.ts +29 -0
- package/dist/tileHash.js +50 -0
- package/dist/timeNodes.d.ts +26 -0
- package/dist/timeNodes.js +48 -0
- package/package.json +59 -0
- package/src/decodeCpu.ts +308 -0
- package/src/decodeGraph.ts +214 -0
- package/src/half.ts +86 -0
- package/src/index.ts +175 -0
- package/src/inference.ts +278 -0
- package/src/materialArray.ts +67 -0
- package/src/mipNdf.ts +63 -0
- package/src/overlay/journal.ts +218 -0
- package/src/overlay/sparse.ts +275 -0
- package/src/progressive.ts +60 -0
- package/src/residency/pageCache.ts +233 -0
- package/src/residency/predict.ts +74 -0
- package/src/residency/predictor.ts +62 -0
- package/src/residency/queue.ts +78 -0
- package/src/residency/stream.ts +194 -0
- package/src/residency/table.ts +89 -0
- package/src/residency/viewTiles.ts +553 -0
- package/src/semantics.ts +114 -0
- package/src/tensor/architecture.ts +140 -0
- package/src/tensor/attention.ts +75 -0
- package/src/tensor/denseOperators.ts +153 -0
- package/src/tensor/graph.ts +244 -0
- package/src/tensor/linear.ts +153 -0
- package/src/tensor/operatorKit.ts +76 -0
- package/src/tensor/operators.ts +28 -0
- package/src/tensor/resize.ts +140 -0
- package/src/tensor/shapeOperators.ts +173 -0
- package/src/tensor/spatial.ts +178 -0
- package/src/tensor/spatialOperators.ts +182 -0
- package/src/tileHash.ts +60 -0
- package/src/timeNodes.ts +60 -0
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The dense operators a transformer is built from: matrix multiply, bias, layer norm, GELU, softmax.
|
|
3
|
+
*
|
|
4
|
+
* **The references the device kernels are held to**, which is why they are written for exactness
|
|
5
|
+
* rather than speed: every sum accumulates in double precision and is rounded once, on the way into
|
|
6
|
+
* the caller's single-precision output. A kernel that disagrees with one of these is wrong, not
|
|
7
|
+
* this.
|
|
8
|
+
*
|
|
9
|
+
* **Contiguous row-major arrays with their dimensions, and nothing that allocates.** A strided view
|
|
10
|
+
* that permutes without copying would make every operator handle strides, and every device kernel
|
|
11
|
+
* would materialise the permutation anyway; so a permutation is a copy into a buffer the caller
|
|
12
|
+
* owns, and the operators stay one loop each.
|
|
13
|
+
*
|
|
14
|
+
* **The exact GELU, `x·Φ(x)`, and never the tanh approximation.** The two differ by 4e-4 at 3, and
|
|
15
|
+
* a port of a network trained with one and evaluated with the other is a different network. `erf`
|
|
16
|
+
* is here for that reason, accurate to double precision.
|
|
17
|
+
*/
|
|
18
|
+
/**
|
|
19
|
+
* `out[m×n] = a[m×k] · b`, where `b` is `k×n`, or `n×k` read as its transpose when `transposeB`.
|
|
20
|
+
* The offsets are where each matrix starts in its array, which is how one head of many is taken.
|
|
21
|
+
*/
|
|
22
|
+
export function matmul(out, a, b, m, k, n, transposeB = false, outAt = 0, aAt = 0, bAt = 0) {
|
|
23
|
+
for (let row = 0; row < m; row += 1) {
|
|
24
|
+
const aRow = aAt + row * k;
|
|
25
|
+
for (let col = 0; col < n; col += 1) {
|
|
26
|
+
let sum = 0;
|
|
27
|
+
if (transposeB) {
|
|
28
|
+
const bRow = bAt + col * k;
|
|
29
|
+
for (let i = 0; i < k; i += 1)
|
|
30
|
+
sum += a[aRow + i] * b[bRow + i];
|
|
31
|
+
}
|
|
32
|
+
else {
|
|
33
|
+
for (let i = 0; i < k; i += 1) {
|
|
34
|
+
sum += a[aRow + i] * b[bAt + i * n + col];
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
out[outAt + row * n + col] = sum;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
/** Add `bias[cols]` to every one of `rows` rows of `x`, in place. */
|
|
42
|
+
export function addBias(x, rows, cols, bias) {
|
|
43
|
+
for (let row = 0; row < rows; row += 1) {
|
|
44
|
+
for (let col = 0; col < cols; col += 1) {
|
|
45
|
+
x[row * cols + col] = x[row * cols + col] + bias[col];
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Each row normalised to zero mean and unit variance, then scaled by `gamma` and shifted by `beta`.
|
|
51
|
+
*
|
|
52
|
+
* **The population variance, divided by the row's length and not one less**, which is what the
|
|
53
|
+
* upstream frameworks compute and therefore what their trained weights expect.
|
|
54
|
+
*/
|
|
55
|
+
export function layerNorm(x, rows, cols, gamma, beta, epsilon, out) {
|
|
56
|
+
for (let row = 0; row < rows; row += 1) {
|
|
57
|
+
const at = row * cols;
|
|
58
|
+
let mean = 0;
|
|
59
|
+
for (let col = 0; col < cols; col += 1)
|
|
60
|
+
mean += x[at + col];
|
|
61
|
+
mean /= cols;
|
|
62
|
+
let variance = 0;
|
|
63
|
+
for (let col = 0; col < cols; col += 1)
|
|
64
|
+
variance += (x[at + col] - mean) ** 2;
|
|
65
|
+
variance /= cols;
|
|
66
|
+
const scale = 1 / Math.sqrt(variance + epsilon);
|
|
67
|
+
for (let col = 0; col < cols; col += 1) {
|
|
68
|
+
const normal = variance + epsilon === 0 ? 0 : (x[at + col] - mean) * scale;
|
|
69
|
+
out[at + col] = normal * gamma[col] + beta[col];
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
const TWO_OVER_ROOT_PI = 2 / Math.sqrt(Math.PI);
|
|
74
|
+
/**
|
|
75
|
+
* The error function, to double precision.
|
|
76
|
+
*
|
|
77
|
+
* A Maclaurin series below 3, where its alternating terms peak near 170 at `x = 3` and cost about
|
|
78
|
+
* three digits to cancellation — still 1e-13; a continued fraction for the complement above, which
|
|
79
|
+
* converges fastest exactly where the series is worst; and ±1 past 6, where the complement is below
|
|
80
|
+
* a double's last place.
|
|
81
|
+
*/
|
|
82
|
+
export function erf(x) {
|
|
83
|
+
const sign = x < 0 ? -1 : 1;
|
|
84
|
+
const a = Math.abs(x);
|
|
85
|
+
if (a >= 6)
|
|
86
|
+
return sign;
|
|
87
|
+
if (a < 3) {
|
|
88
|
+
let term = a;
|
|
89
|
+
let sum = a;
|
|
90
|
+
const square = a * a;
|
|
91
|
+
for (let n = 1; n < 100; n += 1) {
|
|
92
|
+
term *= -square / n;
|
|
93
|
+
const next = term / (2 * n + 1);
|
|
94
|
+
sum += next;
|
|
95
|
+
if (Math.abs(next) < 1e-17 * Math.abs(sum))
|
|
96
|
+
break;
|
|
97
|
+
}
|
|
98
|
+
return sign * TWO_OVER_ROOT_PI * sum;
|
|
99
|
+
}
|
|
100
|
+
/* erfc(a) = exp(−a²)/√π · 1/(a + ½/(a + 1/(a + 3/2/(a + ...)))), evaluated from the tail up. */
|
|
101
|
+
let fraction = 0;
|
|
102
|
+
for (let n = 60; n >= 1; n -= 1)
|
|
103
|
+
fraction = n / 2 / (a + fraction);
|
|
104
|
+
const complement = Math.exp(-a * a) / Math.sqrt(Math.PI) / (a + fraction);
|
|
105
|
+
return sign * (1 - complement);
|
|
106
|
+
}
|
|
107
|
+
/** `x·Φ(x)`, the exact GELU, from `x` into `out`. */
|
|
108
|
+
export function gelu(x, out) {
|
|
109
|
+
for (let i = 0; i < x.length; i += 1) {
|
|
110
|
+
const value = x[i];
|
|
111
|
+
out[i] = 0.5 * value * (1 + erf(value / Math.SQRT2));
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
/** The logistic, `1/(1 + e^−x)`, from `x` into `out`: 1 far to the right, and 0 far to the left. */
|
|
115
|
+
export function sigmoid(x, out) {
|
|
116
|
+
for (let i = 0; i < x.length; i += 1)
|
|
117
|
+
out[i] = 1 / (1 + Math.exp(-x[i]));
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Each row of `x` to a distribution, stably: the row's largest value is subtracted before the
|
|
121
|
+
* exponential, so logits near a thousand give their logistic rather than `Infinity / Infinity`.
|
|
122
|
+
*/
|
|
123
|
+
export function softmax(x, rows, cols, out) {
|
|
124
|
+
for (let row = 0; row < rows; row += 1) {
|
|
125
|
+
const at = row * cols;
|
|
126
|
+
let largest = -Infinity;
|
|
127
|
+
for (let col = 0; col < cols; col += 1)
|
|
128
|
+
largest = Math.max(largest, x[at + col]);
|
|
129
|
+
let total = 0;
|
|
130
|
+
for (let col = 0; col < cols; col += 1)
|
|
131
|
+
total += Math.exp(x[at + col] - largest);
|
|
132
|
+
for (let col = 0; col < cols; col += 1) {
|
|
133
|
+
out[at + col] = Math.exp(x[at + col] - largest) / total;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What every operator table shares: the shape of an operator, and the few helpers they all read
|
|
3
|
+
* their attributes and shapes with.
|
|
4
|
+
*/
|
|
5
|
+
export type AttributeValue = number | string | boolean | readonly number[];
|
|
6
|
+
export type Attributes = Readonly<Record<string, AttributeValue>>;
|
|
7
|
+
export type Shape = readonly number[];
|
|
8
|
+
export interface Operator {
|
|
9
|
+
/** How many inputs it takes, fewest and most. */
|
|
10
|
+
readonly arity: readonly [number, number];
|
|
11
|
+
/**
|
|
12
|
+
* The rank each input must have, in order, where it matters. **Checked before `shape` runs**,
|
|
13
|
+
* because an operator reading an image as `[channels, height, width]` reads a value with one more
|
|
14
|
+
* axis as its first three and validates it — and then declares an output it fills only part of.
|
|
15
|
+
*/
|
|
16
|
+
readonly ranks?: readonly number[];
|
|
17
|
+
shape(inputs: readonly Shape[], attributes: Attributes): number[] | string;
|
|
18
|
+
/** Scratch values it needs beyond its output, if any. */
|
|
19
|
+
scratch?(inputs: readonly Shape[], attributes: Attributes): number;
|
|
20
|
+
evaluate(inputs: readonly Float32Array[], shapes: readonly Shape[], attributes: Attributes, out: Float32Array, scratch: Float32Array): void;
|
|
21
|
+
}
|
|
22
|
+
export declare const num: (attributes: Attributes, name: string, fallback?: number) => number;
|
|
23
|
+
export declare const list: (attributes: Attributes, name: string) => readonly number[];
|
|
24
|
+
export declare const same: (a: Shape, b: Shape) => boolean;
|
|
25
|
+
export declare const elementwise: (apply: (a: number, b: number) => number) => Operator;
|
|
26
|
+
export declare function strides(shape: Shape): number[];
|
|
27
|
+
export declare const product: (shape: Shape, from?: number, to?: number) => number;
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
export const num = (attributes, name, fallback) => {
|
|
2
|
+
const value = attributes[name];
|
|
3
|
+
if (typeof value === 'number')
|
|
4
|
+
return value;
|
|
5
|
+
if (fallback !== undefined)
|
|
6
|
+
return fallback;
|
|
7
|
+
throw new RangeError(`attribute "${name}" must be a number`);
|
|
8
|
+
};
|
|
9
|
+
export const list = (attributes, name) => {
|
|
10
|
+
const value = attributes[name];
|
|
11
|
+
return Array.isArray(value) ? value : [];
|
|
12
|
+
};
|
|
13
|
+
export const same = (a, b) => a.length === b.length && a.every((d, i) => d === b[i]);
|
|
14
|
+
export const elementwise = (apply) => ({
|
|
15
|
+
arity: [2, 2],
|
|
16
|
+
shape: ([a, b]) => {
|
|
17
|
+
const x = a;
|
|
18
|
+
const y = b;
|
|
19
|
+
if (same(x, y) || (y.length === 1 && y[0] === x[x.length - 1]))
|
|
20
|
+
return [...x];
|
|
21
|
+
return `[${y.join(', ')}] is neither [${x.join(', ')}] nor its last dimension`;
|
|
22
|
+
},
|
|
23
|
+
evaluate: ([a, b], _shapes, _attributes, out) => {
|
|
24
|
+
const left = a;
|
|
25
|
+
const right = b;
|
|
26
|
+
const n = right.length;
|
|
27
|
+
for (let i = 0; i < left.length; i += 1) {
|
|
28
|
+
out[i] = apply(left[i], right[i % n]);
|
|
29
|
+
}
|
|
30
|
+
},
|
|
31
|
+
});
|
|
32
|
+
/* Strides of a row-major shape. */
|
|
33
|
+
export function strides(shape) {
|
|
34
|
+
const out = new Array(shape.length).fill(1);
|
|
35
|
+
for (let d = shape.length - 2; d >= 0; d -= 1) {
|
|
36
|
+
out[d] = out[d + 1] * shape[d + 1];
|
|
37
|
+
}
|
|
38
|
+
return out;
|
|
39
|
+
}
|
|
40
|
+
export const product = (shape, from = 0, to = shape.length) => {
|
|
41
|
+
let total = 1;
|
|
42
|
+
for (let d = from; d < to; d += 1)
|
|
43
|
+
total *= shape[d];
|
|
44
|
+
return total;
|
|
45
|
+
};
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The operators a graph may name: what each one's output shape is, and how it is evaluated.
|
|
3
|
+
*
|
|
4
|
+
* **One table, and a graph naming anything not in it is refused at validation, by name** — the
|
|
5
|
+
* runtime lacking an operator is a fact about the model, and a caller learns it when the model is
|
|
6
|
+
* loaded rather than at its first frame. Adding an operator is adding a row here, with its reference
|
|
7
|
+
* in `linear.ts`, `attention.ts` or `spatial.ts` and its device kernel in core.
|
|
8
|
+
*
|
|
9
|
+
* **`linear` is the upstream layer, `x·Wᵀ + b`, with its weight in the upstream layout `[out][in]`,
|
|
10
|
+
* and the bias added in double precision before the one rounding** — which is what makes a
|
|
11
|
+
* perceptron written as a graph agree with `evalNetwork` bit for bit, and is the evidence there is
|
|
12
|
+
* one runtime rather than two.
|
|
13
|
+
*
|
|
14
|
+
* Shapes are checked before anything runs; a shape function returns the output shape, or a sentence
|
|
15
|
+
* saying why there is none.
|
|
16
|
+
*/
|
|
17
|
+
import { DENSE_OPERATORS } from './denseOperators.js';
|
|
18
|
+
import { SHAPE_OPERATORS } from './shapeOperators.js';
|
|
19
|
+
import { SPATIAL_OPERATORS } from './spatialOperators.js';
|
|
20
|
+
export const OPERATORS = new Map([
|
|
21
|
+
...DENSE_OPERATORS,
|
|
22
|
+
...SPATIAL_OPERATORS,
|
|
23
|
+
...SHAPE_OPERATORS,
|
|
24
|
+
]);
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `out[channels][outH][outW]` resized from `input[channels][inH][inW]`. `step`, without aligned
|
|
3
|
+
* corners, is the source pixels a destination pixel covers, vertically and then horizontally, where
|
|
4
|
+
* it is not the ratio of the sizes.
|
|
5
|
+
*/
|
|
6
|
+
export declare function resize(out: Float32Array, input: Float32Array, channels: number, inH: number, inW: number, outH: number, outW: number, mode: 'bilinear' | 'bicubic' | 'nearest', alignCorners: boolean, step?: readonly [number, number]): void;
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resizing a channel-major image, `[channels][height][width]`, by the upstream's conventions.
|
|
3
|
+
*
|
|
4
|
+
* **They are followed exactly, because they are the bug every port has.** Without aligned corners
|
|
5
|
+
* a destination pixel reads source `(i + ½)·scale − ½`, and for bilinear a negative source is
|
|
6
|
+
* clamped to zero; with aligned corners it reads `i·(in − 1)/(out − 1)`, and an output of one reads
|
|
7
|
+
* source zero. Bicubic uses the cubic convolution kernel with `a = −0.75`, not the −0.5 of the
|
|
8
|
+
* textbooks, and clamps its four taps to the border rather than its coordinate. Nearest reads
|
|
9
|
+
* `⌊i·scale⌋`, PyTorch's legacy `nearest` rather than `nearest-exact`.
|
|
10
|
+
*/
|
|
11
|
+
const CUBIC_A = -0.75;
|
|
12
|
+
/* The cubic convolution kernel's two pieces, for |x| ≤ 1 and 1 < |x| < 2. */
|
|
13
|
+
function cubicNear(x) {
|
|
14
|
+
return ((CUBIC_A + 2) * x - (CUBIC_A + 3)) * x * x + 1;
|
|
15
|
+
}
|
|
16
|
+
function cubicFar(x) {
|
|
17
|
+
return ((CUBIC_A * x - 5 * CUBIC_A) * x + 8 * CUBIC_A) * x - 4 * CUBIC_A;
|
|
18
|
+
}
|
|
19
|
+
/*
|
|
20
|
+
* Where destination index `i` reads from, in the upstream's convention. Without aligned corners a
|
|
21
|
+
* destination pixel covers `input / output` source pixels — unless a step is given, which is what
|
|
22
|
+
* PyTorch's interpolate uses when it is handed a scale factor: the factor's inverse, not the ratio
|
|
23
|
+
* of the sizes it rounded to.
|
|
24
|
+
*/
|
|
25
|
+
function sourceOf(i, input, output, align, cubic, step) {
|
|
26
|
+
if (align)
|
|
27
|
+
return output <= 1 ? 0 : (i * (input - 1)) / (output - 1);
|
|
28
|
+
const source = step === undefined ? ((i + 0.5) * input) / output - 0.5 : (i + 0.5) * step - 0.5;
|
|
29
|
+
return !cubic && source < 0 ? 0 : source;
|
|
30
|
+
}
|
|
31
|
+
function clampIndex(i, size) {
|
|
32
|
+
return i < 0 ? 0 : i >= size ? size - 1 : i;
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* `out[channels][outH][outW]` resized from `input[channels][inH][inW]`. `step`, without aligned
|
|
36
|
+
* corners, is the source pixels a destination pixel covers, vertically and then horizontally, where
|
|
37
|
+
* it is not the ratio of the sizes.
|
|
38
|
+
*/
|
|
39
|
+
export function resize(out, input, channels, inH, inW, outH, outW, mode, alignCorners, step) {
|
|
40
|
+
if (mode === 'nearest') {
|
|
41
|
+
nearest(out, input, channels, inH, inW, outH, outW, step);
|
|
42
|
+
return;
|
|
43
|
+
}
|
|
44
|
+
const cubic = mode === 'bicubic';
|
|
45
|
+
const wx = [0, 0, 0, 0];
|
|
46
|
+
const wy = [0, 0, 0, 0];
|
|
47
|
+
for (let c = 0; c < channels; c += 1) {
|
|
48
|
+
const plane = c * inH * inW;
|
|
49
|
+
for (let y = 0; y < outH; y += 1) {
|
|
50
|
+
const sy = sourceOf(y, inH, outH, alignCorners, cubic, step?.[0]);
|
|
51
|
+
const y0 = Math.floor(sy);
|
|
52
|
+
const ty = sy - y0;
|
|
53
|
+
for (let x = 0; x < outW; x += 1) {
|
|
54
|
+
const sx = sourceOf(x, inW, outW, alignCorners, cubic, step?.[1]);
|
|
55
|
+
const x0 = Math.floor(sx);
|
|
56
|
+
const tx = sx - x0;
|
|
57
|
+
let value = 0;
|
|
58
|
+
if (cubic) {
|
|
59
|
+
wx[0] = cubicFar(tx + 1);
|
|
60
|
+
wx[1] = cubicNear(tx);
|
|
61
|
+
wx[2] = cubicNear(1 - tx);
|
|
62
|
+
wx[3] = cubicFar(2 - tx);
|
|
63
|
+
wy[0] = cubicFar(ty + 1);
|
|
64
|
+
wy[1] = cubicNear(ty);
|
|
65
|
+
wy[2] = cubicNear(1 - ty);
|
|
66
|
+
wy[3] = cubicFar(2 - ty);
|
|
67
|
+
for (let j = 0; j < 4; j += 1) {
|
|
68
|
+
const row = plane + clampIndex(y0 - 1 + j, inH) * inW;
|
|
69
|
+
let across = 0;
|
|
70
|
+
for (let i = 0; i < 4; i += 1) {
|
|
71
|
+
across += wx[i] * input[row + clampIndex(x0 - 1 + i, inW)];
|
|
72
|
+
}
|
|
73
|
+
value += wy[j] * across;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
else {
|
|
77
|
+
const x1 = Math.min(x0 + 1, inW - 1);
|
|
78
|
+
const y1 = Math.min(y0 + 1, inH - 1);
|
|
79
|
+
const top = input[plane + y0 * inW + x0] * (1 - tx) +
|
|
80
|
+
input[plane + y0 * inW + x1] * tx;
|
|
81
|
+
const bottom = input[plane + y1 * inW + x0] * (1 - tx) +
|
|
82
|
+
input[plane + y1 * inW + x1] * tx;
|
|
83
|
+
value = top * (1 - ty) + bottom * ty;
|
|
84
|
+
}
|
|
85
|
+
out[(c * outH + y) * outW + x] = value;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
/*
|
|
91
|
+
* PyTorch's `nearest`: destination `i` reads source `⌊i·scale⌋`, clamped to the last, the product
|
|
92
|
+
* taken in single precision as its kernel takes it — `scale` the step where the node gives one, as
|
|
93
|
+
* a resize by a scale factor does, and the sizes' ratio otherwise.
|
|
94
|
+
*/
|
|
95
|
+
function nearest(out, input, channels, inH, inW, outH, outW, step) {
|
|
96
|
+
const scaleY = Math.fround(step?.[0] ?? inH / outH);
|
|
97
|
+
const scaleX = Math.fround(step?.[1] ?? inW / outW);
|
|
98
|
+
for (let c = 0; c < channels; c += 1) {
|
|
99
|
+
for (let y = 0; y < outH; y += 1) {
|
|
100
|
+
const sy = Math.min(Math.floor(Math.fround(y * scaleY)), inH - 1);
|
|
101
|
+
for (let x = 0; x < outW; x += 1) {
|
|
102
|
+
const sx = Math.min(Math.floor(Math.fround(x * scaleX)), inW - 1);
|
|
103
|
+
out[(c * outH + y) * outW + x] = input[(c * inH + sy) * inW + sx];
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Which values of a graph may share one buffer, as one rule both sides plan by: the reference
|
|
3
|
+
* evaluator here, and the device's runner in `@driftengine/core`.
|
|
4
|
+
*
|
|
5
|
+
* **A value's buffer is given back after the last node that reads it, and not before** — so a
|
|
6
|
+
* node's output never shares a buffer with any of its own inputs, since the inputs are read while
|
|
7
|
+
* the output is written. What is `held` keeps a buffer of its own for the life of the plan, and so
|
|
8
|
+
* does every output; a value that is neither written by a node nor held is not planned at all,
|
|
9
|
+
* which is how the evaluator leaves weights and inputs in the caller's own arrays.
|
|
10
|
+
*
|
|
11
|
+
* **Two implementations of this would drift**, and the failure is a network that runs, validates
|
|
12
|
+
* and answers wrongly — a value overwritten before its last reader reads it.
|
|
13
|
+
*
|
|
14
|
+
* What it gives up: a free buffer is taken first-come rather than by the size that fits best, so a
|
|
15
|
+
* small value may hold a large buffer while a larger one waits. The transformers this runs repeat a
|
|
16
|
+
* block of identically shaped values, where first-come and best-fit choose the same buffers.
|
|
17
|
+
*/
|
|
18
|
+
export interface BufferReuse {
|
|
19
|
+
/** The buffer each planned value lives in. */
|
|
20
|
+
readonly slotOf: ReadonlyMap<string, number>;
|
|
21
|
+
/** Each buffer's size, in values: the largest value it ever holds. */
|
|
22
|
+
readonly sizes: readonly number[];
|
|
23
|
+
}
|
|
24
|
+
export declare function planReuse(graph: {
|
|
25
|
+
readonly nodes: readonly {
|
|
26
|
+
readonly inputs: readonly string[];
|
|
27
|
+
readonly output: string;
|
|
28
|
+
}[];
|
|
29
|
+
/** Values given a buffer each before any node runs, in this order. */
|
|
30
|
+
readonly held: readonly string[];
|
|
31
|
+
/** Values never given back, claimed where they are written. */
|
|
32
|
+
readonly outputs: readonly string[];
|
|
33
|
+
}, sizeOf: (name: string) => number): BufferReuse;
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Which values of a graph may share one buffer, as one rule both sides plan by: the reference
|
|
3
|
+
* evaluator here, and the device's runner in `@driftengine/core`.
|
|
4
|
+
*
|
|
5
|
+
* **A value's buffer is given back after the last node that reads it, and not before** — so a
|
|
6
|
+
* node's output never shares a buffer with any of its own inputs, since the inputs are read while
|
|
7
|
+
* the output is written. What is `held` keeps a buffer of its own for the life of the plan, and so
|
|
8
|
+
* does every output; a value that is neither written by a node nor held is not planned at all,
|
|
9
|
+
* which is how the evaluator leaves weights and inputs in the caller's own arrays.
|
|
10
|
+
*
|
|
11
|
+
* **Two implementations of this would drift**, and the failure is a network that runs, validates
|
|
12
|
+
* and answers wrongly — a value overwritten before its last reader reads it.
|
|
13
|
+
*
|
|
14
|
+
* What it gives up: a free buffer is taken first-come rather than by the size that fits best, so a
|
|
15
|
+
* small value may hold a large buffer while a larger one waits. The transformers this runs repeat a
|
|
16
|
+
* block of identically shaped values, where first-come and best-fit choose the same buffers.
|
|
17
|
+
*/
|
|
18
|
+
export function planReuse(graph, sizeOf) {
|
|
19
|
+
const slotOf = new Map();
|
|
20
|
+
const sizes = [];
|
|
21
|
+
const pinned = new Set([...graph.held, ...graph.outputs]);
|
|
22
|
+
const claim = (name) => {
|
|
23
|
+
sizes.push(sizeOf(name));
|
|
24
|
+
slotOf.set(name, sizes.length - 1);
|
|
25
|
+
return sizes.length - 1;
|
|
26
|
+
};
|
|
27
|
+
for (const name of graph.held)
|
|
28
|
+
claim(name);
|
|
29
|
+
/* The last node reading each value; a value never read is last read where it is written. */
|
|
30
|
+
const lastRead = new Map();
|
|
31
|
+
graph.nodes.forEach((node, at) => {
|
|
32
|
+
lastRead.set(node.output, at);
|
|
33
|
+
for (const input of node.inputs)
|
|
34
|
+
lastRead.set(input, at);
|
|
35
|
+
});
|
|
36
|
+
const free = [];
|
|
37
|
+
graph.nodes.forEach((node, at) => {
|
|
38
|
+
if (pinned.has(node.output)) {
|
|
39
|
+
claim(node.output);
|
|
40
|
+
}
|
|
41
|
+
else {
|
|
42
|
+
const reuse = free.shift();
|
|
43
|
+
if (reuse === undefined)
|
|
44
|
+
claim(node.output);
|
|
45
|
+
else {
|
|
46
|
+
slotOf.set(node.output, reuse);
|
|
47
|
+
sizes[reuse] = Math.max(sizes[reuse], sizeOf(node.output));
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
/* Only after the node has run: its inputs are read while its output is written. */
|
|
51
|
+
for (const value of new Set([...node.inputs, node.output])) {
|
|
52
|
+
const slot = slotOf.get(value);
|
|
53
|
+
if (slot === undefined || pinned.has(value) || lastRead.get(value) !== at)
|
|
54
|
+
continue;
|
|
55
|
+
free.push(slot);
|
|
56
|
+
}
|
|
57
|
+
});
|
|
58
|
+
return { slotOf, sizes };
|
|
59
|
+
}
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
/** The shape operators' rows: they move values and compute nothing. */
|
|
2
|
+
import { list, num, product, strides } from './operatorKit.js';
|
|
3
|
+
export const SHAPE_OPERATORS = [
|
|
4
|
+
[
|
|
5
|
+
'permute',
|
|
6
|
+
{
|
|
7
|
+
arity: [1, 1],
|
|
8
|
+
shape: ([x], attributes) => {
|
|
9
|
+
const order = list(attributes, 'order');
|
|
10
|
+
const shape = x;
|
|
11
|
+
const valid = order.length === shape.length &&
|
|
12
|
+
[...order].sort((p, q) => p - q).every((d, i) => d === i);
|
|
13
|
+
return valid
|
|
14
|
+
? order.map((d) => shape[d])
|
|
15
|
+
: `order [${order.join(', ')}] is not a permutation of ${shape.length} axes`;
|
|
16
|
+
},
|
|
17
|
+
evaluate: ([x], [xs], attributes, out) => {
|
|
18
|
+
const shape = xs;
|
|
19
|
+
const order = list(attributes, 'order');
|
|
20
|
+
const from = strides(shape);
|
|
21
|
+
const to = order.map((d) => shape[d]);
|
|
22
|
+
const index = new Array(shape.length).fill(0);
|
|
23
|
+
for (let at = 0; at < out.length; at += 1) {
|
|
24
|
+
let source = 0;
|
|
25
|
+
for (let d = 0; d < order.length; d += 1)
|
|
26
|
+
source += index[d] * from[order[d]];
|
|
27
|
+
out[at] = x[source];
|
|
28
|
+
for (let d = order.length - 1; d >= 0; d -= 1) {
|
|
29
|
+
index[d] = index[d] + 1;
|
|
30
|
+
if (index[d] < to[d])
|
|
31
|
+
break;
|
|
32
|
+
index[d] = 0;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
],
|
|
38
|
+
[
|
|
39
|
+
'reshape',
|
|
40
|
+
{
|
|
41
|
+
arity: [1, 1],
|
|
42
|
+
shape: ([x], attributes) => {
|
|
43
|
+
const shape = list(attributes, 'shape');
|
|
44
|
+
return product(shape) === product(x)
|
|
45
|
+
? [...shape]
|
|
46
|
+
: `[${shape.join(', ')}] does not hold [${x.join(', ')}]`;
|
|
47
|
+
},
|
|
48
|
+
evaluate: ([x], _shapes, _attributes, out) => out.set(x),
|
|
49
|
+
},
|
|
50
|
+
],
|
|
51
|
+
[
|
|
52
|
+
'concat',
|
|
53
|
+
{
|
|
54
|
+
arity: [2, 16],
|
|
55
|
+
shape: (inputs, attributes) => {
|
|
56
|
+
const axis = num(attributes, 'axis');
|
|
57
|
+
const first = inputs[0];
|
|
58
|
+
let along = 0;
|
|
59
|
+
for (const shape of inputs) {
|
|
60
|
+
if (shape.length !== first.length || shape.some((d, i) => i !== axis && d !== first[i])) {
|
|
61
|
+
return `inputs differ off axis ${axis}`;
|
|
62
|
+
}
|
|
63
|
+
along += shape[axis];
|
|
64
|
+
}
|
|
65
|
+
return first.map((d, i) => (i === axis ? along : d));
|
|
66
|
+
},
|
|
67
|
+
evaluate: (inputs, shapes, attributes, out) => {
|
|
68
|
+
const axis = num(attributes, 'axis');
|
|
69
|
+
const first = shapes[0];
|
|
70
|
+
const outer = product(first, 0, axis);
|
|
71
|
+
let at = 0;
|
|
72
|
+
for (let o = 0; o < outer; o += 1) {
|
|
73
|
+
for (let i = 0; i < inputs.length; i += 1) {
|
|
74
|
+
const block = product(shapes[i], axis);
|
|
75
|
+
out.set(inputs[i].subarray(o * block, (o + 1) * block), at);
|
|
76
|
+
at += block;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
},
|
|
80
|
+
},
|
|
81
|
+
],
|
|
82
|
+
[
|
|
83
|
+
'slice',
|
|
84
|
+
{
|
|
85
|
+
arity: [1, 1],
|
|
86
|
+
shape: ([x], attributes) => {
|
|
87
|
+
const axis = num(attributes, 'axis');
|
|
88
|
+
const start = num(attributes, 'start');
|
|
89
|
+
const end = num(attributes, 'end');
|
|
90
|
+
const shape = x;
|
|
91
|
+
return start >= 0 && end <= shape[axis] && start < end
|
|
92
|
+
? shape.map((d, i) => (i === axis ? end - start : d))
|
|
93
|
+
: `[${start}, ${end}) is outside axis ${axis} of length ${shape[axis]}`;
|
|
94
|
+
},
|
|
95
|
+
evaluate: ([x], [xs], attributes, out) => {
|
|
96
|
+
const axis = num(attributes, 'axis');
|
|
97
|
+
const start = num(attributes, 'start');
|
|
98
|
+
const end = num(attributes, 'end');
|
|
99
|
+
const shape = xs;
|
|
100
|
+
const inner = product(shape, axis + 1);
|
|
101
|
+
const outer = product(shape, 0, axis);
|
|
102
|
+
const length = shape[axis];
|
|
103
|
+
const take = (end - start) * inner;
|
|
104
|
+
for (let o = 0; o < outer; o += 1) {
|
|
105
|
+
const from = (o * length + start) * inner;
|
|
106
|
+
out.set(x.subarray(from, from + take), o * take);
|
|
107
|
+
}
|
|
108
|
+
},
|
|
109
|
+
},
|
|
110
|
+
],
|
|
111
|
+
[
|
|
112
|
+
/*
|
|
113
|
+
* Zeros after the end of each axis, `after[axis]` of them, as a window partition pads a grid it
|
|
114
|
+
* does not divide. Every value keeps its index along every axis.
|
|
115
|
+
*/
|
|
116
|
+
'pad',
|
|
117
|
+
{
|
|
118
|
+
arity: [1, 1],
|
|
119
|
+
shape: ([x], attributes) => {
|
|
120
|
+
const after = list(attributes, 'after');
|
|
121
|
+
const shape = x;
|
|
122
|
+
if (after.length !== shape.length) {
|
|
123
|
+
return `the pad names ${after.length} axes and the value has ${shape.length} axes`;
|
|
124
|
+
}
|
|
125
|
+
if (after.some((d) => d < 0))
|
|
126
|
+
return `a pad of [${after.join(', ')}] is negative`;
|
|
127
|
+
return shape.map((d, i) => d + after[i]);
|
|
128
|
+
},
|
|
129
|
+
evaluate: ([x], [xs], attributes, out) => {
|
|
130
|
+
const shape = xs;
|
|
131
|
+
const after = list(attributes, 'after');
|
|
132
|
+
const outShape = shape.map((d, i) => d + after[i]);
|
|
133
|
+
const from = strides(shape);
|
|
134
|
+
const to = strides(outShape);
|
|
135
|
+
out.fill(0);
|
|
136
|
+
const values = x;
|
|
137
|
+
for (let i = 0; i < values.length; i += 1) {
|
|
138
|
+
let at = 0;
|
|
139
|
+
for (let axis = 0; axis < shape.length; axis += 1) {
|
|
140
|
+
at +=
|
|
141
|
+
(Math.floor(i / from[axis]) % shape[axis]) *
|
|
142
|
+
to[axis];
|
|
143
|
+
}
|
|
144
|
+
out[at] = values[i];
|
|
145
|
+
}
|
|
146
|
+
},
|
|
147
|
+
},
|
|
148
|
+
],
|
|
149
|
+
[
|
|
150
|
+
/*
|
|
151
|
+
* Rows of a table by index, as a token's embedding is looked up: `indices` holds whole numbers
|
|
152
|
+
* as values, and row i of the output is row `indices[i]` of the table. An index that is not one
|
|
153
|
+
* of the table's rows is refused here; a device cannot refuse, and clamps it to the last row.
|
|
154
|
+
*/
|
|
155
|
+
'gather',
|
|
156
|
+
{
|
|
157
|
+
ranks: [2, 1],
|
|
158
|
+
arity: [2, 2],
|
|
159
|
+
shape: ([table, indices]) => [indices[0], table[1]],
|
|
160
|
+
evaluate: ([table, indices], [ts], _attributes, out) => {
|
|
161
|
+
const [rows, width] = ts;
|
|
162
|
+
const at = indices;
|
|
163
|
+
for (let i = 0; i < at.length; i += 1) {
|
|
164
|
+
const row = at[i];
|
|
165
|
+
if (!Number.isInteger(row) || row < 0 || row >= rows) {
|
|
166
|
+
throw new RangeError(`gather: index ${row} at ${i} is not a row of a table of ${rows}`);
|
|
167
|
+
}
|
|
168
|
+
out.set(table.subarray(row * width, (row + 1) * width), i * width);
|
|
169
|
+
}
|
|
170
|
+
},
|
|
171
|
+
},
|
|
172
|
+
],
|
|
173
|
+
];
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The spatial operators: convolution, its transpose, a patch embedding and max pooling. Resizing is
|
|
3
|
+
* `resize.ts`'s.
|
|
4
|
+
*
|
|
5
|
+
* **Channel-major images, one at a time** — `[channels][height][width]`, the layout the upstream
|
|
6
|
+
* frameworks call NCHW with a batch of one — and **weights in the upstream layout**: a convolution's
|
|
7
|
+
* `[out][in][kh][kw]`, a transposed convolution's `[in][out][kh][kw]`. A converted checkpoint is
|
|
8
|
+
* then read as it was stored, and a rearrangement is never a place for a transposed weight to hide.
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* `out[cout][oh][ow]` from `input[cin][h][w]`, with `oh = ⌊(h + 2·padding − kh)/stride⌋ + 1` and
|
|
12
|
+
* likewise `ow`. `bias` may be null.
|
|
13
|
+
*
|
|
14
|
+
* **Grouped, the channels split into `groups` runs** and output `o` reads only the run `o` falls in:
|
|
15
|
+
* the weight is `[cout][cin / groups][kh][kw]`, as the upstream stores it, and a depthwise
|
|
16
|
+
* convolution is `groups = cin = cout`.
|
|
17
|
+
*/
|
|
18
|
+
export declare function conv2d(out: Float32Array, input: Float32Array, cin: number, h: number, w: number, weight: Float32Array, bias: Float32Array | null, cout: number, kh: number, kw: number, stride: number, padding: number, groups?: number): void;
|
|
19
|
+
/**
|
|
20
|
+
* `out[cout][oh][ow]` from `input[cin][h][w]`, with `oh = (h − 1)·stride − 2·padding + kh`: each input
|
|
21
|
+
* pixel spread over the kernel at its strided place, which is how a decoder head upsamples.
|
|
22
|
+
*/
|
|
23
|
+
export declare function convTranspose2d(out: Float32Array, input: Float32Array, cin: number, h: number, w: number, weight: Float32Array, bias: Float32Array | null, cout: number, kh: number, kw: number, stride: number, padding: number): void;
|
|
24
|
+
/**
|
|
25
|
+
* A patch embedding: a convolution with stride and kernel both `patch`, written token-major —
|
|
26
|
+
* `out[tokens][dim]`, tokens row by row — which is the order a transformer reads them in.
|
|
27
|
+
*/
|
|
28
|
+
export declare function patchEmbed(out: Float32Array, image: Float32Array, cin: number, h: number, w: number, weight: Float32Array, bias: Float32Array | null, dim: number, patch: number): void;
|
|
29
|
+
/**
|
|
30
|
+
* `out[channels][oh][ow]`, each the largest of its `kernel`-square window at `stride`, with
|
|
31
|
+
* `oh = ⌊(h − kernel)/stride⌋ + 1`: no padding, and a ragged last row or column dropped, as
|
|
32
|
+
* PyTorch's `max_pool2d` does without `ceil_mode`.
|
|
33
|
+
*/
|
|
34
|
+
export declare function maxPool2d(out: Float32Array, input: Float32Array, channels: number, h: number, w: number, kernel: number, stride: number): void;
|