wgblas 2.0.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -18
- package/dist/wgblas.browser.js +2172 -1174
- package/index.d.mts +49 -44
- package/index.mjs +11 -0
- package/package.json +133 -63
- package/src/classes/Complex32.d.mts +43 -0
- package/src/classes/Complex32.mjs +82 -0
- package/src/classes/Complex64.d.mts +44 -0
- package/src/classes/Complex64.mjs +76 -0
- package/src/classes/GpuMatrix.d.mts +41 -41
- package/src/classes/GpuMatrix.mjs +126 -17
- package/src/classes/GpuVector.d.mts +36 -40
- package/src/classes/GpuVector.mjs +66 -11
- package/src/cscal/cscal.d.mts +47 -0
- package/src/cscal/cscal.mjs +98 -0
- package/src/dasum/dasum.d.mts +4 -4
- package/src/dasum/dasum.mjs +38 -20
- package/src/daxpy/daxpy.d.mts +56 -0
- package/src/daxpy/daxpy.mjs +150 -0
- package/src/dcopy/dcopy.d.mts +52 -0
- package/src/dcopy/dcopy.mjs +140 -0
- package/src/ddot/ddot.d.mts +62 -0
- package/src/ddot/ddot.mjs +184 -0
- package/src/devdocs.mjs +13 -0
- package/src/dnrm2/dnrm2.d.mts +50 -0
- package/src/dnrm2/dnrm2.mjs +189 -0
- package/src/drot/drot.d.mts +67 -0
- package/src/drot/drot.mjs +170 -0
- package/src/drotm/drotm.d.mts +67 -0
- package/src/drotm/drotm.mjs +171 -0
- package/src/dscal/dscal.d.mts +52 -0
- package/src/dscal/dscal.mjs +119 -0
- package/src/dswap/dswap.d.mts +57 -0
- package/src/dswap/dswap.mjs +155 -0
- package/src/idamax/idamax.d.mts +20 -2
- package/src/idamax/idamax.mjs +56 -24
- package/src/init.mjs +117 -56
- package/src/isamax/isamax.d.mts +20 -2
- package/src/isamax/isamax.mjs +21 -16
- package/src/random/random.d.mts +37 -39
- package/src/random/random.mjs +39 -7
- package/src/sasum/sasum.d.mts +2 -2
- package/src/sasum/sasum.mjs +20 -16
- package/src/saxpy/saxpy.d.mts +2 -2
- package/src/saxpy/saxpy.mjs +14 -11
- package/src/scopy/scopy.d.mts +2 -2
- package/src/scopy/scopy.mjs +13 -9
- package/src/sdot/sdot.d.mts +2 -2
- package/src/sdot/sdot.mjs +21 -17
- package/src/sgemm/sgemm.d.mts +2 -2
- package/src/sgemm/sgemm.mjs +109 -40
- package/src/sgemmtr/sgemmtr.d.mts +3 -2
- package/src/sgemmtr/sgemmtr.mjs +98 -40
- package/src/sgemv/sgemv.d.mts +2 -2
- package/src/sgemv/sgemv.mjs +69 -41
- package/src/sger/sger.d.mts +2 -2
- package/src/sger/sger.mjs +43 -19
- package/src/shaders/__test_pipeline_a.wgsl +3 -0
- package/src/shaders/__test_pipeline_b.wgsl +2 -0
- package/src/shaders/cscal.wgsl +33 -0
- package/src/shaders/daxpy.wgsl +66 -0
- package/src/shaders/dcopy.wgsl +34 -0
- package/src/shaders/ddot.wgsl +106 -0
- package/src/shaders/dnrm2.wgsl +167 -0
- package/src/shaders/drot.wgsl +81 -0
- package/src/shaders/drotm.wgsl +99 -0
- package/src/shaders/dscal.wgsl +60 -0
- package/src/shaders/dswap.wgsl +38 -0
- package/src/shaders/f64/utils/add.wgsl +6 -0
- package/src/shaders/f64/utils/divide.wgsl +45 -0
- package/src/shaders/f64/utils/multiply.wgsl +19 -10
- package/src/shaders/f64/utils/sqrt.wgsl +45 -0
- package/src/shaders/index.mjs +233 -14
- package/src/shaders/reduction/scaledSum.wgsl +65 -0
- package/src/shaders/reduction/scaledSumF64.wgsl +93 -0
- package/src/shaders/sgemm_large.wgsl +107 -18
- package/src/shaders/sgemm_small.wgsl +115 -15
- package/src/shaders/sgemmtr_large.wgsl +4 -1
- package/src/shaders/sgemmtr_small.wgsl +4 -1
- package/src/shaders/sgemv_n.wgsl +3 -1
- package/src/shaders/sgemv_t.wgsl +3 -1
- package/src/shaders/snrm2.wgsl +72 -23
- package/src/shaders/ssymv.wgsl +3 -1
- package/src/snrm2/snrm2.d.mts +2 -2
- package/src/snrm2/snrm2.mjs +41 -23
- package/src/srot/srot.d.mts +2 -4
- package/src/srot/srot.mjs +16 -11
- package/src/srotm/srotm.d.mts +2 -4
- package/src/srotm/srotm.mjs +17 -11
- package/src/sscal/sscal.d.mts +3 -3
- package/src/sscal/sscal.mjs +14 -12
- package/src/sswap/sswap.d.mts +2 -2
- package/src/sswap/sswap.mjs +18 -10
- package/src/ssymm/ssymm.d.mts +5 -4
- package/src/ssymm/ssymm.mjs +150 -54
- package/src/ssymv/ssymv.d.mts +2 -2
- package/src/ssymv/ssymv.mjs +47 -26
- package/src/ssyr/ssyr.d.mts +2 -2
- package/src/ssyr/ssyr.mjs +38 -17
- package/src/ssyr2/ssyr2.d.mts +2 -2
- package/src/ssyr2/ssyr2.mjs +48 -21
- package/src/ssyr2k/ssyr2k.d.mts +3 -2
- package/src/ssyr2k/ssyr2k.mjs +140 -62
- package/src/ssyrk/ssyrk.d.mts +3 -2
- package/src/ssyrk/ssyrk.mjs +91 -39
- package/src/strmm/strmm.d.mts +5 -4
- package/src/strmm/strmm.mjs +174 -60
- package/src/strmv/strmv.d.mts +2 -2
- package/src/strmv/strmv.mjs +42 -20
- package/src/strsm/strsm.d.mts +6 -4
- package/src/strsm/strsm.mjs +438 -174
- package/src/strsv/strsv.d.mts +5 -3
- package/src/strsv/strsv.mjs +89 -34
- package/src/util/benchmark.mjs +9 -9
- package/src/util/bindgroup.mjs +1 -3
- package/src/util/buffer.mjs +139 -24
- package/src/util/complex.mjs +87 -0
- package/src/util/compute.mjs +19 -16
- package/src/util/constants.mjs +57 -0
- package/src/util/device.mjs +49 -0
- package/src/util/pipeline.mjs +44 -10
- package/src/util/workgroup.mjs +72 -7
- package/src/shaders/browser-shaders.mjs +0 -81
- package/src/shaders/f64add.wgsl +0 -281
package/src/strsv/strsv.d.mts
CHANGED
|
@@ -2,8 +2,9 @@ import { GpuVector } from "../classes/GpuVector.mjs";
|
|
|
2
2
|
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
|
-
* Solves the triangular system
|
|
6
|
-
*
|
|
5
|
+
* Solves the triangular system for $x$, in place — x holds b on input, the
|
|
6
|
+
* solution on output:
|
|
7
|
+
* $$\mathrm{op}(A) x = b$$
|
|
7
8
|
*
|
|
8
9
|
* A is an n×n triangular matrix stored in row-major order. Only the triangle
|
|
9
10
|
* specified by `uplo` is referenced; the other triangle is not accessed.
|
|
@@ -42,7 +43,8 @@ export declare function strsv(
|
|
|
42
43
|
): Promise<{ x: Float32Array; gpuTimeMs?: number }>;
|
|
43
44
|
|
|
44
45
|
/**
|
|
45
|
-
* Solves the triangular system
|
|
46
|
+
* Solves the triangular system for $x$, in place:
|
|
47
|
+
* $$\mathrm{op}(A) x = b$$
|
|
46
48
|
*
|
|
47
49
|
* x is kept resident on the GPU (mutated in place). A must be a GpuMatrix;
|
|
48
50
|
* its own `layout` (set at `GpuMatrix.from` time) determines the operation —
|
package/src/strsv/strsv.mjs
CHANGED
|
@@ -12,9 +12,10 @@ import { resolveTimestamp, extractTimestamp } from "../util/benchmark.mjs";
|
|
|
12
12
|
import { getPipeline } from "../util/pipeline.mjs";
|
|
13
13
|
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
14
14
|
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
15
|
+
import { BLOCK_SIZE } from "../util/constants.mjs";
|
|
16
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
15
17
|
|
|
16
18
|
// Blocked triangular solve via explicit block inversion (invert/apply/update passes) instead of barrier-per-row substitution.
|
|
17
|
-
const BLOCK_SIZE = 64;
|
|
18
19
|
|
|
19
20
|
// One shared buffer holds all blocks' params (offset blockIndex*stride) instead of one buffer per block — avoids the O(numBlocks) createBuffer/writeBuffer calls that dominated CPU time.
|
|
20
21
|
function packBlockParams(numBlocks, stride, fieldsPerBlock) {
|
|
@@ -38,13 +39,24 @@ function createSharedParamsBuffer(device, data, label) {
|
|
|
38
39
|
return buffer;
|
|
39
40
|
}
|
|
40
41
|
|
|
41
|
-
export async function strsv(
|
|
42
|
+
export async function strsv(
|
|
43
|
+
device,
|
|
44
|
+
uplo,
|
|
45
|
+
trans,
|
|
46
|
+
diag,
|
|
47
|
+
n,
|
|
48
|
+
A,
|
|
49
|
+
lda,
|
|
50
|
+
x,
|
|
51
|
+
incx,
|
|
52
|
+
layout = "row-major",
|
|
53
|
+
) {
|
|
42
54
|
const xIsGpu = x instanceof GpuVector;
|
|
43
55
|
const AIsGpu = A instanceof GpuMatrix;
|
|
44
56
|
const isUnit = diag === "unit";
|
|
45
57
|
|
|
46
|
-
|
|
47
|
-
|
|
58
|
+
requireGpuDevice(device);
|
|
59
|
+
requireSameDevice(device, "strsv", { A, x });
|
|
48
60
|
if (uplo !== "lower" && uplo !== "upper")
|
|
49
61
|
throw new Error("uplo must be 'lower' or 'upper'.");
|
|
50
62
|
if (trans !== "no-transpose" && trans !== "transpose")
|
|
@@ -65,6 +77,8 @@ export async function strsv(device, uplo, trans, diag, n, A, lda, x, incx, layou
|
|
|
65
77
|
throw new Error("A must be a GpuMatrix when x is a GpuVector.");
|
|
66
78
|
if (AIsGpu && !xIsGpu)
|
|
67
79
|
throw new Error("x must be a GpuVector when A is a GpuMatrix.");
|
|
80
|
+
if (AIsGpu && xIsGpu && A._buf === x._buf)
|
|
81
|
+
throw new Error("A and x must not reference the same GPU buffer.");
|
|
68
82
|
if (AIsGpu && lda !== A.lda)
|
|
69
83
|
throw new Error("lda must match A.lda when A is a GpuMatrix.");
|
|
70
84
|
if (AIsGpu && (A.rows < n || A.cols < n))
|
|
@@ -73,9 +87,7 @@ export async function strsv(device, uplo, trans, diag, n, A, lda, x, incx, layou
|
|
|
73
87
|
if (n === 0) return xIsGpu ? {} : { x };
|
|
74
88
|
|
|
75
89
|
if (!AIsGpu && A.length < (n - 1) * lda + n)
|
|
76
|
-
throw new Error(
|
|
77
|
-
"A does not have enough elements for the given n and lda.",
|
|
78
|
-
);
|
|
90
|
+
throw new Error("A does not have enough elements for the given n and lda.");
|
|
79
91
|
if (x.length < (n - 1) * incx + 1)
|
|
80
92
|
throw new Error(
|
|
81
93
|
"x does not have enough elements for the given n and incx.",
|
|
@@ -85,7 +97,9 @@ export async function strsv(device, uplo, trans, diag, n, A, lda, x, incx, layou
|
|
|
85
97
|
const effLayout = AIsGpu ? A.layout : layout;
|
|
86
98
|
const isColMajor = effLayout === "column-major";
|
|
87
99
|
const isLower = isColMajor ? uplo === "upper" : uplo === "lower";
|
|
88
|
-
const isNoTrans = isColMajor
|
|
100
|
+
const isNoTrans = isColMajor
|
|
101
|
+
? trans === "transpose"
|
|
102
|
+
: trans === "no-transpose";
|
|
89
103
|
|
|
90
104
|
const invertPipeline = await getPipeline(device, "strsv_invert_block");
|
|
91
105
|
const applyPipeline = await getPipeline(device, "strsv_apply_inverse");
|
|
@@ -109,11 +123,12 @@ export async function strsv(device, uplo, trans, diag, n, A, lda, x, incx, layou
|
|
|
109
123
|
let invertParams = null;
|
|
110
124
|
|
|
111
125
|
try {
|
|
112
|
-
ABuffer = AIsGpu ? A._buf : uploadBuffer(A, "strsv-A", false);
|
|
113
|
-
xBuffer = xIsGpu ? x._buf : uploadBuffer(x, "strsv-x", true);
|
|
126
|
+
ABuffer = AIsGpu ? A._buf : uploadBuffer(device, A, "strsv-A", false);
|
|
127
|
+
xBuffer = xIsGpu ? x._buf : uploadBuffer(device, x, "strsv-x", true);
|
|
114
128
|
// One BLOCK_SIZE x BLOCK_SIZE dense region per block (row-major), even
|
|
115
129
|
// though only a triangular half is ever nonzero — see strsv_invert_block.wgsl.
|
|
116
130
|
AinvBuffer = createStorageBuffer(
|
|
131
|
+
device,
|
|
117
132
|
numBlocks * BLOCK_SIZE * BLOCK_SIZE * 4,
|
|
118
133
|
"strsv-Ainv",
|
|
119
134
|
);
|
|
@@ -126,35 +141,60 @@ export async function strsv(device, uplo, trans, diag, n, A, lda, x, incx, layou
|
|
|
126
141
|
const blockEnd = Math.min(blockStart + BLOCK_SIZE, n);
|
|
127
142
|
return [incx, blockIndex, blockStart, blockEnd];
|
|
128
143
|
});
|
|
129
|
-
applyParamsBuffer = createSharedParamsBuffer(
|
|
144
|
+
applyParamsBuffer = createSharedParamsBuffer(
|
|
145
|
+
device,
|
|
146
|
+
applyData,
|
|
147
|
+
"strsv-apply-params",
|
|
148
|
+
);
|
|
130
149
|
|
|
131
150
|
const updateData = packBlockParams(numBlocks, stride, (blockIndex) => {
|
|
132
151
|
const blockStart = blockIndex * BLOCK_SIZE;
|
|
133
152
|
const blockEnd = Math.min(blockStart + BLOCK_SIZE, n);
|
|
134
|
-
return [
|
|
153
|
+
return [
|
|
154
|
+
n,
|
|
155
|
+
incx,
|
|
156
|
+
lda,
|
|
157
|
+
isNoTrans ? 0 : 1,
|
|
158
|
+
isLower ? 0 : 1,
|
|
159
|
+
blockStart,
|
|
160
|
+
blockEnd,
|
|
161
|
+
];
|
|
135
162
|
});
|
|
136
|
-
updateParamsBuffer = createSharedParamsBuffer(
|
|
163
|
+
updateParamsBuffer = createSharedParamsBuffer(
|
|
164
|
+
device,
|
|
165
|
+
updateData,
|
|
166
|
+
"strsv-update-params",
|
|
167
|
+
);
|
|
137
168
|
|
|
138
|
-
const { commandEncoder, querySet } = beginTimedEncoder();
|
|
169
|
+
const { commandEncoder, querySet } = beginTimedEncoder(device);
|
|
139
170
|
|
|
140
171
|
// Pre-pass: every block's inverse, fully parallel, one dispatch.
|
|
141
172
|
invertParams = createParamsBuffer(
|
|
173
|
+
device,
|
|
142
174
|
[
|
|
143
|
-
{ value: n,
|
|
144
|
-
{ value: lda,
|
|
175
|
+
{ value: n, type: "u32" },
|
|
176
|
+
{ value: lda, type: "u32" },
|
|
145
177
|
{ value: isNoTrans ? 0 : 1, type: "u32" },
|
|
146
|
-
{ value: isLower ? 0 : 1,
|
|
147
|
-
{ value: isUnit ? 1 : 0,
|
|
178
|
+
{ value: isLower ? 0 : 1, type: "u32" },
|
|
179
|
+
{ value: isUnit ? 1 : 0, type: "u32" },
|
|
148
180
|
],
|
|
149
181
|
"strsv-invert-params",
|
|
150
182
|
);
|
|
151
|
-
const invertBindGroup = createBindGroup(
|
|
152
|
-
|
|
153
|
-
|
|
183
|
+
const invertBindGroup = createBindGroup(
|
|
184
|
+
device,
|
|
185
|
+
invertPipeline.getBindGroupLayout(0),
|
|
186
|
+
[ABuffer, AinvBuffer, invertParams],
|
|
187
|
+
);
|
|
154
188
|
const invertDesc = querySet
|
|
155
189
|
? { timestampWrites: { querySet, beginningOfPassWriteIndex: 0 } }
|
|
156
190
|
: undefined;
|
|
157
|
-
encodePass(
|
|
191
|
+
encodePass(
|
|
192
|
+
commandEncoder,
|
|
193
|
+
invertPipeline,
|
|
194
|
+
invertBindGroup,
|
|
195
|
+
{ x: BLOCK_SIZE, y: numBlocks },
|
|
196
|
+
invertDesc,
|
|
197
|
+
);
|
|
158
198
|
|
|
159
199
|
for (let bi = 0; bi < blockStarts.length; bi++) {
|
|
160
200
|
const blockStart = blockStarts[bi];
|
|
@@ -163,30 +203,45 @@ export async function strsv(device, uplo, trans, diag, n, A, lda, x, incx, layou
|
|
|
163
203
|
const isLastPass = bi === blockStarts.length - 1;
|
|
164
204
|
const paramsOffset = blockIndex * stride;
|
|
165
205
|
|
|
166
|
-
const applyBindGroup = createBindGroup(
|
|
167
|
-
|
|
168
|
-
|
|
206
|
+
const applyBindGroup = createBindGroup(
|
|
207
|
+
device,
|
|
208
|
+
applyPipeline.getBindGroupLayout(0),
|
|
209
|
+
[
|
|
210
|
+
AinvBuffer,
|
|
211
|
+
xBuffer,
|
|
212
|
+
{ buffer: applyParamsBuffer, offset: paramsOffset, size: 16 },
|
|
213
|
+
],
|
|
214
|
+
);
|
|
169
215
|
|
|
170
|
-
const applyDesc =
|
|
171
|
-
|
|
172
|
-
|
|
216
|
+
const applyDesc =
|
|
217
|
+
isLastPass && querySet
|
|
218
|
+
? { timestampWrites: { querySet, endOfPassWriteIndex: 1 } }
|
|
219
|
+
: undefined;
|
|
173
220
|
encodePass(commandEncoder, applyPipeline, applyBindGroup, 1, applyDesc);
|
|
174
221
|
|
|
175
222
|
const remaining = forward ? n - blockEnd : blockStart;
|
|
176
223
|
if (remaining === 0) continue;
|
|
177
224
|
|
|
178
|
-
const updateBindGroup = createBindGroup(
|
|
179
|
-
|
|
180
|
-
|
|
225
|
+
const updateBindGroup = createBindGroup(
|
|
226
|
+
device,
|
|
227
|
+
updatePipeline.getBindGroupLayout(0),
|
|
228
|
+
[
|
|
229
|
+
ABuffer,
|
|
230
|
+
xBuffer,
|
|
231
|
+
{ buffer: updateParamsBuffer, offset: paramsOffset, size: 32 },
|
|
232
|
+
],
|
|
233
|
+
);
|
|
181
234
|
|
|
182
235
|
const wgCount = Math.min(remaining, maxWg);
|
|
183
236
|
encodePass(commandEncoder, updatePipeline, updateBindGroup, wgCount);
|
|
184
237
|
}
|
|
185
238
|
|
|
186
|
-
const ts = resolveTimestamp(commandEncoder, querySet);
|
|
187
|
-
const readBuffer = xIsGpu
|
|
239
|
+
const ts = resolveTimestamp(device, commandEncoder, querySet);
|
|
240
|
+
const readBuffer = xIsGpu
|
|
241
|
+
? null
|
|
242
|
+
: stageReadback(device, commandEncoder, xBuffer);
|
|
188
243
|
|
|
189
|
-
submit(commandEncoder);
|
|
244
|
+
submit(device, commandEncoder);
|
|
190
245
|
|
|
191
246
|
const gpuTimeMs = await extractTimestamp(ts);
|
|
192
247
|
|
package/src/util/benchmark.mjs
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/** @module devdocs/utility-functions/benchmark */
|
|
2
|
-
import {
|
|
2
|
+
import { isBenchmarkEnabled } from "../init.mjs";
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* Returns the `requestDevice` descriptor to pass to `adapter.requestDevice()`.
|
|
@@ -29,10 +29,9 @@ export function benchmarkMode(adapter, enabled) {
|
|
|
29
29
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUQuerySet GPUQuerySet}
|
|
30
30
|
* @see {@link https://developer.chrome.com/blog/new-in-webgpu-121 Chrome 121 — timestamp queries (querySet + timestampWrites pattern, quantization caveat)}
|
|
31
31
|
*/
|
|
32
|
-
export function beginTimestamp() {
|
|
33
|
-
if (!isBenchmarkEnabled())
|
|
32
|
+
export function beginTimestamp(device) {
|
|
33
|
+
if (!isBenchmarkEnabled(device))
|
|
34
34
|
return { querySet: null, passDescriptor: undefined };
|
|
35
|
-
const device = getDevice();
|
|
36
35
|
// Two slots: index 0 written when the pass begins, index 1 when it ends.
|
|
37
36
|
const querySet = device.createQuerySet({ type: "timestamp", count: 2 });
|
|
38
37
|
const passDescriptor = {
|
|
@@ -55,9 +54,8 @@ export function beginTimestamp() {
|
|
|
55
54
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/resolveQuerySet GPUCommandEncoder.resolveQuerySet()}
|
|
56
55
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/copyBufferToBuffer GPUCommandEncoder.copyBufferToBuffer()}
|
|
57
56
|
*/
|
|
58
|
-
export function resolveTimestamp(commandEncoder, querySet) {
|
|
57
|
+
export function resolveTimestamp(device, commandEncoder, querySet) {
|
|
59
58
|
if (!querySet) return null;
|
|
60
|
-
const device = getDevice();
|
|
61
59
|
// QUERY_RESOLVE and MAP_READ cannot be combined — two buffers are required.
|
|
62
60
|
// resolveBuffer: GPU writes resolved nanosecond timestamps here.
|
|
63
61
|
const resolveBuffer = device.createBuffer({
|
|
@@ -73,9 +71,11 @@ export function resolveTimestamp(commandEncoder, querySet) {
|
|
|
73
71
|
usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ,
|
|
74
72
|
});
|
|
75
73
|
commandEncoder.copyBufferToBuffer(
|
|
76
|
-
resolveBuffer,
|
|
77
|
-
|
|
78
|
-
|
|
74
|
+
resolveBuffer,
|
|
75
|
+
0, // src, srcOffset
|
|
76
|
+
tsReadBuffer,
|
|
77
|
+
0, // dst, dstOffset
|
|
78
|
+
16, // full 16 bytes (both timestamps)
|
|
79
79
|
);
|
|
80
80
|
// resolveBuffer is returned to prevent GC — the copy command is only encoded here, not yet executed.
|
|
81
81
|
return { tsReadBuffer, resolveBuffer, querySet };
|
package/src/util/bindgroup.mjs
CHANGED
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
/** @module devdocs/utility-functions/bindgroup */
|
|
2
|
-
import { getDevice } from "../init.mjs";
|
|
3
2
|
|
|
4
3
|
/**
|
|
5
4
|
* Creates a `GPUBindGroup` by mapping each buffer to sequential binding indices
|
|
@@ -16,8 +15,7 @@ import { getDevice } from "../init.mjs";
|
|
|
16
15
|
* @returns {GPUBindGroup}
|
|
17
16
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBindGroup GPUDevice.createBindGroup()}
|
|
18
17
|
*/
|
|
19
|
-
export function createBindGroup(layout, buffers, startBinding = 0) {
|
|
20
|
-
const device = getDevice();
|
|
18
|
+
export function createBindGroup(device, layout, buffers, startBinding = 0) {
|
|
21
19
|
const entries = buffers.map((buffer, i) => ({
|
|
22
20
|
binding: startBinding + i,
|
|
23
21
|
resource: buffer instanceof GPUBuffer ? { buffer } : buffer,
|
package/src/util/buffer.mjs
CHANGED
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
/** @module devdocs/utility-functions/buffer */
|
|
2
|
-
import { getDevice } from "../init.mjs";
|
|
3
2
|
|
|
4
3
|
/**
|
|
5
4
|
* Destroys one or more GPU buffers. Accepts individual buffers or arrays of buffers.
|
|
@@ -10,12 +9,38 @@ export function destroyBuffers(...buffers) {
|
|
|
10
9
|
buffers.flat().forEach((b) => b.destroy());
|
|
11
10
|
}
|
|
12
11
|
|
|
12
|
+
/**
|
|
13
|
+
* Throws if `byteSize` is more than this device can bind as a storage buffer.
|
|
14
|
+
*
|
|
15
|
+
* Every storage buffer the library creates goes through here. WebGPU accepts an
|
|
16
|
+
* oversized `createBuffer` and only rejects it later, when it is bound — as a
|
|
17
|
+
* `GPUValidationError` naming a bind group index rather than an operand, which
|
|
18
|
+
* gives no clue which allocation was at fault. Failing at creation, with the
|
|
19
|
+
* buffer's own label, points straight at it.
|
|
20
|
+
*
|
|
21
|
+
* @param {GPUDevice} device
|
|
22
|
+
* @param {number} byteSize
|
|
23
|
+
* @param {string} label - the buffer's debug label, quoted in the error
|
|
24
|
+
* @throws {Error} if `byteSize` exceeds `maxStorageBufferBindingSize`
|
|
25
|
+
* @internal
|
|
26
|
+
*/
|
|
27
|
+
function requireStorageSize(device, byteSize, label) {
|
|
28
|
+
const maxSize = device.limits.maxStorageBufferBindingSize;
|
|
29
|
+
if (byteSize > maxSize) {
|
|
30
|
+
throw new Error(
|
|
31
|
+
`Buffer "${label}" needs ${byteSize} bytes, exceeding this device's ` +
|
|
32
|
+
`maxStorageBufferBindingSize (${maxSize} bytes). The operands are too large for this device.`,
|
|
33
|
+
);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
|
|
13
37
|
/**
|
|
14
38
|
* Creates a GPU storage buffer and uploads `data` into it via mapped-at-creation.
|
|
15
39
|
* The mapped view is constructed from `data`'s own typed-array constructor, so
|
|
16
40
|
* bits are copied as-is regardless of element type (e.g. a Uint32Array's raw
|
|
17
41
|
* bit patterns are preserved — critical for dasum's aux half, which must
|
|
18
42
|
* never pass through a Float32Array view and risk NaN-bit-pattern canonicalization).
|
|
43
|
+
* @param {GPUDevice} device
|
|
19
44
|
* @param {Float32Array|Uint32Array|Int32Array} data
|
|
20
45
|
* @param {string} [label] - debug label visible in browser DevTools GPU inspection
|
|
21
46
|
* @param {boolean} [readback=false] - add `COPY_SRC` so the buffer can be copied to a readback buffer
|
|
@@ -26,17 +51,14 @@ export function destroyBuffers(...buffers) {
|
|
|
26
51
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/unmap GPUBuffer.unmap()}
|
|
27
52
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUSupportedLimits GPUSupportedLimits} (`maxStorageBufferBindingSize`)
|
|
28
53
|
*/
|
|
29
|
-
export function uploadBuffer(
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
54
|
+
export function uploadBuffer(
|
|
55
|
+
device,
|
|
56
|
+
data,
|
|
57
|
+
label = "blas-input",
|
|
58
|
+
readback = false,
|
|
59
|
+
) {
|
|
34
60
|
const byteSize = data.byteLength;
|
|
35
|
-
|
|
36
|
-
throw new Error(
|
|
37
|
-
`Buffer size ${byteSize} bytes exceeds device limit of ${maxSize} bytes.`,
|
|
38
|
-
);
|
|
39
|
-
}
|
|
61
|
+
requireStorageSize(device, byteSize, label);
|
|
40
62
|
|
|
41
63
|
const usage = readback
|
|
42
64
|
? GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC
|
|
@@ -60,15 +82,22 @@ export function uploadBuffer(data, label = "blas-input", readback = false) {
|
|
|
60
82
|
/**
|
|
61
83
|
* Creates an uninitialised GPU storage buffer. Used for intermediate buffers
|
|
62
84
|
* that are written by a shader before being read.
|
|
85
|
+
* @param {GPUDevice} device
|
|
63
86
|
* @param {number} size - byte size
|
|
64
87
|
* @param {string} [label] - debug label visible in browser DevTools GPU inspection
|
|
65
88
|
* @param {number} [extraUsage=0] - additional `GPUBufferUsage` flags OR'd in alongside `STORAGE`
|
|
66
89
|
* (e.g. `GPUBufferUsage.COPY_DST` for a buffer that's also a `copyBufferToBuffer` destination)
|
|
67
90
|
* @returns {GPUBuffer}
|
|
91
|
+
* @throws {Error} if `size` exceeds the device's `maxStorageBufferBindingSize`
|
|
68
92
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBuffer GPUDevice.createBuffer()}
|
|
69
93
|
*/
|
|
70
|
-
export function createStorageBuffer(
|
|
71
|
-
|
|
94
|
+
export function createStorageBuffer(
|
|
95
|
+
device,
|
|
96
|
+
size,
|
|
97
|
+
label = "blas-storage",
|
|
98
|
+
extraUsage = 0,
|
|
99
|
+
) {
|
|
100
|
+
requireStorageSize(device, size, label);
|
|
72
101
|
return device.createBuffer({
|
|
73
102
|
label,
|
|
74
103
|
size,
|
|
@@ -79,13 +108,15 @@ export function createStorageBuffer(size, label = "blas-storage", extraUsage = 0
|
|
|
79
108
|
/**
|
|
80
109
|
* Creates a GPU storage buffer with `COPY_SRC` so its contents can be
|
|
81
110
|
* copied to a CPU-readable readback buffer after the shader runs.
|
|
111
|
+
* @param {GPUDevice} device
|
|
82
112
|
* @param {number} size - byte size
|
|
83
113
|
* @param {string} [label] - debug label visible in browser DevTools GPU inspection
|
|
84
114
|
* @returns {GPUBuffer}
|
|
115
|
+
* @throws {Error} if `size` exceeds the device's `maxStorageBufferBindingSize`
|
|
85
116
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBuffer GPUDevice.createBuffer()}
|
|
86
117
|
*/
|
|
87
|
-
export function createResultBuffer(size, label = "blas-result") {
|
|
88
|
-
|
|
118
|
+
export function createResultBuffer(device, size, label = "blas-result") {
|
|
119
|
+
requireStorageSize(device, size, label);
|
|
89
120
|
return device.createBuffer({
|
|
90
121
|
label,
|
|
91
122
|
size,
|
|
@@ -97,14 +128,13 @@ export function createResultBuffer(size, label = "blas-result") {
|
|
|
97
128
|
* Appends a `copyBufferToBuffer` command to `commandEncoder` that copies
|
|
98
129
|
* `sourceBuffer` into a new `MAP_READ` buffer. Returns that readback buffer;
|
|
99
130
|
* call `readBuffer.mapAsync(GPUMapMode.READ)` after submitting the encoder.
|
|
131
|
+
* @param {GPUDevice} device
|
|
100
132
|
* @param {GPUCommandEncoder} commandEncoder
|
|
101
133
|
* @param {GPUBuffer} sourceBuffer
|
|
102
134
|
* @returns {GPUBuffer}
|
|
103
135
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/copyBufferToBuffer GPUCommandEncoder.copyBufferToBuffer()}
|
|
104
136
|
*/
|
|
105
|
-
export function stageReadback(commandEncoder, sourceBuffer) {
|
|
106
|
-
const device = getDevice();
|
|
107
|
-
|
|
137
|
+
export function stageReadback(device, commandEncoder, sourceBuffer) {
|
|
108
138
|
// COPY_DST: receives the copyBufferToBuffer transfer; MAP_READ: lets the CPU map and read it back.
|
|
109
139
|
const readBuffer = device.createBuffer({
|
|
110
140
|
label: "blas-readback",
|
|
@@ -113,26 +143,111 @@ export function stageReadback(commandEncoder, sourceBuffer) {
|
|
|
113
143
|
});
|
|
114
144
|
|
|
115
145
|
commandEncoder.copyBufferToBuffer(
|
|
116
|
-
sourceBuffer,
|
|
117
|
-
|
|
118
|
-
|
|
146
|
+
sourceBuffer,
|
|
147
|
+
0, // src, srcOffset
|
|
148
|
+
readBuffer,
|
|
149
|
+
0, // dst, dstOffset
|
|
150
|
+
sourceBuffer.size, // full copy, no partial reads
|
|
119
151
|
);
|
|
120
152
|
|
|
121
153
|
return readBuffer;
|
|
122
154
|
}
|
|
123
155
|
|
|
156
|
+
// Minimum bindable size for an array<vec4<f32>> view: one 16-byte element.
|
|
157
|
+
const VEC4_ELEM_BYTES = 16;
|
|
158
|
+
|
|
159
|
+
// Dummy STORAGE buffer bound into a kernel's unused vec4-view slot when the
|
|
160
|
+
// real operand can't host even one vec4 (tiny-matrix edge cases, where the
|
|
161
|
+
// stride check forces the scalar path anyway). Cached per device.
|
|
162
|
+
const _vec4Fallbacks = new WeakMap();
|
|
163
|
+
function vec4FallbackBuffer(device) {
|
|
164
|
+
let b = _vec4Fallbacks.get(device);
|
|
165
|
+
if (!b) {
|
|
166
|
+
b = device.createBuffer({
|
|
167
|
+
label: "blas-vec4-fallback",
|
|
168
|
+
size: VEC4_ELEM_BYTES,
|
|
169
|
+
usage: GPUBufferUsage.STORAGE,
|
|
170
|
+
});
|
|
171
|
+
_vec4Fallbacks.set(device, b);
|
|
172
|
+
}
|
|
173
|
+
return b;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* Bind-group entry exposing `buffer`'s bytes as an `array<vec4<f32>>` view —
|
|
178
|
+
* the twin binding that lets a shader issue 16-byte vector loads alongside
|
|
179
|
+
* scalar access of the same storage (bind the same GPUBuffer twice: once as
|
|
180
|
+
* `array<f32>`, once through this). The view's size is rounded down to a
|
|
181
|
+
* multiple of 16 because some backends reject non-multiple-of-16 ranges for
|
|
182
|
+
* vec4 arrays; whenever a kernel's vector path is usable (stride % 4 == 0)
|
|
183
|
+
* the buffer size is itself a multiple of 16, so the rounding never truncates
|
|
184
|
+
* a component the vector path would actually read. Buffers smaller than one
|
|
185
|
+
* vec4 element get a shared dummy storage buffer bound instead — the shader
|
|
186
|
+
* never dereferences it on those shapes.
|
|
187
|
+
* @param {GPUBuffer|{buffer: GPUBuffer, offset?: number, size?: number}} entry - whole buffer or sub-range, matching what the scalar slot binds
|
|
188
|
+
* @returns {{buffer: GPUBuffer, offset: number, size: number}}
|
|
189
|
+
*/
|
|
190
|
+
export function vec4ViewBinding(device, entry) {
|
|
191
|
+
const buffer = entry instanceof GPUBuffer ? entry : entry.buffer;
|
|
192
|
+
const offset = entry instanceof GPUBuffer ? 0 : (entry.offset ?? 0);
|
|
193
|
+
const avail =
|
|
194
|
+
entry instanceof GPUBuffer
|
|
195
|
+
? entry.size
|
|
196
|
+
: (entry.size ?? buffer.size - offset);
|
|
197
|
+
const size = Math.floor(avail / VEC4_ELEM_BYTES) * VEC4_ELEM_BYTES;
|
|
198
|
+
if (size < VEC4_ELEM_BYTES) {
|
|
199
|
+
return {
|
|
200
|
+
buffer: vec4FallbackBuffer(device),
|
|
201
|
+
offset: 0,
|
|
202
|
+
size: VEC4_ELEM_BYTES,
|
|
203
|
+
};
|
|
204
|
+
}
|
|
205
|
+
return { buffer, offset, size };
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Whether every in-bounds element of a matrix operand is reachable through
|
|
210
|
+
* the `array<vec4<f32>>` view that {@link vec4ViewBinding} produces for it.
|
|
211
|
+
* The view truncates the binding down to a multiple of 16 bytes, so a
|
|
212
|
+
* tightly-uploaded array can hold valid matrix elements past the view's end
|
|
213
|
+
* even while its stride is 4-aligned (e.g. a column-major m×1 operand with
|
|
214
|
+
* padded lda uploaded without padding cells). Call this with the kernel-side
|
|
215
|
+
* dimensions and only take a shader's vectorized path when it returns true;
|
|
216
|
+
* the scalar fallback reads the full storage and is always correct.
|
|
217
|
+
* @param {GPUBuffer|{buffer: GPUBuffer, offset?: number, size?: number}} entry - what the scalar slot binds
|
|
218
|
+
* @param {number} stride - the operand's leading dimension as seen by the kernel
|
|
219
|
+
* @param {number} outerCount - extent of the stride-multiplied dimension
|
|
220
|
+
* @param {number} innerCount - extent of the contiguous dimension
|
|
221
|
+
* @returns {boolean}
|
|
222
|
+
*/
|
|
223
|
+
export function vec4Usable(entry, stride, outerCount, innerCount) {
|
|
224
|
+
if (stride % 4 !== 0) return false;
|
|
225
|
+
const buffer = entry instanceof GPUBuffer ? entry : entry.buffer;
|
|
226
|
+
const offset = entry instanceof GPUBuffer ? 0 : (entry.offset ?? 0);
|
|
227
|
+
const avail =
|
|
228
|
+
entry instanceof GPUBuffer
|
|
229
|
+
? buffer.size
|
|
230
|
+
: (entry.size ?? buffer.size - offset);
|
|
231
|
+
const viewFloats = Math.floor(avail / VEC4_ELEM_BYTES) * 4;
|
|
232
|
+
if (viewFloats <= 0) return false;
|
|
233
|
+
// Highest flat index any masked-in component can touch; usable iff its
|
|
234
|
+
// containing vec4 ends within the view.
|
|
235
|
+
const maxFlat =
|
|
236
|
+
(Math.max(outerCount, 1) - 1) * stride + (Math.max(innerCount, 1) - 1);
|
|
237
|
+
return Math.floor(maxFlat / 4) * 4 + 4 <= viewFloats;
|
|
238
|
+
}
|
|
239
|
+
|
|
124
240
|
/**
|
|
125
241
|
* Packs an array of typed scalar values into a uniform buffer aligned to 16 bytes.
|
|
126
242
|
* Each entry specifies the value and its WGSL type (`"f32"`, `"u32"`, or `"i32"`).
|
|
127
243
|
* The order of entries must match the field order in the shader's `Params` struct.
|
|
244
|
+
* @param {GPUDevice} device
|
|
128
245
|
* @param {{ value: number, type: "f32"|"u32"|"i32" }[]} params
|
|
129
246
|
* @param {string} [label] - debug label visible in browser DevTools GPU inspection
|
|
130
247
|
* @returns {GPUBuffer}
|
|
131
248
|
* @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUQueue/writeBuffer GPUQueue.writeBuffer()}
|
|
132
249
|
*/
|
|
133
|
-
export function createParamsBuffer(params, label = "blas-params") {
|
|
134
|
-
const device = getDevice();
|
|
135
|
-
|
|
250
|
+
export function createParamsBuffer(device, params, label = "blas-params") {
|
|
136
251
|
const rawSize = params.length * 4;
|
|
137
252
|
const size = Math.ceil(rawSize / 16) * 16;
|
|
138
253
|
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
/** @module devdocs/utility-functions/complex */
|
|
2
|
+
import { splitDoubleDouble, mergeDoubleDouble } from "./f64.mjs";
|
|
3
|
+
import { Complex64, Complex64Array } from "../classes/Complex64.mjs";
|
|
4
|
+
|
|
5
|
+
// Complex32Array <-> interleaved [re0, im0, re1, im1, ...] f32 buffer — one
|
|
6
|
+
// buffer, not a (hi, lo) pair like Float64Array (see f64.mjs), since re/im
|
|
7
|
+
// need no error compensation. Matches cuBLAS/stdlib's own complex layout.
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Interleaves the first `n` elements of a Complex32Array into a flat f32
|
|
11
|
+
* buffer ready for `uploadBuffer`.
|
|
12
|
+
* @param {import("../classes/Complex32.mjs").Complex32Array} data
|
|
13
|
+
* @param {number} [n] - element count to interleave (default: data.length)
|
|
14
|
+
* @returns {Float32Array} length `2*n`, [re0, im0, re1, im1, ...]
|
|
15
|
+
* @public
|
|
16
|
+
*/
|
|
17
|
+
export function interleaveComplex32(data, n = data.length) {
|
|
18
|
+
const flat = new Float32Array(n * 2);
|
|
19
|
+
for (let i = 0; i < n; i++) {
|
|
20
|
+
flat[i * 2] = data[i].re;
|
|
21
|
+
flat[i * 2 + 1] = data[i].im;
|
|
22
|
+
}
|
|
23
|
+
return flat;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// Complex64Array <-> a double-double (hi, lo) pair of interleaved f32
|
|
27
|
+
// buffers. re and im each need their own (hi, lo) split (see f64.mjs) —
|
|
28
|
+
// zipped together per channel, [reHi0,imHi0,...] / [reLo0,imLo0,...],
|
|
29
|
+
// rather than four separate buffers, so GpuVector/GpuMatrix's existing
|
|
30
|
+
// two-buffer shape covers this dtype with no restructuring.
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Splits the first `n` elements of a Complex64Array into an interleaved
|
|
34
|
+
* double-double (hi, lo) pair of f32 buffers.
|
|
35
|
+
* @param {Complex64Array} data
|
|
36
|
+
* @param {number} [n] - element count to split (default: data.length)
|
|
37
|
+
* @returns {{hi: Float32Array, lo: Float32Array}} each length `2*n`,
|
|
38
|
+
* [reHi0, imHi0, reHi1, imHi1, ...] / [reLo0, imLo0, reLo1, imLo1, ...]
|
|
39
|
+
* @public
|
|
40
|
+
*/
|
|
41
|
+
export function splitComplex64(data, n = data.length) {
|
|
42
|
+
const re = new Float64Array(n);
|
|
43
|
+
const im = new Float64Array(n);
|
|
44
|
+
for (let i = 0; i < n; i++) {
|
|
45
|
+
re[i] = data[i].re;
|
|
46
|
+
im[i] = data[i].im;
|
|
47
|
+
}
|
|
48
|
+
const { hi: reHi, lo: reLo } = splitDoubleDouble(re);
|
|
49
|
+
const { hi: imHi, lo: imLo } = splitDoubleDouble(im);
|
|
50
|
+
|
|
51
|
+
const hi = new Float32Array(n * 2);
|
|
52
|
+
const lo = new Float32Array(n * 2);
|
|
53
|
+
for (let i = 0; i < n; i++) {
|
|
54
|
+
hi[i * 2] = reHi[i];
|
|
55
|
+
hi[i * 2 + 1] = imHi[i];
|
|
56
|
+
lo[i * 2] = reLo[i];
|
|
57
|
+
lo[i * 2 + 1] = imLo[i];
|
|
58
|
+
}
|
|
59
|
+
return { hi, lo };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Reassembles a Complex64Array from an interleaved double-double (hi, lo)
|
|
64
|
+
* pair of f32 buffers — the inverse of splitComplex64.
|
|
65
|
+
* @param {Float32Array} hi - [reHi0, imHi0, reHi1, imHi1, ...]
|
|
66
|
+
* @param {Float32Array} lo - [reLo0, imLo0, reLo1, imLo1, ...]
|
|
67
|
+
* @returns {Complex64Array}
|
|
68
|
+
* @public
|
|
69
|
+
*/
|
|
70
|
+
export function mergeComplex64(hi, lo) {
|
|
71
|
+
const n = hi.length / 2;
|
|
72
|
+
const reHi = new Float32Array(n),
|
|
73
|
+
reLo = new Float32Array(n);
|
|
74
|
+
const imHi = new Float32Array(n),
|
|
75
|
+
imLo = new Float32Array(n);
|
|
76
|
+
for (let i = 0; i < n; i++) {
|
|
77
|
+
reHi[i] = hi[i * 2];
|
|
78
|
+
imHi[i] = hi[i * 2 + 1];
|
|
79
|
+
reLo[i] = lo[i * 2];
|
|
80
|
+
imLo[i] = lo[i * 2 + 1];
|
|
81
|
+
}
|
|
82
|
+
const re = mergeDoubleDouble(reHi, reLo);
|
|
83
|
+
const im = mergeDoubleDouble(imHi, imLo);
|
|
84
|
+
const out = new Complex64Array(n);
|
|
85
|
+
for (let i = 0; i < n; i++) out[i] = new Complex64(re[i], im[i]);
|
|
86
|
+
return out;
|
|
87
|
+
}
|