wgblas 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +161 -0
  3. package/dist/wgblas.browser.js +422 -0
  4. package/index.d.mts +94 -0
  5. package/index.mjs +16 -0
  6. package/package.json +127 -0
  7. package/src/classes/GpuVector.d.mts +69 -0
  8. package/src/classes/GpuVector.mjs +33 -0
  9. package/src/devdocs.mjs +81 -0
  10. package/src/index.mjs +4 -0
  11. package/src/init.mjs +105 -0
  12. package/src/isamax/isamax.d.mts +46 -0
  13. package/src/isamax/isamax.mjs +115 -0
  14. package/src/random/random.d.mts +61 -0
  15. package/src/random/random.mjs +11 -0
  16. package/src/sasum/sasum.d.mts +44 -0
  17. package/src/sasum/sasum.mjs +98 -0
  18. package/src/saxpy/saxpy.d.mts +54 -0
  19. package/src/saxpy/saxpy.mjs +90 -0
  20. package/src/scopy/scopy.d.mts +50 -0
  21. package/src/scopy/scopy.mjs +87 -0
  22. package/src/sdot/sdot.d.mts +52 -0
  23. package/src/sdot/sdot.mjs +118 -0
  24. package/src/shaders/browser-shaders.mjs +27 -0
  25. package/src/shaders/index.mjs +27 -0
  26. package/src/snrm2/snrm2.d.mts +44 -0
  27. package/src/snrm2/snrm2.mjs +100 -0
  28. package/src/srot/srot.d.mts +62 -0
  29. package/src/srot/srot.mjs +95 -0
  30. package/src/srotm/srotm.d.mts +60 -0
  31. package/src/srotm/srotm.mjs +94 -0
  32. package/src/sscal/sscal.d.mts +46 -0
  33. package/src/sscal/sscal.mjs +71 -0
  34. package/src/sswap/sswap.d.mts +50 -0
  35. package/src/sswap/sswap.mjs +90 -0
  36. package/src/util/benchmark.mjs +103 -0
  37. package/src/util/bindgroup.mjs +22 -0
  38. package/src/util/buffer.mjs +160 -0
  39. package/src/util/compute.mjs +52 -0
  40. package/src/util/index.mjs +12 -0
  41. package/src/util/pipeline.mjs +82 -0
  42. package/src/util/result.mjs +19 -0
  43. package/src/util/workgroup.mjs +31 -0
@@ -0,0 +1,90 @@
1
+ import {
2
+ uploadBuffer,
3
+ createParamsBuffer,
4
+ stageReadback,
5
+ destroyBuffers,
6
+ } from "../util/buffer.mjs";
7
+ import { createBindGroup } from "../util/bindgroup.mjs";
8
+ import { runComputePass, submit } from "../util/compute.mjs";
9
+ import { extractResult } from "../util/result.mjs";
10
+ import { extractTimestamp } from "../util/benchmark.mjs";
11
+ import { getPipeline } from "../util/pipeline.mjs";
12
+ import { calcWorkgroups } from "../util/workgroup.mjs";
13
+ import { GpuVector } from "../classes/GpuVector.mjs";
14
+
15
+ export async function sswap(device, n, x, incx, y, incy) {
16
+ const xIsGpu = x instanceof GpuVector;
17
+ const yIsGpu = y instanceof GpuVector;
18
+
19
+ if (!(device instanceof GPUDevice))
20
+ throw new Error("device must be a GPUDevice.");
21
+ if (
22
+ !Number.isInteger(n) ||
23
+ !Number.isInteger(incx) ||
24
+ !Number.isInteger(incy)
25
+ )
26
+ throw new Error("n, incx, and incy must be integers.");
27
+ if (incx <= 0 || incy <= 0)
28
+ throw new Error("incx and incy must be positive.");
29
+ if (!(x instanceof Float32Array) && !(x instanceof GpuVector))
30
+ throw new Error("x must be a Float32Array or GpuVector.");
31
+ if (!(y instanceof Float32Array) && !(y instanceof GpuVector))
32
+ throw new Error("y must be a Float32Array or GpuVector.");
33
+ if (x.constructor !== y.constructor)
34
+ throw new Error(
35
+ "x and y must be the same type (both Float32Array or both GpuVector).",
36
+ );
37
+ if (n <= 0) return xIsGpu ? {} : { x, y };
38
+ if (x.length < (n - 1) * incx + 1)
39
+ throw new Error(
40
+ "x does not have enough elements for the given n and incx.",
41
+ );
42
+ if (y.length < (n - 1) * incy + 1)
43
+ throw new Error(
44
+ "y does not have enough elements for the given n and incy.",
45
+ );
46
+
47
+ const pipeline = await getPipeline(device, "sswap");
48
+
49
+ const xBuffer = xIsGpu ? x._buf : uploadBuffer(x, "sswap-x", true);
50
+ const yBuffer = yIsGpu ? y._buf : uploadBuffer(y, "sswap-y", true);
51
+ const paramsBuffer = createParamsBuffer(
52
+ [
53
+ { value: n, type: "u32" },
54
+ { value: incx, type: "u32" },
55
+ { value: incy, type: "u32" },
56
+ ],
57
+ "sswap-params",
58
+ );
59
+
60
+ const bindGroup = createBindGroup(pipeline.getBindGroupLayout(0), [
61
+ xBuffer,
62
+ yBuffer,
63
+ paramsBuffer,
64
+ ]);
65
+ const { commandEncoder, ts } = runComputePass(
66
+ pipeline,
67
+ bindGroup,
68
+ calcWorkgroups(n),
69
+ );
70
+ const xReadBuffer = xIsGpu ? null : stageReadback(commandEncoder, xBuffer);
71
+ const yReadBuffer = yIsGpu ? null : stageReadback(commandEncoder, yBuffer);
72
+
73
+ submit(commandEncoder);
74
+
75
+ const gpuTimeMs = await extractTimestamp(ts);
76
+
77
+ if (xIsGpu && yIsGpu) {
78
+ destroyBuffers(paramsBuffer);
79
+ if (gpuTimeMs !== undefined) return { gpuTimeMs };
80
+ return {};
81
+ }
82
+
83
+ const resultX = await extractResult(xReadBuffer, Float32Array);
84
+ const resultY = await extractResult(yReadBuffer, Float32Array);
85
+
86
+ destroyBuffers(xBuffer, xReadBuffer, yBuffer, yReadBuffer, paramsBuffer);
87
+
88
+ if (gpuTimeMs !== undefined) return { x: resultX, y: resultY, gpuTimeMs };
89
+ return { x: resultX, y: resultY };
90
+ }
@@ -0,0 +1,103 @@
1
+ /** @module devdocs/utility-functions/benchmark */
2
+ import { getDevice, isBenchmarkEnabled } from "../init.mjs";
3
+
4
+ /**
5
+ * Returns the `requestDevice` descriptor to pass to `adapter.requestDevice()`.
6
+ * Includes `requiredFeatures: ["timestamp-query"]` when benchmark is enabled and supported;
7
+ * returns an empty descriptor otherwise so the device is still created without benchmark support.
8
+ * @param {GPUAdapter} adapter
9
+ * @param {boolean} enabled
10
+ * @returns {{ requiredFeatures?: string[] }}
11
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUSupportedFeatures GPUSupportedFeatures.has()} (`timestamp-query`)
12
+ * @see {@link https://developer.chrome.com/blog/new-in-webgpu-121 Chrome 121 — timestamp queries}
13
+ */
14
+ export function benchmarkMode(adapter, enabled) {
15
+ if (!enabled) return {};
16
+ if (adapter.features.has("timestamp-query"))
17
+ return { requiredFeatures: ["timestamp-query"] };
18
+ console.warn(
19
+ "timestamp-query not supported on this device — benchmark mode disabled.",
20
+ );
21
+ return {};
22
+ }
23
+
24
+ /**
25
+ * Creates a `GPUQuerySet` with two timestamp slots and a `passDescriptor` to pass
26
+ * to `beginComputePass`. Index 0 = beginning of pass, index 1 = end of pass.
27
+ * Returns nulls if benchmark mode is off.
28
+ * @returns {{ querySet: GPUQuerySet|null, passDescriptor: object|undefined }}
29
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUQuerySet GPUQuerySet}
30
+ * @see {@link https://developer.chrome.com/blog/new-in-webgpu-121 Chrome 121 — timestamp queries (querySet + timestampWrites pattern, quantization caveat)}
31
+ */
32
+ export function beginTimestamp() {
33
+ if (!isBenchmarkEnabled())
34
+ return { querySet: null, passDescriptor: undefined };
35
+ const device = getDevice();
36
+ // Two slots: index 0 written when the pass begins, index 1 when it ends.
37
+ const querySet = device.createQuerySet({ type: "timestamp", count: 2 });
38
+ const passDescriptor = {
39
+ timestampWrites: {
40
+ querySet,
41
+ beginningOfPassWriteIndex: 0,
42
+ endOfPassWriteIndex: 1,
43
+ },
44
+ };
45
+ return { querySet, passDescriptor };
46
+ }
47
+
48
+ /**
49
+ * Encodes commands to resolve the two timestamp slots into a CPU-readable buffer.
50
+ * Timestamps are stored as 64-bit unsigned integers (nanoseconds); `extractTimestamp`
51
+ * reads them back. Returns `null` if `querySet` is null (benchmark mode off).
52
+ * @param {GPUCommandEncoder} commandEncoder
53
+ * @param {GPUQuerySet|null} querySet
54
+ * @returns {{ tsReadBuffer: GPUBuffer, resolveBuffer: GPUBuffer, querySet: GPUQuerySet }|null}
55
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/resolveQuerySet GPUCommandEncoder.resolveQuerySet()}
56
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/copyBufferToBuffer GPUCommandEncoder.copyBufferToBuffer()}
57
+ */
58
+ export function resolveTimestamp(commandEncoder, querySet) {
59
+ if (!querySet) return null;
60
+ const device = getDevice();
61
+ // QUERY_RESOLVE and MAP_READ cannot be combined — two buffers are required.
62
+ // resolveBuffer: GPU writes resolved nanosecond timestamps here.
63
+ const resolveBuffer = device.createBuffer({
64
+ label: "timestamp-resolve",
65
+ size: 16, // 2 × BigInt64 (8 bytes each)
66
+ usage: GPUBufferUsage.QUERY_RESOLVE | GPUBufferUsage.COPY_SRC,
67
+ });
68
+ commandEncoder.resolveQuerySet(querySet, 0, 2, resolveBuffer, 0);
69
+ // tsReadBuffer: CPU-readable copy of resolveBuffer, mapped in extractTimestamp.
70
+ const tsReadBuffer = device.createBuffer({
71
+ label: "timestamp-readback",
72
+ size: 16,
73
+ usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ,
74
+ });
75
+ commandEncoder.copyBufferToBuffer(
76
+ resolveBuffer, 0, // src, srcOffset
77
+ tsReadBuffer, 0, // dst, dstOffset
78
+ 16, // full 16 bytes (both timestamps)
79
+ );
80
+ // resolveBuffer is returned to prevent GC — the copy command is only encoded here, not yet executed.
81
+ return { tsReadBuffer, resolveBuffer, querySet };
82
+ }
83
+
84
+ /**
85
+ * Maps the readback buffer, computes `timestamps[1] - timestamps[0]` (nanoseconds),
86
+ * converts to milliseconds, then destroys all three handles.
87
+ * Returns `undefined` if `ts` is null (benchmark mode off).
88
+ * @param {{ tsReadBuffer: GPUBuffer, resolveBuffer: GPUBuffer, querySet: GPUQuerySet }|null} ts
89
+ * @returns {Promise<number|undefined>} elapsed GPU time in ms
90
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/mapAsync GPUBuffer.mapAsync()}
91
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/getMappedRange GPUBuffer.getMappedRange()}
92
+ */
93
+ export async function extractTimestamp(ts) {
94
+ if (!ts) return undefined;
95
+ const { tsReadBuffer, resolveBuffer, querySet } = ts;
96
+ await tsReadBuffer.mapAsync(GPUMapMode.READ);
97
+ const timestamps = new BigInt64Array(tsReadBuffer.getMappedRange().slice());
98
+ tsReadBuffer.unmap();
99
+ tsReadBuffer.destroy();
100
+ resolveBuffer.destroy(); // never mapped — no unmap() needed
101
+ querySet.destroy();
102
+ return Number(timestamps[1] - timestamps[0]) / 1e6;
103
+ }
@@ -0,0 +1,22 @@
1
+ /** @module devdocs/utility-functions/bindgroup */
2
+ import { getDevice } from "../init.mjs";
3
+
4
+ /**
5
+ * Creates a `GPUBindGroup` by mapping each buffer to sequential binding indices (0, 1, 2 …).
6
+ * `resultBuffer` is appended last so its binding index follows all input buffers.
7
+ * The order of `buffers` must match the `@binding` indices declared in the shader.
8
+ * @param {GPUBindGroupLayout} layout - from `pipeline.getBindGroupLayout(0)`; the layout is derived automatically from the shader because `loadShader` uses `layout: "auto"`
9
+ * @param {GPUBuffer[]} buffers - input and intermediate buffers in binding order
10
+ * @param {GPUBuffer|null} [resultBuffer=null] - appended as the final binding if provided
11
+ * @returns {GPUBindGroup}
12
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBindGroup GPUDevice.createBindGroup()}
13
+ */
14
+ export function createBindGroup(layout, buffers, resultBuffer = null) {
15
+ const device = getDevice();
16
+ const allBuffers = resultBuffer ? [...buffers, resultBuffer] : [...buffers];
17
+ const entries = allBuffers.map((buffer, i) => ({
18
+ binding: i,
19
+ resource: { buffer },
20
+ }));
21
+ return device.createBindGroup({ layout, entries });
22
+ }
@@ -0,0 +1,160 @@
1
+ /** @module devdocs/utility-functions/buffer */
2
+ import { getDevice } from "../init.mjs";
3
+
4
+ /**
5
+ * Destroys one or more GPU buffers. Accepts individual buffers or arrays of buffers.
6
+ * @param {...(GPUBuffer|GPUBuffer[])} buffers
7
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/destroy GPUBuffer.destroy()}
8
+ */
9
+ export function destroyBuffers(...buffers) {
10
+ buffers.flat().forEach((b) => b.destroy());
11
+ }
12
+
13
+ /**
14
+ * Creates a GPU storage buffer and uploads `data` into it via mapped-at-creation.
15
+ * @param {Float32Array} data
16
+ * @param {string} [label] - debug label visible in browser DevTools GPU inspection
17
+ * @param {boolean} [readback=false] - add `COPY_SRC` so the buffer can be copied to a readback buffer
18
+ * @throws {Error} if `data.byteLength` exceeds the device's `maxStorageBufferBindingSize`
19
+ * @returns {GPUBuffer}
20
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBuffer GPUDevice.createBuffer()}
21
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/getMappedRange GPUBuffer.getMappedRange()}
22
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/unmap GPUBuffer.unmap()}
23
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUSupportedLimits GPUSupportedLimits} (`maxStorageBufferBindingSize`)
24
+ */
25
+ export function uploadBuffer(data, label = "blas-input", readback = false) {
26
+ const device = getDevice();
27
+
28
+ // User-facing boundary: give a clear error instead of a cryptic GPUValidationError.
29
+ const maxSize = device.limits.maxStorageBufferBindingSize;
30
+ const byteSize = data.byteLength;
31
+ if (byteSize > maxSize) {
32
+ throw new Error(
33
+ `Buffer size ${byteSize} bytes exceeds device limit of ${maxSize} bytes.`,
34
+ );
35
+ }
36
+
37
+ const usage = readback
38
+ ? GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC
39
+ : GPUBufferUsage.STORAGE;
40
+
41
+ const buffer = device.createBuffer({
42
+ label,
43
+ size: byteSize,
44
+ usage,
45
+ mappedAtCreation: true,
46
+ });
47
+
48
+ const mappedArray = new Float32Array(buffer.getMappedRange());
49
+ mappedArray.set(data);
50
+ buffer.unmap();
51
+
52
+ return buffer;
53
+ }
54
+
55
+ /**
56
+ * Creates an uninitialised GPU storage buffer. Used for intermediate buffers
57
+ * that are written by a shader before being read.
58
+ * @param {number} size - byte size
59
+ * @param {string} [label] - debug label visible in browser DevTools GPU inspection
60
+ * @returns {GPUBuffer}
61
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBuffer GPUDevice.createBuffer()}
62
+ */
63
+ export function createStorageBuffer(size, label = "blas-storage") {
64
+ const device = getDevice();
65
+ return device.createBuffer({
66
+ label,
67
+ size,
68
+ usage: GPUBufferUsage.STORAGE,
69
+ });
70
+ }
71
+
72
+ /**
73
+ * Creates a GPU storage buffer with `COPY_SRC` so its contents can be
74
+ * copied to a CPU-readable readback buffer after the shader runs.
75
+ * @param {number} size - byte size
76
+ * @param {string} [label] - debug label visible in browser DevTools GPU inspection
77
+ * @returns {GPUBuffer}
78
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createBuffer GPUDevice.createBuffer()}
79
+ */
80
+ export function createResultBuffer(size, label = "blas-result") {
81
+ const device = getDevice();
82
+ return device.createBuffer({
83
+ label,
84
+ size,
85
+ usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC,
86
+ });
87
+ }
88
+
89
+ /**
90
+ * Appends a `copyBufferToBuffer` command to `commandEncoder` that copies
91
+ * `sourceBuffer` into a new `MAP_READ` buffer. Returns that readback buffer;
92
+ * call `readBuffer.mapAsync(GPUMapMode.READ)` after submitting the encoder.
93
+ * @param {GPUCommandEncoder} commandEncoder
94
+ * @param {GPUBuffer} sourceBuffer
95
+ * @returns {GPUBuffer}
96
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/copyBufferToBuffer GPUCommandEncoder.copyBufferToBuffer()}
97
+ */
98
+ export function stageReadback(commandEncoder, sourceBuffer) {
99
+ const device = getDevice();
100
+
101
+ // COPY_DST: receives the copyBufferToBuffer transfer; MAP_READ: lets the CPU map and read it back.
102
+ const readBuffer = device.createBuffer({
103
+ label: "blas-readback",
104
+ size: sourceBuffer.size,
105
+ usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ,
106
+ });
107
+
108
+ commandEncoder.copyBufferToBuffer(
109
+ sourceBuffer, 0, // src, srcOffset
110
+ readBuffer, 0, // dst, dstOffset
111
+ sourceBuffer.size, // full copy, no partial reads
112
+ );
113
+
114
+ return readBuffer;
115
+ }
116
+
117
+ /**
118
+ * Packs an array of typed scalar values into a uniform buffer aligned to 16 bytes.
119
+ * Each entry specifies the value and its WGSL type (`"f32"`, `"u32"`, or `"i32"`).
120
+ * The order of entries must match the field order in the shader's `Params` struct.
121
+ * @param {{ value: number, type: "f32"|"u32"|"i32" }[]} params
122
+ * @param {string} [label] - debug label visible in browser DevTools GPU inspection
123
+ * @returns {GPUBuffer}
124
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUQueue/writeBuffer GPUQueue.writeBuffer()}
125
+ */
126
+ export function createParamsBuffer(params, label = "blas-params") {
127
+ const device = getDevice();
128
+
129
+ const rawSize = params.length * 4;
130
+ const size = Math.ceil(rawSize / 16) * 16;
131
+
132
+ const arrayBuffer = new ArrayBuffer(size);
133
+ const view = new DataView(arrayBuffer);
134
+
135
+ params.forEach(({ value, type }, i) => {
136
+ const offset = i * 4;
137
+ if (type === "u32") {
138
+ view.setUint32(offset, value, true);
139
+ } else if (type === "i32") {
140
+ view.setInt32(offset, value, true);
141
+ } else if (type === "f32") {
142
+ view.setFloat32(offset, value, true);
143
+ } else {
144
+ throw new Error(
145
+ `Unknown param type "${type}". Use "f32", "u32", or "i32".`,
146
+ );
147
+ }
148
+ });
149
+
150
+ // UNIFORM: binds as var<uniform> in the shader; COPY_DST: allows writeBuffer to upload the packed data.
151
+ const buffer = device.createBuffer({
152
+ label,
153
+ size,
154
+ usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST,
155
+ });
156
+
157
+ device.queue.writeBuffer(buffer, 0, arrayBuffer);
158
+
159
+ return buffer;
160
+ }
@@ -0,0 +1,52 @@
1
+ /** @module devdocs/utility-functions/compute */
2
+ import { getDevice } from "../init.mjs";
3
+ import { beginTimestamp, resolveTimestamp } from "./benchmark.mjs";
4
+
5
+ /**
6
+ * Finalises `commandEncoder` into a command buffer and submits it to the GPU queue.
7
+ * @param {GPUCommandEncoder} commandEncoder
8
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUQueue/submit GPUQueue.submit()}
9
+ */
10
+ export function submit(commandEncoder) {
11
+ const device = getDevice();
12
+ device.queue.submit([commandEncoder.finish()]);
13
+ }
14
+
15
+ /**
16
+ * Encodes and submits a single compute pass: sets the pipeline and bind group,
17
+ * dispatches workgroups, and optionally wraps the pass in GPU timestamp queries.
18
+ * @param {GPUComputePipeline} pipeline
19
+ * @param {GPUBindGroup} bindGroup
20
+ * @param {number | { x: number, y: number }} workgroups - workgroup count; number for 1D dispatch, `{x, y}` for 2D
21
+ * @returns {{ commandEncoder: GPUCommandEncoder, ts: any }} encoded commands and timestamp handle
22
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createCommandEncoder GPUDevice.createCommandEncoder()}
23
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUCommandEncoder/beginComputePass GPUCommandEncoder.beginComputePass()}
24
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUComputePassEncoder/setPipeline GPUComputePassEncoder.setPipeline()}
25
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUComputePassEncoder/setBindGroup GPUComputePassEncoder.setBindGroup()}
26
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUComputePassEncoder/dispatchWorkgroups GPUComputePassEncoder.dispatchWorkgroups()}
27
+ */
28
+ export function runComputePass(pipeline, bindGroup, workgroups) {
29
+ const device = getDevice();
30
+
31
+ const { querySet, passDescriptor } = beginTimestamp();
32
+
33
+ const commandEncoder = device.createCommandEncoder();
34
+ const passEncoder = commandEncoder.beginComputePass(passDescriptor);
35
+
36
+ passEncoder.setPipeline(pipeline);
37
+ passEncoder.setBindGroup(0, bindGroup);
38
+
39
+ if (typeof workgroups === "number") {
40
+ passEncoder.dispatchWorkgroups(workgroups);
41
+ } else {
42
+ passEncoder.dispatchWorkgroups(workgroups.x, workgroups.y);
43
+ }
44
+
45
+ passEncoder.end();
46
+
47
+ const ts = resolveTimestamp(commandEncoder, querySet);
48
+
49
+ commandEncoder._passEncoder = passEncoder; // anchor — GC'd passEncoder may crash native encoder
50
+
51
+ return { commandEncoder, ts };
52
+ }
@@ -0,0 +1,12 @@
1
+ /**
2
+ * WebGPU is intentionally low-level. Every routine needs to create buffers, set up bind groups,
3
+ * compile a pipeline, encode a compute pass, and map results back to the CPU. Writing that inline
4
+ * for each BLAS routine means dozens of near-identical code paths that are hard to read, test,
5
+ * or change consistently.
6
+ *
7
+ * These utilities factor out each concern into a focused function. A routine calls
8
+ * `createStorageBuffer`, `createBindGroup`, `loadShader`, `runComputePass`, and `extractResult`
9
+ * — one line per step — rather than managing raw WebGPU objects directly.
10
+ *
11
+ * @module devdocs/utility-functions
12
+ */
@@ -0,0 +1,82 @@
1
+ /** @module devdocs/utility-functions/pipeline */
2
+ import { getDevice } from "../init.mjs";
3
+
4
+ // WeakMap keyed by GPUDevice so pipelines are released automatically when the device is destroyed.
5
+ const _pipelines = new WeakMap();
6
+
7
+ /**
8
+ * Returns a cached `GPUComputePipeline` for the given shader, compiling it on first use.
9
+ * Pipelines are cached per device so reinitialization (new device) always recompiles.
10
+ * @param {GPUDevice} device
11
+ * @param {string} shaderName - filename without `.wgsl` extension (e.g. `"sscal"`)
12
+ * @returns {Promise<GPUComputePipeline>}
13
+ */
14
+ export async function getPipeline(device, shaderName) {
15
+ if (!_pipelines.has(device)) {
16
+ _pipelines.set(device, new Map());
17
+ }
18
+ const byName = _pipelines.get(device);
19
+ if (!byName.has(shaderName)) {
20
+ byName.set(shaderName, await loadShader(shaderName));
21
+ }
22
+ return byName.get(shaderName);
23
+ }
24
+
25
+ /**
26
+ * Loads WGSL source for `shaderName`. In the browser, reads from the inline bundle
27
+ * (`browser-shaders.mjs`); in Node.js, reads the `.wgsl` file directly from disk.
28
+ * @param {string} shaderName
29
+ * @returns {Promise<string>}
30
+ */
31
+ async function loadCode(shaderName) {
32
+ // typeof guards against ReferenceError in Node.js — accessing window directly throws if it doesn't exist.
33
+ if (typeof window !== "undefined") {
34
+ const { shaderSources } = await import("../shaders/browser-shaders.mjs");
35
+ const src = shaderSources[shaderName];
36
+ if (!src) throw new Error(`Shader "${shaderName}" not found in browser bundle.`);
37
+ return src;
38
+ } else {
39
+ const { readFileSync } = await import("fs");
40
+ const { fileURLToPath } = await import("url");
41
+ const { dirname, join } = await import("path");
42
+ const dir = dirname(fileURLToPath(import.meta.url));
43
+ return readFileSync(join(dir, `../shaders/${shaderName}.wgsl`), "utf8");
44
+ }
45
+ }
46
+
47
+ /**
48
+ * Compiles a WGSL shader into a `GPUComputePipeline`. Throws with line-level detail if compilation fails, rather than surfacing a raw GPU error.
49
+ * Uses `layout: "auto"` so the pipeline derives its bind group layout from the shader —
50
+ * no manual layout definition needed.
51
+ * @param {string} shaderName
52
+ * @returns {Promise<GPUComputePipeline>}
53
+ * @throws {Error} if the shader has compilation errors
54
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createShaderModule GPUDevice.createShaderModule()}
55
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUShaderModule/getCompilationInfo GPUShaderModule.getCompilationInfo()}
56
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUDevice/createComputePipeline GPUDevice.createComputePipeline()}
57
+ */
58
+ export async function loadShader(shaderName) {
59
+ const device = getDevice();
60
+ const code = await loadCode(shaderName);
61
+
62
+ const shaderModule = device.createShaderModule({ label: shaderName, code });
63
+
64
+ const info = await shaderModule.getCompilationInfo();
65
+ // GPUCompilationMessage: https://developer.mozilla.org/en-US/docs/Web/API/GPUCompilationMessage
66
+ const errors = info.messages.filter((m) => m.type === "error");
67
+ if (errors.length > 0) {
68
+ throw new Error(
69
+ `Shader "${shaderName}" compilation failed:\n${errors.map((m) => ` line ${m.lineNum}: ${m.message}`).join("\n")}`,
70
+ );
71
+ }
72
+
73
+ const pipeline = device.createComputePipeline({
74
+ label: shaderName,
75
+ layout: "auto",
76
+ compute: { module: shaderModule },
77
+ });
78
+
79
+ pipeline._shaderModule = shaderModule; // anchor — GC'd shaderModule crashes native pipeline
80
+
81
+ return pipeline;
82
+ }
@@ -0,0 +1,19 @@
1
+ /** @module devdocs/utility-functions/result */
2
+
3
+ /**
4
+ * Maps `readBuffer` for CPU access, copies its contents into a new typed array, then unmaps.
5
+ * The `.slice()` is required — `getMappedRange()` returns a view into the mapped GPU memory
6
+ * which becomes invalid after `unmap()`, so we copy it out before unmapping.
7
+ * @param {GPUBuffer} readBuffer - a `MAP_READ` buffer produced by `stageReadback`
8
+ * @param {typeof Float32Array | typeof Uint32Array} [readbackType=Float32Array] - typed array constructor for the result
9
+ * @returns {Promise<Float32Array | Uint32Array>}
10
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/mapAsync GPUBuffer.mapAsync()}
11
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/getMappedRange GPUBuffer.getMappedRange()}
12
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer/unmap GPUBuffer.unmap()}
13
+ */
14
+ export async function extractResult(readBuffer, readbackType = Float32Array) {
15
+ await readBuffer.mapAsync(GPUMapMode.READ);
16
+ const result = new readbackType(readBuffer.getMappedRange().slice());
17
+ readBuffer.unmap();
18
+ return result;
19
+ }
@@ -0,0 +1,31 @@
1
+ /** @module devdocs/utility-functions/workgroup */
2
+ import { getDevice } from "../init.mjs";
3
+
4
+ // Fixed sizes match the shader declarations (WGS = 64 for 1D, 8×8 = 64 threads for 2D).
5
+ const WORKGROUP_SIZE_1D = 64;
6
+ const WORKGROUP_SIZE_2D = 8;
7
+
8
+ /**
9
+ * Calculates the number of workgroups to dispatch, clamped to the device's
10
+ * `maxComputeWorkgroupsPerDimension` limit (default 65535 across most devices).
11
+ *
12
+ * - 1D (pass only `dim1`): returns a single count for `dispatchWorkgroups(n)`.
13
+ * - 2D (pass both `dim1` and `dim2`): returns `{ x, y }` for `dispatchWorkgroups(x, y)`.
14
+ * `dim1` maps to rows (y) and `dim2` maps to columns (x).
15
+ *
16
+ * @param {number} dim1 - row count (1D: element count)
17
+ * @param {number} [dim2] - column count; omit for a 1D dispatch
18
+ * @returns {number | { x: number, y: number }}
19
+ * @see {@link https://developer.mozilla.org/en-US/docs/Web/API/GPUSupportedLimits GPUSupportedLimits} (`maxComputeWorkgroupsPerDimension`)
20
+ */
21
+ export function calcWorkgroups(dim1, dim2) {
22
+ const max = getDevice().limits.maxComputeWorkgroupsPerDimension;
23
+ if (dim2 === undefined) {
24
+ return Math.min(Math.ceil(dim1 / WORKGROUP_SIZE_1D), max);
25
+ } else {
26
+ return {
27
+ x: Math.min(Math.ceil(dim2 / WORKGROUP_SIZE_2D), max),
28
+ y: Math.min(Math.ceil(dim1 / WORKGROUP_SIZE_2D), max),
29
+ };
30
+ }
31
+ }