wgblas 0.1.2 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +3 -0
  2. package/dist/wgblas.browser.js +1249 -37
  3. package/index.d.mts +9 -0
  4. package/index.mjs +9 -0
  5. package/package.json +47 -1
  6. package/src/classes/GpuMatrix.d.mts +98 -0
  7. package/src/classes/GpuMatrix.mjs +109 -0
  8. package/src/classes/GpuVector.d.mts +11 -7
  9. package/src/classes/GpuVector.mjs +31 -5
  10. package/src/dasum/dasum.d.mts +48 -0
  11. package/src/dasum/dasum.mjs +121 -0
  12. package/src/init.mjs +3 -1
  13. package/src/isamax/isamax.mjs +66 -67
  14. package/src/random/random.d.mts +37 -0
  15. package/src/random/random.mjs +17 -0
  16. package/src/sasum/sasum.mjs +55 -51
  17. package/src/saxpy/saxpy.mjs +48 -35
  18. package/src/scopy/scopy.mjs +43 -32
  19. package/src/sdot/sdot.mjs +60 -55
  20. package/src/sgemv/sgemv.d.mts +127 -0
  21. package/src/sgemv/sgemv.mjs +148 -0
  22. package/src/sger/sger.d.mts +111 -0
  23. package/src/sger/sger.mjs +136 -0
  24. package/src/shaders/browser-shaders.mjs +26 -0
  25. package/src/shaders/dasum.wgsl +98 -0
  26. package/src/shaders/f64add.wgsl +281 -0
  27. package/src/shaders/isamax.wgsl +32 -9
  28. package/src/shaders/reduction/sumF64.wgsl +49 -0
  29. package/src/shaders/sasum.wgsl +18 -4
  30. package/src/shaders/sdot.wgsl +18 -4
  31. package/src/shaders/sgemv_n.wgsl +75 -0
  32. package/src/shaders/sgemv_t.wgsl +65 -0
  33. package/src/shaders/sger.wgsl +48 -0
  34. package/src/shaders/snrm2.wgsl +22 -4
  35. package/src/shaders/ssymv.wgsl +69 -0
  36. package/src/shaders/ssyr.wgsl +60 -0
  37. package/src/shaders/ssyr2.wgsl +63 -0
  38. package/src/shaders/strmv.wgsl +103 -0
  39. package/src/shaders/strsv_apply_inverse.wgsl +47 -0
  40. package/src/shaders/strsv_invert_block.wgsl +109 -0
  41. package/src/shaders/strsv_update.wgsl +75 -0
  42. package/src/snrm2/snrm2.mjs +56 -52
  43. package/src/srot/srot.mjs +57 -41
  44. package/src/srotm/srotm.mjs +54 -38
  45. package/src/sscal/sscal.mjs +43 -32
  46. package/src/sswap/sswap.mjs +49 -34
  47. package/src/ssymv/ssymv.d.mts +117 -0
  48. package/src/ssymv/ssymv.mjs +135 -0
  49. package/src/ssyr/ssyr.d.mts +100 -0
  50. package/src/ssyr/ssyr.mjs +106 -0
  51. package/src/ssyr2/ssyr2.d.mts +112 -0
  52. package/src/ssyr2/ssyr2.mjs +130 -0
  53. package/src/strmv/strmv.d.mts +117 -0
  54. package/src/strmv/strmv.mjs +138 -0
  55. package/src/strsv/strsv.d.mts +106 -0
  56. package/src/strsv/strsv.mjs +207 -0
  57. package/src/util/benchmark.mjs +1 -1
  58. package/src/util/bindgroup.mjs +14 -10
  59. package/src/util/buffer.mjs +7 -2
  60. package/src/util/compute.mjs +41 -15
  61. package/src/util/f64pack.mjs +152 -0
  62. package/src/util/pipeline.mjs +32 -17
  63. package/src/util/result.mjs +8 -4
  64. package/src/util/workgroup.mjs +10 -10
@@ -0,0 +1,100 @@
1
+ import { GpuVector } from "../classes/GpuVector.mjs";
2
+ import { GpuMatrix } from "../classes/GpuMatrix.mjs";
3
+
4
+ /**
5
+ * Performs the symmetric rank-1 update A = alpha * x * x^T + A
6
+ *
7
+ * A is an n×n symmetric matrix stored in row-major order, updated in place.
8
+ * Only the triangle specified by `uplo` is referenced and updated; the other
9
+ * triangle is left untouched (implied by symmetry).
10
+ *
11
+ * {@includeCode ../../examples/ssyr/ssyr.js}
12
+ *
13
+ * **Browser (standalone HTML):**
14
+ * {@includeCode ../../examples/ssyr/web/ssyr.html}
15
+ *
16
+ * @param device - GPUDevice from `init()`
17
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
18
+ * @param n - order of the matrix A (number of rows and columns)
19
+ * @param alpha - scalar multiplier for x*x^T
20
+ * @param x - Float32Array input vector, length at least (n-1)*incx+1
21
+ * @param incx - stride for x (must be a positive integer)
22
+ * @param A - Float32Array, row-major or column-major (see `layout`), at least (n-1)*lda+n elements
23
+ * @param lda - leading dimension of A (>= n either way — A is square)
24
+ * @param layout - storage layout of `A` (default: `'row-major'`); for a symmetric
25
+ * matrix, column-major storage just means the *other* triangle is the one
26
+ * physically referenced for a given `uplo`
27
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/ssyr/ssyr.mjs#L15">Source code: ssyr.mjs (L15)</a>
28
+ * @category BLAS Level 2
29
+ */
30
+ export declare function ssyr(
31
+ device: GPUDevice,
32
+ uplo: 'lower' | 'upper',
33
+ n: number,
34
+ alpha: number,
35
+ x: Float32Array,
36
+ incx: number,
37
+ A: Float32Array,
38
+ lda: number,
39
+ layout?: 'row-major' | 'column-major',
40
+ ): Promise<{ A: Float32Array; gpuTimeMs?: number }>;
41
+
42
+ /**
43
+ * Performs the symmetric rank-1 update A = alpha * x * x^T + A
44
+ *
45
+ * A is kept GPU-resident; x is a CPU Float32Array. `A`'s own `layout` (set at
46
+ * `GpuMatrix.from` time) determines the operation — there is no separate
47
+ * `layout` argument here.
48
+ *
49
+ * @param device - GPUDevice from `init()`
50
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
51
+ * @param n - order of the matrix A
52
+ * @param alpha - scalar multiplier for x*x^T
53
+ * @param x - Float32Array input vector
54
+ * @param incx - stride for x (must be a positive integer)
55
+ * @param A - GpuMatrix, GPU-resident
56
+ * @param lda - leading dimension of A (must equal A.lda)
57
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/ssyr/ssyr.mjs#L15">Source code: ssyr.mjs (L15)</a>
58
+ * @category BLAS Level 2
59
+ */
60
+ export declare function ssyr(
61
+ device: GPUDevice,
62
+ uplo: 'lower' | 'upper',
63
+ n: number,
64
+ alpha: number,
65
+ x: Float32Array,
66
+ incx: number,
67
+ A: GpuMatrix,
68
+ lda: number,
69
+ ): Promise<{ gpuTimeMs?: number }>;
70
+
71
+ /**
72
+ * Performs the symmetric rank-1 update A = alpha * x * x^T + A
73
+ *
74
+ * x and A are both kept resident on the GPU. `A`'s own `layout` (set at
75
+ * `GpuMatrix.from` time) determines the operation — there is no separate
76
+ * `layout` argument here.
77
+ *
78
+ * {@includeCode ../../examples/ssyr/gpuvec.ssyr.js}
79
+ *
80
+ * @param device - GPUDevice from `init()`
81
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
82
+ * @param n - order of the matrix A
83
+ * @param alpha - scalar multiplier for x*x^T
84
+ * @param x - GpuVector input vector (not mutated)
85
+ * @param incx - stride for x (must be a positive integer)
86
+ * @param A - GpuMatrix, mutated in place
87
+ * @param lda - leading dimension of A (must equal A.lda)
88
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/ssyr/ssyr.mjs#L15">Source code: ssyr.mjs (L15)</a>
89
+ * @category BLAS Level 2
90
+ */
91
+ export declare function ssyr(
92
+ device: GPUDevice,
93
+ uplo: 'lower' | 'upper',
94
+ n: number,
95
+ alpha: number,
96
+ x: GpuVector,
97
+ incx: number,
98
+ A: GpuMatrix,
99
+ lda: number,
100
+ ): Promise<{ gpuTimeMs?: number }>;
@@ -0,0 +1,106 @@
1
+ import {
2
+ uploadBuffer,
3
+ createParamsBuffer,
4
+ stageReadback,
5
+ destroyBuffers,
6
+ } from "../util/buffer.mjs";
7
+ import { createBindGroup } from "../util/bindgroup.mjs";
8
+ import { runComputePass, submit } from "../util/compute.mjs";
9
+ import { extractResult } from "../util/result.mjs";
10
+ import { extractTimestamp } from "../util/benchmark.mjs";
11
+ import { getPipeline } from "../util/pipeline.mjs";
12
+ import { GpuVector } from "../classes/GpuVector.mjs";
13
+ import { GpuMatrix } from "../classes/GpuMatrix.mjs";
14
+
15
+ export async function ssyr(device, uplo, n, alpha, x, incx, A, lda, layout = "row-major") {
16
+ const xIsGpu = x instanceof GpuVector;
17
+ const AIsGpu = A instanceof GpuMatrix;
18
+
19
+ if (!(device instanceof GPUDevice))
20
+ throw new Error("device must be a GPUDevice.");
21
+ if (uplo !== "lower" && uplo !== "upper")
22
+ throw new Error("uplo must be 'lower' or 'upper'.");
23
+ if (layout !== "row-major" && layout !== "column-major")
24
+ throw new Error("layout must be 'row-major' or 'column-major'.");
25
+ if (!Number.isInteger(n) || !Number.isInteger(incx) || !Number.isInteger(lda))
26
+ throw new Error("n, incx, and lda must be integers.");
27
+ if (typeof alpha !== "number")
28
+ throw new Error("alpha must be a number.");
29
+ if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
30
+ if (!Number.isFinite(alpha)) throw new Error("alpha must be finite.");
31
+ if (incx <= 0) throw new Error("incx must be positive.");
32
+ if (lda < n) throw new Error("lda must be >= n.");
33
+ if (!AIsGpu && !(A instanceof Float32Array))
34
+ throw new Error("A must be a Float32Array or GpuMatrix.");
35
+ if (!xIsGpu && !(x instanceof Float32Array))
36
+ throw new Error("x must be a Float32Array or GpuVector.");
37
+ if (xIsGpu && !AIsGpu)
38
+ throw new Error("A must be a GpuMatrix when x is a GpuVector.");
39
+ if (AIsGpu && xIsGpu && A._buf === x._buf)
40
+ throw new Error("A and x must not reference the same GPU buffer.");
41
+ if (AIsGpu && lda !== A.lda)
42
+ throw new Error("lda must match A.lda when A is a GpuMatrix.");
43
+ if (AIsGpu && (A.rows < n || A.cols < n))
44
+ throw new Error("A is too small for the given n.");
45
+ if (n < 0) throw new Error("n must be non-negative.");
46
+ if (n === 0) return AIsGpu ? {} : { A };
47
+
48
+ if (!AIsGpu && A.length < (n - 1) * lda + n)
49
+ throw new Error("A does not have enough elements for the given n and lda.");
50
+ if (x.length < (n - 1) * incx + 1)
51
+ throw new Error("x does not have enough elements for the given n and incx.");
52
+
53
+ // GpuMatrix's own layout wins over the argument; A is symmetric, so column-major A reinterpreted row-major just flips which triangle is stored — flip uplo to match.
54
+ const effLayout = AIsGpu ? A.layout : layout;
55
+ const isLower = effLayout === "column-major" ? uplo === "upper" : uplo === "lower";
56
+
57
+ const pipeline = await getPipeline(device, "ssyr");
58
+
59
+ let xBuffer = null;
60
+ let ABuffer = null;
61
+ let paramsBuffer = null;
62
+
63
+ try {
64
+ xBuffer = xIsGpu ? x._buf : uploadBuffer(x, "ssyr-x", false);
65
+ ABuffer = AIsGpu ? A._buf : uploadBuffer(A, "ssyr-A", true);
66
+ paramsBuffer = createParamsBuffer(
67
+ [
68
+ { value: n, type: "u32" },
69
+ { value: alpha, type: "f32" },
70
+ { value: incx, type: "u32" },
71
+ { value: lda, type: "u32" },
72
+ { value: isLower ? 0 : 1, type: "u32" },
73
+ ],
74
+ "ssyr-params",
75
+ );
76
+
77
+ const bindGroup = createBindGroup(pipeline.getBindGroupLayout(0), [
78
+ xBuffer,
79
+ ABuffer,
80
+ paramsBuffer,
81
+ ]);
82
+
83
+ // One workgroup per row of A; clamped to device limit — the shader's
84
+ // grid-stride loop handles remaining rows when n > dispatch count.
85
+ const wgCount = Math.min(n, device.limits.maxComputeWorkgroupsPerDimension);
86
+ const { commandEncoder, ts } = runComputePass(pipeline, bindGroup, wgCount);
87
+ const readBuffer = AIsGpu ? null : stageReadback(commandEncoder, ABuffer);
88
+
89
+ submit(commandEncoder);
90
+
91
+ const gpuTimeMs = await extractTimestamp(ts);
92
+
93
+ if (AIsGpu) {
94
+ if (gpuTimeMs !== undefined) return { gpuTimeMs };
95
+ return {};
96
+ }
97
+
98
+ const result = await extractResult(readBuffer, Float32Array);
99
+ if (gpuTimeMs !== undefined) return { A: result, gpuTimeMs };
100
+ return { A: result };
101
+ } finally {
102
+ if (!xIsGpu && xBuffer) destroyBuffers(xBuffer);
103
+ if (!AIsGpu && ABuffer) destroyBuffers(ABuffer);
104
+ if (paramsBuffer) destroyBuffers(paramsBuffer);
105
+ }
106
+ }
@@ -0,0 +1,112 @@
1
+ import { GpuVector } from "../classes/GpuVector.mjs";
2
+ import { GpuMatrix } from "../classes/GpuMatrix.mjs";
3
+
4
+ /**
5
+ * Performs the symmetric rank-2 update A = alpha * x * y^T + alpha * y * x^T + A
6
+ *
7
+ * A is an n×n symmetric matrix stored in row-major order, updated in place.
8
+ * Only the triangle specified by `uplo` is referenced and updated; the other
9
+ * triangle is left untouched (implied by symmetry).
10
+ *
11
+ * {@includeCode ../../examples/ssyr2/ssyr2.js}
12
+ *
13
+ * **Browser (standalone HTML):**
14
+ * {@includeCode ../../examples/ssyr2/web/ssyr2.html}
15
+ *
16
+ * @param device - GPUDevice from `init()`
17
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
18
+ * @param n - order of the matrix A (number of rows and columns)
19
+ * @param alpha - scalar multiplier for x*y^T + y*x^T
20
+ * @param x - Float32Array input vector, length at least (n-1)*incx+1
21
+ * @param incx - stride for x (must be a positive integer)
22
+ * @param y - Float32Array input vector, length at least (n-1)*incy+1
23
+ * @param incy - stride for y (must be a positive integer)
24
+ * @param A - Float32Array, row-major or column-major (see `layout`), at least (n-1)*lda+n elements
25
+ * @param lda - leading dimension of A (>= n either way — A is square)
26
+ * @param layout - storage layout of `A` (default: `'row-major'`); for a symmetric
27
+ * matrix, column-major storage just means the *other* triangle is the one
28
+ * physically referenced for a given `uplo`
29
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/ssyr2/ssyr2.mjs#L15">Source code: ssyr2.mjs (L15)</a>
30
+ * @category BLAS Level 2
31
+ */
32
+ export declare function ssyr2(
33
+ device: GPUDevice,
34
+ uplo: 'lower' | 'upper',
35
+ n: number,
36
+ alpha: number,
37
+ x: Float32Array,
38
+ incx: number,
39
+ y: Float32Array,
40
+ incy: number,
41
+ A: Float32Array,
42
+ lda: number,
43
+ layout?: 'row-major' | 'column-major',
44
+ ): Promise<{ A: Float32Array; gpuTimeMs?: number }>;
45
+
46
+ /**
47
+ * Performs the symmetric rank-2 update A = alpha * x * y^T + alpha * y * x^T + A
48
+ *
49
+ * A is kept GPU-resident; x and y are CPU Float32Arrays. `A`'s own `layout`
50
+ * (set at `GpuMatrix.from` time) determines the operation — there is no
51
+ * separate `layout` argument here.
52
+ *
53
+ * @param device - GPUDevice from `init()`
54
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
55
+ * @param n - order of the matrix A
56
+ * @param alpha - scalar multiplier for x*y^T + y*x^T
57
+ * @param x - Float32Array input vector
58
+ * @param incx - stride for x (must be a positive integer)
59
+ * @param y - Float32Array input vector
60
+ * @param incy - stride for y (must be a positive integer)
61
+ * @param A - GpuMatrix, row-major, GPU-resident, Float32-backed
62
+ * @param lda - leading dimension of A (must equal A.lda)
63
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/ssyr2/ssyr2.mjs#L15">Source code: ssyr2.mjs (L15)</a>
64
+ * @category BLAS Level 2
65
+ */
66
+ export declare function ssyr2(
67
+ device: GPUDevice,
68
+ uplo: 'lower' | 'upper',
69
+ n: number,
70
+ alpha: number,
71
+ x: Float32Array,
72
+ incx: number,
73
+ y: Float32Array,
74
+ incy: number,
75
+ A: GpuMatrix,
76
+ lda: number,
77
+ ): Promise<{ gpuTimeMs?: number }>;
78
+
79
+ /**
80
+ * Performs the symmetric rank-2 update A = alpha * x * y^T + alpha * y * x^T + A
81
+ *
82
+ * x, y, and A are all kept resident on the GPU. `A`'s own `layout` (set at
83
+ * `GpuMatrix.from` time) determines the operation — there is no separate
84
+ * `layout` argument here.
85
+ *
86
+ * {@includeCode ../../examples/ssyr2/gpuvec.ssyr2.js}
87
+ *
88
+ * @param device - GPUDevice from `init()`
89
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
90
+ * @param n - order of the matrix A
91
+ * @param alpha - scalar multiplier for x*y^T + y*x^T
92
+ * @param x - GpuVector input vector (not mutated), Float32-backed
93
+ * @param incx - stride for x (must be a positive integer)
94
+ * @param y - GpuVector input vector (not mutated), Float32-backed
95
+ * @param incy - stride for y (must be a positive integer)
96
+ * @param A - GpuMatrix, row-major, mutated in place, Float32-backed
97
+ * @param lda - leading dimension of A (must equal A.lda)
98
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/ssyr2/ssyr2.mjs#L15">Source code: ssyr2.mjs (L15)</a>
99
+ * @category BLAS Level 2
100
+ */
101
+ export declare function ssyr2(
102
+ device: GPUDevice,
103
+ uplo: 'lower' | 'upper',
104
+ n: number,
105
+ alpha: number,
106
+ x: GpuVector,
107
+ incx: number,
108
+ y: GpuVector,
109
+ incy: number,
110
+ A: GpuMatrix,
111
+ lda: number,
112
+ ): Promise<{ gpuTimeMs?: number }>;
@@ -0,0 +1,130 @@
1
+ import {
2
+ uploadBuffer,
3
+ createParamsBuffer,
4
+ stageReadback,
5
+ destroyBuffers,
6
+ } from "../util/buffer.mjs";
7
+ import { createBindGroup } from "../util/bindgroup.mjs";
8
+ import { runComputePass, submit } from "../util/compute.mjs";
9
+ import { extractResult } from "../util/result.mjs";
10
+ import { extractTimestamp } from "../util/benchmark.mjs";
11
+ import { getPipeline } from "../util/pipeline.mjs";
12
+ import { GpuVector } from "../classes/GpuVector.mjs";
13
+ import { GpuMatrix } from "../classes/GpuMatrix.mjs";
14
+
15
+ export async function ssyr2(device, uplo, n, alpha, x, incx, y, incy, A, lda, layout = "row-major") {
16
+ const xIsGpu = x instanceof GpuVector;
17
+ const yIsGpu = y instanceof GpuVector;
18
+ const AIsGpu = A instanceof GpuMatrix;
19
+
20
+ if (!(device instanceof GPUDevice))
21
+ throw new Error("device must be a GPUDevice.");
22
+ if (uplo !== "lower" && uplo !== "upper")
23
+ throw new Error("uplo must be 'lower' or 'upper'.");
24
+ if (layout !== "row-major" && layout !== "column-major")
25
+ throw new Error("layout must be 'row-major' or 'column-major'.");
26
+ if (
27
+ !Number.isInteger(n) ||
28
+ !Number.isInteger(incx) ||
29
+ !Number.isInteger(incy) ||
30
+ !Number.isInteger(lda)
31
+ )
32
+ throw new Error("n, incx, incy, and lda must be integers.");
33
+ if (typeof alpha !== "number")
34
+ throw new Error("alpha must be a number.");
35
+ if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
36
+ if (!Number.isFinite(alpha)) throw new Error("alpha must be finite.");
37
+ if (incx <= 0 || incy <= 0)
38
+ throw new Error("incx and incy must be positive.");
39
+ if (lda < n) throw new Error("lda must be >= n.");
40
+ if (!AIsGpu && !(A instanceof Float32Array))
41
+ throw new Error("A must be a Float32Array or GpuMatrix.");
42
+ if (!xIsGpu && !(x instanceof Float32Array))
43
+ throw new Error("x must be a Float32Array or GpuVector.");
44
+ if (!yIsGpu && !(y instanceof Float32Array))
45
+ throw new Error("y must be a Float32Array or GpuVector.");
46
+ if (xIsGpu !== yIsGpu)
47
+ throw new Error(
48
+ "x and y must be the same type (both Float32Array or both GpuVector).",
49
+ );
50
+ if (xIsGpu && !AIsGpu)
51
+ throw new Error("A must be a GpuMatrix when x and y are GpuVectors.");
52
+ if (AIsGpu && xIsGpu && A._buf === x._buf)
53
+ throw new Error("A and x must not reference the same GPU buffer.");
54
+ if (AIsGpu && yIsGpu && A._buf === y._buf)
55
+ throw new Error("A and y must not reference the same GPU buffer.");
56
+ if (xIsGpu && x._buf === y._buf)
57
+ throw new Error("x and y must not reference the same GPU buffer when both are GpuVectors.");
58
+ if (AIsGpu && lda !== A.lda)
59
+ throw new Error("lda must match A.lda when A is a GpuMatrix.");
60
+ if (AIsGpu && (A.rows < n || A.cols < n))
61
+ throw new Error("A is too small for the given n.");
62
+ if (n < 0) throw new Error("n must be non-negative.");
63
+ if (n === 0) return AIsGpu ? {} : { A };
64
+
65
+ if (!AIsGpu && A.length < (n - 1) * lda + n)
66
+ throw new Error("A does not have enough elements for the given n and lda.");
67
+ if (x.length < (n - 1) * incx + 1)
68
+ throw new Error("x does not have enough elements for the given n and incx.");
69
+ if (y.length < (n - 1) * incy + 1)
70
+ throw new Error("y does not have enough elements for the given n and incy.");
71
+
72
+ // GpuMatrix's own layout wins over the argument; A is symmetric, so column-major A reinterpreted row-major just flips which triangle is stored (no x/y swap needed — x*y^T+y*x^T is already symmetric under swapping them).
73
+ const effLayout = AIsGpu ? A.layout : layout;
74
+ const isLower = effLayout === "column-major" ? uplo === "upper" : uplo === "lower";
75
+
76
+ const pipeline = await getPipeline(device, "ssyr2");
77
+
78
+ let xBuffer = null;
79
+ let yBuffer = null;
80
+ let ABuffer = null;
81
+ let paramsBuffer = null;
82
+
83
+ try {
84
+ xBuffer = xIsGpu ? x._buf : uploadBuffer(x, "ssyr2-x", false);
85
+ yBuffer = yIsGpu ? y._buf : uploadBuffer(y, "ssyr2-y", false);
86
+ ABuffer = AIsGpu ? A._buf : uploadBuffer(A, "ssyr2-A", true);
87
+ paramsBuffer = createParamsBuffer(
88
+ [
89
+ { value: n, type: "u32" },
90
+ { value: alpha, type: "f32" },
91
+ { value: incx, type: "u32" },
92
+ { value: incy, type: "u32" },
93
+ { value: lda, type: "u32" },
94
+ { value: isLower ? 0 : 1, type: "u32" },
95
+ ],
96
+ "ssyr2-params",
97
+ );
98
+
99
+ const bindGroup = createBindGroup(pipeline.getBindGroupLayout(0), [
100
+ xBuffer,
101
+ yBuffer,
102
+ ABuffer,
103
+ paramsBuffer,
104
+ ]);
105
+
106
+ // One workgroup per row of A; clamped to device limit — the shader's
107
+ // grid-stride loop handles remaining rows when n > dispatch count.
108
+ const wgCount = Math.min(n, device.limits.maxComputeWorkgroupsPerDimension);
109
+ const { commandEncoder, ts } = runComputePass(pipeline, bindGroup, wgCount);
110
+ const readBuffer = AIsGpu ? null : stageReadback(commandEncoder, ABuffer);
111
+
112
+ submit(commandEncoder);
113
+
114
+ const gpuTimeMs = await extractTimestamp(ts);
115
+
116
+ if (AIsGpu) {
117
+ if (gpuTimeMs !== undefined) return { gpuTimeMs };
118
+ return {};
119
+ }
120
+
121
+ const result = await extractResult(readBuffer, Float32Array);
122
+ if (gpuTimeMs !== undefined) return { A: result, gpuTimeMs };
123
+ return { A: result };
124
+ } finally {
125
+ if (!xIsGpu && xBuffer) destroyBuffers(xBuffer);
126
+ if (!yIsGpu && yBuffer) destroyBuffers(yBuffer);
127
+ if (!AIsGpu && ABuffer) destroyBuffers(ABuffer);
128
+ if (paramsBuffer) destroyBuffers(paramsBuffer);
129
+ }
130
+ }
@@ -0,0 +1,117 @@
1
+ import { GpuVector } from "../classes/GpuVector.mjs";
2
+ import { GpuMatrix } from "../classes/GpuMatrix.mjs";
3
+
4
+ /**
5
+ * Performs the triangular matrix-vector operation y = op(A) * x
6
+ *
7
+ * A is an n×n triangular matrix stored in row-major order. Only the triangle
8
+ * specified by `uplo` is referenced; the other triangle is not accessed.
9
+ *
10
+ * {@includeCode ../../examples/strmv/strmv.js}
11
+ *
12
+ * **Browser (standalone HTML):**
13
+ * {@includeCode ../../examples/strmv/web/strmv.html}
14
+ *
15
+ * @param device - GPUDevice from `init()`
16
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
17
+ * @param trans - `'no-transpose'` for A, `'transpose'` for A^T
18
+ * @param diag - `'unit'` to treat the diagonal as all-ones (A's diagonal is not read), `'non-unit'` to read it
19
+ * @param n - order of the matrix A (number of rows and columns)
20
+ * @param A - Float32Array, row-major or column-major (see `layout`), at least (n-1)*lda+n elements
21
+ * @param lda - leading dimension of A (>= n either way — A is square)
22
+ * @param x - Float32Array input vector, length at least (n-1)*incx+1
23
+ * @param incx - stride for x (must be a positive integer)
24
+ * @param y - Float32Array output vector, length at least (n-1)*incy+1
25
+ * @param incy - stride for y (must be a positive integer)
26
+ * @param layout - storage layout of `A` (default: `'row-major'`); column-major
27
+ * flips both the stored triangle and the effective `trans` (op(A) stays
28
+ * what you asked for either way)
29
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/strmv/strmv.mjs#L15">Source code: strmv.mjs (L15)</a>
30
+ * @category BLAS Level 2
31
+ */
32
+ export declare function strmv(
33
+ device: GPUDevice,
34
+ uplo: 'lower' | 'upper',
35
+ trans: 'no-transpose' | 'transpose',
36
+ diag: 'unit' | 'non-unit',
37
+ n: number,
38
+ A: Float32Array,
39
+ lda: number,
40
+ x: Float32Array,
41
+ incx: number,
42
+ y: Float32Array,
43
+ incy: number,
44
+ layout?: 'row-major' | 'column-major',
45
+ ): Promise<{ y: Float32Array; gpuTimeMs?: number }>;
46
+
47
+ /**
48
+ * Performs the triangular matrix-vector operation y = op(A) * x
49
+ *
50
+ * A is kept GPU-resident; x and y are CPU Float32Arrays. `A`'s own `layout`
51
+ * (set at `GpuMatrix.from` time) determines the operation — there is no
52
+ * separate `layout` argument here.
53
+ *
54
+ * @param device - GPUDevice from `init()`
55
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
56
+ * @param trans - `'no-transpose'` for A, `'transpose'` for A^T
57
+ * @param diag - `'unit'` to treat the diagonal as all-ones (A's diagonal is not read), `'non-unit'` to read it
58
+ * @param n - order of the matrix A
59
+ * @param A - GpuMatrix, GPU-resident
60
+ * @param lda - leading dimension of A (must equal A.lda)
61
+ * @param x - Float32Array input vector
62
+ * @param incx - stride for x (must be a positive integer)
63
+ * @param y - Float32Array output vector
64
+ * @param incy - stride for y (must be a positive integer)
65
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/strmv/strmv.mjs#L15">Source code: strmv.mjs (L15)</a>
66
+ * @category BLAS Level 2
67
+ */
68
+ export declare function strmv(
69
+ device: GPUDevice,
70
+ uplo: 'lower' | 'upper',
71
+ trans: 'no-transpose' | 'transpose',
72
+ diag: 'unit' | 'non-unit',
73
+ n: number,
74
+ A: GpuMatrix,
75
+ lda: number,
76
+ x: Float32Array,
77
+ incx: number,
78
+ y: Float32Array,
79
+ incy: number,
80
+ ): Promise<{ y: Float32Array; gpuTimeMs?: number }>;
81
+
82
+ /**
83
+ * Performs the triangular matrix-vector operation y = op(A) * x
84
+ *
85
+ * x and y are kept resident on the GPU. A must be a GpuMatrix; its own
86
+ * `layout` (set at `GpuMatrix.from` time) determines the operation — there is
87
+ * no separate `layout` argument here.
88
+ *
89
+ * {@includeCode ../../examples/strmv/gpuvec.strmv.js}
90
+ *
91
+ * @param device - GPUDevice from `init()`
92
+ * @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
93
+ * @param trans - `'no-transpose'` for A, `'transpose'` for A^T
94
+ * @param diag - `'unit'` to treat the diagonal as all-ones (A's diagonal is not read), `'non-unit'` to read it
95
+ * @param n - order of the matrix A
96
+ * @param A - GpuMatrix, GPU-resident
97
+ * @param lda - leading dimension of A (must equal A.lda)
98
+ * @param x - GpuVector input vector (not mutated)
99
+ * @param incx - stride for x (must be a positive integer)
100
+ * @param y - GpuVector output vector (mutated in place)
101
+ * @param incy - stride for y (must be a positive integer)
102
+ * @see <a href="https://github.com/manit2004/wgblas/blob/main/src/strmv/strmv.mjs#L15">Source code: strmv.mjs (L15)</a>
103
+ * @category BLAS Level 2
104
+ */
105
+ export declare function strmv(
106
+ device: GPUDevice,
107
+ uplo: 'lower' | 'upper',
108
+ trans: 'no-transpose' | 'transpose',
109
+ diag: 'unit' | 'non-unit',
110
+ n: number,
111
+ A: GpuMatrix,
112
+ lda: number,
113
+ x: GpuVector,
114
+ incx: number,
115
+ y: GpuVector,
116
+ incy: number,
117
+ ): Promise<{ gpuTimeMs?: number }>;