wgblas 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/wgblas.browser.js +197 -26
- package/index.d.mts +3 -0
- package/index.mjs +3 -0
- package/package.json +16 -1
- package/src/classes/GpuMatrix.d.mts +35 -22
- package/src/classes/GpuMatrix.mjs +40 -22
- package/src/sgemv/sgemv.d.mts +17 -8
- package/src/sgemv/sgemv.mjs +25 -18
- package/src/sger/sger.d.mts +111 -0
- package/src/sger/sger.mjs +136 -0
- package/src/shaders/browser-shaders.mjs +6 -0
- package/src/shaders/sger.wgsl +48 -0
- package/src/shaders/ssyr.wgsl +60 -0
- package/src/shaders/ssyr2.wgsl +63 -0
- package/src/ssymv/ssymv.d.mts +15 -7
- package/src/ssymv/ssymv.mjs +8 -3
- package/src/ssyr/ssyr.d.mts +100 -0
- package/src/ssyr/ssyr.mjs +106 -0
- package/src/ssyr2/ssyr2.d.mts +112 -0
- package/src/ssyr2/ssyr2.mjs +130 -0
- package/src/strmv/strmv.d.mts +15 -7
- package/src/strmv/strmv.mjs +11 -5
- package/src/strsv/strsv.d.mts +15 -7
- package/src/strsv/strsv.mjs +13 -18
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "wgblas",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.1.0",
|
|
4
4
|
"description": "BLAS on WebGPU",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.mjs",
|
|
@@ -82,6 +82,18 @@
|
|
|
82
82
|
"./strsv": {
|
|
83
83
|
"import": "./src/strsv/strsv.mjs",
|
|
84
84
|
"types": "./src/strsv/strsv.d.mts"
|
|
85
|
+
},
|
|
86
|
+
"./sger": {
|
|
87
|
+
"import": "./src/sger/sger.mjs",
|
|
88
|
+
"types": "./src/sger/sger.d.mts"
|
|
89
|
+
},
|
|
90
|
+
"./ssyr": {
|
|
91
|
+
"import": "./src/ssyr/ssyr.mjs",
|
|
92
|
+
"types": "./src/ssyr/ssyr.d.mts"
|
|
93
|
+
},
|
|
94
|
+
"./ssyr2": {
|
|
95
|
+
"import": "./src/ssyr2/ssyr2.mjs",
|
|
96
|
+
"types": "./src/ssyr2/ssyr2.d.mts"
|
|
85
97
|
}
|
|
86
98
|
},
|
|
87
99
|
"files": [
|
|
@@ -123,12 +135,15 @@
|
|
|
123
135
|
"@stdlib/blas-base-scopy": "^0.3.1",
|
|
124
136
|
"@stdlib/blas-base-sdot": "^0.3.1",
|
|
125
137
|
"@stdlib/blas-base-sgemv": "^0.1.1",
|
|
138
|
+
"@stdlib/blas-base-sger": "^0.1.1",
|
|
126
139
|
"@stdlib/blas-base-snrm2": "^0.3.1",
|
|
127
140
|
"@stdlib/blas-base-srot": "^0.2.1",
|
|
128
141
|
"@stdlib/blas-base-srotm": "^0.2.1",
|
|
129
142
|
"@stdlib/blas-base-sscal": "^0.3.1",
|
|
130
143
|
"@stdlib/blas-base-sswap": "^0.3.1",
|
|
131
144
|
"@stdlib/blas-base-ssymv": "^0.1.1",
|
|
145
|
+
"@stdlib/blas-base-ssyr": "^0.1.1",
|
|
146
|
+
"@stdlib/blas-base-ssyr2": "^0.1.1",
|
|
132
147
|
"@stdlib/blas-base-strmv": "^0.1.0",
|
|
133
148
|
"@stdlib/blas-base-strsv": "^0.1.1",
|
|
134
149
|
"@stdlib/random-array-uniform": "^0.2.2",
|
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Represents a
|
|
2
|
+
* Represents a Float32Array (or Float64Array) matrix stored in GPU memory,
|
|
3
|
+
* row-major or column-major.
|
|
3
4
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
5
|
+
* `rows`/`cols` always describe the logical shape regardless of layout.
|
|
6
|
+
* `lda` (leading dimension) is the stride between consecutive rows
|
|
7
|
+
* (row-major) or columns (column-major) — must be >= `cols` (row-major) or
|
|
8
|
+
* >= `rows` (column-major). When `lda` equals that minimum the matrix is
|
|
9
|
+
* dense with no padding.
|
|
7
10
|
*
|
|
8
11
|
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/classes/GpuMatrix.mjs#L7">Source code: GpuMatrix.mjs (L7)</a>
|
|
9
12
|
* @see [MDN: GPUBuffer](https://developer.mozilla.org/en-US/docs/Web/API/GPUBuffer)
|
|
@@ -15,28 +18,34 @@ export declare class GpuMatrix {
|
|
|
15
18
|
/** @internal */
|
|
16
19
|
readonly _buf: GPUBuffer;
|
|
17
20
|
|
|
18
|
-
/** Number of rows. */
|
|
21
|
+
/** Number of rows (logical shape, independent of layout). */
|
|
19
22
|
readonly rows: number;
|
|
20
23
|
|
|
21
|
-
/** Number of columns. */
|
|
24
|
+
/** Number of columns (logical shape, independent of layout). */
|
|
22
25
|
readonly cols: number;
|
|
23
26
|
|
|
24
|
-
/** Leading dimension — stride between row starts (
|
|
27
|
+
/** Leading dimension — stride between row starts (row-major) or column starts (column-major). */
|
|
25
28
|
readonly lda: number;
|
|
26
29
|
|
|
30
|
+
/** Storage layout this matrix was created with — every routine that accepts a GpuMatrix reads this automatically. */
|
|
31
|
+
readonly layout: 'row-major' | 'column-major';
|
|
32
|
+
|
|
27
33
|
/**
|
|
28
|
-
* Uploads a
|
|
29
|
-
* Float64Array is packed as two f32s per element (WGSL
|
|
30
|
-
* and stored across two GPU buffers internally; `read()`
|
|
31
|
-
* original doubles.
|
|
34
|
+
* Uploads a Float32Array or Float64Array matrix to GPU memory, row-major
|
|
35
|
+
* or column-major. A Float64Array is packed as two f32s per element (WGSL
|
|
36
|
+
* has no f64 type) and stored across two GPU buffers internally; `read()`
|
|
37
|
+
* reassembles the original doubles.
|
|
32
38
|
*
|
|
33
|
-
* `
|
|
34
|
-
* `
|
|
39
|
+
* `rows`/`cols` always describe the logical shape regardless of layout.
|
|
40
|
+
* `lda` defaults to `cols` (row-major) or `rows` (column-major) — dense, no
|
|
41
|
+
* padding. `data` must have at least `rows * lda` (row-major) or
|
|
42
|
+
* `cols * lda` (column-major) elements.
|
|
35
43
|
*
|
|
36
|
-
* @param data
|
|
37
|
-
* @param rows
|
|
38
|
-
* @param cols
|
|
39
|
-
* @param lda
|
|
44
|
+
* @param data - matrix data, in the order matching `layout`
|
|
45
|
+
* @param rows - number of rows
|
|
46
|
+
* @param cols - number of columns
|
|
47
|
+
* @param lda - leading dimension (default: `cols` for row-major, `rows` for column-major)
|
|
48
|
+
* @param layout - storage layout (default: `'row-major'`)
|
|
40
49
|
*
|
|
41
50
|
* @example
|
|
42
51
|
* ```js
|
|
@@ -46,15 +55,19 @@ export declare class GpuMatrix {
|
|
|
46
55
|
* // 2×3 matrix: [[1,2,3],[4,5,6]]
|
|
47
56
|
* const mat = GpuMatrix.from(new Float32Array([1,2,3,4,5,6]), 2, 3);
|
|
48
57
|
* console.log(mat.rows, mat.cols, mat.lda); // 2 3 3
|
|
58
|
+
*
|
|
59
|
+
* // Same logical matrix, column-major storage
|
|
60
|
+
* const matCol = GpuMatrix.from(new Float32Array([1,4,2,5,3,6]), 2, 3, undefined, "column-major");
|
|
49
61
|
* ```
|
|
50
62
|
*/
|
|
51
|
-
static from(data: Float32Array | Float64Array, rows: number, cols: number, lda?: number): GpuMatrix;
|
|
63
|
+
static from(data: Float32Array | Float64Array, rows: number, cols: number, lda?: number, layout?: 'row-major' | 'column-major'): GpuMatrix;
|
|
52
64
|
|
|
53
65
|
/**
|
|
54
|
-
* Downloads the matrix from GPU memory and returns a dense
|
|
55
|
-
*
|
|
56
|
-
* this matrix was created from one. If `lda
|
|
57
|
-
*
|
|
66
|
+
* Downloads the matrix from GPU memory and returns a dense array of shape
|
|
67
|
+
* `rows × cols`, in the same layout it was created with — a `Float32Array`,
|
|
68
|
+
* or a `Float64Array` if this matrix was created from one. If `lda` exceeds
|
|
69
|
+
* the dense minimum, the leading-dimension padding is stripped so the
|
|
70
|
+
* returned array is always tightly packed.
|
|
58
71
|
*
|
|
59
72
|
* @example
|
|
60
73
|
* ```js
|
|
@@ -4,7 +4,7 @@ import { extractResult } from "../util/result.mjs";
|
|
|
4
4
|
import { packF64, unpackF64 } from "../util/f64pack.mjs";
|
|
5
5
|
|
|
6
6
|
export class GpuMatrix {
|
|
7
|
-
constructor(buffer, rows, cols, lda, auxBuffer = null) {
|
|
7
|
+
constructor(buffer, rows, cols, lda, auxBuffer = null, layout = "row-major") {
|
|
8
8
|
this._buf = buffer;
|
|
9
9
|
// Non-null only for Float64Array-backed matrices — see GpuVector for why
|
|
10
10
|
// (packF64 splits each element into a "main"/_buf f32 and "aux"/_auxBuf
|
|
@@ -13,29 +13,43 @@ export class GpuMatrix {
|
|
|
13
13
|
this.rows = rows;
|
|
14
14
|
this.cols = cols;
|
|
15
15
|
this.lda = lda;
|
|
16
|
+
this.layout = layout;
|
|
16
17
|
}
|
|
17
18
|
|
|
18
19
|
/**
|
|
19
|
-
* Uploads a
|
|
20
|
-
*
|
|
21
|
-
*
|
|
20
|
+
* Uploads a Float32Array or Float64Array matrix to GPU memory, row-major or
|
|
21
|
+
* column-major. `rows`/`cols` always describe the logical shape regardless
|
|
22
|
+
* of layout. `lda` is the stride between consecutive rows (row-major) or
|
|
23
|
+
* columns (column-major) — defaults to `cols`/`rows` respectively (dense,
|
|
24
|
+
* no padding). `data` must have at least `rows * lda` (row-major) or
|
|
25
|
+
* `cols * lda` (column-major) elements.
|
|
22
26
|
*/
|
|
23
|
-
static from(data, rows, cols, lda =
|
|
27
|
+
static from(data, rows, cols, lda, layout = "row-major") {
|
|
28
|
+
if (layout !== "row-major" && layout !== "column-major")
|
|
29
|
+
throw new Error("layout must be 'row-major' or 'column-major'.");
|
|
30
|
+
const isRowMajor = layout === "row-major";
|
|
31
|
+
if (lda === undefined) lda = isRowMajor ? cols : rows;
|
|
32
|
+
|
|
24
33
|
if (!(data instanceof Float32Array) && !(data instanceof Float64Array))
|
|
25
34
|
throw new Error("GpuMatrix.from expects a Float32Array or Float64Array.");
|
|
26
35
|
if (!Number.isInteger(rows) || rows <= 0)
|
|
27
36
|
throw new Error("rows must be a positive integer.");
|
|
28
37
|
if (!Number.isInteger(cols) || cols <= 0)
|
|
29
38
|
throw new Error("cols must be a positive integer.");
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
39
|
+
const minLda = isRowMajor ? cols : rows;
|
|
40
|
+
if (!Number.isInteger(lda) || lda < minLda)
|
|
41
|
+
throw new Error(`lda must be an integer >= ${isRowMajor ? "cols" : "rows"}.`);
|
|
42
|
+
|
|
43
|
+
// Row-major: `rows` chunks of length `lda` (only the first `cols` of each used).
|
|
44
|
+
// Column-major: `cols` chunks of length `lda` (only the first `rows` of each used).
|
|
45
|
+
const outerCount = isRowMajor ? rows : cols;
|
|
46
|
+
if (data.length < outerCount * lda)
|
|
33
47
|
throw new Error(
|
|
34
|
-
"data does not have enough elements for the given rows and lda.",
|
|
48
|
+
"data does not have enough elements for the given rows, cols, and lda.",
|
|
35
49
|
);
|
|
36
50
|
|
|
37
51
|
if (data instanceof Float64Array) {
|
|
38
|
-
const n =
|
|
52
|
+
const n = outerCount * lda;
|
|
39
53
|
const main = new Float32Array(n);
|
|
40
54
|
const aux = new Uint32Array(n);
|
|
41
55
|
for (let i = 0; i < n; i++) {
|
|
@@ -45,11 +59,11 @@ export class GpuMatrix {
|
|
|
45
59
|
}
|
|
46
60
|
const mainBuf = uploadBuffer(main, "gpu-matrix-f64-main", true);
|
|
47
61
|
const auxBuf = uploadBuffer(aux, "gpu-matrix-f64-aux", true);
|
|
48
|
-
return new GpuMatrix(mainBuf, rows, cols, lda, auxBuf);
|
|
62
|
+
return new GpuMatrix(mainBuf, rows, cols, lda, auxBuf, layout);
|
|
49
63
|
}
|
|
50
64
|
|
|
51
|
-
const buf = uploadBuffer(data.subarray(0,
|
|
52
|
-
return new GpuMatrix(buf, rows, cols, lda);
|
|
65
|
+
const buf = uploadBuffer(data.subarray(0, outerCount * lda), "gpu-matrix", true);
|
|
66
|
+
return new GpuMatrix(buf, rows, cols, lda, null, layout);
|
|
53
67
|
}
|
|
54
68
|
|
|
55
69
|
async read() {
|
|
@@ -58,6 +72,10 @@ export class GpuMatrix {
|
|
|
58
72
|
const rb = stageReadback(enc, this._buf);
|
|
59
73
|
device.queue.submit([enc.finish()]);
|
|
60
74
|
|
|
75
|
+
const isRowMajor = this.layout !== "column-major";
|
|
76
|
+
const outerCount = isRowMajor ? this.rows : this.cols;
|
|
77
|
+
const innerLen = isRowMajor ? this.cols : this.rows;
|
|
78
|
+
|
|
61
79
|
if (this._auxBuf) {
|
|
62
80
|
const encAux = device.createCommandEncoder();
|
|
63
81
|
const rbAux = stageReadback(encAux, this._auxBuf);
|
|
@@ -67,20 +85,20 @@ export class GpuMatrix {
|
|
|
67
85
|
extractResult(rb, Float32Array),
|
|
68
86
|
extractResult(rbAux, Uint32Array),
|
|
69
87
|
]);
|
|
70
|
-
const raw = new Float64Array(
|
|
88
|
+
const raw = new Float64Array(outerCount * this.lda);
|
|
71
89
|
for (let i = 0; i < raw.length; i++) raw[i] = unpackF64(main[i], aux[i]);
|
|
72
|
-
if (this.lda ===
|
|
73
|
-
const out = new Float64Array(
|
|
74
|
-
for (let r = 0; r <
|
|
75
|
-
out.set(raw.subarray(r * this.lda, r * this.lda +
|
|
90
|
+
if (this.lda === innerLen) return raw;
|
|
91
|
+
const out = new Float64Array(outerCount * innerLen);
|
|
92
|
+
for (let r = 0; r < outerCount; r++)
|
|
93
|
+
out.set(raw.subarray(r * this.lda, r * this.lda + innerLen), r * innerLen);
|
|
76
94
|
return out;
|
|
77
95
|
}
|
|
78
96
|
|
|
79
97
|
const raw = await extractResult(rb, Float32Array);
|
|
80
|
-
if (this.lda ===
|
|
81
|
-
const out = new Float32Array(
|
|
82
|
-
for (let r = 0; r <
|
|
83
|
-
out.set(raw.subarray(r * this.lda, r * this.lda +
|
|
98
|
+
if (this.lda === innerLen) return raw;
|
|
99
|
+
const out = new Float32Array(outerCount * innerLen);
|
|
100
|
+
for (let r = 0; r < outerCount; r++)
|
|
101
|
+
out.set(raw.subarray(r * this.lda, r * this.lda + innerLen), r * innerLen);
|
|
84
102
|
return out;
|
|
85
103
|
}
|
|
86
104
|
|
package/src/sgemv/sgemv.d.mts
CHANGED
|
@@ -20,13 +20,17 @@ import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
|
20
20
|
* @param m - number of rows in A
|
|
21
21
|
* @param n - number of columns in A
|
|
22
22
|
* @param alpha - scalar multiplier for op(A)*x
|
|
23
|
-
* @param A - Float32Array
|
|
24
|
-
*
|
|
23
|
+
* @param A - Float32Array, row-major or column-major (see `layout`), at least
|
|
24
|
+
* (m-1)*lda+n elements for row-major or (n-1)*lda+m elements for column-major
|
|
25
|
+
* @param lda - leading dimension of A (>= n for row-major, >= m for column-major)
|
|
25
26
|
* @param x - Float32Array input vector
|
|
26
27
|
* @param incx - stride for x (must be a positive integer)
|
|
27
28
|
* @param beta - scalar multiplier for y
|
|
28
29
|
* @param y - Float32Array input/output vector
|
|
29
30
|
* @param incy - stride for y (must be a positive integer)
|
|
31
|
+
* @param layout - storage layout of `A` (default: `'row-major'`); column-major
|
|
32
|
+
* swaps the effective `m`/`n` and flips `trans` internally (op(A) stays
|
|
33
|
+
* what you asked for either way — x/y keep their original lengths)
|
|
30
34
|
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/sgemv/sgemv.mjs#L16">Source code: sgemv.mjs (L16)</a>
|
|
31
35
|
* @category BLAS Level 2
|
|
32
36
|
*/
|
|
@@ -36,26 +40,29 @@ export declare function sgemv(
|
|
|
36
40
|
m: number,
|
|
37
41
|
n: number,
|
|
38
42
|
alpha: number,
|
|
39
|
-
A: Float32Array
|
|
43
|
+
A: Float32Array,
|
|
40
44
|
lda: number,
|
|
41
45
|
x: Float32Array,
|
|
42
46
|
incx: number,
|
|
43
47
|
beta: number,
|
|
44
48
|
y: Float32Array,
|
|
45
49
|
incy: number,
|
|
50
|
+
layout?: 'row-major' | 'column-major',
|
|
46
51
|
): Promise<{ y: Float32Array; gpuTimeMs?: number }>;
|
|
47
52
|
|
|
48
53
|
/**
|
|
49
54
|
* Performs the matrix-vector operation y = alpha * op(A) * x + beta * y
|
|
50
55
|
*
|
|
51
|
-
* A is kept GPU-resident; x and y are CPU Float32Arrays.
|
|
56
|
+
* A is kept GPU-resident; x and y are CPU Float32Arrays. `A`'s own `layout`
|
|
57
|
+
* (set at `GpuMatrix.from` time) determines the operation — there is no
|
|
58
|
+
* separate `layout` argument here.
|
|
52
59
|
*
|
|
53
60
|
* @param device - GPUDevice from `init()`
|
|
54
61
|
* @param trans - `'no-transpose'` for A, `'transpose'` for A^T
|
|
55
62
|
* @param m - number of rows in A
|
|
56
63
|
* @param n - number of columns in A
|
|
57
64
|
* @param alpha - scalar multiplier for op(A)*x
|
|
58
|
-
* @param A - GpuMatrix,
|
|
65
|
+
* @param A - GpuMatrix, GPU-resident
|
|
59
66
|
* @param lda - leading dimension of A (must equal A.lda)
|
|
60
67
|
* @param x - Float32Array input vector
|
|
61
68
|
* @param incx - stride for x (must be a positive integer)
|
|
@@ -83,7 +90,9 @@ export declare function sgemv(
|
|
|
83
90
|
/**
|
|
84
91
|
* Performs the matrix-vector operation y = alpha * op(A) * x + beta * y
|
|
85
92
|
*
|
|
86
|
-
* x and y are kept resident on the GPU. A must be a GpuMatrix
|
|
93
|
+
* x and y are kept resident on the GPU. A must be a GpuMatrix; its own
|
|
94
|
+
* `layout` (set at `GpuMatrix.from` time) determines the operation — there is
|
|
95
|
+
* no separate `layout` argument here.
|
|
87
96
|
*
|
|
88
97
|
* {@includeCode ../../examples/sgemv/gpuvec.sgemv.js}
|
|
89
98
|
*
|
|
@@ -92,8 +101,8 @@ export declare function sgemv(
|
|
|
92
101
|
* @param m - number of rows in A
|
|
93
102
|
* @param n - number of columns in A
|
|
94
103
|
* @param alpha - scalar multiplier for op(A)*x
|
|
95
|
-
* @param A - GpuMatrix
|
|
96
|
-
* @param lda - leading dimension of A (
|
|
104
|
+
* @param A - GpuMatrix
|
|
105
|
+
* @param lda - leading dimension of A (must equal A.lda)
|
|
97
106
|
* @param x - GpuVector input vector (not mutated)
|
|
98
107
|
* @param incx - stride for x (must be a positive integer)
|
|
99
108
|
* @param beta - scalar multiplier for y
|
package/src/sgemv/sgemv.mjs
CHANGED
|
@@ -13,24 +13,17 @@ import { calcWorkgroups } from "../util/workgroup.mjs";
|
|
|
13
13
|
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
14
14
|
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
15
15
|
|
|
16
|
-
export async function sgemv(device, trans, m, n, alpha, A, lda, x, incx, beta, y, incy) {
|
|
16
|
+
export async function sgemv(device, trans, m, n, alpha, A, lda, x, incx, beta, y, incy, layout = "row-major") {
|
|
17
|
+
const AIsGpu = A instanceof GpuMatrix;
|
|
17
18
|
const xIsGpu = x instanceof GpuVector;
|
|
18
19
|
const yIsGpu = y instanceof GpuVector;
|
|
19
|
-
const AIsGpu = A instanceof GpuMatrix;
|
|
20
|
-
const isNoTrans = trans === "no-transpose";
|
|
21
20
|
|
|
22
21
|
if (!(device instanceof GPUDevice))
|
|
23
22
|
throw new Error("device must be a GPUDevice.");
|
|
24
|
-
if (
|
|
23
|
+
if (trans !== "no-transpose" && trans !== "transpose")
|
|
25
24
|
throw new Error("trans must be 'no-transpose' or 'transpose'.");
|
|
26
|
-
if (
|
|
27
|
-
|
|
28
|
-
!Number.isInteger(n) ||
|
|
29
|
-
!Number.isInteger(incx) ||
|
|
30
|
-
!Number.isInteger(incy) ||
|
|
31
|
-
!Number.isInteger(lda)
|
|
32
|
-
)
|
|
33
|
-
throw new Error("m, n, incx, incy, and lda must be integers.");
|
|
25
|
+
if (layout !== "row-major" && layout !== "column-major")
|
|
26
|
+
throw new Error("layout must be 'row-major' or 'column-major'.");
|
|
34
27
|
if (typeof alpha !== "number")
|
|
35
28
|
throw new Error("alpha must be a number.");
|
|
36
29
|
if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
|
|
@@ -39,9 +32,16 @@ export async function sgemv(device, trans, m, n, alpha, A, lda, x, incx, beta, y
|
|
|
39
32
|
throw new Error("beta must be a number.");
|
|
40
33
|
if (Number.isNaN(beta)) throw new Error("beta must not be NaN.");
|
|
41
34
|
if (!Number.isFinite(beta)) throw new Error("beta must be finite.");
|
|
35
|
+
if (
|
|
36
|
+
!Number.isInteger(m) ||
|
|
37
|
+
!Number.isInteger(n) ||
|
|
38
|
+
!Number.isInteger(incx) ||
|
|
39
|
+
!Number.isInteger(incy) ||
|
|
40
|
+
!Number.isInteger(lda)
|
|
41
|
+
)
|
|
42
|
+
throw new Error("m, n, incx, incy, and lda must be integers.");
|
|
42
43
|
if (incx <= 0 || incy <= 0)
|
|
43
44
|
throw new Error("incx and incy must be positive.");
|
|
44
|
-
if (lda < n) throw new Error("lda must be >= n.");
|
|
45
45
|
if (!AIsGpu && !(A instanceof Float32Array))
|
|
46
46
|
throw new Error("A must be a Float32Array or GpuMatrix.");
|
|
47
47
|
if (!xIsGpu && !(x instanceof Float32Array))
|
|
@@ -65,11 +65,20 @@ export async function sgemv(device, trans, m, n, alpha, A, lda, x, incx, beta, y
|
|
|
65
65
|
if (m < 0 || n < 0) throw new Error("m and n must be non-negative.");
|
|
66
66
|
if (m === 0 || n === 0) return yIsGpu ? {} : { y };
|
|
67
67
|
|
|
68
|
-
//
|
|
69
|
-
//
|
|
68
|
+
// Column-major A reinterpreted row-major is A^T — swap to the kernel's
|
|
69
|
+
// view of m/n/trans now that the shape checks above are done.
|
|
70
|
+
const effLayout = AIsGpu ? A.layout : layout;
|
|
71
|
+
if (effLayout === "column-major") {
|
|
72
|
+
[m, n] = [n, m];
|
|
73
|
+
trans = trans === "no-transpose" ? "transpose" : "no-transpose";
|
|
74
|
+
}
|
|
75
|
+
const isNoTrans = trans === "no-transpose";
|
|
76
|
+
|
|
77
|
+
// NoTrans: x has n elements, y has m elements; Trans: x has m elements, y has n elements.
|
|
70
78
|
const xLen = isNoTrans ? n : m;
|
|
71
79
|
const yLen = isNoTrans ? m : n;
|
|
72
80
|
|
|
81
|
+
if (lda < n) throw new Error("lda must be >= n.");
|
|
73
82
|
if (!AIsGpu && A.length < (m - 1) * lda + n)
|
|
74
83
|
throw new Error(
|
|
75
84
|
"A does not have enough elements for the given m, n, and lda.",
|
|
@@ -110,9 +119,7 @@ export async function sgemv(device, trans, m, n, alpha, A, lda, x, incx, beta, y
|
|
|
110
119
|
paramsBuffer,
|
|
111
120
|
]);
|
|
112
121
|
|
|
113
|
-
// NoTrans: one workgroup per row;
|
|
114
|
-
// grid-stride loop handles remaining rows when m > dispatch count.
|
|
115
|
-
// Trans: one thread per output column → dispatch ceil(n/64)
|
|
122
|
+
// NoTrans: one workgroup per row (grid-stride handles overflow); Trans: one thread per output column.
|
|
116
123
|
const wgCount = isNoTrans
|
|
117
124
|
? Math.min(m, device.limits.maxComputeWorkgroupsPerDimension)
|
|
118
125
|
: calcWorkgroups(yLen);
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
2
|
+
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Performs the rank-1 update A = alpha * x * y^T + A
|
|
6
|
+
*
|
|
7
|
+
* A is an m×n matrix stored in row-major order, updated in place. `lda` is
|
|
8
|
+
* the leading dimension (number of floats between the start of consecutive
|
|
9
|
+
* rows — must be >= n).
|
|
10
|
+
*
|
|
11
|
+
* {@includeCode ../../examples/sger/sger.js}
|
|
12
|
+
*
|
|
13
|
+
* **Browser (standalone HTML):**
|
|
14
|
+
* {@includeCode ../../examples/sger/web/sger.html}
|
|
15
|
+
*
|
|
16
|
+
* @param device - GPUDevice from `init()`
|
|
17
|
+
* @param m - number of rows in A (length of x)
|
|
18
|
+
* @param n - number of columns in A (length of y)
|
|
19
|
+
* @param alpha - scalar multiplier for x*y^T
|
|
20
|
+
* @param x - Float32Array input vector, length at least (m-1)*incx+1
|
|
21
|
+
* @param incx - stride for x (must be a positive integer)
|
|
22
|
+
* @param y - Float32Array input vector, length at least (n-1)*incy+1
|
|
23
|
+
* @param incy - stride for y (must be a positive integer)
|
|
24
|
+
* @param A - Float32Array, row-major or column-major (see `layout`), at least
|
|
25
|
+
* (m-1)*lda+n elements for row-major or (n-1)*lda+m elements for column-major
|
|
26
|
+
* @param lda - leading dimension of A (>= n for row-major, >= m for column-major)
|
|
27
|
+
* @param layout - storage layout of `A` (default: `'row-major'`)
|
|
28
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/sger/sger.mjs#L15">Source code: sger.mjs (L15)</a>
|
|
29
|
+
* @category BLAS Level 2
|
|
30
|
+
*/
|
|
31
|
+
export declare function sger(
|
|
32
|
+
device: GPUDevice,
|
|
33
|
+
m: number,
|
|
34
|
+
n: number,
|
|
35
|
+
alpha: number,
|
|
36
|
+
x: Float32Array,
|
|
37
|
+
incx: number,
|
|
38
|
+
y: Float32Array,
|
|
39
|
+
incy: number,
|
|
40
|
+
A: Float32Array,
|
|
41
|
+
lda: number,
|
|
42
|
+
layout?: 'row-major' | 'column-major',
|
|
43
|
+
): Promise<{ A: Float32Array; gpuTimeMs?: number }>;
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Performs the rank-1 update A = alpha * x * y^T + A
|
|
47
|
+
*
|
|
48
|
+
* A is kept GPU-resident; x and y are CPU Float32Arrays. `A`'s own `layout`
|
|
49
|
+
* (set at `GpuMatrix.from` time) determines the operation — there is no
|
|
50
|
+
* separate `layout` argument here.
|
|
51
|
+
*
|
|
52
|
+
* @param device - GPUDevice from `init()`
|
|
53
|
+
* @param m - number of rows in A
|
|
54
|
+
* @param n - number of columns in A
|
|
55
|
+
* @param alpha - scalar multiplier for x*y^T
|
|
56
|
+
* @param x - Float32Array input vector
|
|
57
|
+
* @param incx - stride for x (must be a positive integer)
|
|
58
|
+
* @param y - Float32Array input vector
|
|
59
|
+
* @param incy - stride for y (must be a positive integer)
|
|
60
|
+
* @param A - GpuMatrix, GPU-resident
|
|
61
|
+
* @param lda - leading dimension of A (must equal A.lda)
|
|
62
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/sger/sger.mjs#L15">Source code: sger.mjs (L15)</a>
|
|
63
|
+
* @category BLAS Level 2
|
|
64
|
+
*/
|
|
65
|
+
export declare function sger(
|
|
66
|
+
device: GPUDevice,
|
|
67
|
+
m: number,
|
|
68
|
+
n: number,
|
|
69
|
+
alpha: number,
|
|
70
|
+
x: Float32Array,
|
|
71
|
+
incx: number,
|
|
72
|
+
y: Float32Array,
|
|
73
|
+
incy: number,
|
|
74
|
+
A: GpuMatrix,
|
|
75
|
+
lda: number,
|
|
76
|
+
): Promise<{ gpuTimeMs?: number }>;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Performs the rank-1 update A = alpha * x * y^T + A
|
|
80
|
+
*
|
|
81
|
+
* x, y, and A are all kept resident on the GPU. `A`'s own `layout` (set at
|
|
82
|
+
* `GpuMatrix.from` time) determines the operation — there is no separate
|
|
83
|
+
* `layout` argument here.
|
|
84
|
+
*
|
|
85
|
+
* {@includeCode ../../examples/sger/gpuvec.sger.js}
|
|
86
|
+
*
|
|
87
|
+
* @param device - GPUDevice from `init()`
|
|
88
|
+
* @param m - number of rows in A
|
|
89
|
+
* @param n - number of columns in A
|
|
90
|
+
* @param alpha - scalar multiplier for x*y^T
|
|
91
|
+
* @param x - GpuVector input vector (not mutated)
|
|
92
|
+
* @param incx - stride for x (must be a positive integer)
|
|
93
|
+
* @param y - GpuVector input vector (not mutated)
|
|
94
|
+
* @param incy - stride for y (must be a positive integer)
|
|
95
|
+
* @param A - GpuMatrix, mutated in place
|
|
96
|
+
* @param lda - leading dimension of A (must equal A.lda)
|
|
97
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/sger/sger.mjs#L15">Source code: sger.mjs (L15)</a>
|
|
98
|
+
* @category BLAS Level 2
|
|
99
|
+
*/
|
|
100
|
+
export declare function sger(
|
|
101
|
+
device: GPUDevice,
|
|
102
|
+
m: number,
|
|
103
|
+
n: number,
|
|
104
|
+
alpha: number,
|
|
105
|
+
x: GpuVector,
|
|
106
|
+
incx: number,
|
|
107
|
+
y: GpuVector,
|
|
108
|
+
incy: number,
|
|
109
|
+
A: GpuMatrix,
|
|
110
|
+
lda: number,
|
|
111
|
+
): Promise<{ gpuTimeMs?: number }>;
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
import {
|
|
2
|
+
uploadBuffer,
|
|
3
|
+
createParamsBuffer,
|
|
4
|
+
stageReadback,
|
|
5
|
+
destroyBuffers,
|
|
6
|
+
} from "../util/buffer.mjs";
|
|
7
|
+
import { createBindGroup } from "../util/bindgroup.mjs";
|
|
8
|
+
import { runComputePass, submit } from "../util/compute.mjs";
|
|
9
|
+
import { extractResult } from "../util/result.mjs";
|
|
10
|
+
import { extractTimestamp } from "../util/benchmark.mjs";
|
|
11
|
+
import { getPipeline } from "../util/pipeline.mjs";
|
|
12
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
13
|
+
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
14
|
+
|
|
15
|
+
export async function sger(device, m, n, alpha, x, incx, y, incy, A, lda, layout = "row-major") {
|
|
16
|
+
const AIsGpu = A instanceof GpuMatrix;
|
|
17
|
+
|
|
18
|
+
if (!(device instanceof GPUDevice))
|
|
19
|
+
throw new Error("device must be a GPUDevice.");
|
|
20
|
+
if (layout !== "row-major" && layout !== "column-major")
|
|
21
|
+
throw new Error("layout must be 'row-major' or 'column-major'.");
|
|
22
|
+
if (typeof alpha !== "number")
|
|
23
|
+
throw new Error("alpha must be a number.");
|
|
24
|
+
if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
|
|
25
|
+
if (!Number.isFinite(alpha)) throw new Error("alpha must be finite.");
|
|
26
|
+
if (
|
|
27
|
+
!Number.isInteger(m) ||
|
|
28
|
+
!Number.isInteger(n) ||
|
|
29
|
+
!Number.isInteger(incx) ||
|
|
30
|
+
!Number.isInteger(incy) ||
|
|
31
|
+
!Number.isInteger(lda)
|
|
32
|
+
)
|
|
33
|
+
throw new Error("m, n, incx, incy, and lda must be integers.");
|
|
34
|
+
if (incx <= 0 || incy <= 0)
|
|
35
|
+
throw new Error("incx and incy must be positive.");
|
|
36
|
+
if (!AIsGpu && !(A instanceof Float32Array))
|
|
37
|
+
throw new Error("A must be a Float32Array or GpuMatrix.");
|
|
38
|
+
if (AIsGpu && lda !== A.lda)
|
|
39
|
+
throw new Error("lda must match A.lda when A is a GpuMatrix.");
|
|
40
|
+
// A.rows/A.cols are fixed regardless of layout — check against original m/n before the swap below.
|
|
41
|
+
if (AIsGpu && (A.rows < m || A.cols < n))
|
|
42
|
+
throw new Error("A is too small for the given m and n.");
|
|
43
|
+
|
|
44
|
+
// GpuMatrix's own layout wins over the argument; column-major A reinterpreted row-major is A^T, so swap x/y and m/n to reproduce it (transpose of a rank-1 update swaps its two vectors).
|
|
45
|
+
const effLayout = AIsGpu ? A.layout : layout;
|
|
46
|
+
if (effLayout === "column-major") {
|
|
47
|
+
[m, n] = [n, m];
|
|
48
|
+
[x, y] = [y, x];
|
|
49
|
+
[incx, incy] = [incy, incx];
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const xIsGpu = x instanceof GpuVector;
|
|
53
|
+
const yIsGpu = y instanceof GpuVector;
|
|
54
|
+
|
|
55
|
+
if (lda < n) throw new Error("lda must be >= n.");
|
|
56
|
+
if (!xIsGpu && !(x instanceof Float32Array))
|
|
57
|
+
throw new Error("x must be a Float32Array or GpuVector.");
|
|
58
|
+
if (!yIsGpu && !(y instanceof Float32Array))
|
|
59
|
+
throw new Error("y must be a Float32Array or GpuVector.");
|
|
60
|
+
if (xIsGpu !== yIsGpu)
|
|
61
|
+
throw new Error(
|
|
62
|
+
"x and y must be the same type (both Float32Array or both GpuVector).",
|
|
63
|
+
);
|
|
64
|
+
if (xIsGpu && !AIsGpu)
|
|
65
|
+
throw new Error("A must be a GpuMatrix when x and y are GpuVectors.");
|
|
66
|
+
if (AIsGpu && xIsGpu && A._buf === x._buf)
|
|
67
|
+
throw new Error("A and x must not reference the same GPU buffer.");
|
|
68
|
+
if (AIsGpu && yIsGpu && A._buf === y._buf)
|
|
69
|
+
throw new Error("A and y must not reference the same GPU buffer.");
|
|
70
|
+
if (m < 0 || n < 0) throw new Error("m and n must be non-negative.");
|
|
71
|
+
if (m === 0 || n === 0) return AIsGpu ? {} : { A };
|
|
72
|
+
|
|
73
|
+
if (!AIsGpu && A.length < (m - 1) * lda + n)
|
|
74
|
+
throw new Error(
|
|
75
|
+
"A does not have enough elements for the given m, n, and lda.",
|
|
76
|
+
);
|
|
77
|
+
if (x.length < (m - 1) * incx + 1)
|
|
78
|
+
throw new Error("x does not have enough elements for the given m and incx.");
|
|
79
|
+
if (y.length < (n - 1) * incy + 1)
|
|
80
|
+
throw new Error("y does not have enough elements for the given n and incy.");
|
|
81
|
+
|
|
82
|
+
const pipeline = await getPipeline(device, "sger");
|
|
83
|
+
|
|
84
|
+
let xBuffer = null;
|
|
85
|
+
let yBuffer = null;
|
|
86
|
+
let ABuffer = null;
|
|
87
|
+
let paramsBuffer = null;
|
|
88
|
+
|
|
89
|
+
try {
|
|
90
|
+
xBuffer = xIsGpu ? x._buf : uploadBuffer(x, "sger-x", false);
|
|
91
|
+
yBuffer = yIsGpu ? y._buf : uploadBuffer(y, "sger-y", false);
|
|
92
|
+
ABuffer = AIsGpu ? A._buf : uploadBuffer(A, "sger-A", true);
|
|
93
|
+
paramsBuffer = createParamsBuffer(
|
|
94
|
+
[
|
|
95
|
+
{ value: m, type: "u32" },
|
|
96
|
+
{ value: n, type: "u32" },
|
|
97
|
+
{ value: alpha, type: "f32" },
|
|
98
|
+
{ value: incx, type: "u32" },
|
|
99
|
+
{ value: incy, type: "u32" },
|
|
100
|
+
{ value: lda, type: "u32" },
|
|
101
|
+
],
|
|
102
|
+
"sger-params",
|
|
103
|
+
);
|
|
104
|
+
|
|
105
|
+
const bindGroup = createBindGroup(pipeline.getBindGroupLayout(0), [
|
|
106
|
+
xBuffer,
|
|
107
|
+
yBuffer,
|
|
108
|
+
ABuffer,
|
|
109
|
+
paramsBuffer,
|
|
110
|
+
]);
|
|
111
|
+
|
|
112
|
+
// One workgroup per row of A; clamped to device limit — the shader's
|
|
113
|
+
// grid-stride loop handles remaining rows when m > dispatch count.
|
|
114
|
+
const wgCount = Math.min(m, device.limits.maxComputeWorkgroupsPerDimension);
|
|
115
|
+
const { commandEncoder, ts } = runComputePass(pipeline, bindGroup, wgCount);
|
|
116
|
+
const readBuffer = AIsGpu ? null : stageReadback(commandEncoder, ABuffer);
|
|
117
|
+
|
|
118
|
+
submit(commandEncoder);
|
|
119
|
+
|
|
120
|
+
const gpuTimeMs = await extractTimestamp(ts);
|
|
121
|
+
|
|
122
|
+
if (AIsGpu) {
|
|
123
|
+
if (gpuTimeMs !== undefined) return { gpuTimeMs };
|
|
124
|
+
return {};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const result = await extractResult(readBuffer, Float32Array);
|
|
128
|
+
if (gpuTimeMs !== undefined) return { A: result, gpuTimeMs };
|
|
129
|
+
return { A: result };
|
|
130
|
+
} finally {
|
|
131
|
+
if (!xIsGpu && xBuffer) destroyBuffers(xBuffer);
|
|
132
|
+
if (!yIsGpu && yBuffer) destroyBuffers(yBuffer);
|
|
133
|
+
if (!AIsGpu && ABuffer) destroyBuffers(ABuffer);
|
|
134
|
+
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
135
|
+
}
|
|
136
|
+
}
|