wgblas 2.2.0 → 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/dist/wgblas.browser.js +1092 -58
- package/index.d.mts +7 -0
- package/index.mjs +7 -0
- package/package.json +40 -4
- package/src/dgemv/dgemv.d.mts +92 -0
- package/src/dgemv/dgemv.mjs +247 -0
- package/src/dger/dger.d.mts +80 -0
- package/src/dger/dger.mjs +213 -0
- package/src/dsymv/dsymv.d.mts +84 -0
- package/src/dsymv/dsymv.mjs +227 -0
- package/src/dsyr/dsyr.d.mts +73 -0
- package/src/dsyr/dsyr.mjs +171 -0
- package/src/dsyr2/dsyr2.d.mts +81 -0
- package/src/dsyr2/dsyr2.mjs +214 -0
- package/src/dtrmv/dtrmv.d.mts +84 -0
- package/src/dtrmv/dtrmv.mjs +219 -0
- package/src/dtrsv/dtrsv.d.mts +79 -0
- package/src/dtrsv/dtrsv.mjs +331 -0
- package/src/shaders/dgemv_n.wgsl +112 -0
- package/src/shaders/dgemv_t.wgsl +102 -0
- package/src/shaders/dger.wgsl +77 -0
- package/src/shaders/dsymv.wgsl +125 -0
- package/src/shaders/dsyr.wgsl +84 -0
- package/src/shaders/dsyr2.wgsl +95 -0
- package/src/shaders/dtrmv.wgsl +111 -0
- package/src/shaders/dtrsv_apply_inverse.wgsl +60 -0
- package/src/shaders/dtrsv_invert_block.wgsl +148 -0
- package/src/shaders/dtrsv_update.wgsl +111 -0
- package/src/shaders/f64/utils/multiply.wgsl +10 -1
- package/src/shaders/index.mjs +63 -0
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
import {
|
|
2
|
+
uploadBuffer,
|
|
3
|
+
createParamsBuffer,
|
|
4
|
+
stageReadback,
|
|
5
|
+
destroyBuffers,
|
|
6
|
+
} from "../util/buffer.mjs";
|
|
7
|
+
import { createBindGroup } from "../util/bindgroup.mjs";
|
|
8
|
+
import { runComputePass, submit } from "../util/compute.mjs";
|
|
9
|
+
import { extractResult } from "../util/result.mjs";
|
|
10
|
+
import { extractTimestamp } from "../util/benchmark.mjs";
|
|
11
|
+
import { getPipeline } from "../util/pipeline.mjs";
|
|
12
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
13
|
+
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
14
|
+
import { splitDoubleDouble, mergeDoubleDouble } from "../util/f64.mjs";
|
|
15
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
16
|
+
|
|
17
|
+
// dsyr: A := alpha * x * x^T + A, double-double (Dekker) f64 emulation of
|
|
18
|
+
// ssyr — x, A, and alpha are each split into an f32 (hi, lo) pair; WGSL has
|
|
19
|
+
// no f64 type.
|
|
20
|
+
export async function dsyr(
|
|
21
|
+
device,
|
|
22
|
+
uplo,
|
|
23
|
+
n,
|
|
24
|
+
alpha,
|
|
25
|
+
x,
|
|
26
|
+
incx,
|
|
27
|
+
A,
|
|
28
|
+
lda,
|
|
29
|
+
layout = "row-major",
|
|
30
|
+
) {
|
|
31
|
+
const xIsGpu = x instanceof GpuVector;
|
|
32
|
+
const AIsGpu = A instanceof GpuMatrix;
|
|
33
|
+
|
|
34
|
+
requireGpuDevice(device);
|
|
35
|
+
requireSameDevice(device, "dsyr", { A, x });
|
|
36
|
+
if (uplo !== "lower" && uplo !== "upper")
|
|
37
|
+
throw new Error("uplo must be 'lower' or 'upper'.");
|
|
38
|
+
if (layout !== "row-major" && layout !== "column-major")
|
|
39
|
+
throw new Error("layout must be 'row-major' or 'column-major'.");
|
|
40
|
+
if (!Number.isInteger(n) || !Number.isInteger(incx) || !Number.isInteger(lda))
|
|
41
|
+
throw new Error("n, incx, and lda must be integers.");
|
|
42
|
+
if (typeof alpha !== "number") throw new Error("alpha must be a number.");
|
|
43
|
+
if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
|
|
44
|
+
if (!Number.isFinite(alpha)) throw new Error("alpha must be finite.");
|
|
45
|
+
if (incx <= 0) throw new Error("incx must be positive.");
|
|
46
|
+
if (lda < n) throw new Error("lda must be >= n.");
|
|
47
|
+
if (!AIsGpu && !(A instanceof Float64Array))
|
|
48
|
+
throw new Error("A must be a Float64Array or GpuMatrix.");
|
|
49
|
+
if (AIsGpu && A.dtype !== Float64Array)
|
|
50
|
+
throw new Error("A must be a Float64Array-backed GpuMatrix.");
|
|
51
|
+
if (!xIsGpu && !(x instanceof Float64Array))
|
|
52
|
+
throw new Error("x must be a Float64Array or GpuVector.");
|
|
53
|
+
if (xIsGpu && x.dtype !== Float64Array)
|
|
54
|
+
throw new Error("x must be a Float64Array-backed GpuVector.");
|
|
55
|
+
if (xIsGpu && !AIsGpu)
|
|
56
|
+
throw new Error("A must be a GpuMatrix when x is a GpuVector.");
|
|
57
|
+
if (AIsGpu && !xIsGpu)
|
|
58
|
+
throw new Error("x must be a GpuVector when A is a GpuMatrix.");
|
|
59
|
+
if (AIsGpu && xIsGpu && A._buf === x._buf)
|
|
60
|
+
throw new Error("A and x must not reference the same GPU buffer.");
|
|
61
|
+
if (AIsGpu && lda !== A.lda)
|
|
62
|
+
throw new Error("lda must match A.lda when A is a GpuMatrix.");
|
|
63
|
+
if (AIsGpu && (A.rows < n || A.cols < n))
|
|
64
|
+
throw new Error("A is too small for the given n.");
|
|
65
|
+
if (n < 0) throw new Error("n must be non-negative.");
|
|
66
|
+
if (n === 0) return AIsGpu ? {} : { A };
|
|
67
|
+
|
|
68
|
+
if (!AIsGpu && A.length < (n - 1) * lda + n)
|
|
69
|
+
throw new Error("A does not have enough elements for the given n and lda.");
|
|
70
|
+
if (x.length < (n - 1) * incx + 1)
|
|
71
|
+
throw new Error(
|
|
72
|
+
"x does not have enough elements for the given n and incx.",
|
|
73
|
+
);
|
|
74
|
+
|
|
75
|
+
// GpuMatrix's own layout wins over the argument; A is symmetric, so column-major A reinterpreted row-major just flips which triangle is stored — flip uplo to match.
|
|
76
|
+
const effLayout = AIsGpu ? A.layout : layout;
|
|
77
|
+
const isLower =
|
|
78
|
+
effLayout === "column-major" ? uplo === "upper" : uplo === "lower";
|
|
79
|
+
|
|
80
|
+
const f64Deps = ["f64/dekker", "f64/utils/add", "f64/utils/multiply"];
|
|
81
|
+
const pipeline = await getPipeline(device, [...f64Deps, "dsyr"], "dsyr_main");
|
|
82
|
+
|
|
83
|
+
const { hi: alphaHi, lo: alphaLo } = splitDoubleDouble(
|
|
84
|
+
new Float64Array([alpha]),
|
|
85
|
+
);
|
|
86
|
+
|
|
87
|
+
let xHiBuffer = null;
|
|
88
|
+
let xLoBuffer = null;
|
|
89
|
+
let AHiBuffer = null;
|
|
90
|
+
let ALoBuffer = null;
|
|
91
|
+
let paramsBuffer = null;
|
|
92
|
+
let readHiBuffer = null;
|
|
93
|
+
let readLoBuffer = null;
|
|
94
|
+
|
|
95
|
+
try {
|
|
96
|
+
if (xIsGpu) {
|
|
97
|
+
xHiBuffer = x._buf;
|
|
98
|
+
xLoBuffer = x._loBuf;
|
|
99
|
+
AHiBuffer = A._buf;
|
|
100
|
+
ALoBuffer = A._loBuf;
|
|
101
|
+
} else {
|
|
102
|
+
const xSplit = splitDoubleDouble(x);
|
|
103
|
+
const ASplit = splitDoubleDouble(A);
|
|
104
|
+
xHiBuffer = uploadBuffer(device, xSplit.hi, "dsyr-xHi", false);
|
|
105
|
+
xLoBuffer = uploadBuffer(device, xSplit.lo, "dsyr-xLo", false);
|
|
106
|
+
AHiBuffer = uploadBuffer(device, ASplit.hi, "dsyr-AHi", true);
|
|
107
|
+
ALoBuffer = uploadBuffer(device, ASplit.lo, "dsyr-ALo", true);
|
|
108
|
+
}
|
|
109
|
+
paramsBuffer = createParamsBuffer(
|
|
110
|
+
device,
|
|
111
|
+
[
|
|
112
|
+
{ value: n, type: "u32" },
|
|
113
|
+
{ value: alphaHi[0], type: "f32" },
|
|
114
|
+
{ value: alphaLo[0], type: "f32" },
|
|
115
|
+
{ value: incx, type: "u32" },
|
|
116
|
+
{ value: lda, type: "u32" },
|
|
117
|
+
{ value: isLower ? 0 : 1, type: "u32" },
|
|
118
|
+
],
|
|
119
|
+
"dsyr-params",
|
|
120
|
+
);
|
|
121
|
+
|
|
122
|
+
const bindGroup = createBindGroup(device, pipeline.getBindGroupLayout(0), [
|
|
123
|
+
xHiBuffer,
|
|
124
|
+
xLoBuffer,
|
|
125
|
+
AHiBuffer,
|
|
126
|
+
ALoBuffer,
|
|
127
|
+
paramsBuffer,
|
|
128
|
+
]);
|
|
129
|
+
|
|
130
|
+
// One workgroup per row of A; clamped to device limit — the shader's
|
|
131
|
+
// grid-stride loop handles remaining rows when n > dispatch count.
|
|
132
|
+
const wgCount = Math.min(n, device.limits.maxComputeWorkgroupsPerDimension);
|
|
133
|
+
const { commandEncoder, ts } = runComputePass(
|
|
134
|
+
device,
|
|
135
|
+
pipeline,
|
|
136
|
+
bindGroup,
|
|
137
|
+
wgCount,
|
|
138
|
+
);
|
|
139
|
+
readHiBuffer = AIsGpu
|
|
140
|
+
? null
|
|
141
|
+
: stageReadback(device, commandEncoder, AHiBuffer);
|
|
142
|
+
readLoBuffer = AIsGpu
|
|
143
|
+
? null
|
|
144
|
+
: stageReadback(device, commandEncoder, ALoBuffer);
|
|
145
|
+
|
|
146
|
+
submit(device, commandEncoder);
|
|
147
|
+
|
|
148
|
+
const gpuTimeMs = await extractTimestamp(ts);
|
|
149
|
+
|
|
150
|
+
if (AIsGpu) {
|
|
151
|
+
if (gpuTimeMs !== undefined) return { gpuTimeMs };
|
|
152
|
+
return {};
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const hi = await extractResult(readHiBuffer, Float32Array);
|
|
156
|
+
readHiBuffer = null;
|
|
157
|
+
const lo = await extractResult(readLoBuffer, Float32Array);
|
|
158
|
+
readLoBuffer = null;
|
|
159
|
+
const result = mergeDoubleDouble(hi, lo);
|
|
160
|
+
if (gpuTimeMs !== undefined) return { A: result, gpuTimeMs };
|
|
161
|
+
return { A: result };
|
|
162
|
+
} finally {
|
|
163
|
+
if (!xIsGpu && xHiBuffer) destroyBuffers(xHiBuffer);
|
|
164
|
+
if (!xIsGpu && xLoBuffer) destroyBuffers(xLoBuffer);
|
|
165
|
+
if (!AIsGpu && AHiBuffer) destroyBuffers(AHiBuffer);
|
|
166
|
+
if (!AIsGpu && ALoBuffer) destroyBuffers(ALoBuffer);
|
|
167
|
+
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
168
|
+
if (readHiBuffer) destroyBuffers(readHiBuffer);
|
|
169
|
+
if (readLoBuffer) destroyBuffers(readLoBuffer);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
2
|
+
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Performs the symmetric rank-2 update $$A \leftarrow \alpha x y^{T} + \alpha y x^{T} + A$$
|
|
6
|
+
* in double precision (double-double emulation — WGSL has no native f64 type).
|
|
7
|
+
*
|
|
8
|
+
* A is an n×n symmetric matrix stored in row-major order, updated in place.
|
|
9
|
+
* Only the triangle specified by `uplo` is referenced and updated; the other
|
|
10
|
+
* triangle is left untouched (implied by symmetry).
|
|
11
|
+
*
|
|
12
|
+
* {@includeCode ../../examples/dsyr2/dsyr2.js}
|
|
13
|
+
*
|
|
14
|
+
* **Browser (standalone HTML):**
|
|
15
|
+
* {@includeCode ../../examples/dsyr2/web/dsyr2.html}
|
|
16
|
+
*
|
|
17
|
+
* @param device - GPUDevice from `init()`
|
|
18
|
+
* @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
|
|
19
|
+
* @param n - order of the matrix A (number of rows and columns)
|
|
20
|
+
* @param alpha - scalar multiplier for x*y^T + y*x^T
|
|
21
|
+
* @param x - Float64Array input vector, length at least (n-1)*incx+1
|
|
22
|
+
* @param incx - stride for x (must be a positive integer)
|
|
23
|
+
* @param y - Float64Array input vector, length at least (n-1)*incy+1
|
|
24
|
+
* @param incy - stride for y (must be a positive integer)
|
|
25
|
+
* @param A - Float64Array, row-major or column-major (see `layout`), at least (n-1)*lda+n elements
|
|
26
|
+
* @param lda - leading dimension of A (>= n either way — A is square)
|
|
27
|
+
* @param layout - storage layout of `A` (default: `'row-major'`); for a symmetric
|
|
28
|
+
* matrix, column-major storage just means the *other* triangle is the one
|
|
29
|
+
* physically referenced for a given `uplo`
|
|
30
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/dsyr2/dsyr2.mjs#L19">Source code: dsyr2.mjs (L19)</a>
|
|
31
|
+
* @category BLAS Level 2
|
|
32
|
+
*/
|
|
33
|
+
export declare function dsyr2(
|
|
34
|
+
device: GPUDevice,
|
|
35
|
+
uplo: 'lower' | 'upper',
|
|
36
|
+
n: number,
|
|
37
|
+
alpha: number,
|
|
38
|
+
x: Float64Array,
|
|
39
|
+
incx: number,
|
|
40
|
+
y: Float64Array,
|
|
41
|
+
incy: number,
|
|
42
|
+
A: Float64Array,
|
|
43
|
+
lda: number,
|
|
44
|
+
layout?: 'row-major' | 'column-major',
|
|
45
|
+
): Promise<{ A: Float64Array; gpuTimeMs?: number }>;
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Performs the symmetric rank-2 update $$A \leftarrow \alpha x y^{T} + \alpha y x^{T} + A$$
|
|
49
|
+
* in double precision (double-double emulation).
|
|
50
|
+
*
|
|
51
|
+
* x, y, and A are all kept resident on the GPU. `A`'s own `layout` (set at
|
|
52
|
+
* `GpuMatrix.from` time) determines the operation — there is no separate
|
|
53
|
+
* `layout` argument here.
|
|
54
|
+
*
|
|
55
|
+
* {@includeCode ../../examples/dsyr2/gpu.dsyr2.js}
|
|
56
|
+
*
|
|
57
|
+
* @param device - GPUDevice from `init()`
|
|
58
|
+
* @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
|
|
59
|
+
* @param n - order of the matrix A
|
|
60
|
+
* @param alpha - scalar multiplier for x*y^T + y*x^T
|
|
61
|
+
* @param x - GpuVector input vector (not mutated), Float64-backed
|
|
62
|
+
* @param incx - stride for x (must be a positive integer)
|
|
63
|
+
* @param y - GpuVector input vector (not mutated), Float64-backed
|
|
64
|
+
* @param incy - stride for y (must be a positive integer)
|
|
65
|
+
* @param A - GpuMatrix (Float64Array-backed), mutated in place
|
|
66
|
+
* @param lda - leading dimension of A (must equal A.lda)
|
|
67
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/dsyr2/dsyr2.mjs#L19">Source code: dsyr2.mjs (L19)</a>
|
|
68
|
+
* @category BLAS Level 2
|
|
69
|
+
*/
|
|
70
|
+
export declare function dsyr2(
|
|
71
|
+
device: GPUDevice,
|
|
72
|
+
uplo: 'lower' | 'upper',
|
|
73
|
+
n: number,
|
|
74
|
+
alpha: number,
|
|
75
|
+
x: GpuVector,
|
|
76
|
+
incx: number,
|
|
77
|
+
y: GpuVector,
|
|
78
|
+
incy: number,
|
|
79
|
+
A: GpuMatrix,
|
|
80
|
+
lda: number,
|
|
81
|
+
): Promise<{ gpuTimeMs?: number }>;
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
import {
|
|
2
|
+
uploadBuffer,
|
|
3
|
+
createParamsBuffer,
|
|
4
|
+
stageReadback,
|
|
5
|
+
destroyBuffers,
|
|
6
|
+
} from "../util/buffer.mjs";
|
|
7
|
+
import { createBindGroup } from "../util/bindgroup.mjs";
|
|
8
|
+
import { runComputePass, submit } from "../util/compute.mjs";
|
|
9
|
+
import { extractResult } from "../util/result.mjs";
|
|
10
|
+
import { extractTimestamp } from "../util/benchmark.mjs";
|
|
11
|
+
import { getPipeline } from "../util/pipeline.mjs";
|
|
12
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
13
|
+
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
14
|
+
import { splitDoubleDouble, mergeDoubleDouble } from "../util/f64.mjs";
|
|
15
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
16
|
+
|
|
17
|
+
// dsyr2: A := alpha * x * y^T + alpha * y * x^T + A, double-double (Dekker)
|
|
18
|
+
// f64 emulation of ssyr2 — x, y, A, and alpha are each split into an f32
|
|
19
|
+
// (hi, lo) pair; WGSL has no f64 type.
|
|
20
|
+
export async function dsyr2(
|
|
21
|
+
device,
|
|
22
|
+
uplo,
|
|
23
|
+
n,
|
|
24
|
+
alpha,
|
|
25
|
+
x,
|
|
26
|
+
incx,
|
|
27
|
+
y,
|
|
28
|
+
incy,
|
|
29
|
+
A,
|
|
30
|
+
lda,
|
|
31
|
+
layout = "row-major",
|
|
32
|
+
) {
|
|
33
|
+
const xIsGpu = x instanceof GpuVector;
|
|
34
|
+
const yIsGpu = y instanceof GpuVector;
|
|
35
|
+
const AIsGpu = A instanceof GpuMatrix;
|
|
36
|
+
|
|
37
|
+
requireGpuDevice(device);
|
|
38
|
+
requireSameDevice(device, "dsyr2", { A, x, y });
|
|
39
|
+
if (uplo !== "lower" && uplo !== "upper")
|
|
40
|
+
throw new Error("uplo must be 'lower' or 'upper'.");
|
|
41
|
+
if (layout !== "row-major" && layout !== "column-major")
|
|
42
|
+
throw new Error("layout must be 'row-major' or 'column-major'.");
|
|
43
|
+
if (
|
|
44
|
+
!Number.isInteger(n) ||
|
|
45
|
+
!Number.isInteger(incx) ||
|
|
46
|
+
!Number.isInteger(incy) ||
|
|
47
|
+
!Number.isInteger(lda)
|
|
48
|
+
)
|
|
49
|
+
throw new Error("n, incx, incy, and lda must be integers.");
|
|
50
|
+
if (typeof alpha !== "number") throw new Error("alpha must be a number.");
|
|
51
|
+
if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
|
|
52
|
+
if (!Number.isFinite(alpha)) throw new Error("alpha must be finite.");
|
|
53
|
+
if (incx <= 0 || incy <= 0)
|
|
54
|
+
throw new Error("incx and incy must be positive.");
|
|
55
|
+
if (lda < n) throw new Error("lda must be >= n.");
|
|
56
|
+
if (!AIsGpu && !(A instanceof Float64Array))
|
|
57
|
+
throw new Error("A must be a Float64Array or GpuMatrix.");
|
|
58
|
+
if (AIsGpu && A.dtype !== Float64Array)
|
|
59
|
+
throw new Error("A must be a Float64Array-backed GpuMatrix.");
|
|
60
|
+
if (!xIsGpu && !(x instanceof Float64Array))
|
|
61
|
+
throw new Error("x must be a Float64Array or GpuVector.");
|
|
62
|
+
if (!yIsGpu && !(y instanceof Float64Array))
|
|
63
|
+
throw new Error("y must be a Float64Array or GpuVector.");
|
|
64
|
+
if (xIsGpu && x.dtype !== Float64Array)
|
|
65
|
+
throw new Error("x must be a Float64Array-backed GpuVector.");
|
|
66
|
+
if (yIsGpu && y.dtype !== Float64Array)
|
|
67
|
+
throw new Error("y must be a Float64Array-backed GpuVector.");
|
|
68
|
+
if (xIsGpu !== yIsGpu)
|
|
69
|
+
throw new Error(
|
|
70
|
+
"x and y must be the same type (both Float64Array or both GpuVector).",
|
|
71
|
+
);
|
|
72
|
+
if (xIsGpu && !AIsGpu)
|
|
73
|
+
throw new Error("A must be a GpuMatrix when x and y are GpuVectors.");
|
|
74
|
+
if (AIsGpu && !xIsGpu)
|
|
75
|
+
throw new Error("x and y must be GpuVectors when A is a GpuMatrix.");
|
|
76
|
+
if (AIsGpu && xIsGpu && A._buf === x._buf)
|
|
77
|
+
throw new Error("A and x must not reference the same GPU buffer.");
|
|
78
|
+
if (AIsGpu && yIsGpu && A._buf === y._buf)
|
|
79
|
+
throw new Error("A and y must not reference the same GPU buffer.");
|
|
80
|
+
if (xIsGpu && x._buf === y._buf)
|
|
81
|
+
throw new Error(
|
|
82
|
+
"x and y must not reference the same GPU buffer when both are GpuVectors.",
|
|
83
|
+
);
|
|
84
|
+
if (AIsGpu && lda !== A.lda)
|
|
85
|
+
throw new Error("lda must match A.lda when A is a GpuMatrix.");
|
|
86
|
+
if (AIsGpu && (A.rows < n || A.cols < n))
|
|
87
|
+
throw new Error("A is too small for the given n.");
|
|
88
|
+
if (n < 0) throw new Error("n must be non-negative.");
|
|
89
|
+
if (n === 0) return AIsGpu ? {} : { A };
|
|
90
|
+
|
|
91
|
+
if (!AIsGpu && A.length < (n - 1) * lda + n)
|
|
92
|
+
throw new Error("A does not have enough elements for the given n and lda.");
|
|
93
|
+
if (x.length < (n - 1) * incx + 1)
|
|
94
|
+
throw new Error(
|
|
95
|
+
"x does not have enough elements for the given n and incx.",
|
|
96
|
+
);
|
|
97
|
+
if (y.length < (n - 1) * incy + 1)
|
|
98
|
+
throw new Error(
|
|
99
|
+
"y does not have enough elements for the given n and incy.",
|
|
100
|
+
);
|
|
101
|
+
|
|
102
|
+
// GpuMatrix's own layout wins over the argument; A is symmetric, so column-major A reinterpreted row-major just flips which triangle is stored — flip uplo to match (no x/y swap needed — x*y^T+y*x^T is already symmetric under swapping them).
|
|
103
|
+
const effLayout = AIsGpu ? A.layout : layout;
|
|
104
|
+
const isLower =
|
|
105
|
+
effLayout === "column-major" ? uplo === "upper" : uplo === "lower";
|
|
106
|
+
|
|
107
|
+
const f64Deps = ["f64/dekker", "f64/utils/add", "f64/utils/multiply"];
|
|
108
|
+
const pipeline = await getPipeline(
|
|
109
|
+
device,
|
|
110
|
+
[...f64Deps, "dsyr2"],
|
|
111
|
+
"dsyr2_main",
|
|
112
|
+
);
|
|
113
|
+
|
|
114
|
+
const { hi: alphaHi, lo: alphaLo } = splitDoubleDouble(
|
|
115
|
+
new Float64Array([alpha]),
|
|
116
|
+
);
|
|
117
|
+
|
|
118
|
+
let xHiBuffer = null;
|
|
119
|
+
let xLoBuffer = null;
|
|
120
|
+
let yHiBuffer = null;
|
|
121
|
+
let yLoBuffer = null;
|
|
122
|
+
let AHiBuffer = null;
|
|
123
|
+
let ALoBuffer = null;
|
|
124
|
+
let paramsBuffer = null;
|
|
125
|
+
let readHiBuffer = null;
|
|
126
|
+
let readLoBuffer = null;
|
|
127
|
+
|
|
128
|
+
try {
|
|
129
|
+
if (xIsGpu) {
|
|
130
|
+
xHiBuffer = x._buf;
|
|
131
|
+
xLoBuffer = x._loBuf;
|
|
132
|
+
yHiBuffer = y._buf;
|
|
133
|
+
yLoBuffer = y._loBuf;
|
|
134
|
+
AHiBuffer = A._buf;
|
|
135
|
+
ALoBuffer = A._loBuf;
|
|
136
|
+
} else {
|
|
137
|
+
const xSplit = splitDoubleDouble(x);
|
|
138
|
+
const ySplit = splitDoubleDouble(y);
|
|
139
|
+
const ASplit = splitDoubleDouble(A);
|
|
140
|
+
xHiBuffer = uploadBuffer(device, xSplit.hi, "dsyr2-xHi", false);
|
|
141
|
+
xLoBuffer = uploadBuffer(device, xSplit.lo, "dsyr2-xLo", false);
|
|
142
|
+
yHiBuffer = uploadBuffer(device, ySplit.hi, "dsyr2-yHi", false);
|
|
143
|
+
yLoBuffer = uploadBuffer(device, ySplit.lo, "dsyr2-yLo", false);
|
|
144
|
+
AHiBuffer = uploadBuffer(device, ASplit.hi, "dsyr2-AHi", true);
|
|
145
|
+
ALoBuffer = uploadBuffer(device, ASplit.lo, "dsyr2-ALo", true);
|
|
146
|
+
}
|
|
147
|
+
paramsBuffer = createParamsBuffer(
|
|
148
|
+
device,
|
|
149
|
+
[
|
|
150
|
+
{ value: n, type: "u32" },
|
|
151
|
+
{ value: alphaHi[0], type: "f32" },
|
|
152
|
+
{ value: alphaLo[0], type: "f32" },
|
|
153
|
+
{ value: incx, type: "u32" },
|
|
154
|
+
{ value: incy, type: "u32" },
|
|
155
|
+
{ value: lda, type: "u32" },
|
|
156
|
+
{ value: isLower ? 0 : 1, type: "u32" },
|
|
157
|
+
],
|
|
158
|
+
"dsyr2-params",
|
|
159
|
+
);
|
|
160
|
+
|
|
161
|
+
const bindGroup = createBindGroup(device, pipeline.getBindGroupLayout(0), [
|
|
162
|
+
xHiBuffer,
|
|
163
|
+
xLoBuffer,
|
|
164
|
+
yHiBuffer,
|
|
165
|
+
yLoBuffer,
|
|
166
|
+
AHiBuffer,
|
|
167
|
+
ALoBuffer,
|
|
168
|
+
paramsBuffer,
|
|
169
|
+
]);
|
|
170
|
+
|
|
171
|
+
// One workgroup per row of A; clamped to device limit — the shader's
|
|
172
|
+
// grid-stride loop handles remaining rows when n > dispatch count.
|
|
173
|
+
const wgCount = Math.min(n, device.limits.maxComputeWorkgroupsPerDimension);
|
|
174
|
+
const { commandEncoder, ts } = runComputePass(
|
|
175
|
+
device,
|
|
176
|
+
pipeline,
|
|
177
|
+
bindGroup,
|
|
178
|
+
wgCount,
|
|
179
|
+
);
|
|
180
|
+
readHiBuffer = AIsGpu
|
|
181
|
+
? null
|
|
182
|
+
: stageReadback(device, commandEncoder, AHiBuffer);
|
|
183
|
+
readLoBuffer = AIsGpu
|
|
184
|
+
? null
|
|
185
|
+
: stageReadback(device, commandEncoder, ALoBuffer);
|
|
186
|
+
|
|
187
|
+
submit(device, commandEncoder);
|
|
188
|
+
|
|
189
|
+
const gpuTimeMs = await extractTimestamp(ts);
|
|
190
|
+
|
|
191
|
+
if (AIsGpu) {
|
|
192
|
+
if (gpuTimeMs !== undefined) return { gpuTimeMs };
|
|
193
|
+
return {};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const hi = await extractResult(readHiBuffer, Float32Array);
|
|
197
|
+
readHiBuffer = null;
|
|
198
|
+
const lo = await extractResult(readLoBuffer, Float32Array);
|
|
199
|
+
readLoBuffer = null;
|
|
200
|
+
const result = mergeDoubleDouble(hi, lo);
|
|
201
|
+
if (gpuTimeMs !== undefined) return { A: result, gpuTimeMs };
|
|
202
|
+
return { A: result };
|
|
203
|
+
} finally {
|
|
204
|
+
if (!xIsGpu && xHiBuffer) destroyBuffers(xHiBuffer);
|
|
205
|
+
if (!xIsGpu && xLoBuffer) destroyBuffers(xLoBuffer);
|
|
206
|
+
if (!yIsGpu && yHiBuffer) destroyBuffers(yHiBuffer);
|
|
207
|
+
if (!yIsGpu && yLoBuffer) destroyBuffers(yLoBuffer);
|
|
208
|
+
if (!AIsGpu && AHiBuffer) destroyBuffers(AHiBuffer);
|
|
209
|
+
if (!AIsGpu && ALoBuffer) destroyBuffers(ALoBuffer);
|
|
210
|
+
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
211
|
+
if (readHiBuffer) destroyBuffers(readHiBuffer);
|
|
212
|
+
if (readLoBuffer) destroyBuffers(readLoBuffer);
|
|
213
|
+
}
|
|
214
|
+
}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
2
|
+
import { GpuMatrix } from "../classes/GpuMatrix.mjs";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Performs the triangular matrix-vector operation $$y \leftarrow \mathrm{op}(A) x$$
|
|
6
|
+
* in double precision (double-double emulation — WGSL has no native f64 type).
|
|
7
|
+
*
|
|
8
|
+
* A is an n×n triangular matrix stored in row-major order. Only the triangle
|
|
9
|
+
* specified by `uplo` is referenced; the other triangle is not accessed.
|
|
10
|
+
*
|
|
11
|
+
* {@includeCode ../../examples/dtrmv/dtrmv.js}
|
|
12
|
+
*
|
|
13
|
+
* **Browser (standalone HTML):**
|
|
14
|
+
* {@includeCode ../../examples/dtrmv/web/dtrmv.html}
|
|
15
|
+
*
|
|
16
|
+
* @param device - GPUDevice from `init()`
|
|
17
|
+
* @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
|
|
18
|
+
* @param trans - `'no-transpose'` for A, `'transpose'` for A^T
|
|
19
|
+
* @param diag - `'unit'` to treat the diagonal as all-ones (A's diagonal is not read), `'non-unit'` to read it
|
|
20
|
+
* @param n - order of the matrix A (number of rows and columns)
|
|
21
|
+
* @param A - Float64Array, row-major or column-major (see `layout`), at least (n-1)*lda+n elements
|
|
22
|
+
* @param lda - leading dimension of A (>= n either way — A is square)
|
|
23
|
+
* @param x - Float64Array input vector, length at least (n-1)*incx+1
|
|
24
|
+
* @param incx - stride for x (must be a positive integer)
|
|
25
|
+
* @param y - Float64Array output vector, length at least (n-1)*incy+1
|
|
26
|
+
* @param incy - stride for y (must be a positive integer)
|
|
27
|
+
* @param layout - storage layout of `A` (default: `'row-major'`); column-major
|
|
28
|
+
* flips both the stored triangle and the effective `trans` (op(A) stays
|
|
29
|
+
* what you asked for either way)
|
|
30
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/dtrmv/dtrmv.mjs#L18">Source code: dtrmv.mjs (L18)</a>
|
|
31
|
+
* @category BLAS Level 2
|
|
32
|
+
*/
|
|
33
|
+
export declare function dtrmv(
|
|
34
|
+
device: GPUDevice,
|
|
35
|
+
uplo: 'lower' | 'upper',
|
|
36
|
+
trans: 'no-transpose' | 'transpose',
|
|
37
|
+
diag: 'unit' | 'non-unit',
|
|
38
|
+
n: number,
|
|
39
|
+
A: Float64Array,
|
|
40
|
+
lda: number,
|
|
41
|
+
x: Float64Array,
|
|
42
|
+
incx: number,
|
|
43
|
+
y: Float64Array,
|
|
44
|
+
incy: number,
|
|
45
|
+
layout?: 'row-major' | 'column-major',
|
|
46
|
+
): Promise<{ y: Float64Array; gpuTimeMs?: number }>;
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Performs the triangular matrix-vector operation $$y \leftarrow \mathrm{op}(A) x$$
|
|
50
|
+
* in double precision (double-double emulation).
|
|
51
|
+
*
|
|
52
|
+
* x and y are kept resident on the GPU. A must be a GpuMatrix (Float64Array-
|
|
53
|
+
* backed); its own `layout` (set at `GpuMatrix.from` time) determines the
|
|
54
|
+
* operation — there is no separate `layout` argument here.
|
|
55
|
+
*
|
|
56
|
+
* {@includeCode ../../examples/dtrmv/gpu.dtrmv.js}
|
|
57
|
+
*
|
|
58
|
+
* @param device - GPUDevice from `init()`
|
|
59
|
+
* @param uplo - `'lower'` to use the lower triangle, `'upper'` to use the upper triangle
|
|
60
|
+
* @param trans - `'no-transpose'` for A, `'transpose'` for A^T
|
|
61
|
+
* @param diag - `'unit'` to treat the diagonal as all-ones (A's diagonal is not read), `'non-unit'` to read it
|
|
62
|
+
* @param n - order of the matrix A
|
|
63
|
+
* @param A - GpuMatrix (Float64Array-backed), GPU-resident
|
|
64
|
+
* @param lda - leading dimension of A (must equal A.lda)
|
|
65
|
+
* @param x - GpuVector input vector (Float64Array-backed, not mutated)
|
|
66
|
+
* @param incx - stride for x (must be a positive integer)
|
|
67
|
+
* @param y - GpuVector output vector (Float64Array-backed, mutated in place)
|
|
68
|
+
* @param incy - stride for y (must be a positive integer)
|
|
69
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/dtrmv/dtrmv.mjs#L18">Source code: dtrmv.mjs (L18)</a>
|
|
70
|
+
* @category BLAS Level 2
|
|
71
|
+
*/
|
|
72
|
+
export declare function dtrmv(
|
|
73
|
+
device: GPUDevice,
|
|
74
|
+
uplo: 'lower' | 'upper',
|
|
75
|
+
trans: 'no-transpose' | 'transpose',
|
|
76
|
+
diag: 'unit' | 'non-unit',
|
|
77
|
+
n: number,
|
|
78
|
+
A: GpuMatrix,
|
|
79
|
+
lda: number,
|
|
80
|
+
x: GpuVector,
|
|
81
|
+
incx: number,
|
|
82
|
+
y: GpuVector,
|
|
83
|
+
incy: number,
|
|
84
|
+
): Promise<{ gpuTimeMs?: number }>;
|