wgblas 2.0.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -18
- package/dist/wgblas.browser.js +2172 -1174
- package/index.d.mts +49 -44
- package/index.mjs +11 -0
- package/package.json +133 -63
- package/src/classes/Complex32.d.mts +43 -0
- package/src/classes/Complex32.mjs +82 -0
- package/src/classes/Complex64.d.mts +44 -0
- package/src/classes/Complex64.mjs +76 -0
- package/src/classes/GpuMatrix.d.mts +41 -41
- package/src/classes/GpuMatrix.mjs +126 -17
- package/src/classes/GpuVector.d.mts +36 -40
- package/src/classes/GpuVector.mjs +66 -11
- package/src/cscal/cscal.d.mts +47 -0
- package/src/cscal/cscal.mjs +98 -0
- package/src/dasum/dasum.d.mts +4 -4
- package/src/dasum/dasum.mjs +38 -20
- package/src/daxpy/daxpy.d.mts +56 -0
- package/src/daxpy/daxpy.mjs +150 -0
- package/src/dcopy/dcopy.d.mts +52 -0
- package/src/dcopy/dcopy.mjs +140 -0
- package/src/ddot/ddot.d.mts +62 -0
- package/src/ddot/ddot.mjs +184 -0
- package/src/devdocs.mjs +13 -0
- package/src/dnrm2/dnrm2.d.mts +50 -0
- package/src/dnrm2/dnrm2.mjs +189 -0
- package/src/drot/drot.d.mts +67 -0
- package/src/drot/drot.mjs +170 -0
- package/src/drotm/drotm.d.mts +67 -0
- package/src/drotm/drotm.mjs +171 -0
- package/src/dscal/dscal.d.mts +52 -0
- package/src/dscal/dscal.mjs +119 -0
- package/src/dswap/dswap.d.mts +57 -0
- package/src/dswap/dswap.mjs +155 -0
- package/src/idamax/idamax.d.mts +20 -2
- package/src/idamax/idamax.mjs +56 -24
- package/src/init.mjs +117 -56
- package/src/isamax/isamax.d.mts +20 -2
- package/src/isamax/isamax.mjs +21 -16
- package/src/random/random.d.mts +37 -39
- package/src/random/random.mjs +39 -7
- package/src/sasum/sasum.d.mts +2 -2
- package/src/sasum/sasum.mjs +20 -16
- package/src/saxpy/saxpy.d.mts +2 -2
- package/src/saxpy/saxpy.mjs +14 -11
- package/src/scopy/scopy.d.mts +2 -2
- package/src/scopy/scopy.mjs +13 -9
- package/src/sdot/sdot.d.mts +2 -2
- package/src/sdot/sdot.mjs +21 -17
- package/src/sgemm/sgemm.d.mts +2 -2
- package/src/sgemm/sgemm.mjs +109 -40
- package/src/sgemmtr/sgemmtr.d.mts +3 -2
- package/src/sgemmtr/sgemmtr.mjs +98 -40
- package/src/sgemv/sgemv.d.mts +2 -2
- package/src/sgemv/sgemv.mjs +69 -41
- package/src/sger/sger.d.mts +2 -2
- package/src/sger/sger.mjs +43 -19
- package/src/shaders/__test_pipeline_a.wgsl +3 -0
- package/src/shaders/__test_pipeline_b.wgsl +2 -0
- package/src/shaders/cscal.wgsl +33 -0
- package/src/shaders/daxpy.wgsl +66 -0
- package/src/shaders/dcopy.wgsl +34 -0
- package/src/shaders/ddot.wgsl +106 -0
- package/src/shaders/dnrm2.wgsl +167 -0
- package/src/shaders/drot.wgsl +81 -0
- package/src/shaders/drotm.wgsl +99 -0
- package/src/shaders/dscal.wgsl +60 -0
- package/src/shaders/dswap.wgsl +38 -0
- package/src/shaders/f64/utils/add.wgsl +6 -0
- package/src/shaders/f64/utils/divide.wgsl +45 -0
- package/src/shaders/f64/utils/multiply.wgsl +19 -10
- package/src/shaders/f64/utils/sqrt.wgsl +45 -0
- package/src/shaders/index.mjs +233 -14
- package/src/shaders/reduction/scaledSum.wgsl +65 -0
- package/src/shaders/reduction/scaledSumF64.wgsl +93 -0
- package/src/shaders/sgemm_large.wgsl +107 -18
- package/src/shaders/sgemm_small.wgsl +115 -15
- package/src/shaders/sgemmtr_large.wgsl +4 -1
- package/src/shaders/sgemmtr_small.wgsl +4 -1
- package/src/shaders/sgemv_n.wgsl +3 -1
- package/src/shaders/sgemv_t.wgsl +3 -1
- package/src/shaders/snrm2.wgsl +72 -23
- package/src/shaders/ssymv.wgsl +3 -1
- package/src/snrm2/snrm2.d.mts +2 -2
- package/src/snrm2/snrm2.mjs +41 -23
- package/src/srot/srot.d.mts +2 -4
- package/src/srot/srot.mjs +16 -11
- package/src/srotm/srotm.d.mts +2 -4
- package/src/srotm/srotm.mjs +17 -11
- package/src/sscal/sscal.d.mts +3 -3
- package/src/sscal/sscal.mjs +14 -12
- package/src/sswap/sswap.d.mts +2 -2
- package/src/sswap/sswap.mjs +18 -10
- package/src/ssymm/ssymm.d.mts +5 -4
- package/src/ssymm/ssymm.mjs +150 -54
- package/src/ssymv/ssymv.d.mts +2 -2
- package/src/ssymv/ssymv.mjs +47 -26
- package/src/ssyr/ssyr.d.mts +2 -2
- package/src/ssyr/ssyr.mjs +38 -17
- package/src/ssyr2/ssyr2.d.mts +2 -2
- package/src/ssyr2/ssyr2.mjs +48 -21
- package/src/ssyr2k/ssyr2k.d.mts +3 -2
- package/src/ssyr2k/ssyr2k.mjs +140 -62
- package/src/ssyrk/ssyrk.d.mts +3 -2
- package/src/ssyrk/ssyrk.mjs +91 -39
- package/src/strmm/strmm.d.mts +5 -4
- package/src/strmm/strmm.mjs +174 -60
- package/src/strmv/strmv.d.mts +2 -2
- package/src/strmv/strmv.mjs +42 -20
- package/src/strsm/strsm.d.mts +6 -4
- package/src/strsm/strsm.mjs +438 -174
- package/src/strsv/strsv.d.mts +5 -3
- package/src/strsv/strsv.mjs +89 -34
- package/src/util/benchmark.mjs +9 -9
- package/src/util/bindgroup.mjs +1 -3
- package/src/util/buffer.mjs +139 -24
- package/src/util/complex.mjs +87 -0
- package/src/util/compute.mjs +19 -16
- package/src/util/constants.mjs +57 -0
- package/src/util/device.mjs +49 -0
- package/src/util/pipeline.mjs +44 -10
- package/src/util/workgroup.mjs +72 -7
- package/src/shaders/browser-shaders.mjs +0 -81
- package/src/shaders/f64add.wgsl +0 -281
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
2
|
+
import { Complex32, Complex32Array } from "../classes/Complex32.mjs";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Scales a complex vector by a complex constant: $$x \leftarrow \alpha x$$
|
|
6
|
+
*
|
|
7
|
+
* {@includeCode ../../examples/cscal/cscal.js}
|
|
8
|
+
*
|
|
9
|
+
* **Browser (standalone HTML):**
|
|
10
|
+
* {@includeCode ../../examples/cscal/web/cscal.html}
|
|
11
|
+
*
|
|
12
|
+
* @param device - GPUDevice from `init()`
|
|
13
|
+
* @param n - number of elements to scale (must be a positive integer)
|
|
14
|
+
* @param alpha - complex scalar multiplier
|
|
15
|
+
* @param x - Complex32Array input/output vector
|
|
16
|
+
* @param incx - stride for x (must be a positive integer)
|
|
17
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/cscal/cscal.mjs">Source code: cscal.mjs</a>
|
|
18
|
+
* @category BLAS Level 1
|
|
19
|
+
*/
|
|
20
|
+
export declare function cscal(
|
|
21
|
+
device: GPUDevice,
|
|
22
|
+
n: number,
|
|
23
|
+
alpha: Complex32,
|
|
24
|
+
x: Complex32Array,
|
|
25
|
+
incx: number,
|
|
26
|
+
): Promise<{ x: Complex32Array } | { x: Complex32Array; gpuTimeMs: number }>;
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Scales a complex vector by a complex constant: $$x \leftarrow \alpha x$$
|
|
30
|
+
*
|
|
31
|
+
* {@includeCode ../../examples/cscal/gpu.cscal.js}
|
|
32
|
+
*
|
|
33
|
+
* @param device - GPUDevice from `init()`
|
|
34
|
+
* @param n - number of elements to scale (must be a positive integer)
|
|
35
|
+
* @param alpha - complex scalar multiplier
|
|
36
|
+
* @param x - Complex32Array-backed GpuVector input/output vector (mutated in place)
|
|
37
|
+
* @param incx - stride for x (must be a positive integer)
|
|
38
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/cscal/cscal.mjs">Source code: cscal.mjs</a>
|
|
39
|
+
* @category BLAS Level 1
|
|
40
|
+
*/
|
|
41
|
+
export declare function cscal(
|
|
42
|
+
device: GPUDevice,
|
|
43
|
+
n: number,
|
|
44
|
+
alpha: Complex32,
|
|
45
|
+
x: GpuVector,
|
|
46
|
+
incx: number,
|
|
47
|
+
): Promise<{} | { gpuTimeMs: number }>;
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import {
|
|
2
|
+
uploadBuffer,
|
|
3
|
+
createParamsBuffer,
|
|
4
|
+
stageReadback,
|
|
5
|
+
destroyBuffers,
|
|
6
|
+
} from "../util/buffer.mjs";
|
|
7
|
+
import { createBindGroup } from "../util/bindgroup.mjs";
|
|
8
|
+
import { runComputePass, submit } from "../util/compute.mjs";
|
|
9
|
+
import { extractResult } from "../util/result.mjs";
|
|
10
|
+
import { extractTimestamp } from "../util/benchmark.mjs";
|
|
11
|
+
import { getPipeline } from "../util/pipeline.mjs";
|
|
12
|
+
import { calcWorkgroups } from "../util/workgroup.mjs";
|
|
13
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
14
|
+
import { Complex32, Complex32Array } from "../classes/Complex32.mjs";
|
|
15
|
+
import { interleaveComplex32 } from "../util/complex.mjs";
|
|
16
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
17
|
+
|
|
18
|
+
// cscal: x := alpha * x, complex. alpha is a Complex32; x is a Complex32Array
|
|
19
|
+
// or a Complex32Array-backed GpuVector, interleaved [re0, im0, re1, im1, ...]
|
|
20
|
+
// to match cscal.wgsl's single buffer.
|
|
21
|
+
export async function cscal(device, n, alpha, x, incx) {
|
|
22
|
+
const xIsGpu = x instanceof GpuVector;
|
|
23
|
+
|
|
24
|
+
requireGpuDevice(device);
|
|
25
|
+
requireSameDevice(device, "cscal", { x });
|
|
26
|
+
if (!Number.isInteger(n) || !Number.isInteger(incx))
|
|
27
|
+
throw new Error("n and incx must be integers.");
|
|
28
|
+
if (!(alpha instanceof Complex32))
|
|
29
|
+
throw new Error("alpha must be a Complex32.");
|
|
30
|
+
if (Number.isNaN(alpha.re) || Number.isNaN(alpha.im))
|
|
31
|
+
throw new Error("alpha must not be NaN.");
|
|
32
|
+
if (!Number.isFinite(alpha.re) || !Number.isFinite(alpha.im))
|
|
33
|
+
throw new Error("alpha must be finite.");
|
|
34
|
+
if (incx <= 0) throw new Error("incx must be positive.");
|
|
35
|
+
if (!(x instanceof Complex32Array) && !xIsGpu)
|
|
36
|
+
throw new Error("x must be a Complex32Array or GpuVector.");
|
|
37
|
+
if (xIsGpu && x.dtype !== Complex32Array)
|
|
38
|
+
throw new Error("x must be a Complex32Array-backed GpuVector.");
|
|
39
|
+
if (n <= 0) return xIsGpu ? {} : { x };
|
|
40
|
+
if (x.length < (n - 1) * incx + 1)
|
|
41
|
+
throw new Error(
|
|
42
|
+
"x does not have enough elements for the given n and incx.",
|
|
43
|
+
);
|
|
44
|
+
|
|
45
|
+
const pipeline = await getPipeline(device, "cscal");
|
|
46
|
+
|
|
47
|
+
let xBuffer = null;
|
|
48
|
+
let paramsBuffer = null;
|
|
49
|
+
let readBuffer = null;
|
|
50
|
+
|
|
51
|
+
try {
|
|
52
|
+
xBuffer = xIsGpu
|
|
53
|
+
? x._buf
|
|
54
|
+
: uploadBuffer(device, interleaveComplex32(x), "cscal-x", true);
|
|
55
|
+
paramsBuffer = createParamsBuffer(
|
|
56
|
+
device,
|
|
57
|
+
[
|
|
58
|
+
{ value: n, type: "u32" },
|
|
59
|
+
{ value: alpha.re, type: "f32" },
|
|
60
|
+
{ value: alpha.im, type: "f32" },
|
|
61
|
+
{ value: incx, type: "u32" },
|
|
62
|
+
],
|
|
63
|
+
"cscal-params",
|
|
64
|
+
);
|
|
65
|
+
|
|
66
|
+
const bindGroup = createBindGroup(device, pipeline.getBindGroupLayout(0), [
|
|
67
|
+
xBuffer,
|
|
68
|
+
paramsBuffer,
|
|
69
|
+
]);
|
|
70
|
+
const { commandEncoder, ts } = runComputePass(
|
|
71
|
+
device,
|
|
72
|
+
pipeline,
|
|
73
|
+
bindGroup,
|
|
74
|
+
calcWorkgroups(device, n),
|
|
75
|
+
);
|
|
76
|
+
readBuffer = xIsGpu ? null : stageReadback(device, commandEncoder, xBuffer);
|
|
77
|
+
|
|
78
|
+
submit(device, commandEncoder);
|
|
79
|
+
|
|
80
|
+
const gpuTimeMs = await extractTimestamp(ts);
|
|
81
|
+
|
|
82
|
+
if (xIsGpu) {
|
|
83
|
+
if (gpuTimeMs !== undefined) return { gpuTimeMs };
|
|
84
|
+
return {};
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const flat = await extractResult(readBuffer, Float32Array);
|
|
88
|
+
readBuffer = null; // extractResult already destroyed it
|
|
89
|
+
const result = new Complex32Array(flat);
|
|
90
|
+
if (gpuTimeMs !== undefined) return { x: result, gpuTimeMs };
|
|
91
|
+
return { x: result };
|
|
92
|
+
} finally {
|
|
93
|
+
if (!xIsGpu && xBuffer) destroyBuffers(xBuffer);
|
|
94
|
+
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
95
|
+
// Only reached if extractTimestamp threw before extractResult ran.
|
|
96
|
+
if (readBuffer) destroyBuffers(readBuffer);
|
|
97
|
+
}
|
|
98
|
+
}
|
package/src/dasum/dasum.d.mts
CHANGED
|
@@ -2,9 +2,9 @@ import { GpuVector } from "../classes/GpuVector.mjs";
|
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* Computes the sum of absolute values of a vector of doubles in extended
|
|
5
|
-
* precision: result =
|
|
6
|
-
* then is split into a (hi, lo) double-double f32 pair
|
|
7
|
-
* `splitDoubleDouble`/`f64.mjs`) since WGSL has no f64 type; accumulation
|
|
5
|
+
* precision: $$\text{result} = \sum_{i} |x_i|$$
|
|
6
|
+
* Each element of `x` has abs() applied, then is split into a (hi, lo) double-double f32 pair
|
|
7
|
+
* (see `splitDoubleDouble`/`f64.mjs`) since WGSL has no f64 type; accumulation
|
|
8
8
|
* uses Dekker's double-double algorithm (see `shaders/f64/`), giving ~48 bits
|
|
9
9
|
* of mantissa — more than a single f32 (24 bits) but less than true f64
|
|
10
10
|
* (52 bits), so results are not bit-exact with a CPU double.
|
|
@@ -31,7 +31,7 @@ export declare function dasum(
|
|
|
31
31
|
|
|
32
32
|
/**
|
|
33
33
|
* Computes the sum of absolute values of a vector of doubles in double
|
|
34
|
-
* precision: result =
|
|
34
|
+
* precision: $$\text{result} = \sum_{i} |x_i|$$.
|
|
35
35
|
*
|
|
36
36
|
* {@includeCode ../../examples/dasum/gpu.dasum.js}
|
|
37
37
|
*
|
package/src/dasum/dasum.mjs
CHANGED
|
@@ -13,14 +13,14 @@ import { extractResult } from "../util/result.mjs";
|
|
|
13
13
|
import { getPipeline } from "../util/pipeline.mjs";
|
|
14
14
|
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
15
15
|
import { splitDoubleDouble, mergeDoubleDouble } from "../util/f64.mjs";
|
|
16
|
-
|
|
17
|
-
|
|
16
|
+
import { WGS } from "../util/constants.mjs";
|
|
17
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
18
18
|
|
|
19
19
|
export async function dasum(device, n, x, incx) {
|
|
20
20
|
const xIsGpu = x instanceof GpuVector;
|
|
21
21
|
|
|
22
|
-
|
|
23
|
-
|
|
22
|
+
requireGpuDevice(device);
|
|
23
|
+
requireSameDevice(device, "dasum", { x });
|
|
24
24
|
if (!Number.isInteger(n) || !Number.isInteger(incx))
|
|
25
25
|
throw new Error("n and incx must be integers.");
|
|
26
26
|
if (incx <= 0) throw new Error("incx must be positive.");
|
|
@@ -39,7 +39,10 @@ export async function dasum(device, n, x, incx) {
|
|
|
39
39
|
// #include; entryPoint omitted since each module has only one @compute.
|
|
40
40
|
const f64Deps = ["f64/dekker", "f64/utils/abs", "f64/utils/add"];
|
|
41
41
|
const pipelineMain = await getPipeline(device, [...f64Deps, "dasum"]);
|
|
42
|
-
const pipelineReduce = await getPipeline(device, [
|
|
42
|
+
const pipelineReduce = await getPipeline(device, [
|
|
43
|
+
...f64Deps,
|
|
44
|
+
"reduction/sumF64",
|
|
45
|
+
]);
|
|
43
46
|
|
|
44
47
|
let xHiBuffer = null;
|
|
45
48
|
let xLoBuffer = null;
|
|
@@ -57,14 +60,23 @@ export async function dasum(device, n, x, incx) {
|
|
|
57
60
|
xLoBuffer = x._loBuf;
|
|
58
61
|
} else {
|
|
59
62
|
const { hi, lo } = splitDoubleDouble(x.map(Math.abs));
|
|
60
|
-
xHiBuffer = uploadBuffer(hi, "dasum-xHi", false);
|
|
61
|
-
xLoBuffer = uploadBuffer(lo, "dasum-xLo", false);
|
|
63
|
+
xHiBuffer = uploadBuffer(device, hi, "dasum-xHi", false);
|
|
64
|
+
xLoBuffer = uploadBuffer(device, lo, "dasum-xLo", false);
|
|
62
65
|
}
|
|
63
|
-
partialsHiBuffer = createStorageBuffer(
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
66
|
+
partialsHiBuffer = createStorageBuffer(
|
|
67
|
+
device,
|
|
68
|
+
2 * WGS * 4,
|
|
69
|
+
"dasum-partialsHi",
|
|
70
|
+
);
|
|
71
|
+
partialsLoBuffer = createStorageBuffer(
|
|
72
|
+
device,
|
|
73
|
+
2 * WGS * 4,
|
|
74
|
+
"dasum-partialsLo",
|
|
75
|
+
);
|
|
76
|
+
resultHiBuffer = createResultBuffer(device, 4, "dasum-result-hi");
|
|
77
|
+
resultLoBuffer = createResultBuffer(device, 4, "dasum-result-lo");
|
|
67
78
|
paramsBuffer = createParamsBuffer(
|
|
79
|
+
device,
|
|
68
80
|
[
|
|
69
81
|
{ value: n, type: "u32" },
|
|
70
82
|
{ value: incx, type: "u32" },
|
|
@@ -72,31 +84,37 @@ export async function dasum(device, n, x, incx) {
|
|
|
72
84
|
"dasum-params",
|
|
73
85
|
);
|
|
74
86
|
|
|
75
|
-
const bgMain = createBindGroup(
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
87
|
+
const bgMain = createBindGroup(device, pipelineMain.getBindGroupLayout(0), [
|
|
88
|
+
xHiBuffer,
|
|
89
|
+
xLoBuffer,
|
|
90
|
+
partialsHiBuffer,
|
|
91
|
+
partialsLoBuffer,
|
|
92
|
+
paramsBuffer,
|
|
93
|
+
]);
|
|
79
94
|
const { commandEncoder: enc1, ts: ts1 } = runComputePass(
|
|
95
|
+
device,
|
|
80
96
|
pipelineMain,
|
|
81
97
|
bgMain,
|
|
82
98
|
2 * WGS,
|
|
83
99
|
); // dispatch 2*WGS workgroups
|
|
84
100
|
|
|
85
|
-
submit(enc1);
|
|
101
|
+
submit(device, enc1);
|
|
86
102
|
|
|
87
103
|
const bgReduce = createBindGroup(
|
|
104
|
+
device,
|
|
88
105
|
pipelineReduce.getBindGroupLayout(0),
|
|
89
106
|
[partialsHiBuffer, partialsLoBuffer, resultHiBuffer, resultLoBuffer],
|
|
90
107
|
);
|
|
91
108
|
const { commandEncoder: enc2, ts: ts2 } = runComputePass(
|
|
109
|
+
device,
|
|
92
110
|
pipelineReduce,
|
|
93
111
|
bgReduce,
|
|
94
112
|
1,
|
|
95
113
|
); // dispatch 1 workgroup to reduce the partial sums to a single result
|
|
96
|
-
readHiBuffer = stageReadback(enc2, resultHiBuffer);
|
|
97
|
-
readLoBuffer = stageReadback(enc2, resultLoBuffer);
|
|
114
|
+
readHiBuffer = stageReadback(device, enc2, resultHiBuffer);
|
|
115
|
+
readLoBuffer = stageReadback(device, enc2, resultLoBuffer);
|
|
98
116
|
|
|
99
|
-
submit(enc2);
|
|
117
|
+
submit(device, enc2);
|
|
100
118
|
|
|
101
119
|
const hiPromise = extractResult(readHiBuffer, Float32Array);
|
|
102
120
|
const loPromise = extractResult(readLoBuffer, Float32Array);
|
|
@@ -123,7 +141,7 @@ export async function dasum(device, n, x, incx) {
|
|
|
123
141
|
if (resultHiBuffer) destroyBuffers(resultHiBuffer);
|
|
124
142
|
if (resultLoBuffer) destroyBuffers(resultLoBuffer);
|
|
125
143
|
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
126
|
-
// Only reached if submit(enc2) threw before ownership was transferred above.
|
|
144
|
+
// Only reached if submit(device, enc2) threw before ownership was transferred above.
|
|
127
145
|
if (readHiBuffer) destroyBuffers(readHiBuffer);
|
|
128
146
|
if (readLoBuffer) destroyBuffers(readLoBuffer);
|
|
129
147
|
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Performs the operation $$y \leftarrow \alpha x + y$$ — double-double
|
|
5
|
+
* (Dekker) f64 emulation of {@link saxpy}, since WGSL has no native f64 type.
|
|
6
|
+
*
|
|
7
|
+
* {@includeCode ../../examples/daxpy/daxpy.js}
|
|
8
|
+
*
|
|
9
|
+
* **Browser (standalone HTML):**
|
|
10
|
+
* {@includeCode ../../examples/daxpy/web/daxpy.html}
|
|
11
|
+
*
|
|
12
|
+
* @param device - GPUDevice from `init()`
|
|
13
|
+
* @param n - number of elements (must be a positive integer)
|
|
14
|
+
* @param alpha - scalar multiplier
|
|
15
|
+
* @param x - Float64Array input vector
|
|
16
|
+
* @param incx - stride for x (must be a positive integer)
|
|
17
|
+
* @param y - Float64Array input/output vector
|
|
18
|
+
* @param incy - stride for y (must be a positive integer)
|
|
19
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/daxpy/daxpy.mjs#L18">Source code: daxpy.mjs (L18)</a>
|
|
20
|
+
* @category BLAS Level 1
|
|
21
|
+
*/
|
|
22
|
+
export declare function daxpy(
|
|
23
|
+
device: GPUDevice,
|
|
24
|
+
n: number,
|
|
25
|
+
alpha: number,
|
|
26
|
+
x: Float64Array,
|
|
27
|
+
incx: number,
|
|
28
|
+
y: Float64Array,
|
|
29
|
+
incy: number,
|
|
30
|
+
): Promise<{ y: Float64Array } | { y: Float64Array; gpuTimeMs: number }>;
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Performs the operation $$y \leftarrow \alpha x + y$$ — GPU-resident
|
|
34
|
+
* overload; see the Float64Array overload above for the routine itself.
|
|
35
|
+
*
|
|
36
|
+
* {@includeCode ../../examples/daxpy/gpu.daxpy.js}
|
|
37
|
+
*
|
|
38
|
+
* @param device - GPUDevice from `init()`
|
|
39
|
+
* @param n - number of elements (must be a positive integer)
|
|
40
|
+
* @param alpha - scalar multiplier
|
|
41
|
+
* @param x - GpuVector input vector (must be Float64Array-backed)
|
|
42
|
+
* @param incx - stride for x (must be a positive integer)
|
|
43
|
+
* @param y - GpuVector input/output vector (must be Float64Array-backed, mutated in place)
|
|
44
|
+
* @param incy - stride for y (must be a positive integer)
|
|
45
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/daxpy/daxpy.mjs#L18">Source code: daxpy.mjs (L18)</a>
|
|
46
|
+
* @category BLAS Level 1
|
|
47
|
+
*/
|
|
48
|
+
export declare function daxpy(
|
|
49
|
+
device: GPUDevice,
|
|
50
|
+
n: number,
|
|
51
|
+
alpha: number,
|
|
52
|
+
x: GpuVector,
|
|
53
|
+
incx: number,
|
|
54
|
+
y: GpuVector,
|
|
55
|
+
incy: number,
|
|
56
|
+
): Promise<{} | { gpuTimeMs: number }>;
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
import {
|
|
2
|
+
uploadBuffer,
|
|
3
|
+
createParamsBuffer,
|
|
4
|
+
stageReadback,
|
|
5
|
+
destroyBuffers,
|
|
6
|
+
} from "../util/buffer.mjs";
|
|
7
|
+
import { createBindGroup } from "../util/bindgroup.mjs";
|
|
8
|
+
import { runComputePass, submit } from "../util/compute.mjs";
|
|
9
|
+
import { extractResult } from "../util/result.mjs";
|
|
10
|
+
import { extractTimestamp } from "../util/benchmark.mjs";
|
|
11
|
+
import { getPipeline } from "../util/pipeline.mjs";
|
|
12
|
+
import { calcWorkgroups } from "../util/workgroup.mjs";
|
|
13
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
14
|
+
import { splitDoubleDouble, mergeDoubleDouble } from "../util/f64.mjs";
|
|
15
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
16
|
+
|
|
17
|
+
// daxpy: y := alpha * x + y, double-double (Dekker) f64 emulation of saxpy —
|
|
18
|
+
// x, y, and alpha are each split into an f32 (hi, lo) pair; WGSL has no f64 type.
|
|
19
|
+
export async function daxpy(device, n, alpha, x, incx, y, incy) {
|
|
20
|
+
const xIsGpu = x instanceof GpuVector;
|
|
21
|
+
const yIsGpu = y instanceof GpuVector;
|
|
22
|
+
|
|
23
|
+
requireGpuDevice(device);
|
|
24
|
+
if (
|
|
25
|
+
!Number.isInteger(n) ||
|
|
26
|
+
!Number.isInteger(incx) ||
|
|
27
|
+
!Number.isInteger(incy)
|
|
28
|
+
)
|
|
29
|
+
throw new Error("n, incx, and incy must be integers.");
|
|
30
|
+
if (typeof alpha !== "number") throw new Error("alpha must be a number.");
|
|
31
|
+
if (Number.isNaN(alpha)) throw new Error("alpha must not be NaN.");
|
|
32
|
+
if (!Number.isFinite(alpha)) throw new Error("alpha must be finite.");
|
|
33
|
+
if (!(x instanceof Float64Array) && !xIsGpu)
|
|
34
|
+
throw new Error("x must be a Float64Array or GpuVector.");
|
|
35
|
+
if (!(y instanceof Float64Array) && !yIsGpu)
|
|
36
|
+
throw new Error("y must be a Float64Array or GpuVector.");
|
|
37
|
+
if (xIsGpu && x.dtype !== Float64Array)
|
|
38
|
+
throw new Error("x must be a Float64Array-backed GpuVector.");
|
|
39
|
+
if (yIsGpu && y.dtype !== Float64Array)
|
|
40
|
+
throw new Error("y must be a Float64Array-backed GpuVector.");
|
|
41
|
+
if (xIsGpu !== yIsGpu)
|
|
42
|
+
throw new Error(
|
|
43
|
+
"x and y must be the same type (both Float64Array or both GpuVector).",
|
|
44
|
+
);
|
|
45
|
+
if (incx <= 0 || incy <= 0)
|
|
46
|
+
throw new Error("incx and incy must be positive.");
|
|
47
|
+
requireSameDevice(device, "daxpy", { x, y });
|
|
48
|
+
if (n <= 0) return yIsGpu ? {} : { y };
|
|
49
|
+
if (x.length < (n - 1) * incx + 1)
|
|
50
|
+
throw new Error(
|
|
51
|
+
"x does not have enough elements for the given n and incx.",
|
|
52
|
+
);
|
|
53
|
+
if (y.length < (n - 1) * incy + 1)
|
|
54
|
+
throw new Error(
|
|
55
|
+
"y does not have enough elements for the given n and incy.",
|
|
56
|
+
);
|
|
57
|
+
|
|
58
|
+
// Concatenated with f64/dekker.wgsl (DD struct), f64/utils/add.wgsl
|
|
59
|
+
// (fsub/negf/fastTwoSumProtected/ddAddProtected), and f64/utils/multiply.wgsl
|
|
60
|
+
// (ddMulProtected) — WGSL has no #include.
|
|
61
|
+
const f64Deps = ["f64/dekker", "f64/utils/add", "f64/utils/multiply"];
|
|
62
|
+
const pipeline = await getPipeline(device, [...f64Deps, "daxpy"]);
|
|
63
|
+
|
|
64
|
+
const { hi: alphaHi, lo: alphaLo } = splitDoubleDouble(
|
|
65
|
+
new Float64Array([alpha]),
|
|
66
|
+
);
|
|
67
|
+
|
|
68
|
+
let xHiBuffer = null;
|
|
69
|
+
let xLoBuffer = null;
|
|
70
|
+
let yHiBuffer = null;
|
|
71
|
+
let yLoBuffer = null;
|
|
72
|
+
let paramsBuffer = null;
|
|
73
|
+
let readHiBuffer = null;
|
|
74
|
+
let readLoBuffer = null;
|
|
75
|
+
|
|
76
|
+
try {
|
|
77
|
+
if (xIsGpu) {
|
|
78
|
+
xHiBuffer = x._buf;
|
|
79
|
+
xLoBuffer = x._loBuf;
|
|
80
|
+
yHiBuffer = y._buf;
|
|
81
|
+
yLoBuffer = y._loBuf;
|
|
82
|
+
} else {
|
|
83
|
+
const xSplit = splitDoubleDouble(x);
|
|
84
|
+
const ySplit = splitDoubleDouble(y);
|
|
85
|
+
xHiBuffer = uploadBuffer(device, xSplit.hi, "daxpy-xHi", false);
|
|
86
|
+
xLoBuffer = uploadBuffer(device, xSplit.lo, "daxpy-xLo", false);
|
|
87
|
+
yHiBuffer = uploadBuffer(device, ySplit.hi, "daxpy-yHi", true);
|
|
88
|
+
yLoBuffer = uploadBuffer(device, ySplit.lo, "daxpy-yLo", true);
|
|
89
|
+
}
|
|
90
|
+
paramsBuffer = createParamsBuffer(
|
|
91
|
+
device,
|
|
92
|
+
[
|
|
93
|
+
{ value: n, type: "u32" },
|
|
94
|
+
{ value: alphaHi[0], type: "f32" },
|
|
95
|
+
{ value: alphaLo[0], type: "f32" },
|
|
96
|
+
{ value: incx, type: "u32" },
|
|
97
|
+
{ value: incy, type: "u32" },
|
|
98
|
+
],
|
|
99
|
+
"daxpy-params",
|
|
100
|
+
);
|
|
101
|
+
|
|
102
|
+
const bindGroup = createBindGroup(device, pipeline.getBindGroupLayout(0), [
|
|
103
|
+
xHiBuffer,
|
|
104
|
+
xLoBuffer,
|
|
105
|
+
yHiBuffer,
|
|
106
|
+
yLoBuffer,
|
|
107
|
+
paramsBuffer,
|
|
108
|
+
]);
|
|
109
|
+
const { commandEncoder, ts } = runComputePass(
|
|
110
|
+
device,
|
|
111
|
+
pipeline,
|
|
112
|
+
bindGroup,
|
|
113
|
+
calcWorkgroups(device, n),
|
|
114
|
+
);
|
|
115
|
+
readHiBuffer = yIsGpu
|
|
116
|
+
? null
|
|
117
|
+
: stageReadback(device, commandEncoder, yHiBuffer);
|
|
118
|
+
readLoBuffer = yIsGpu
|
|
119
|
+
? null
|
|
120
|
+
: stageReadback(device, commandEncoder, yLoBuffer);
|
|
121
|
+
|
|
122
|
+
submit(device, commandEncoder);
|
|
123
|
+
|
|
124
|
+
const gpuTimeMs = await extractTimestamp(ts);
|
|
125
|
+
|
|
126
|
+
if (yIsGpu) {
|
|
127
|
+
// xIsGpu === yIsGpu, enforced above
|
|
128
|
+
if (gpuTimeMs !== undefined) return { gpuTimeMs };
|
|
129
|
+
return {};
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
const hi = await extractResult(readHiBuffer, Float32Array);
|
|
133
|
+
readHiBuffer = null; // extractResult already destroyed it
|
|
134
|
+
const lo = await extractResult(readLoBuffer, Float32Array);
|
|
135
|
+
readLoBuffer = null;
|
|
136
|
+
const result = mergeDoubleDouble(hi, lo);
|
|
137
|
+
if (gpuTimeMs !== undefined) return { y: result, gpuTimeMs };
|
|
138
|
+
return { y: result };
|
|
139
|
+
} finally {
|
|
140
|
+
if (!xIsGpu && xHiBuffer) destroyBuffers(xHiBuffer);
|
|
141
|
+
if (!xIsGpu && xLoBuffer) destroyBuffers(xLoBuffer);
|
|
142
|
+
if (!yIsGpu && yHiBuffer) destroyBuffers(yHiBuffer);
|
|
143
|
+
if (!yIsGpu && yLoBuffer) destroyBuffers(yLoBuffer);
|
|
144
|
+
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
145
|
+
// Only reached if extractTimestamp or extractResult threw before
|
|
146
|
+
// clearing these — on the success path they're already null.
|
|
147
|
+
if (readHiBuffer) destroyBuffers(readHiBuffer);
|
|
148
|
+
if (readLoBuffer) destroyBuffers(readLoBuffer);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Performs the operation $$y \leftarrow x$$ — double-double (Dekker) f64
|
|
5
|
+
* emulation of {@link scopy}, since WGSL has no native f64 type.
|
|
6
|
+
*
|
|
7
|
+
* {@includeCode ../../examples/dcopy/dcopy.js}
|
|
8
|
+
*
|
|
9
|
+
* **Browser (standalone HTML):**
|
|
10
|
+
* {@includeCode ../../examples/dcopy/web/dcopy.html}
|
|
11
|
+
*
|
|
12
|
+
* @param device - GPUDevice from `init()`
|
|
13
|
+
* @param n - number of elements (must be a positive integer)
|
|
14
|
+
* @param x - Float64Array input vector
|
|
15
|
+
* @param incx - stride for x (must be a positive integer)
|
|
16
|
+
* @param y - Float64Array output vector
|
|
17
|
+
* @param incy - stride for y (must be a positive integer)
|
|
18
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/dcopy/dcopy.mjs#L20">Source code: dcopy.mjs (L20)</a>
|
|
19
|
+
* @category BLAS Level 1
|
|
20
|
+
*/
|
|
21
|
+
export declare function dcopy(
|
|
22
|
+
device: GPUDevice,
|
|
23
|
+
n: number,
|
|
24
|
+
x: Float64Array,
|
|
25
|
+
incx: number,
|
|
26
|
+
y: Float64Array,
|
|
27
|
+
incy: number,
|
|
28
|
+
): Promise<{ y: Float64Array } | { y: Float64Array; gpuTimeMs: number }>;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Performs the operation $$y \leftarrow x$$ — GPU-resident overload; see the
|
|
32
|
+
* Float64Array overload above for the routine itself.
|
|
33
|
+
*
|
|
34
|
+
* {@includeCode ../../examples/dcopy/gpu.dcopy.js}
|
|
35
|
+
*
|
|
36
|
+
* @param device - GPUDevice from `init()`
|
|
37
|
+
* @param n - number of elements (must be a positive integer)
|
|
38
|
+
* @param x - GpuVector input vector (must be Float64Array-backed)
|
|
39
|
+
* @param incx - stride for x (must be a positive integer)
|
|
40
|
+
* @param y - GpuVector output vector (must be Float64Array-backed, mutated in place)
|
|
41
|
+
* @param incy - stride for y (must be a positive integer)
|
|
42
|
+
* @see <a href="https://github.com/manit2004/wgblas/blob/main/src/dcopy/dcopy.mjs#L20">Source code: dcopy.mjs (L20)</a>
|
|
43
|
+
* @category BLAS Level 1
|
|
44
|
+
*/
|
|
45
|
+
export declare function dcopy(
|
|
46
|
+
device: GPUDevice,
|
|
47
|
+
n: number,
|
|
48
|
+
x: GpuVector,
|
|
49
|
+
incx: number,
|
|
50
|
+
y: GpuVector,
|
|
51
|
+
incy: number,
|
|
52
|
+
): Promise<{} | { gpuTimeMs: number }>;
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import {
|
|
2
|
+
uploadBuffer,
|
|
3
|
+
createParamsBuffer,
|
|
4
|
+
stageReadback,
|
|
5
|
+
destroyBuffers,
|
|
6
|
+
} from "../util/buffer.mjs";
|
|
7
|
+
import { createBindGroup } from "../util/bindgroup.mjs";
|
|
8
|
+
import { runComputePass, submit } from "../util/compute.mjs";
|
|
9
|
+
import { extractResult } from "../util/result.mjs";
|
|
10
|
+
import { extractTimestamp } from "../util/benchmark.mjs";
|
|
11
|
+
import { getPipeline } from "../util/pipeline.mjs";
|
|
12
|
+
import { calcWorkgroups } from "../util/workgroup.mjs";
|
|
13
|
+
import { GpuVector } from "../classes/GpuVector.mjs";
|
|
14
|
+
import { splitDoubleDouble, mergeDoubleDouble } from "../util/f64.mjs";
|
|
15
|
+
import { requireGpuDevice, requireSameDevice } from "../util/device.mjs";
|
|
16
|
+
|
|
17
|
+
// dcopy: y := x, double-double (Dekker) f64 emulation of scopy — x and y are
|
|
18
|
+
// each split into an f32 (hi, lo) pair; WGSL has no f64 type. Unlike
|
|
19
|
+
// dscal/daxpy/ddot, a copy has no arithmetic at all, so it needs no Dekker
|
|
20
|
+
// shader dependencies (dekker.wgsl/add.wgsl/multiply.wgsl) — dcopy.wgsl is
|
|
21
|
+
// entirely self-contained.
|
|
22
|
+
export async function dcopy(device, n, x, incx, y, incy) {
|
|
23
|
+
const xIsGpu = x instanceof GpuVector;
|
|
24
|
+
const yIsGpu = y instanceof GpuVector;
|
|
25
|
+
|
|
26
|
+
requireGpuDevice(device);
|
|
27
|
+
requireSameDevice(device, "dcopy", { x, y });
|
|
28
|
+
if (
|
|
29
|
+
!Number.isInteger(n) ||
|
|
30
|
+
!Number.isInteger(incx) ||
|
|
31
|
+
!Number.isInteger(incy)
|
|
32
|
+
)
|
|
33
|
+
throw new Error("n, incx, and incy must be integers.");
|
|
34
|
+
if (incx <= 0 || incy <= 0)
|
|
35
|
+
throw new Error("incx and incy must be positive.");
|
|
36
|
+
if (!(x instanceof Float64Array) && !xIsGpu)
|
|
37
|
+
throw new Error("x must be a Float64Array or GpuVector.");
|
|
38
|
+
if (!(y instanceof Float64Array) && !yIsGpu)
|
|
39
|
+
throw new Error("y must be a Float64Array or GpuVector.");
|
|
40
|
+
if (xIsGpu && x.dtype !== Float64Array)
|
|
41
|
+
throw new Error("x must be a Float64Array-backed GpuVector.");
|
|
42
|
+
if (yIsGpu && y.dtype !== Float64Array)
|
|
43
|
+
throw new Error("y must be a Float64Array-backed GpuVector.");
|
|
44
|
+
if (xIsGpu !== yIsGpu)
|
|
45
|
+
throw new Error(
|
|
46
|
+
"x and y must be the same type (both Float64Array or both GpuVector).",
|
|
47
|
+
);
|
|
48
|
+
if (n <= 0) return yIsGpu ? {} : { y };
|
|
49
|
+
if (x.length < (n - 1) * incx + 1)
|
|
50
|
+
throw new Error(
|
|
51
|
+
"x does not have enough elements for the given n and incx.",
|
|
52
|
+
);
|
|
53
|
+
if (y.length < (n - 1) * incy + 1)
|
|
54
|
+
throw new Error(
|
|
55
|
+
"y does not have enough elements for the given n and incy.",
|
|
56
|
+
);
|
|
57
|
+
|
|
58
|
+
const pipeline = await getPipeline(device, "dcopy");
|
|
59
|
+
|
|
60
|
+
let xHiBuffer = null;
|
|
61
|
+
let xLoBuffer = null;
|
|
62
|
+
let yHiBuffer = null;
|
|
63
|
+
let yLoBuffer = null;
|
|
64
|
+
let paramsBuffer = null;
|
|
65
|
+
let readHiBuffer = null;
|
|
66
|
+
let readLoBuffer = null;
|
|
67
|
+
|
|
68
|
+
try {
|
|
69
|
+
if (xIsGpu) {
|
|
70
|
+
xHiBuffer = x._buf;
|
|
71
|
+
xLoBuffer = x._loBuf;
|
|
72
|
+
yHiBuffer = y._buf;
|
|
73
|
+
yLoBuffer = y._loBuf;
|
|
74
|
+
} else {
|
|
75
|
+
const xSplit = splitDoubleDouble(x);
|
|
76
|
+
const ySplit = splitDoubleDouble(y);
|
|
77
|
+
xHiBuffer = uploadBuffer(device, xSplit.hi, "dcopy-xHi", false);
|
|
78
|
+
xLoBuffer = uploadBuffer(device, xSplit.lo, "dcopy-xLo", false);
|
|
79
|
+
yHiBuffer = uploadBuffer(device, ySplit.hi, "dcopy-yHi", true);
|
|
80
|
+
yLoBuffer = uploadBuffer(device, ySplit.lo, "dcopy-yLo", true);
|
|
81
|
+
}
|
|
82
|
+
paramsBuffer = createParamsBuffer(
|
|
83
|
+
device,
|
|
84
|
+
[
|
|
85
|
+
{ value: n, type: "u32" },
|
|
86
|
+
{ value: incx, type: "u32" },
|
|
87
|
+
{ value: incy, type: "u32" },
|
|
88
|
+
],
|
|
89
|
+
"dcopy-params",
|
|
90
|
+
);
|
|
91
|
+
|
|
92
|
+
const bindGroup = createBindGroup(device, pipeline.getBindGroupLayout(0), [
|
|
93
|
+
xHiBuffer,
|
|
94
|
+
xLoBuffer,
|
|
95
|
+
yHiBuffer,
|
|
96
|
+
yLoBuffer,
|
|
97
|
+
paramsBuffer,
|
|
98
|
+
]);
|
|
99
|
+
const { commandEncoder, ts } = runComputePass(
|
|
100
|
+
device,
|
|
101
|
+
pipeline,
|
|
102
|
+
bindGroup,
|
|
103
|
+
calcWorkgroups(device, n),
|
|
104
|
+
);
|
|
105
|
+
readHiBuffer = yIsGpu
|
|
106
|
+
? null
|
|
107
|
+
: stageReadback(device, commandEncoder, yHiBuffer);
|
|
108
|
+
readLoBuffer = yIsGpu
|
|
109
|
+
? null
|
|
110
|
+
: stageReadback(device, commandEncoder, yLoBuffer);
|
|
111
|
+
|
|
112
|
+
submit(device, commandEncoder);
|
|
113
|
+
|
|
114
|
+
const gpuTimeMs = await extractTimestamp(ts);
|
|
115
|
+
|
|
116
|
+
if (yIsGpu) {
|
|
117
|
+
// xIsGpu === yIsGpu, enforced above
|
|
118
|
+
if (gpuTimeMs !== undefined) return { gpuTimeMs };
|
|
119
|
+
return {};
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
const hi = await extractResult(readHiBuffer, Float32Array);
|
|
123
|
+
readHiBuffer = null; // extractResult already destroyed it
|
|
124
|
+
const lo = await extractResult(readLoBuffer, Float32Array);
|
|
125
|
+
readLoBuffer = null;
|
|
126
|
+
const result = mergeDoubleDouble(hi, lo);
|
|
127
|
+
if (gpuTimeMs !== undefined) return { y: result, gpuTimeMs };
|
|
128
|
+
return { y: result };
|
|
129
|
+
} finally {
|
|
130
|
+
if (!xIsGpu && xHiBuffer) destroyBuffers(xHiBuffer);
|
|
131
|
+
if (!xIsGpu && xLoBuffer) destroyBuffers(xLoBuffer);
|
|
132
|
+
if (!yIsGpu && yHiBuffer) destroyBuffers(yHiBuffer);
|
|
133
|
+
if (!yIsGpu && yLoBuffer) destroyBuffers(yLoBuffer);
|
|
134
|
+
if (paramsBuffer) destroyBuffers(paramsBuffer);
|
|
135
|
+
// Only reached if extractTimestamp or extractResult threw before
|
|
136
|
+
// clearing these — on the success path they're already null.
|
|
137
|
+
if (readHiBuffer) destroyBuffers(readHiBuffer);
|
|
138
|
+
if (readLoBuffer) destroyBuffers(readLoBuffer);
|
|
139
|
+
}
|
|
140
|
+
}
|