@danielsimonjr/mathts-functions 0.19.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -45,9 +45,12 @@ declare const WGSL_OP_BODY: {
45
45
  };
46
46
  /** Ops that have a GPU kernel. A chain outside this set falls back. */
47
47
  export type GpuElementwiseOp = keyof typeof WGSL_OP_BODY;
48
- export declare const GPU_ELEMENTWISE_OPS: GpuElementwiseOp[];
48
+ export declare const GPU_ELEMENTWISE_OPS: readonly GpuElementwiseOp[];
49
49
  /** Whether every op in the chain has a GPU kernel. */
50
50
  export declare function isGpuChainSupported(ops: readonly string[]): ops is readonly GpuElementwiseOp[];
51
+ /** Reductions that can be fused onto the end of a chain. */
52
+ export type GpuReduceOp = 'sum' | 'max' | 'min';
53
+ export declare const GPU_REDUCE_OPS: readonly ["sum", "max", "min"];
51
54
  /** Drop the cached shaders/buffers (device loss, or between tests). */
52
55
  export declare function resetGpuElementwise(): void;
53
56
  /** Options for a GPU element-wise dispatch. */
@@ -69,5 +72,49 @@ export interface GpuChainOptions extends GPUContextOptions {
69
72
  * @returns the f32 results, or `null` to signal "fall back to another tier"
70
73
  */
71
74
  export declare function elementwiseChainGpuDispatch(ops: readonly string[], xs: Float64Array | Float32Array, options?: GpuChainOptions): Promise<Float32Array | null>;
75
+ /**
76
+ * Apply `ops` on the GPU and **reduce the result on-device**, returning a single
77
+ * number instead of an array.
78
+ *
79
+ * The point is the readback, not the arithmetic: reducing on the device replaces
80
+ * an **n-float** transfer back to the CPU with an **n/256-float** one. Measured
81
+ * end-to-end through THIS function (not a prototype), NVIDIA Pascal,
82
+ * `sum(exp(sin(x)))`:
83
+ *
84
+ * | n | WASM chain + JS sum | GPU chain + JS sum | fused GPU reduce |
85
+ * | --------- | ------------------- | ------------------ | ---------------- |
86
+ * | 262,144 | 25.6 ms | 16.7 ms | **9.9 ms** |
87
+ * | 1,048,576 | 96.8 ms | 34.3 ms | **25.4 ms** |
88
+ * | 4,194,304 | 260.0 ms | 100.0 ms | **72.2 ms** |
89
+ *
90
+ * **1.35-1.7x** over the shipped GPU path, **2.6-3.8x** over the CPU tier.
91
+ *
92
+ * Quote the **1.39x at n=2^22** if you quote one number: it is the only ratio here that
93
+ * reproduces run to run (1.31-1.39x over four runs). The 1.7x is the n=262,144 row, and
94
+ * that size swings 1.19-2.83x between runs — the GPU work is short enough that fixed
95
+ * costs dominate. A headline should not be a lucky sample.
96
+ *
97
+ * Why not more: a bare-WGSL prototype of this hit ~2x, but it pre-converted its
98
+ * input outside the timed region. The real f64->f32 conversion is an n-scaling cost
99
+ * that BOTH paths pay, so it dilutes the ratio as n grows (the absolute saving is
100
+ * steady: ~28 ms at n=2^22). The prototype's number was not a lie, it was measuring
101
+ * a workload no caller has. Quote the numbers above, not those.
102
+ *
103
+ * **An empty `ops` is declined on purpose.** A *standalone* GPU reduction uploads
104
+ * n floats to produce one number — pure transfer tax, measured 3-9x SLOWER than a
105
+ * plain JS sum. There is no chain to amortise the upload against, so this returns
106
+ * `null` and lets the caller use the CPU, which is genuinely the faster path. The
107
+ * upload is only worth paying for when real work rides along with it.
108
+ *
109
+ * Same never-throw contract as {@link elementwiseChainGpuDispatch}: returns `null`
110
+ * — never rejects — whenever the GPU is unavailable, not opted into, the input is
111
+ * below `GPU_MIN_ELEMENTS`, an op has no kernel, or a device limit is exceeded.
112
+ *
113
+ * Precision: f32, like every GPU path here. For `sum` the tree reduction is
114
+ * pairwise, so its error grows O(log n) rather than the O(n) of a sequential
115
+ * accumulate — it is better-conditioned than the JS loop it replaces, even though
116
+ * it works in f32.
117
+ */
118
+ export declare function elementwiseChainReduceGpuDispatch(ops: readonly string[], xs: Float64Array | Float32Array, reduce: GpuReduceOp, options?: GpuChainOptions): Promise<number | null>;
72
119
  export {};
73
120
  //# sourceMappingURL=elementwise-gpu.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"elementwise-gpu.d.ts","sourceRoot":"","sources":["../../src/gpu/elementwise-gpu.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEH,OAAO,EAOL,KAAK,iBAAiB,EACvB,MAAM,2BAA2B,CAAC;AA+CnC,QAAA,MAAM,YAAY;;;;;;;;;;;;;;;;CAgCR,CAAC;AAEX,uEAAuE;AACvE,MAAM,MAAM,gBAAgB,GAAG,MAAM,OAAO,YAAY,CAAC;AAEzD,eAAO,MAAM,mBAAmB,EAAgC,gBAAgB,EAAE,CAAC;AAEnF,sDAAsD;AACtD,wBAAgB,mBAAmB,CAAC,GAAG,EAAE,SAAS,MAAM,EAAE,GAAG,GAAG,IAAI,SAAS,gBAAgB,EAAE,CAE9F;AAsED,uEAAuE;AACvE,wBAAgB,mBAAmB,IAAI,IAAI,CAG1C;AAED,+CAA+C;AAC/C,MAAM,WAAW,eAAgB,SAAQ,iBAAiB;IACxD;;;;;;OAMG;IACH,GAAG,CAAC,EAAE,OAAO,CAAC;CACf;AAED;;;;;;GAMG;AACH,wBAAsB,2BAA2B,CAC/C,GAAG,EAAE,SAAS,MAAM,EAAE,EACtB,EAAE,EAAE,YAAY,GAAG,YAAY,EAC/B,OAAO,CAAC,EAAE,eAAe,GACxB,OAAO,CAAC,YAAY,GAAG,IAAI,CAAC,CA6H9B"}
1
+ {"version":3,"file":"elementwise-gpu.d.ts","sourceRoot":"","sources":["../../src/gpu/elementwise-gpu.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEH,OAAO,EAOL,KAAK,iBAAiB,EACvB,MAAM,2BAA2B,CAAC;AA+CnC,QAAA,MAAM,YAAY;;;;;;;;;;;;;;;;CAgCR,CAAC;AAEX,uEAAuE;AACvE,MAAM,MAAM,gBAAgB,GAAG,MAAM,OAAO,YAAY,CAAC;AAEzD,eAAO,MAAM,mBAAmB,EAAE,SAAS,gBAAgB,EAEpC,CAAC;AAExB,sDAAsD;AACtD,wBAAgB,mBAAmB,CAAC,GAAG,EAAE,SAAS,MAAM,EAAE,GAAG,GAAG,IAAI,SAAS,gBAAgB,EAAE,CAE9F;AAoCD,4DAA4D;AAC5D,MAAM,MAAM,WAAW,GAAG,KAAK,GAAG,KAAK,GAAG,KAAK,CAAC;AAIhD,eAAO,MAAM,cAAc,gCAAkE,CAAC;AAwG9F,uEAAuE;AACvE,wBAAgB,mBAAmB,IAAI,IAAI,CAG1C;AAED,+CAA+C;AAC/C,MAAM,WAAW,eAAgB,SAAQ,iBAAiB;IACxD;;;;;;OAMG;IACH,GAAG,CAAC,EAAE,OAAO,CAAC;CACf;AA4BD;;;;;;GAMG;AACH,wBAAgB,2BAA2B,CACzC,GAAG,EAAE,SAAS,MAAM,EAAE,EACtB,EAAE,EAAE,YAAY,GAAG,YAAY,EAC/B,OAAO,CAAC,EAAE,eAAe,GACxB,OAAO,CAAC,YAAY,GAAG,IAAI,CAAC,CAE9B;AA8ID;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA0CG;AACH,wBAAgB,iCAAiC,CAC/C,GAAG,EAAE,SAAS,MAAM,EAAE,EACtB,EAAE,EAAE,YAAY,GAAG,YAAY,EAC/B,MAAM,EAAE,WAAW,EACnB,OAAO,CAAC,EAAE,eAAe,GACxB,OAAO,CAAC,MAAM,GAAG,IAAI,CAAC,CAExB"}
package/dist/index.js CHANGED
@@ -24,6 +24,7 @@ __export(typed_exports, {
24
24
  GAUSS_WORKER_THRESHOLD: () => GAUSS_WORKER_THRESHOLD,
25
25
  GPU_ELEMENTWISE_OPS: () => GPU_ELEMENTWISE_OPS,
26
26
  GPU_MIN_ELEMENTS: () => GPU_MIN_ELEMENTS2,
27
+ GPU_REDUCE_OPS: () => GPU_REDUCE_OPS,
27
28
  WASM_INTERP_THRESHOLD: () => WASM_INTERP_THRESHOLD,
28
29
  abs: () => abs,
29
30
  acos: () => acos,
@@ -153,6 +154,7 @@ __export(typed_exports, {
153
154
  eigenvectorCentrality: () => eigenvectorCentrality,
154
155
  element: () => element,
155
156
  elementwiseChainGpuDispatch: () => elementwiseChainGpuDispatch,
157
+ elementwiseChainReduceGpuDispatch: () => elementwiseChainReduceGpuDispatch,
156
158
  eliminate: () => eliminate,
157
159
  ellipticE: () => ellipticE,
158
160
  ellipticEIncomplete: () => ellipticEIncomplete,
@@ -195,6 +197,7 @@ __export(typed_exports, {
195
197
  functionExpand: () => functionExpand,
196
198
  fuseUnaryChain: () => fuseUnaryChain,
197
199
  fuseUnaryChainAsync: () => fuseUnaryChainAsync,
200
+ fuseUnaryChainReduceAsync: () => fuseUnaryChainReduceAsync,
198
201
  gammaDist: () => gammaDist,
199
202
  gammaPDF: () => gammaPDF,
200
203
  gammainc: () => gammainc,
@@ -7413,7 +7416,9 @@ var WGSL_OP_BODY = {
7413
7416
  // An f32 fast path may be less precise. It may not be WRONG. Chains containing
7414
7417
  // these fall back to the exact WASM/JS tiers.
7415
7418
  };
7416
- var GPU_ELEMENTWISE_OPS = Object.keys(WGSL_OP_BODY);
7419
+ var GPU_ELEMENTWISE_OPS = Object.keys(
7420
+ WGSL_OP_BODY
7421
+ );
7417
7422
  function isGpuChainSupported(ops) {
7418
7423
  return ops.every((op) => op in WGSL_OP_BODY);
7419
7424
  }
@@ -7438,6 +7443,48 @@ function wgslFor(op) {
7438
7443
  }
7439
7444
  `;
7440
7445
  }
7446
+ var GPU_REDUCE_OPS = ["sum", "max", "min"];
7447
+ var reduceKey = (r) => `reduce:${r}`;
7448
+ var WGSL_REDUCE = {
7449
+ sum: { identity: "0.0", combine: "a + b" },
7450
+ max: { identity: "neg_inf()", combine: "max(a, b)" },
7451
+ min: { identity: "pos_inf()", combine: "min(a, b)" }
7452
+ };
7453
+ function wgslForReduce(r) {
7454
+ const { identity: identity2, combine: combine2 } = WGSL_REDUCE[r];
7455
+ return `
7456
+ @group(0) @binding(0) var<storage, read> inp: array<f32>;
7457
+ @group(0) @binding(1) var<storage, read_write> partials: array<f32>;
7458
+ @group(0) @binding(2) var<uniform> params: vec4<u32>; // n, nanBits, +infBits, -infBits
7459
+
7460
+ ${WGSL_IEEE}
7461
+
7462
+ // NOTE: "shared" is a RESERVED WORD in WGSL \u2014 this must not be named that.
7463
+ var<workgroup> sdata: array<f32, ${WORKGROUP_SIZE}>;
7464
+
7465
+ fn identity() -> f32 { return ${identity2}; }
7466
+ fn combine(a: f32, b: f32) -> f32 { return ${combine2}; }
7467
+
7468
+ @compute @workgroup_size(${WORKGROUP_SIZE})
7469
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>,
7470
+ @builtin(local_invocation_id) lid: vec3<u32>,
7471
+ @builtin(workgroup_id) wid: vec3<u32>) {
7472
+ var v: f32 = identity();
7473
+ if (gid.x < params.x) { v = inp[gid.x]; }
7474
+ sdata[lid.x] = v;
7475
+ workgroupBarrier();
7476
+
7477
+ var s: u32 = ${WORKGROUP_SIZE}u / 2u;
7478
+ loop {
7479
+ if (s == 0u) { break; }
7480
+ if (lid.x < s) { sdata[lid.x] = combine(sdata[lid.x], sdata[lid.x + s]); }
7481
+ workgroupBarrier();
7482
+ s = s >> 1u;
7483
+ }
7484
+ if (lid.x == 0u) { partials[wid.x] = sdata[0]; }
7485
+ }
7486
+ `;
7487
+ }
7441
7488
  var resources = null;
7442
7489
  async function getResources(options) {
7443
7490
  const device = await getGpuDevice(options);
@@ -7448,6 +7495,9 @@ async function getResources(options) {
7448
7495
  for (const op of GPU_ELEMENTWISE_OPS) {
7449
7496
  shaders.registerShader(op, wgslFor(op));
7450
7497
  }
7498
+ for (const r of GPU_REDUCE_OPS) {
7499
+ shaders.registerShader(reduceKey(r), wgslForReduce(r));
7500
+ }
7451
7501
  shaders.precompileRegistered();
7452
7502
  resources = { device, shaders, pool: new BufferPool(ctx) };
7453
7503
  return resources;
@@ -7456,7 +7506,16 @@ function resetGpuElementwise() {
7456
7506
  resources?.pool.destroy();
7457
7507
  resources = null;
7458
7508
  }
7459
- async function elementwiseChainGpuDispatch(ops, xs, options) {
7509
+ var gpuQueue = Promise.resolve();
7510
+ function serializeGpu(run) {
7511
+ const next = gpuQueue.then(run, run);
7512
+ gpuQueue = next.catch(() => void 0);
7513
+ return next;
7514
+ }
7515
+ function elementwiseChainGpuDispatch(ops, xs, options) {
7516
+ return serializeGpu(() => chainGpuDispatchImpl(ops, xs, options));
7517
+ }
7518
+ async function chainGpuDispatchImpl(ops, xs, options) {
7460
7519
  const n = xs.length;
7461
7520
  const enabled = options?.gpu ?? isGpuEnabled();
7462
7521
  if (!enabled) return null;
@@ -7470,6 +7529,7 @@ async function elementwiseChainGpuDispatch(ops, xs, options) {
7470
7529
  let bufB;
7471
7530
  let staging;
7472
7531
  let params;
7532
+ let scopePushed = false;
7473
7533
  let scopePopped = false;
7474
7534
  try {
7475
7535
  const r = await getResources(options);
@@ -7482,6 +7542,7 @@ async function elementwiseChainGpuDispatch(ops, xs, options) {
7482
7542
  if (bytes > limits.maxBufferSize) return null;
7483
7543
  const input = xs instanceof Float32Array ? xs : new Float32Array(xs);
7484
7544
  device.pushErrorScope("validation");
7545
+ scopePushed = true;
7485
7546
  bufA = pool.acquireStorageBuffer(bytes, "chain-a", true, true);
7486
7547
  bufB = pool.acquireStorageBuffer(bytes, "chain-b", true, true);
7487
7548
  staging = pool.acquireStagingBuffer(bytes, "chain-staging");
@@ -7520,7 +7581,108 @@ async function elementwiseChainGpuDispatch(ops, xs, options) {
7520
7581
  } catch {
7521
7582
  return null;
7522
7583
  } finally {
7523
- if (res && !scopePopped) {
7584
+ if (res && scopePushed && !scopePopped) {
7585
+ try {
7586
+ await res.device.popErrorScope();
7587
+ } catch {
7588
+ }
7589
+ }
7590
+ if (res) {
7591
+ if (bufA) res.pool.release(bufA);
7592
+ if (bufB) res.pool.release(bufB);
7593
+ if (staging) res.pool.release(staging);
7594
+ if (params) res.pool.release(params);
7595
+ }
7596
+ }
7597
+ }
7598
+ function elementwiseChainReduceGpuDispatch(ops, xs, reduce3, options) {
7599
+ return serializeGpu(() => chainReduceGpuDispatchImpl(ops, xs, reduce3, options));
7600
+ }
7601
+ async function chainReduceGpuDispatchImpl(ops, xs, reduce3, options) {
7602
+ const n = xs.length;
7603
+ const enabled = options?.gpu ?? isGpuEnabled();
7604
+ if (!enabled) return null;
7605
+ if (ops.length === 0) return null;
7606
+ if (n < GPU_MIN_ELEMENTS) return null;
7607
+ if (!isGpuChainSupported(ops)) return null;
7608
+ if (!GPU_REDUCE_OPS.includes(reduce3)) return null;
7609
+ const bytes = n * 4;
7610
+ const workgroups = Math.ceil(n / WORKGROUP_SIZE);
7611
+ const partialBytes = workgroups * 4;
7612
+ let res;
7613
+ let bufA;
7614
+ let bufB;
7615
+ let partials;
7616
+ let staging;
7617
+ let params;
7618
+ let scopePushed = false;
7619
+ let scopePopped = false;
7620
+ try {
7621
+ const r = await getResources(options);
7622
+ if (!r) return null;
7623
+ res = r;
7624
+ const { device, shaders, pool } = r;
7625
+ const limits = device.limits;
7626
+ if (workgroups > limits.maxComputeWorkgroupsPerDimension) return null;
7627
+ if (bytes > limits.maxStorageBufferBindingSize) return null;
7628
+ if (bytes > limits.maxBufferSize) return null;
7629
+ const input = xs instanceof Float32Array ? xs : new Float32Array(xs);
7630
+ device.pushErrorScope("validation");
7631
+ scopePushed = true;
7632
+ bufA = pool.acquireStorageBuffer(bytes, "chain-a", true, true);
7633
+ bufB = pool.acquireStorageBuffer(bytes, "chain-b", true, true);
7634
+ partials = pool.acquireStorageBuffer(partialBytes, "reduce-partials", true, true);
7635
+ staging = pool.acquireStagingBuffer(partialBytes, "reduce-staging");
7636
+ params = pool.acquireUniformBuffer(16, "chain-params");
7637
+ device.queue.writeBuffer(bufA, 0, input);
7638
+ device.queue.writeBuffer(params, 0, new Uint32Array([n, 2143289344, 2139095040, 4286578688]));
7639
+ const encoder = device.createCommandEncoder({ label: "elementwise-chain-reduce" });
7640
+ let src = bufA;
7641
+ let dst2 = bufB;
7642
+ for (const op of ops) {
7643
+ const pipeline = shaders.getRegisteredPipeline(op);
7644
+ const bindGroup = device.createBindGroup({
7645
+ layout: pipeline.getBindGroupLayout(0),
7646
+ entries: [
7647
+ { binding: 0, resource: { buffer: src } },
7648
+ { binding: 1, resource: { buffer: dst2 } },
7649
+ { binding: 2, resource: { buffer: params } }
7650
+ ]
7651
+ });
7652
+ const pass = encoder.beginComputePass({ label: `chain:${op}` });
7653
+ pass.setPipeline(pipeline);
7654
+ pass.setBindGroup(0, bindGroup);
7655
+ pass.dispatchWorkgroups(workgroups);
7656
+ pass.end();
7657
+ [src, dst2] = [dst2, src];
7658
+ }
7659
+ const reducePipeline = shaders.getRegisteredPipeline(reduceKey(reduce3));
7660
+ const reduceBind = device.createBindGroup({
7661
+ layout: reducePipeline.getBindGroupLayout(0),
7662
+ entries: [
7663
+ { binding: 0, resource: { buffer: src } },
7664
+ { binding: 1, resource: { buffer: partials } },
7665
+ { binding: 2, resource: { buffer: params } }
7666
+ ]
7667
+ });
7668
+ const reducePass = encoder.beginComputePass({ label: `reduce:${reduce3}` });
7669
+ reducePass.setPipeline(reducePipeline);
7670
+ reducePass.setBindGroup(0, reduceBind);
7671
+ reducePass.dispatchWorkgroups(workgroups);
7672
+ reducePass.end();
7673
+ encoder.copyBufferToBuffer(partials, 0, staging, 0, partialBytes);
7674
+ device.queue.submit([encoder.finish()]);
7675
+ const validationError = await device.popErrorScope();
7676
+ scopePopped = true;
7677
+ if (validationError) return null;
7678
+ await staging.mapAsync(GPUMapMode.READ, 0, partialBytes);
7679
+ const parts = new Float32Array(staging.getMappedRange(0, partialBytes).slice(0));
7680
+ staging.unmap();
7681
+ return foldPartials(parts, reduce3);
7682
+ } catch {
7683
+ return null;
7684
+ } finally {
7685
+ if (res && scopePushed && !scopePopped) {
7524
7686
  try {
7525
7687
  await res.device.popErrorScope();
7526
7688
  } catch {
@@ -7529,11 +7691,24 @@ async function elementwiseChainGpuDispatch(ops, xs, options) {
7529
7691
  if (res) {
7530
7692
  if (bufA) res.pool.release(bufA);
7531
7693
  if (bufB) res.pool.release(bufB);
7694
+ if (partials) res.pool.release(partials);
7532
7695
  if (staging) res.pool.release(staging);
7533
7696
  if (params) res.pool.release(params);
7534
7697
  }
7535
7698
  }
7536
7699
  }
7700
+ function foldPartials(parts, reduce3) {
7701
+ if (reduce3 === "sum") {
7702
+ let acc2 = 0;
7703
+ for (let i = 0; i < parts.length; i++) acc2 += parts[i];
7704
+ return acc2;
7705
+ }
7706
+ let acc = reduce3 === "max" ? -Infinity : Infinity;
7707
+ for (let i = 0; i < parts.length; i++) {
7708
+ acc = reduce3 === "max" ? Math.max(acc, parts[i]) : Math.min(acc, parts[i]);
7709
+ }
7710
+ return acc;
7711
+ }
7537
7712
 
7538
7713
  // src/typed/fused.ts
7539
7714
  var SCALAR = {
@@ -7576,6 +7751,24 @@ async function fuseUnaryChainAsync(ops, xs, options) {
7576
7751
  if (wasm) return wasm;
7577
7752
  return jsChain(ops, xs);
7578
7753
  }
7754
+ async function fuseUnaryChainReduceAsync(ops, xs, reduce3, options) {
7755
+ const gpu = await elementwiseChainReduceGpuDispatch(ops, xs, reduce3, options);
7756
+ if (gpu !== null) return gpu;
7757
+ const chained = elementwiseChainDispatch(ops, xs) ?? jsChain(ops, xs);
7758
+ return reduceF64(chained, reduce3);
7759
+ }
7760
+ function reduceF64(xs, reduce3) {
7761
+ if (reduce3 === "sum") {
7762
+ let acc2 = 0;
7763
+ for (let i = 0; i < xs.length; i++) acc2 += xs[i];
7764
+ return acc2;
7765
+ }
7766
+ let acc = reduce3 === "max" ? -Infinity : Infinity;
7767
+ for (let i = 0; i < xs.length; i++) {
7768
+ acc = reduce3 === "max" ? Math.max(acc, xs[i]) : Math.min(acc, xs[i]);
7769
+ }
7770
+ return acc;
7771
+ }
7579
7772
 
7580
7773
  // src/typed/distributions.ts
7581
7774
  import { mathTyped as mathTyped11 } from "@danielsimonjr/mathts-core";
@@ -44940,6 +45133,7 @@ export {
44940
45133
  GAUSS_WORKER_THRESHOLD,
44941
45134
  GPU_ELEMENTWISE_OPS,
44942
45135
  GPU_MIN_ELEMENTS2 as GPU_MIN_ELEMENTS,
45136
+ GPU_REDUCE_OPS,
44943
45137
  WASM_INTERP_THRESHOLD,
44944
45138
  abs,
44945
45139
  acf,
@@ -45157,6 +45351,7 @@ export {
45157
45351
  element,
45158
45352
  elementaryCharge,
45159
45353
  elementwiseChainGpuDispatch,
45354
+ elementwiseChainReduceGpuDispatch,
45160
45355
  eliminate,
45161
45356
  ellipticE,
45162
45357
  ellipticEIncomplete,
@@ -45285,6 +45480,7 @@ export {
45285
45480
  functionExpand,
45286
45481
  fuseUnaryChain,
45287
45482
  fuseUnaryChainAsync,
45483
+ fuseUnaryChainReduceAsync,
45288
45484
  gamma,
45289
45485
  gammaCDF,
45290
45486
  gammaDist,
@@ -13,7 +13,7 @@
13
13
  * array is below threshold, so the result is always correct.
14
14
  */
15
15
  import { type WasmElementwiseOp } from '../wasm/elementwise/wasm-bridge.js';
16
- import { type GpuChainOptions } from '../gpu/elementwise-gpu.js';
16
+ import { type GpuChainOptions, type GpuReduceOp } from '../gpu/elementwise-gpu.js';
17
17
  /**
18
18
  * Apply `ops` left-to-right over `xs` (i.e. `ops[last](…ops[0](x))`), fused in
19
19
  * WASM when possible. Returns a new `Float64Array`; never mutates `xs`.
@@ -83,4 +83,25 @@ export declare function fuseUnaryChain(ops: WasmElementwiseOp[], xs: Float64Arra
83
83
  * `elementwiseChainGpuDispatch` directly — it is exported for exactly that.
84
84
  */
85
85
  export declare function fuseUnaryChainAsync(ops: WasmElementwiseOp[], xs: Float64Array, options?: GpuChainOptions): Promise<Float64Array>;
86
+ /**
87
+ * Apply `ops` and then reduce to a single number — `sum(exp(sin(x)))` and friends.
88
+ *
89
+ * Tiers: **GPU (f32, opt-in) → WASM chain + JS reduce → JS chain + JS reduce.**
90
+ *
91
+ * When the GPU tier runs, the reduction happens **on the device**, so only n/256
92
+ * floats cross the bus instead of n. That is the whole reason this function exists.
93
+ * Measured end-to-end for `sum(exp(sin(x)))` on an NVIDIA Pascal adapter: **1.35-1.7x**
94
+ * faster than `fuseUnaryChainAsync(...)` followed by a JS loop, and **2.6-3.8x** faster
95
+ * than the CPU tier. (See `elementwiseChainReduceGpuDispatch` for the full table and
96
+ * for why the ratio shrinks as n grows.)
97
+ *
98
+ * Reach for it only when you want the **scalar**. If you also need the transformed
99
+ * array, use `fuseUnaryChainAsync` — you have to pay the n-float readback anyway, and
100
+ * summing it in JS afterwards costs almost nothing on top.
101
+ *
102
+ * Precision follows the tier that ran: f32 (~7 significant digits) on the GPU, exact
103
+ * f64 on WASM/JS. `enableGpu()` is the consent; with the flag off this is a pure f64
104
+ * computation.
105
+ */
106
+ export declare function fuseUnaryChainReduceAsync(ops: WasmElementwiseOp[], xs: Float64Array, reduce: GpuReduceOp, options?: GpuChainOptions): Promise<number>;
86
107
  //# sourceMappingURL=fused.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"fused.d.ts","sourceRoot":"","sources":["../../src/typed/fused.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AACH,OAAO,EAEL,KAAK,iBAAiB,EACvB,MAAM,oCAAoC,CAAC;AAC5C,OAAO,EAA+B,KAAK,eAAe,EAAE,MAAM,2BAA2B,CAAC;AAyB9F;;;GAGG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,iBAAiB,EAAE,EAAE,EAAE,EAAE,YAAY,GAAG,YAAY,CAIvF;AAcD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8DG;AACH,wBAAsB,mBAAmB,CACvC,GAAG,EAAE,iBAAiB,EAAE,EACxB,EAAE,EAAE,YAAY,EAChB,OAAO,CAAC,EAAE,eAAe,GACxB,OAAO,CAAC,YAAY,CAAC,CAgBvB"}
1
+ {"version":3,"file":"fused.d.ts","sourceRoot":"","sources":["../../src/typed/fused.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AACH,OAAO,EAEL,KAAK,iBAAiB,EACvB,MAAM,oCAAoC,CAAC;AAC5C,OAAO,EAGL,KAAK,eAAe,EACpB,KAAK,WAAW,EACjB,MAAM,2BAA2B,CAAC;AAyBnC;;;GAGG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,iBAAiB,EAAE,EAAE,EAAE,EAAE,YAAY,GAAG,YAAY,CAIvF;AAcD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8DG;AACH,wBAAsB,mBAAmB,CACvC,GAAG,EAAE,iBAAiB,EAAE,EACxB,EAAE,EAAE,YAAY,EAChB,OAAO,CAAC,EAAE,eAAe,GACxB,OAAO,CAAC,YAAY,CAAC,CAgBvB;AAED;;;;;;;;;;;;;;;;;;;GAmBG;AACH,wBAAsB,yBAAyB,CAC7C,GAAG,EAAE,iBAAiB,EAAE,EACxB,EAAE,EAAE,YAAY,EAChB,MAAM,EAAE,WAAW,EACnB,OAAO,CAAC,EAAE,eAAe,GACxB,OAAO,CAAC,MAAM,CAAC,CAQjB"}
@@ -47,7 +47,7 @@ export * from './matrix-ops.js';
47
47
  export { cond } from './matrix-ops.js';
48
48
  export * from './gpu.js';
49
49
  export { enableGpu, disableGpu, isGpuEnabled, GPU_MIN_ELEMENTS } from '@danielsimonjr/mathts-gpu';
50
- export { elementwiseChainGpuDispatch, isGpuChainSupported, resetGpuElementwise, GPU_ELEMENTWISE_OPS, type GpuElementwiseOp, type GpuChainOptions, } from '../gpu/elementwise-gpu.js';
50
+ export { elementwiseChainGpuDispatch, elementwiseChainReduceGpuDispatch, GPU_REDUCE_OPS, type GpuReduceOp, isGpuChainSupported, resetGpuElementwise, GPU_ELEMENTWISE_OPS, type GpuElementwiseOp, type GpuChainOptions, } from '../gpu/elementwise-gpu.js';
51
51
  export * from './relational.js';
52
52
  export { typedRelational } from './relational.js';
53
53
  export * from './string.js';
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../src/typed/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAOH,cAAc,iBAAiB,CAAC;AAChC,OAAO,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAGlD,cAAc,mBAAmB,CAAC;AAClC,OAAO,EAAE,iBAAiB,EAAE,MAAM,mBAAmB,CAAC;AAGtD,cAAc,iBAAiB,CAAC;AAChC,OAAO,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAGlD,cAAc,aAAa,CAAC;AAC5B,OAAO,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAG1C,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAK5C,cAAc,UAAU,CAAC;AACzB,OAAO,EAAE,QAAQ,EAAE,MAAM,UAAU,CAAC;AAGpC,cAAc,cAAc,CAAC;AAC7B,cAAc,YAAY,CAAC;AAC3B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,oBAAoB,CAAC;AACnC,OAAO,EAAE,kBAAkB,EAAE,MAAM,oBAAoB,CAAC;AAGxD,cAAc,eAAe,CAAC;AAG9B,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,kBAAkB,CAAC;AAGjC,cAAc,oBAAoB,CAAC;AAGnC,cAAc,cAAc,CAAC;AAG7B,cAAc,oBAAoB,CAAC;AAGnC,cAAc,YAAY,CAAC;AAG3B,cAAc,mBAAmB,CAAC;AAGlC,cAAc,iBAAiB,CAAC;AAWhC,cAAc,iBAAiB,CAAC;AAGhC,OAAO,EAAE,IAAI,EAAE,MAAM,iBAAiB,CAAC;AAGvC,cAAc,UAAU,CAAC;AAIzB,OAAO,EAAE,SAAS,EAAE,UAAU,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AAClG,OAAO,EACL,2BAA2B,EAC3B,mBAAmB,EACnB,mBAAmB,EACnB,mBAAmB,EACnB,KAAK,gBAAgB,EACrB,KAAK,eAAe,GACrB,MAAM,2BAA2B,CAAC;AAInC,cAAc,iBAAiB,CAAC;AAChC,OAAO,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAGlD,cAAc,aAAa,CAAC;AAC5B,OAAO,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAI1C,cAAc,kBAAkB,CAAC;AACjC,OAAO,EAAE,gBAAgB,EAAE,MAAM,kBAAkB,CAAC;AAGpD,cAAc,WAAW,CAAC;AAC1B,OAAO,EAAE,SAAS,EAAE,MAAM,WAAW,CAAC;AAsCtC;;;;;;GAMG;AACH,eAAO,MAAM,cAAc;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA8B1B,CAAC"}
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../src/typed/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAOH,cAAc,iBAAiB,CAAC;AAChC,OAAO,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAGlD,cAAc,mBAAmB,CAAC;AAClC,OAAO,EAAE,iBAAiB,EAAE,MAAM,mBAAmB,CAAC;AAGtD,cAAc,iBAAiB,CAAC;AAChC,OAAO,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAGlD,cAAc,aAAa,CAAC;AAC5B,OAAO,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAG1C,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAK5C,cAAc,UAAU,CAAC;AACzB,OAAO,EAAE,QAAQ,EAAE,MAAM,UAAU,CAAC;AAGpC,cAAc,cAAc,CAAC;AAC7B,cAAc,YAAY,CAAC;AAC3B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,oBAAoB,CAAC;AACnC,OAAO,EAAE,kBAAkB,EAAE,MAAM,oBAAoB,CAAC;AAGxD,cAAc,eAAe,CAAC;AAG9B,cAAc,cAAc,CAAC;AAC7B,OAAO,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAG5C,cAAc,kBAAkB,CAAC;AAGjC,cAAc,oBAAoB,CAAC;AAGnC,cAAc,cAAc,CAAC;AAG7B,cAAc,oBAAoB,CAAC;AAGnC,cAAc,YAAY,CAAC;AAG3B,cAAc,mBAAmB,CAAC;AAGlC,cAAc,iBAAiB,CAAC;AAWhC,cAAc,iBAAiB,CAAC;AAGhC,OAAO,EAAE,IAAI,EAAE,MAAM,iBAAiB,CAAC;AAGvC,cAAc,UAAU,CAAC;AAIzB,OAAO,EAAE,SAAS,EAAE,UAAU,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AAClG,OAAO,EACL,2BAA2B,EAC3B,iCAAiC,EACjC,cAAc,EACd,KAAK,WAAW,EAChB,mBAAmB,EACnB,mBAAmB,EACnB,mBAAmB,EACnB,KAAK,gBAAgB,EACrB,KAAK,eAAe,GACrB,MAAM,2BAA2B,CAAC;AAInC,cAAc,iBAAiB,CAAC;AAChC,OAAO,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAGlD,cAAc,aAAa,CAAC;AAC5B,OAAO,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAI1C,cAAc,kBAAkB,CAAC;AACjC,OAAO,EAAE,gBAAgB,EAAE,MAAM,kBAAkB,CAAC;AAGpD,cAAc,WAAW,CAAC;AAC1B,OAAO,EAAE,SAAS,EAAE,MAAM,WAAW,CAAC;AAsCtC;;;;;;GAMG;AACH,eAAO,MAAM,cAAc;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA8B1B,CAAC"}
package/package.json CHANGED
@@ -1,76 +1,76 @@
1
- {
2
- "name": "@danielsimonjr/mathts-functions",
3
- "version": "0.19.0",
4
- "description": "Mathematical functions for MathTS - arithmetic, algebra, trigonometry, statistics, and more",
5
- "author": "Daniel Simon Jr.",
6
- "license": "MIT",
7
- "type": "module",
8
- "main": "./dist/index.js",
9
- "module": "./dist/index.js",
10
- "types": "./dist/index.d.ts",
11
- "exports": {
12
- ".": {
13
- "import": "./dist/index.js",
14
- "types": "./dist/index.d.ts"
15
- }
16
- },
17
- "files": [
18
- "dist",
19
- "types",
20
- "README.md"
21
- ],
22
- "scripts": {
23
- "build": "tsup src/index.ts --format esm --clean && tsc -p tsconfig.dts.json && node scripts/copy-wasm.mjs",
24
- "dev": "tsup src/index.ts --format esm --dts --watch",
25
- "test": "vitest run",
26
- "test:diff": "node tests/diff-special.test.mjs && node tests/diff-elementwise.test.mjs && node tests/diff-fusion.test.mjs",
27
- "golden:gen": "python -X utf8 tests/golden/gen_special_goldens.py",
28
- "test:watch": "vitest",
29
- "test:coverage": "vitest run --coverage",
30
- "typecheck": "tsc --noEmit",
31
- "lint": "eslint src --ext .ts",
32
- "lint:fix": "eslint src --ext .ts --fix",
33
- "clean": "rm -rf dist",
34
- "build:prod": "tsup src/index.ts --format esm --clean --minify --treeshake"
35
- },
36
- "dependencies": {
37
- "@danielsimonjr/mathts-core": "^0.6.0",
38
- "@danielsimonjr/mathts-expression": "^0.6.0",
39
- "@danielsimonjr/mathts-gpu": "^0.1.1",
40
- "@danielsimonjr/mathts-matrix": "^0.4.0",
41
- "@danielsimonjr/mathts-parallel": "^0.3.4",
42
- "bignumber.js": "^9.1.2",
43
- "complex.js": "^2.2.5",
44
- "decimal.js": "^10.4.3",
45
- "escape-latex": "^1.2.0",
46
- "fraction.js": "^5.2.1",
47
- "javascript-natural-sort": "^0.7.1",
48
- "seedrandom": "^3.0.5",
49
- "tiny-emitter": "^2.1.0",
50
- "typed-function": "github:danielsimonjr/typed-function"
51
- },
52
- "devDependencies": {
53
- "@types/node": "^25.5.2",
54
- "@webgpu/types": "^0.1.67",
55
- "tsup": "^8.0.0",
56
- "typescript": "^5.3.0",
57
- "vitest": "^4.1.5"
58
- },
59
- "publishConfig": {
60
- "access": "public"
61
- },
62
- "repository": {
63
- "type": "git",
64
- "url": "https://github.com/danielsimonjr/mathts",
65
- "directory": "functions"
66
- },
67
- "keywords": [
68
- "math",
69
- "typescript",
70
- "functions",
71
- "arithmetic",
72
- "algebra",
73
- "trigonometry",
74
- "statistics"
75
- ]
76
- }
1
+ {
2
+ "name": "@danielsimonjr/mathts-functions",
3
+ "version": "0.20.0",
4
+ "description": "Mathematical functions for MathTS - arithmetic, algebra, trigonometry, statistics, and more",
5
+ "author": "Daniel Simon Jr.",
6
+ "license": "MIT",
7
+ "type": "module",
8
+ "main": "./dist/index.js",
9
+ "module": "./dist/index.js",
10
+ "types": "./dist/index.d.ts",
11
+ "exports": {
12
+ ".": {
13
+ "import": "./dist/index.js",
14
+ "types": "./dist/index.d.ts"
15
+ }
16
+ },
17
+ "files": [
18
+ "dist",
19
+ "types",
20
+ "README.md"
21
+ ],
22
+ "scripts": {
23
+ "build": "tsup src/index.ts --format esm --clean && tsc -p tsconfig.dts.json && node scripts/copy-wasm.mjs",
24
+ "dev": "tsup src/index.ts --format esm --dts --watch",
25
+ "test": "vitest run",
26
+ "test:diff": "node tests/diff-special.test.mjs && node tests/diff-elementwise.test.mjs && node tests/diff-fusion.test.mjs",
27
+ "golden:gen": "python -X utf8 tests/golden/gen_special_goldens.py",
28
+ "test:watch": "vitest",
29
+ "test:coverage": "vitest run --coverage",
30
+ "typecheck": "tsc --noEmit",
31
+ "lint": "eslint src --ext .ts",
32
+ "lint:fix": "eslint src --ext .ts --fix",
33
+ "clean": "rm -rf dist",
34
+ "build:prod": "tsup src/index.ts --format esm --clean --minify --treeshake"
35
+ },
36
+ "dependencies": {
37
+ "@danielsimonjr/mathts-core": "^0.6.0",
38
+ "@danielsimonjr/mathts-expression": "^0.6.0",
39
+ "@danielsimonjr/mathts-gpu": "^0.1.1",
40
+ "@danielsimonjr/mathts-matrix": "^0.4.0",
41
+ "@danielsimonjr/mathts-parallel": "^0.3.4",
42
+ "bignumber.js": "^9.1.2",
43
+ "complex.js": "^2.2.5",
44
+ "decimal.js": "^10.4.3",
45
+ "escape-latex": "^1.2.0",
46
+ "fraction.js": "^5.2.1",
47
+ "javascript-natural-sort": "^0.7.1",
48
+ "seedrandom": "^3.0.5",
49
+ "tiny-emitter": "^2.1.0",
50
+ "typed-function": "github:danielsimonjr/typed-function"
51
+ },
52
+ "devDependencies": {
53
+ "@types/node": "^25.5.2",
54
+ "@webgpu/types": "^0.1.67",
55
+ "tsup": "^8.0.0",
56
+ "typescript": "^5.3.0",
57
+ "vitest": "^4.1.5"
58
+ },
59
+ "publishConfig": {
60
+ "access": "public"
61
+ },
62
+ "repository": {
63
+ "type": "git",
64
+ "url": "https://github.com/danielsimonjr/mathts",
65
+ "directory": "functions"
66
+ },
67
+ "keywords": [
68
+ "math",
69
+ "typescript",
70
+ "functions",
71
+ "arithmetic",
72
+ "algebra",
73
+ "trigonometry",
74
+ "statistics"
75
+ ]
76
+ }