@tryhamster/gerbil 1.11.4 → 1.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/README.md +79 -0
  2. package/dist/{architectures-DmZMEFsA.mjs → architectures-BH_z6k9d.mjs} +14 -12
  3. package/dist/architectures-BH_z6k9d.mjs.map +1 -0
  4. package/dist/browser/index.d.ts.map +1 -1
  5. package/dist/browser/index.js +11 -0
  6. package/dist/browser/index.js.map +1 -1
  7. package/dist/cli.mjs +8 -8
  8. package/dist/cli.mjs.map +1 -1
  9. package/dist/frameworks/express.mjs +1 -1
  10. package/dist/frameworks/fastify.mjs +1 -1
  11. package/dist/frameworks/hono.mjs +1 -1
  12. package/dist/frameworks/next.d.mts +2 -2
  13. package/dist/frameworks/next.mjs +1 -1
  14. package/dist/frameworks/trpc.mjs +1 -1
  15. package/dist/gerbil-CZvoFo0T.mjs +4 -0
  16. package/dist/{gerbil-CYVmU8sQ.mjs → gerbil-CpS3P240.mjs} +13 -2
  17. package/dist/gerbil-CpS3P240.mjs.map +1 -0
  18. package/dist/{gerbil-5_80K0gA.d.mts → gerbil-DF009Waa.d.mts} +2 -2
  19. package/dist/{gerbil-5_80K0gA.d.mts.map → gerbil-DF009Waa.d.mts.map} +1 -1
  20. package/dist/gpu/hooks.d.mts +1 -1
  21. package/dist/gpu/index.d.mts +2 -2
  22. package/dist/gpu/index.mjs +4 -4
  23. package/dist/{gpu-CrzjQHv2.mjs → gpu-DRFhiv4R.mjs} +818 -137
  24. package/dist/gpu-DRFhiv4R.mjs.map +1 -0
  25. package/dist/{index-h8TDu1qm.d.mts → index-CNYoTRgr.d.mts} +1187 -550
  26. package/dist/index-CNYoTRgr.d.mts.map +1 -0
  27. package/dist/index.d.mts +3 -3
  28. package/dist/index.d.mts.map +1 -1
  29. package/dist/index.mjs +5 -5
  30. package/dist/index.mjs.map +1 -1
  31. package/dist/integrations/ai-sdk.mjs +1 -1
  32. package/dist/integrations/langchain.mjs +1 -1
  33. package/dist/integrations/llamaindex.mjs +1 -1
  34. package/dist/integrations/mcp.d.mts +2 -2
  35. package/dist/integrations/mcp.mjs +4 -4
  36. package/dist/{mcp-CAsD7eCj.mjs → mcp-DAbWO8VS.mjs} +3 -3
  37. package/dist/{mcp-CAsD7eCj.mjs.map → mcp-DAbWO8VS.mjs.map} +1 -1
  38. package/dist/{moonshine-stt-BXoZaHJE.mjs → moonshine-stt-B1kV5n1c.mjs} +5251 -1678
  39. package/dist/moonshine-stt-B1kV5n1c.mjs.map +1 -0
  40. package/dist/moonshine-stt-DA1WuiZb.mjs +4 -0
  41. package/dist/{one-liner-ppgw4jHH.mjs → one-liner-CmP9ktUn.mjs} +2 -2
  42. package/dist/{one-liner-ppgw4jHH.mjs.map → one-liner-CmP9ktUn.mjs.map} +1 -1
  43. package/dist/repl-17NVeIaJ.mjs +9 -0
  44. package/dist/skills/index.d.mts +4 -4
  45. package/dist/skills/index.mjs +3 -3
  46. package/dist/{skills-CQa1Gshd.mjs → skills-BDOEbHSx.mjs} +2 -2
  47. package/dist/{skills-CQa1Gshd.mjs.map → skills-BDOEbHSx.mjs.map} +1 -1
  48. package/dist/tune/index.mjs +1 -1
  49. package/package.json +1 -1
  50. package/dist/architectures-DmZMEFsA.mjs.map +0 -1
  51. package/dist/gerbil-CTefAwKp.mjs +0 -4
  52. package/dist/gerbil-CYVmU8sQ.mjs.map +0 -1
  53. package/dist/gpu-CrzjQHv2.mjs.map +0 -1
  54. package/dist/index-h8TDu1qm.d.mts.map +0 -1
  55. package/dist/moonshine-stt-BXoZaHJE.mjs.map +0 -1
  56. package/dist/moonshine-stt-DZVnKgPO.mjs +0 -4
  57. package/dist/repl-C0Ew7_Z-.mjs +0 -9
@@ -1,6 +1,6 @@
1
1
  import { a as resolveDefaultRepo, i as isTTSRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "./defaults-DfGx4d1m.mjs";
2
- import { A as kaniSinTensor, C as computeKaniPositions, D as kaniAttentionLayerIndices, E as generateNanoCodecDecoderGraph, F as CANONICAL_KEYS, I as DTYPE_BYTES, L as GEMMA4_VIS_KEYS, M as parseKaniConfig, N as DEFAULT_GROUP_SIZE, O as kaniCosTensor, S as buildKaniLayerCosSin, T as generateKaniTtsGraph, a as PARLER_DAC_LATENT_DIM, b as KANI_START_OF_HUMAN, c as PARLER_SAMPLE_RATE, d as generateParlerEncoderGraph, f as parseParlerConfig, i as PARLER_DAC_DECODER_DIM, k as kaniLayerAlpha, l as buildT5RelativeBias, o as PARLER_DECODER_RATES, p as revertDelayPattern, r as PARLER_BOS_TOKEN_ID, s as PARLER_EOS_TOKEN_ID, u as generateParlerDecoderGraph, x as audioTokensToCodes, y as KANI_END_OF_HUMAN } from "./architectures-DmZMEFsA.mjs";
3
- import { C as createUniformBuffer, D as verifyGPU, E as initGPU, S as createStorageBuffer, T as getOrCreatePipeline, a as loadModel, b as clearPipelineCache, c as loadParlerTTS, d as remapPrunedToken, g as fetchAdapter, h as buildLoRADeltas, i as loadKaniTTS, l as quantizeBackboneInt4, p as Executor, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as KERNEL_REGISTRY, w as destroyBuffers, x as createBindGroup, y as MATMUL_BIAS_F16C_SPEC } from "./moonshine-stt-BXoZaHJE.mjs";
2
+ import { A as kaniSinTensor, C as computeKaniPositions, D as kaniAttentionLayerIndices, E as generateNanoCodecDecoderGraph, F as CANONICAL_KEYS, I as DTYPE_BYTES, L as GEMMA4_VIS_KEYS, M as parseKaniConfig, N as DEFAULT_GROUP_SIZE, O as kaniCosTensor, S as buildKaniLayerCosSin, T as generateKaniTtsGraph, a as PARLER_DAC_LATENT_DIM, b as KANI_START_OF_HUMAN, c as PARLER_SAMPLE_RATE, d as generateParlerEncoderGraph, f as parseParlerConfig, i as PARLER_DAC_DECODER_DIM, k as kaniLayerAlpha, l as buildT5RelativeBias, o as PARLER_DECODER_RATES, p as revertDelayPattern, r as PARLER_BOS_TOKEN_ID, s as PARLER_EOS_TOKEN_ID, u as generateParlerDecoderGraph, x as audioTokensToCodes, y as KANI_END_OF_HUMAN } from "./architectures-BH_z6k9d.mjs";
3
+ import { C as createUniformBuffer, D as verifyGPU, E as initGPU, S as createStorageBuffer, T as getOrCreatePipeline, a as loadModel, b as clearPipelineCache, c as loadParlerTTS, d as remapPrunedToken, g as fetchAdapter, h as buildLoRADeltas, i as loadKaniTTS, l as quantizeBackboneInt4, p as Executor, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as KERNEL_REGISTRY, w as destroyBuffers, x as createBindGroup, y as MATMUL_BIAS_F16C_SPEC } from "./moonshine-stt-B1kV5n1c.mjs";
4
4
 
5
5
  //#region src/gpu/architectures/gemma4_vision.ts
6
6
  /**
@@ -1225,6 +1225,571 @@ function cleanSuggestion(raw, typed, options = {}) {
1225
1225
  return startsWithPunct || typedEndsWithSpace ? s : ` ${s}`;
1226
1226
  }
1227
1227
 
1228
+ //#endregion
1229
+ //#region src/gpu/sampler.ts
1230
+ /** mulberry32 — tiny deterministic PRNG for seeded sampling. */
1231
+ function mulberry32(seed) {
1232
+ let a = seed >>> 0;
1233
+ return () => {
1234
+ a = a + 1831565813 | 0;
1235
+ let t = Math.imul(a ^ a >>> 15, 1 | a);
1236
+ t = t + Math.imul(t ^ t >>> 7, 61 | t) ^ t;
1237
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
1238
+ };
1239
+ }
1240
+ /** Per-params seeded RNG streams (one stream per options object identity). */
1241
+ const seededStreams = /* @__PURE__ */ new WeakMap();
1242
+ function nextRandom(params) {
1243
+ if (params.seed === void 0) return Math.random();
1244
+ let rng = seededStreams.get(params);
1245
+ if (!rng) {
1246
+ rng = mulberry32(params.seed);
1247
+ seededStreams.set(params, rng);
1248
+ }
1249
+ return rng();
1250
+ }
1251
+ let _heapIndices = null;
1252
+ let _heapValues = null;
1253
+ /**
1254
+ * Sample a token ID from logits.
1255
+ *
1256
+ * Pipeline: repetition penalty → temperature → top-k (min-heap) → softmax → top-p → sample.
1257
+ */
1258
+ function sampleToken(logits, params = {}, previousTokens) {
1259
+ const temperature = params.temperature ?? .7;
1260
+ const topK = params.topK ?? 50;
1261
+ const topP = params.topP ?? .9;
1262
+ const repetitionPenalty = params.repetitionPenalty ?? 1;
1263
+ if (temperature < 1e-6) return argmax$2(logits);
1264
+ const N = logits.length;
1265
+ const K = Math.min(topK > 0 ? topK : N, N);
1266
+ if (!_heapIndices || _heapIndices.length < K) {
1267
+ _heapIndices = new Uint32Array(K);
1268
+ _heapValues = new Float32Array(K);
1269
+ }
1270
+ const hIdx = _heapIndices;
1271
+ const hVal = _heapValues;
1272
+ let penaltySet = null;
1273
+ if (repetitionPenalty !== 1 && previousTokens?.length) penaltySet = new Set(previousTokens);
1274
+ let heapSize = 0;
1275
+ for (let i = 0; i < N; i++) {
1276
+ let s = logits[i];
1277
+ if (penaltySet?.has(i)) s = s > 0 ? s / repetitionPenalty : s * repetitionPenalty;
1278
+ s /= temperature;
1279
+ if (heapSize < K) {
1280
+ hIdx[heapSize] = i;
1281
+ hVal[heapSize] = s;
1282
+ heapSize++;
1283
+ if (heapSize === K) for (let j = (K >> 1) - 1; j >= 0; j--) siftDown(hIdx, hVal, j, K);
1284
+ } else if (s > hVal[0]) {
1285
+ hIdx[0] = i;
1286
+ hVal[0] = s;
1287
+ siftDown(hIdx, hVal, 0, K);
1288
+ }
1289
+ }
1290
+ for (let i = 1; i < heapSize; i++) {
1291
+ const vi = hVal[i];
1292
+ const ii = hIdx[i];
1293
+ let j = i - 1;
1294
+ while (j >= 0 && hVal[j] < vi) {
1295
+ hVal[j + 1] = hVal[j];
1296
+ hIdx[j + 1] = hIdx[j];
1297
+ j--;
1298
+ }
1299
+ hVal[j + 1] = vi;
1300
+ hIdx[j + 1] = ii;
1301
+ }
1302
+ const maxScore = hVal[0];
1303
+ let sumExp = 0;
1304
+ for (let i = 0; i < heapSize; i++) {
1305
+ const p = Math.exp(hVal[i] - maxScore);
1306
+ hVal[i] = p;
1307
+ sumExp += p;
1308
+ }
1309
+ const invSum = 1 / sumExp;
1310
+ for (let i = 0; i < heapSize; i++) hVal[i] *= invSum;
1311
+ let candidateCount = heapSize;
1312
+ if (topP < 1) {
1313
+ let cumulative$1 = 0;
1314
+ for (let i = 0; i < heapSize; i++) {
1315
+ cumulative$1 += hVal[i];
1316
+ if (cumulative$1 >= topP) {
1317
+ candidateCount = i + 1;
1318
+ break;
1319
+ }
1320
+ }
1321
+ let sum = 0;
1322
+ for (let i = 0; i < candidateCount; i++) sum += hVal[i];
1323
+ const inv = 1 / sum;
1324
+ for (let i = 0; i < candidateCount; i++) hVal[i] *= inv;
1325
+ }
1326
+ const r = nextRandom(params);
1327
+ let cumulative = 0;
1328
+ for (let i = 0; i < candidateCount; i++) {
1329
+ cumulative += hVal[i];
1330
+ if (r <= cumulative) return hIdx[i];
1331
+ }
1332
+ return hIdx[candidateCount - 1];
1333
+ }
1334
+ /**
1335
+ * Return the index of the maximum value (greedy decoding).
1336
+ */
1337
+ function argmax$2(arr) {
1338
+ let maxIdx = 0;
1339
+ let maxVal = arr[0];
1340
+ for (let i = 1; i < arr.length; i++) if (arr[i] > maxVal) {
1341
+ maxVal = arr[i];
1342
+ maxIdx = i;
1343
+ }
1344
+ return maxIdx;
1345
+ }
1346
+ /** Min-heap sift down on parallel index/value typed arrays. */
1347
+ function siftDown(indices, values, i, n) {
1348
+ while (true) {
1349
+ let smallest = i;
1350
+ const left = 2 * i + 1;
1351
+ const right = 2 * i + 2;
1352
+ if (left < n && values[left] < values[smallest]) smallest = left;
1353
+ if (right < n && values[right] < values[smallest]) smallest = right;
1354
+ if (smallest === i) break;
1355
+ const ti = indices[i];
1356
+ indices[i] = indices[smallest];
1357
+ indices[smallest] = ti;
1358
+ const tv = values[i];
1359
+ values[i] = values[smallest];
1360
+ values[smallest] = tv;
1361
+ i = smallest;
1362
+ }
1363
+ }
1364
+
1365
+ //#endregion
1366
+ //#region src/gpu/batch/scheduler.ts
1367
+ const PIPELINE_DEPTH = 2;
1368
+ var RequestScheduler = class {
1369
+ host;
1370
+ batchSize;
1371
+ maxSeqLen;
1372
+ blockSize;
1373
+ prefillChunk;
1374
+ /** Free KV blocks kept unreserved as safety headroom. */
1375
+ blockHeadroom;
1376
+ nextId = 1;
1377
+ queue = [];
1378
+ /** slot → occupying request (prefilling or decoding). */
1379
+ slots;
1380
+ /** Sum of blocksReserved across live (admitted, unfinished) requests. */
1381
+ reservedBlocks = 0;
1382
+ prefilling = null;
1383
+ inFlight = [];
1384
+ /**
1385
+ * The slot→adapter-id assignment last uploaded to the executor (null =
1386
+ * nothing uploaded this batch session). Steps only re-upload when the
1387
+ * assignment changes (admissions/retirements), and the very first upload is
1388
+ * skipped entirely while every lane is bare base — an adapter-free serving
1389
+ * session never touches the LoRA machinery.
1390
+ */
1391
+ lastLaneAdapterIds = null;
1392
+ stepCounter = 0;
1393
+ loop = null;
1394
+ closed = false;
1395
+ stats = {
1396
+ submitted: 0,
1397
+ completed: 0,
1398
+ stepsSubmitted: 0,
1399
+ prefillChunks: 0,
1400
+ tokensGenerated: 0,
1401
+ busyMs: 0,
1402
+ peakActiveLanes: 0
1403
+ };
1404
+ constructor(host, options = {}) {
1405
+ this.host = host;
1406
+ this.batchSize = host.executor.batchSize;
1407
+ if (this.batchSize <= 0) throw new Error("RequestScheduler requires GERBIL_BATCH=N (N >= 2) on the Dawn path");
1408
+ this.maxSeqLen = host.executor.maxSequenceLength;
1409
+ this.blockSize = host.executor.kvBlockSizeTokens;
1410
+ this.slots = new Array(this.batchSize).fill(null);
1411
+ const envChunk = typeof process !== "undefined" ? Number(process.env?.GERBIL_PREFILL_CHUNK) : NaN;
1412
+ this.prefillChunk = options.prefillChunkTokens ?? (Number.isFinite(envChunk) && envChunk >= 8 ? envChunk : 128);
1413
+ this.blockHeadroom = 2;
1414
+ }
1415
+ /**
1416
+ * Submit a request; resolves with the full result when it finishes.
1417
+ * Tokens stream through options.onToken as they are decoded.
1418
+ */
1419
+ submit(prompt, options = {}) {
1420
+ if (this.closed) return Promise.reject(/* @__PURE__ */ new Error("RequestScheduler is closed"));
1421
+ if ((options.sampling?.temperature ?? 0) >= 1e-6) return Promise.reject(/* @__PURE__ */ new Error("RequestScheduler: Phase 5 is greedy-only — pass sampling: { temperature: 0 }"));
1422
+ let adapterReg = null;
1423
+ if (options.adapter != null) {
1424
+ if (!this.host.executor.batchLoraEnabled) return Promise.reject(/* @__PURE__ */ new Error("RequestScheduler: per-request adapters require GERBIL_BATCH_LORA=1 (with GERBIL_BATCH=N)"));
1425
+ try {
1426
+ adapterReg = this.host.resolveAdapter(options.adapter);
1427
+ } catch (err) {
1428
+ return Promise.reject(err instanceof Error ? err : new Error(String(err)));
1429
+ }
1430
+ }
1431
+ const messages = typeof prompt === "string" ? [...options.systemPrompt ? [{
1432
+ role: "system",
1433
+ content: options.systemPrompt
1434
+ }] : [], {
1435
+ role: "user",
1436
+ content: prompt
1437
+ }] : options.systemPrompt ? [{
1438
+ role: "system",
1439
+ content: options.systemPrompt
1440
+ }, ...prompt] : prompt;
1441
+ const promptIds = new Uint32Array(this.host.encodeChat(messages));
1442
+ const roomFor = this.maxSeqLen - promptIds.length - PIPELINE_DEPTH - 1;
1443
+ if (roomFor < 1) return Promise.reject(/* @__PURE__ */ new Error(`RequestScheduler: prompt (${promptIds.length} tokens) leaves no room in maxSeqLen ${this.maxSeqLen}`));
1444
+ const maxTokens = Math.min(options.maxTokens ?? 128, roomFor);
1445
+ return new Promise((resolve, reject) => {
1446
+ const req = {
1447
+ id: this.nextId++,
1448
+ promptIds,
1449
+ maxTokens,
1450
+ stopSequences: options.stopSequences ?? [],
1451
+ adapterName: options.adapter ?? null,
1452
+ adapterId: adapterReg?.id ?? -1,
1453
+ adapterDeltas: adapterReg?.deltas ?? null,
1454
+ onToken: options.onToken,
1455
+ resolve,
1456
+ reject,
1457
+ state: "queued",
1458
+ slot: -1,
1459
+ prefillPos: 0,
1460
+ blocksReserved: 0,
1461
+ generatedIds: [],
1462
+ text: "",
1463
+ finishReason: "max_tokens",
1464
+ submittedAt: performance.now(),
1465
+ admittedAt: 0,
1466
+ firstTokenAt: 0
1467
+ };
1468
+ this.queue.push(req);
1469
+ this.stats.submitted++;
1470
+ this.pump();
1471
+ });
1472
+ }
1473
+ /** Resolves when every submitted request has completed. */
1474
+ async drain() {
1475
+ while (this.loop) await this.loop;
1476
+ }
1477
+ /** Stop accepting requests; resolves when in-flight work drains. */
1478
+ async close() {
1479
+ this.closed = true;
1480
+ await this.drain();
1481
+ }
1482
+ getStats() {
1483
+ return { ...this.stats };
1484
+ }
1485
+ pump() {
1486
+ if (this.loop) return;
1487
+ this.loop = this.run().finally(() => {
1488
+ this.loop = null;
1489
+ if (this.hasWork()) this.pump();
1490
+ });
1491
+ }
1492
+ hasWork() {
1493
+ return this.queue.length > 0 || this.prefilling !== null || this.inFlight.length > 0 || this.slots.some((s) => s !== null);
1494
+ }
1495
+ activeMask() {
1496
+ const mask = new Uint8Array(this.batchSize);
1497
+ const rows = [];
1498
+ for (let b = 0; b < this.batchSize; b++) {
1499
+ const req = this.slots[b];
1500
+ if (req && req.state === "decoding") {
1501
+ mask[b] = 1;
1502
+ rows.push({
1503
+ slot: b,
1504
+ req
1505
+ });
1506
+ }
1507
+ }
1508
+ return {
1509
+ mask,
1510
+ rows
1511
+ };
1512
+ }
1513
+ /** The serving loop. Holds the engine generation lock while live. */
1514
+ async run() {
1515
+ const release = await this.host.acquireLock();
1516
+ const ex = this.host.executor;
1517
+ const loopStart = performance.now();
1518
+ ex.resetBatch();
1519
+ this.stepCounter = 0;
1520
+ this.lastLaneAdapterIds = null;
1521
+ try {
1522
+ while (this.hasWork()) {
1523
+ this.admit();
1524
+ const { mask, rows } = this.activeMask();
1525
+ if (rows.length > 0 && this.inFlight.length < PIPELINE_DEPTH) {
1526
+ const readbackSlot = this.stepCounter % PIPELINE_DEPTH;
1527
+ this.syncLaneAdapters();
1528
+ ex.submitBatchDecodeStepMasked(mask, readbackSlot);
1529
+ this.stepCounter++;
1530
+ this.inFlight.push({
1531
+ readbackSlot,
1532
+ rows
1533
+ });
1534
+ this.stats.stepsSubmitted++;
1535
+ if (rows.length > this.stats.peakActiveLanes) this.stats.peakActiveLanes = rows.length;
1536
+ }
1537
+ if (this.prefilling) await this.prefillChunkStep();
1538
+ const pipelineFull = this.inFlight.length >= PIPELINE_DEPTH;
1539
+ const idleLanes = rows.length === 0 && this.prefilling === null;
1540
+ if (this.inFlight.length > 0 && (pipelineFull || idleLanes)) await this.readStep();
1541
+ }
1542
+ } catch (err) {
1543
+ const error = err instanceof Error ? err : new Error(String(err));
1544
+ if (this.prefilling?.adapterDeltas) try {
1545
+ ex.clearRuntimeLoRA();
1546
+ } catch {}
1547
+ for (const req of [...this.queue, ...this.slots.filter((s) => s !== null)]) if (req.state !== "done") {
1548
+ req.state = "done";
1549
+ req.reject(error);
1550
+ }
1551
+ this.queue = [];
1552
+ this.slots.fill(null);
1553
+ this.prefilling = null;
1554
+ this.inFlight = [];
1555
+ this.reservedBlocks = 0;
1556
+ throw error;
1557
+ } finally {
1558
+ this.stats.busyMs += performance.now() - loopStart;
1559
+ release();
1560
+ }
1561
+ }
1562
+ /**
1563
+ * Upload the slot→adapter assignment for the step about to be submitted,
1564
+ * when it differs from the last uploaded one. Called immediately before
1565
+ * every decode-step submit: `queue.writeBuffer` is queue-ordered, so steps
1566
+ * already in flight keep the assignment they were submitted under, and a
1567
+ * retiring lane's slot can never leak its adapter into the slot's next
1568
+ * occupant — the next occupant's assignment is re-derived from `slots`
1569
+ * before its first decode step. Masked (free/prefilling) lanes are pinned
1570
+ * to bare base; their output is discarded regardless.
1571
+ */
1572
+ syncLaneAdapters() {
1573
+ if (!this.host.executor.batchLoraEnabled) return;
1574
+ const ids = new Int32Array(this.batchSize).fill(-1);
1575
+ let anyAdapter = false;
1576
+ for (let b = 0; b < this.batchSize; b++) {
1577
+ const req = this.slots[b];
1578
+ if (req && req.state === "decoding" && req.adapterId >= 0) {
1579
+ ids[b] = req.adapterId;
1580
+ anyAdapter = true;
1581
+ }
1582
+ }
1583
+ const last = this.lastLaneAdapterIds;
1584
+ if (last === null) {
1585
+ if (!anyAdapter) return;
1586
+ } else if (ids.every((id, b) => id === last[b])) return;
1587
+ this.host.executor.setBatchLaneAdapters(ids);
1588
+ this.lastLaneAdapterIds = ids;
1589
+ }
1590
+ /**
1591
+ * Admission: move the next queued request into a free slot when (a) a slot
1592
+ * is free, (b) no other prefill is staged (the single-sequence prefill path
1593
+ * stages SSM state in singleton buffers — one at a time), and (c) its
1594
+ * worst-case block need fits the unreserved KV pool with headroom.
1595
+ */
1596
+ admit() {
1597
+ if (this.prefilling || this.queue.length === 0) return;
1598
+ const slot = this.slots.indexOf(null);
1599
+ if (slot === -1) return;
1600
+ const req = this.queue[0];
1601
+ const worstTokens = req.promptIds.length + req.maxTokens + PIPELINE_DEPTH;
1602
+ const blocksNeeded = Math.ceil(worstTokens / this.blockSize);
1603
+ const totalBlocks = this.host.executor.kvTotalBlocks;
1604
+ if (this.reservedBlocks + blocksNeeded + this.blockHeadroom > totalBlocks) return;
1605
+ this.queue.shift();
1606
+ req.state = "prefilling";
1607
+ req.slot = slot;
1608
+ req.blocksReserved = blocksNeeded;
1609
+ req.admittedAt = performance.now();
1610
+ this.reservedBlocks += blocksNeeded;
1611
+ this.slots[slot] = req;
1612
+ this.prefilling = req;
1613
+ if (req.adapterDeltas) {
1614
+ const { applied } = this.host.executor.applyRuntimeLoRA(req.adapterDeltas);
1615
+ if (applied === 0) throw new Error(`RequestScheduler: adapter "${req.adapterName}" resolved to 0 applicable prefill targets`);
1616
+ }
1617
+ this.host.executor.beginBatchSlot(slot);
1618
+ }
1619
+ /**
1620
+ * Run one bounded prefill chunk for the staged request on the
1621
+ * single-sequence path (into its slot's block-table row via tableBase).
1622
+ * Intermediate chunks are submit-only; the final chunk reads logits, adopts
1623
+ * the singleton SSM state into the slot's pool row, seeds the lane's first
1624
+ * token, and flips the lane to decoding.
1625
+ */
1626
+ async prefillChunkStep() {
1627
+ const req = this.prefilling;
1628
+ if (!req) return;
1629
+ const ex = this.host.executor;
1630
+ const end = Math.min(req.prefillPos + this.prefillChunk, req.promptIds.length);
1631
+ const isLast = end === req.promptIds.length;
1632
+ const chunk = req.promptIds.subarray(req.prefillPos, end);
1633
+ const { logits } = await ex.forward(chunk, { readLogits: isLast });
1634
+ req.prefillPos = end;
1635
+ this.stats.prefillChunks++;
1636
+ if (!isLast) return;
1637
+ ex.finishBatchSlot(req.slot);
1638
+ if (req.adapterDeltas) ex.clearRuntimeLoRA();
1639
+ this.prefilling = null;
1640
+ const first = this.host.remapToken(sampleToken(logits, { temperature: 0 }));
1641
+ ex.injectBatchToken(req.slot, first);
1642
+ req.state = "decoding";
1643
+ req.firstTokenAt = performance.now();
1644
+ this.consumeToken(req, first);
1645
+ }
1646
+ /** Read the oldest in-flight step and deliver its tokens. */
1647
+ async readStep() {
1648
+ const step = this.inFlight.shift();
1649
+ if (!step) return;
1650
+ const tokens = await this.host.executor.readBatchTokens(step.readbackSlot);
1651
+ for (const { slot, req } of step.rows) {
1652
+ if (this.slots[slot] !== req || req.state !== "decoding") continue;
1653
+ this.consumeToken(req, tokens[slot]);
1654
+ }
1655
+ }
1656
+ /** Append one generated token; finish + retire the lane when terminal. */
1657
+ consumeToken(req, tokenId) {
1658
+ req.generatedIds.push(tokenId);
1659
+ this.stats.tokensGenerated++;
1660
+ if (req.firstTokenAt === 0) req.firstTokenAt = performance.now();
1661
+ const { eosTokenId, eotTokenId } = this.host;
1662
+ if (eosTokenId !== null && tokenId === eosTokenId || eotTokenId !== null && tokenId === eotTokenId) {
1663
+ req.finishReason = "eos";
1664
+ this.finish(req);
1665
+ return;
1666
+ }
1667
+ const piece = this.host.decodeToken(tokenId);
1668
+ req.text += piece;
1669
+ if (req.stopSequences.some((s) => req.text.includes(s))) {
1670
+ for (const s of req.stopSequences) {
1671
+ const idx = req.text.indexOf(s);
1672
+ if (idx !== -1) req.text = req.text.slice(0, idx);
1673
+ }
1674
+ req.finishReason = "stop_sequence";
1675
+ this.finish(req);
1676
+ return;
1677
+ }
1678
+ req.onToken?.(piece, {
1679
+ requestId: req.id,
1680
+ tokenId,
1681
+ generated: req.generatedIds.length
1682
+ });
1683
+ if (req.generatedIds.length >= req.maxTokens) {
1684
+ req.finishReason = "max_tokens";
1685
+ this.finish(req);
1686
+ }
1687
+ }
1688
+ /** Retire the lane immediately: free KV blocks, open the slot, resolve. */
1689
+ finish(req) {
1690
+ req.state = "done";
1691
+ if (req.slot >= 0 && this.slots[req.slot] === req) {
1692
+ this.host.executor.retireBatchSlot(req.slot);
1693
+ this.slots[req.slot] = null;
1694
+ }
1695
+ this.reservedBlocks -= req.blocksReserved;
1696
+ this.stats.completed++;
1697
+ const finishedAt = performance.now();
1698
+ req.resolve({
1699
+ requestId: req.id,
1700
+ text: req.text,
1701
+ tokenIds: [...req.generatedIds],
1702
+ tokensGenerated: req.generatedIds.length,
1703
+ finishReason: req.finishReason,
1704
+ adapter: req.adapterName ?? void 0,
1705
+ submittedAt: req.submittedAt,
1706
+ admittedAt: req.admittedAt,
1707
+ firstTokenAt: req.firstTokenAt,
1708
+ finishedAt,
1709
+ ttftMs: req.firstTokenAt - req.submittedAt,
1710
+ e2eMs: finishedAt - req.submittedAt
1711
+ });
1712
+ }
1713
+ };
1714
+
1715
+ //#endregion
1716
+ //#region src/gpu/batch/swarm.ts
1717
+ /** Default reduce: member-keyed map, disambiguating duplicated members. */
1718
+ function reduceToKeyedMap(results) {
1719
+ const out = {};
1720
+ const seen = /* @__PURE__ */ new Map();
1721
+ for (const { member, result } of results) {
1722
+ const n = (seen.get(member) ?? 0) + 1;
1723
+ seen.set(member, n);
1724
+ out[n === 1 ? member : `${member}#${n}`] = result;
1725
+ }
1726
+ return out;
1727
+ }
1728
+ /**
1729
+ * A swarm handle: registered members + plan/reduce over the engine's
1730
+ * scheduler. Create via {@link WebGPUEngine.createSwarm}.
1731
+ */
1732
+ var Swarm = class {
1733
+ scheduler;
1734
+ memberSources;
1735
+ plan;
1736
+ reduce;
1737
+ constructor(scheduler, options) {
1738
+ if (Object.keys(options.members).length === 0) throw new Error("Swarm: members must contain at least one member");
1739
+ this.scheduler = scheduler;
1740
+ this.memberSources = { ...options.members };
1741
+ this.plan = options.plan;
1742
+ this.reduce = options.reduce;
1743
+ }
1744
+ /** Member names, declaration order. */
1745
+ get members() {
1746
+ return Object.keys(this.memberSources);
1747
+ }
1748
+ /**
1749
+ * Run one input through the swarm: plan → concurrent fan-out (one batched
1750
+ * pass across the scheduler's lanes) → reduce.
1751
+ *
1752
+ * @param input The task input handed to the plan (default plan: broadcast
1753
+ * it to every member as the prompt).
1754
+ * @param options Per-request options applied to every sub-task (a task's
1755
+ * own `options` win field-by-field).
1756
+ */
1757
+ async run(input, options = {}) {
1758
+ const tasks = this.plan ? await this.plan(input, this.members) : this.members.map((member) => ({
1759
+ member,
1760
+ prompt: input
1761
+ }));
1762
+ if (tasks.length === 0) throw new Error("Swarm: plan produced no sub-tasks");
1763
+ const results = await Promise.all(tasks.map(async (task) => {
1764
+ const result = await this.submitFor(task.member, task.prompt, {
1765
+ ...options,
1766
+ ...task.options
1767
+ });
1768
+ return {
1769
+ member: task.member,
1770
+ result
1771
+ };
1772
+ }));
1773
+ if (this.reduce) return this.reduce(results);
1774
+ return reduceToKeyedMap(results);
1775
+ }
1776
+ /**
1777
+ * Mode-A access: decode one prompt with a single member (no plan/reduce).
1778
+ * Still rides the shared scheduler, so it batches with any concurrent work.
1779
+ */
1780
+ generate(member, prompt, options = {}) {
1781
+ return this.submitFor(member, prompt, options);
1782
+ }
1783
+ submitFor(member, prompt, options) {
1784
+ if (!Object.hasOwn(this.memberSources, member)) throw new Error(`Swarm: unknown member "${member}" (members: ${this.members.join(", ")})`);
1785
+ const adapter = this.memberSources[member] == null ? void 0 : member;
1786
+ return this.scheduler.submit(prompt, {
1787
+ ...options,
1788
+ adapter
1789
+ });
1790
+ }
1791
+ };
1792
+
1228
1793
  //#endregion
1229
1794
  //#region src/gpu/architectures/outetts.ts
1230
1795
  const OUTETTS_C1_BASE = 151669;
@@ -2859,7 +3424,7 @@ function selectGraphWeights$1(graph, weights) {
2859
3424
  return out;
2860
3425
  }
2861
3426
  /** Index of the maximum logit (greedy / argmax decode). */
2862
- function argmax$2(logits) {
3427
+ function argmax$1(logits) {
2863
3428
  let best = 0;
2864
3429
  let bestV = logits[0];
2865
3430
  for (let i = 1; i < logits.length; i++) if (logits[i] > bestV) {
@@ -3188,7 +3753,7 @@ var OuteTTS = class OuteTTS {
3188
3753
  const generated = [];
3189
3754
  let logits = prefillLogits;
3190
3755
  for (let step = 0; step < p.maxNewTokens; step++) {
3191
- const next = p.greedy ? argmax$2(logits) : this.sampleToken(logits, {
3756
+ const next = p.greedy ? argmax$1(logits) : this.sampleToken(logits, {
3192
3757
  temperature: p.temperature,
3193
3758
  topP: p.topP,
3194
3759
  topK: p.topK,
@@ -3320,7 +3885,7 @@ function selectGraphWeights(graph, weights) {
3320
3885
  return out;
3321
3886
  }
3322
3887
  /** Argmax over a logit row. */
3323
- function argmax$1(logits) {
3888
+ function argmax(logits) {
3324
3889
  let best = 0;
3325
3890
  let bestV = logits[0];
3326
3891
  for (let i = 1; i < logits.length; i++) if (logits[i] > bestV) {
@@ -3671,7 +4236,7 @@ var ParlerTTS = class ParlerTTS {
3671
4236
  continue;
3672
4237
  }
3673
4238
  if (cb > frontier) logitRows[cb][PARLER_EOS_TOKEN_ID] = Number.NEGATIVE_INFINITY;
3674
- const code = opts.greedy ? argmax$1(logitRows[cb]) : sampleLogits(logitRows[cb], opts.temperature ?? 1, opts.topK ?? 0);
4239
+ const code = opts.greedy ? argmax(logitRows[cb]) : sampleLogits(logitRows[cb], opts.temperature ?? 1, opts.topK ?? 0);
3675
4240
  nextCodes[cb] = code;
3676
4241
  grid[cb * maxSteps + step] = code < PARLER_CODEBOOK_SIZE ? code : 0;
3677
4242
  }
@@ -3742,122 +4307,6 @@ function addPosition(row, posTable, pos, H) {
3742
4307
  for (let i = 0; i < H; i++) row[i] += posTable[base + i];
3743
4308
  }
3744
4309
 
3745
- //#endregion
3746
- //#region src/gpu/sampler.ts
3747
- let _heapIndices = null;
3748
- let _heapValues = null;
3749
- /**
3750
- * Sample a token ID from logits.
3751
- *
3752
- * Pipeline: repetition penalty → temperature → top-k (min-heap) → softmax → top-p → sample.
3753
- */
3754
- function sampleToken(logits, params = {}, previousTokens) {
3755
- const temperature = params.temperature ?? .7;
3756
- const topK = params.topK ?? 50;
3757
- const topP = params.topP ?? .9;
3758
- const repetitionPenalty = params.repetitionPenalty ?? 1;
3759
- if (temperature < 1e-6) return argmax(logits);
3760
- const N = logits.length;
3761
- const K = Math.min(topK > 0 ? topK : N, N);
3762
- if (!_heapIndices || _heapIndices.length < K) {
3763
- _heapIndices = new Uint32Array(K);
3764
- _heapValues = new Float32Array(K);
3765
- }
3766
- const hIdx = _heapIndices;
3767
- const hVal = _heapValues;
3768
- let penaltySet = null;
3769
- if (repetitionPenalty !== 1 && previousTokens?.length) penaltySet = new Set(previousTokens);
3770
- let heapSize = 0;
3771
- for (let i = 0; i < N; i++) {
3772
- let s = logits[i];
3773
- if (penaltySet?.has(i)) s = s > 0 ? s / repetitionPenalty : s * repetitionPenalty;
3774
- s /= temperature;
3775
- if (heapSize < K) {
3776
- hIdx[heapSize] = i;
3777
- hVal[heapSize] = s;
3778
- heapSize++;
3779
- if (heapSize === K) for (let j = (K >> 1) - 1; j >= 0; j--) siftDown(hIdx, hVal, j, K);
3780
- } else if (s > hVal[0]) {
3781
- hIdx[0] = i;
3782
- hVal[0] = s;
3783
- siftDown(hIdx, hVal, 0, K);
3784
- }
3785
- }
3786
- for (let i = 1; i < heapSize; i++) {
3787
- const vi = hVal[i];
3788
- const ii = hIdx[i];
3789
- let j = i - 1;
3790
- while (j >= 0 && hVal[j] < vi) {
3791
- hVal[j + 1] = hVal[j];
3792
- hIdx[j + 1] = hIdx[j];
3793
- j--;
3794
- }
3795
- hVal[j + 1] = vi;
3796
- hIdx[j + 1] = ii;
3797
- }
3798
- const maxScore = hVal[0];
3799
- let sumExp = 0;
3800
- for (let i = 0; i < heapSize; i++) {
3801
- const p = Math.exp(hVal[i] - maxScore);
3802
- hVal[i] = p;
3803
- sumExp += p;
3804
- }
3805
- const invSum = 1 / sumExp;
3806
- for (let i = 0; i < heapSize; i++) hVal[i] *= invSum;
3807
- let candidateCount = heapSize;
3808
- if (topP < 1) {
3809
- let cumulative$1 = 0;
3810
- for (let i = 0; i < heapSize; i++) {
3811
- cumulative$1 += hVal[i];
3812
- if (cumulative$1 >= topP) {
3813
- candidateCount = i + 1;
3814
- break;
3815
- }
3816
- }
3817
- let sum = 0;
3818
- for (let i = 0; i < candidateCount; i++) sum += hVal[i];
3819
- const inv = 1 / sum;
3820
- for (let i = 0; i < candidateCount; i++) hVal[i] *= inv;
3821
- }
3822
- const r = Math.random();
3823
- let cumulative = 0;
3824
- for (let i = 0; i < candidateCount; i++) {
3825
- cumulative += hVal[i];
3826
- if (r <= cumulative) return hIdx[i];
3827
- }
3828
- return hIdx[candidateCount - 1];
3829
- }
3830
- /**
3831
- * Return the index of the maximum value (greedy decoding).
3832
- */
3833
- function argmax(arr) {
3834
- let maxIdx = 0;
3835
- let maxVal = arr[0];
3836
- for (let i = 1; i < arr.length; i++) if (arr[i] > maxVal) {
3837
- maxVal = arr[i];
3838
- maxIdx = i;
3839
- }
3840
- return maxIdx;
3841
- }
3842
- /** Min-heap sift down on parallel index/value typed arrays. */
3843
- function siftDown(indices, values, i, n) {
3844
- while (true) {
3845
- let smallest = i;
3846
- const left = 2 * i + 1;
3847
- const right = 2 * i + 2;
3848
- if (left < n && values[left] < values[smallest]) smallest = left;
3849
- if (right < n && values[right] < values[smallest]) smallest = right;
3850
- if (smallest === i) break;
3851
- const ti = indices[i];
3852
- indices[i] = indices[smallest];
3853
- indices[smallest] = ti;
3854
- const tv = values[i];
3855
- values[i] = values[smallest];
3856
- values[smallest] = tv;
3857
- i = smallest;
3858
- }
3859
- }
3860
-
3861
4310
  //#endregion
3862
4311
  //#region src/gpu/vision-executor.ts
3863
4312
  const MAP_MODE_READ = 1;
@@ -5143,8 +5592,12 @@ var WebGPUEngine = class WebGPUEngine {
5143
5592
  _outeVoice = null;
5144
5593
  /** Source of the runtime LoRA adapter currently applied on the static base, if any. */
5145
5594
  _currentAdapter = null;
5595
+ /** Named adapters registered for per-lane batched decode (GERBIL_BATCH_LORA). */
5596
+ _batchAdapters = /* @__PURE__ */ new Map();
5146
5597
  /** Lazily-created Parler-TTS engine (Flan-T5 encoder + decoder LM + dac_44khz). */
5147
5598
  _parlerTTS = null;
5599
+ /** Lazily-created continuous-batching scheduler (GERBIL_BATCH, Phase 5). */
5600
+ _scheduler = null;
5148
5601
  /**
5149
5602
  * WebKit group-size probe state. When true, a candidate group size is being
5150
5603
  * tried this page-load and must be promoted (or capped) after the FIRST
@@ -5250,6 +5703,35 @@ var WebGPUEngine = class WebGPUEngine {
5250
5703
  return this._currentAdapter;
5251
5704
  }
5252
5705
  /**
5706
+ * Register a named LoRA adapter for per-lane batched decode (SGMV phase 1,
5707
+ * `GERBIL_BATCH=N` + `GERBIL_BATCH_LORA=1`). The adapter's factors are
5708
+ * fetched and packed into the executor's shared factor buffers ONCE; any
5709
+ * number of batch lanes can then run it concurrently by passing its name in
5710
+ * `generateBatch`'s per-lane `adapters` option. Registering the same name
5711
+ * twice is a no-op.
5712
+ */
5713
+ async registerAdapter(name, source) {
5714
+ this.checkDestroyed();
5715
+ if (!this.executor.batchLoraEnabled) throw new Error("registerAdapter requires GERBIL_BATCH=N (N >= 2) and GERBIL_BATCH_LORA=1 on the Dawn path");
5716
+ if (this._batchAdapters.has(name)) return;
5717
+ const adapter = await fetchAdapter(source, {
5718
+ hfToken: this._createOptions.hfToken,
5719
+ revision: this._createOptions.revision
5720
+ });
5721
+ if (!adapter) throw new Error(`No adapter found at ${source}.`);
5722
+ const deltas = buildLoRADeltas(adapter, createKeyMapperForArch(this._architecture));
5723
+ const id = this.executor.registerBatchAdapter(deltas);
5724
+ this._batchAdapters.set(name, {
5725
+ id,
5726
+ deltas
5727
+ });
5728
+ console.log(`[engine] batch adapter "${name}" (${source}) registered as id ${id}.`);
5729
+ }
5730
+ /** Names of the adapters registered for per-lane batched decode. */
5731
+ getRegisteredAdapters() {
5732
+ return [...this._batchAdapters.keys()];
5733
+ }
5734
+ /**
5253
5735
  * Write a coarse crash-phase breadcrumb that survives a GPU-process kill / page
5254
5736
  * reload. The iPad harness reads `localStorage["gerbil-crash-phase"]` after a
5255
5737
  * crash; without these, a describe-time crash only shows the last load phase
@@ -5580,6 +6062,186 @@ var WebGPUEngine = class WebGPUEngine {
5580
6062
  }
5581
6063
  }
5582
6064
  /**
6065
+ * Batched lockstep generation (GERBIL_BATCH=N — Phase 2 of the
6066
+ * continuous-batching campaign, Dawn only).
6067
+ *
6068
+ * Prefills each prompt sequentially into its batch slot (KV into the slot's
6069
+ * block-table row, SSM/conv state adopted into the slot's state-pool rows),
6070
+ * then decodes all N sequences in lockstep: one batched dispatch stream per
6071
+ * step produces one token per row. Greedy-only in Phase 2 (per-row GPU
6072
+ * argmax); rows that hit EOS/stop keep decoding in lockstep with their
6073
+ * output discarded until every row has finished — rows never interact, so
6074
+ * per-row streams are token-exact vs single-sequence generation.
6075
+ *
6076
+ * Requires exactly `executor.batchSize` prompts (fixed lockstep batch; the
6077
+ * Phase 3 scheduler will lift this).
6078
+ */
6079
+ async generateBatch(prompts, options = {}) {
6080
+ this.checkDestroyed();
6081
+ const B = this.executor.batchSize;
6082
+ if (B <= 0) throw new Error("generateBatch requires GERBIL_BATCH=N (N >= 2) on the Dawn path");
6083
+ if (prompts.length !== B) throw new Error(`generateBatch: got ${prompts.length} prompts for a fixed batch of ${B} (set GERBIL_BATCH=${prompts.length})`);
6084
+ if (this._multimodalGraph || this.executor.hasPleSource() || this.executor.hasRuntimeAdapter) throw new Error("generateBatch: multimodal, PLE, and runtime-LoRA models are Phase 2+ work");
6085
+ const { maxTokens = 512, stopSequences = [], sampling = {}, systemPrompt } = options;
6086
+ if ((sampling.temperature ?? .7) >= 1e-6) throw new Error("generateBatch: Phase 2 is greedy-only — pass sampling: { temperature: 0 }");
6087
+ const laneAdapterNames = options.adapters;
6088
+ if (laneAdapterNames && laneAdapterNames.length !== B) throw new Error(`generateBatch: got ${laneAdapterNames.length} lane adapters for a batch of ${B}`);
6089
+ if ((laneAdapterNames?.some((a) => a != null) ?? false) && !this.executor.batchLoraEnabled) throw new Error("generateBatch: per-lane adapters require GERBIL_BATCH_LORA=1");
6090
+ const laneAdapters = new Array(B).fill(null);
6091
+ if (laneAdapterNames) for (let i = 0; i < B; i++) {
6092
+ const name = laneAdapterNames[i];
6093
+ if (name == null) continue;
6094
+ const reg = this._batchAdapters.get(name);
6095
+ if (!reg) throw new Error(`generateBatch: adapter "${name}" not registered (call registerAdapter first)`);
6096
+ laneAdapters[i] = reg;
6097
+ }
6098
+ const release = await this._acquireGenLock();
6099
+ try {
6100
+ const startTime = performance.now();
6101
+ this.executor.resetBatch();
6102
+ const eosId = this.tokenizer.config.eosTokenId;
6103
+ const eotId = this.resolveEndOfTurnId();
6104
+ const keepPlan = this.config.vocabKeepPlan;
6105
+ const remapTok = keepPlan ? (i) => remapPrunedToken(i, keepPlan) : (i) => i;
6106
+ const rows = prompts.map((_p, i) => ({
6107
+ generatedIds: [],
6108
+ text: "",
6109
+ finished: false,
6110
+ finishReason: "max_tokens",
6111
+ limit: options.maxTokensPerRow?.[i] ?? maxTokens
6112
+ }));
6113
+ const firstTokens = new Uint32Array(B);
6114
+ for (let i = 0; i < B; i++) {
6115
+ const p = prompts[i];
6116
+ const messages = typeof p === "string" ? [...systemPrompt ? [{
6117
+ role: "system",
6118
+ content: systemPrompt
6119
+ }] : [], {
6120
+ role: "user",
6121
+ content: p
6122
+ }] : systemPrompt ? [{
6123
+ role: "system",
6124
+ content: systemPrompt
6125
+ }, ...p] : p;
6126
+ const inputIds = this.tokenizer.encodeChat(messages, { addGenerationPrompt: true });
6127
+ const laneReg = laneAdapters[i];
6128
+ if (laneReg) this.executor.applyRuntimeLoRA(laneReg.deltas);
6129
+ try {
6130
+ this.executor.beginBatchSlot(i);
6131
+ const { logits } = await this.executor.forward(new Uint32Array(inputIds));
6132
+ this.executor.finishBatchSlot(i);
6133
+ firstTokens[i] = remapTok(sampleToken(logits, { temperature: 0 }));
6134
+ } finally {
6135
+ if (laneReg) this.executor.clearRuntimeLoRA();
6136
+ }
6137
+ }
6138
+ if (this.executor.batchLoraEnabled) this.executor.setBatchLaneAdapters(laneAdapters.map((a) => a ? a.id : -1));
6139
+ const consume = (row, tok) => {
6140
+ if (row.finished) return;
6141
+ row.generatedIds.push(tok);
6142
+ if (eosId !== null && tok === eosId || eotId !== null && tok === eotId) {
6143
+ row.finished = true;
6144
+ row.finishReason = "eos";
6145
+ return;
6146
+ }
6147
+ row.text += this.tokenizer.decode([tok], true);
6148
+ if (stopSequences.some((s) => row.text.includes(s))) {
6149
+ for (const s of stopSequences) {
6150
+ const idx = row.text.indexOf(s);
6151
+ if (idx !== -1) row.text = row.text.slice(0, idx);
6152
+ }
6153
+ row.finished = true;
6154
+ row.finishReason = "stop_sequence";
6155
+ return;
6156
+ }
6157
+ if (row.generatedIds.length >= row.limit) row.finished = true;
6158
+ };
6159
+ const decodeStart = performance.now();
6160
+ for (let i = 0; i < B; i++) consume(rows[i], firstTokens[i]);
6161
+ const depth = Executor.PIPELINE_DEPTH;
6162
+ const lens = this.executor.batchSlotLengths;
6163
+ let capacity = Number.POSITIVE_INFINITY;
6164
+ for (let i = 0; i < B; i++) capacity = Math.min(capacity, this.maxSeqLen - lens[i] - 1);
6165
+ const maxRowLimit = rows.reduce((a, r) => Math.max(a, r.limit), 0);
6166
+ const stepsNeeded = Math.max(0, Math.min(maxRowLimit - 1, capacity));
6167
+ let submitted = 0;
6168
+ let consumedSteps = 0;
6169
+ while (consumedSteps < stepsNeeded && !rows.every((r) => r.finished)) {
6170
+ while (submitted < stepsNeeded && submitted < consumedSteps + depth) {
6171
+ this.executor.submitBatchDecodeStep(submitted === 0 ? firstTokens : null, submitted % depth);
6172
+ submitted++;
6173
+ }
6174
+ const toks = await this.executor.readBatchTokens(consumedSteps % depth);
6175
+ consumedSteps++;
6176
+ for (let i = 0; i < B; i++) consume(rows[i], toks[i]);
6177
+ }
6178
+ const decodeTime = performance.now() - decodeStart;
6179
+ const totalTime = performance.now() - startTime;
6180
+ return rows.map((r) => ({
6181
+ text: r.text,
6182
+ tokensGenerated: r.generatedIds.length,
6183
+ tokensPerSecond: r.generatedIds.length / (totalTime / 1e3),
6184
+ totalTime,
6185
+ decodeTime,
6186
+ finishReason: r.finishReason,
6187
+ tokenIds: [...r.generatedIds]
6188
+ }));
6189
+ } finally {
6190
+ release();
6191
+ }
6192
+ }
6193
+ /**
6194
+ * Continuous-batching scheduler (GERBIL_BATCH=N — Phase 5 of the batching
6195
+ * campaign, Dawn only). Returns the engine's request scheduler (created on
6196
+ * first call): submit many requests with independent prompts/limits and the
6197
+ * scheduler multiplexes them over the batch lanes with dynamic admission,
6198
+ * immediate lane retirement, and chunked prefill. Greedy-only; per-request
6199
+ * output is token-exact vs single-sequence generate().
6200
+ */
6201
+ createScheduler() {
6202
+ this.checkDestroyed();
6203
+ if (this._scheduler) return this._scheduler;
6204
+ if (this.executor.batchSize <= 0) throw new Error("createScheduler requires GERBIL_BATCH=N (N >= 2) on the Dawn path");
6205
+ if (this._multimodalGraph || this.executor.hasPleSource() || this.executor.hasRuntimeAdapter) throw new Error("createScheduler: multimodal, PLE, and runtime-LoRA models are unsupported");
6206
+ const keepPlan = this.config.vocabKeepPlan;
6207
+ this._scheduler = new RequestScheduler({
6208
+ executor: this.executor,
6209
+ encodeChat: (messages) => this.tokenizer.encodeChat(messages, { addGenerationPrompt: true }),
6210
+ decodeToken: (id) => this.tokenizer.decode([id], true),
6211
+ eosTokenId: this.tokenizer.config.eosTokenId,
6212
+ eotTokenId: this.resolveEndOfTurnId(),
6213
+ remapToken: keepPlan ? (i) => remapPrunedToken(i, keepPlan) : (i) => i,
6214
+ resolveAdapter: (name) => {
6215
+ const reg = this._batchAdapters.get(name);
6216
+ if (!reg) throw new Error(`scheduler: adapter "${name}" not registered (call registerAdapter first)`);
6217
+ return reg;
6218
+ },
6219
+ acquireLock: () => this._acquireGenLock()
6220
+ });
6221
+ return this._scheduler;
6222
+ }
6223
+ /**
6224
+ * Create a swarm — the MoTA mode-B primitive (docs/research/slm-swarms.md):
6225
+ * named members (LoRA adapters on this engine's shared base, or null for the
6226
+ * bare base), an optional plan that fans an input out into per-member
6227
+ * sub-tasks, and an optional reduce that joins the results. `swarm.run()`
6228
+ * decodes every sub-task CONCURRENTLY through the continuous-batching
6229
+ * scheduler with per-request adapters — one batched pass, so the fan-out
6230
+ * costs roughly the slowest member, not the sum.
6231
+ *
6232
+ * Every non-null member source is registered via {@link registerAdapter}
6233
+ * (downloads in parallel; factors packed into shared GPU buffers once).
6234
+ * Requires `GERBIL_BATCH=N` (N >= 2) + `GERBIL_BATCH_LORA=1` on the Dawn
6235
+ * path; greedy-only like the scheduler underneath.
6236
+ */
6237
+ async createSwarm(options) {
6238
+ this.checkDestroyed();
6239
+ if (!this.executor.batchLoraEnabled) throw new Error("createSwarm requires GERBIL_BATCH=N (N >= 2) and GERBIL_BATCH_LORA=1 on the Dawn path");
6240
+ const scheduler = this.createScheduler();
6241
+ await Promise.all(Object.entries(options.members).map(([name, source]) => source == null ? Promise.resolve() : this.registerAdapter(name, source)));
6242
+ return new Swarm(scheduler, options);
6243
+ }
6244
+ /**
5583
6245
  * Resolve the end-of-turn stop token id, or null if the model has none.
5584
6246
  *
5585
6247
  * Chat models like Gemma 4 end an assistant turn with a dedicated end-of-turn
@@ -5665,18 +6327,34 @@ var WebGPUEngine = class WebGPUEngine {
5665
6327
  if (isGreedy && !this.executor.needsMultiEncoder && !mmDecode && !streamsPle) {
5666
6328
  const firstToken = remapTok(sampleToken(logits, sampling, [...inputIds, ...generatedIds]));
5667
6329
  if (!consumeToken(firstToken)) {
5668
- const depth = Executor.PIPELINE_DEPTH;
5669
6330
  const stepsNeeded = Math.min(maxTokens - 1, this.executor.decodeCapacityRemaining());
5670
- let submitted = 0;
5671
- let consumed = 0;
5672
- while (consumed < stepsNeeded) {
5673
- while (submitted < stepsNeeded && submitted < consumed + depth) {
5674
- this.executor.submitGreedyDecodeStep(submitted === 0 ? firstToken : null, submitted % depth);
5675
- submitted++;
6331
+ const windowK = this.executor.decodeWindowK;
6332
+ if (windowK > 0) {
6333
+ let produced = 0;
6334
+ let stopped = false;
6335
+ while (!stopped && produced < stepsNeeded) {
6336
+ const steps = Math.min(windowK, stepsNeeded - produced);
6337
+ this.executor.submitGreedyDecodeWindow(produced === 0 ? firstToken : null, steps, 0);
6338
+ const tokens = await this.executor.readDecodeWindow(0, steps);
6339
+ produced += steps;
6340
+ for (const tok of tokens) if (consumeToken(tok)) {
6341
+ stopped = true;
6342
+ break;
6343
+ }
6344
+ }
6345
+ } else {
6346
+ const depth = Executor.PIPELINE_DEPTH;
6347
+ let submitted = 0;
6348
+ let consumed = 0;
6349
+ while (consumed < stepsNeeded) {
6350
+ while (submitted < stepsNeeded && submitted < consumed + depth) {
6351
+ this.executor.submitGreedyDecodeStep(submitted === 0 ? firstToken : null, submitted % depth);
6352
+ submitted++;
6353
+ }
6354
+ const tok = await this.executor.readDecodeToken(consumed % depth);
6355
+ consumed++;
6356
+ if (consumeToken(tok)) break;
5676
6357
  }
5677
- const tok = await this.executor.readDecodeToken(consumed % depth);
5678
- consumed++;
5679
- if (consumeToken(tok)) break;
5680
6358
  }
5681
6359
  }
5682
6360
  } else {
@@ -5705,7 +6383,8 @@ var WebGPUEngine = class WebGPUEngine {
5705
6383
  tokensGenerated,
5706
6384
  tokensPerSecond,
5707
6385
  totalTime,
5708
- finishReason
6386
+ finishReason,
6387
+ tokenIds: [...generatedIds]
5709
6388
  };
5710
6389
  }
5711
6390
  /**
@@ -6214,7 +6893,8 @@ var WebGPUEngine = class WebGPUEngine {
6214
6893
  tokensGenerated,
6215
6894
  tokensPerSecond: tokensGenerated / (totalTime / 1e3),
6216
6895
  totalTime,
6217
- finishReason
6896
+ finishReason,
6897
+ tokenIds: [...generatedIds]
6218
6898
  };
6219
6899
  }
6220
6900
  /** Prepare + prefill + decode for a fully-specified multimodal token sequence. */
@@ -6299,7 +6979,8 @@ var WebGPUEngine = class WebGPUEngine {
6299
6979
  tokensGenerated,
6300
6980
  tokensPerSecond: tokensGenerated / (totalTime / 1e3),
6301
6981
  totalTime,
6302
- finishReason
6982
+ finishReason,
6983
+ tokenIds: [...generatedIds]
6303
6984
  };
6304
6985
  }
6305
6986
  /**
@@ -6780,5 +7461,5 @@ var WebGPUEngine = class WebGPUEngine {
6780
7461
  };
6781
7462
 
6782
7463
  //#endregion
6783
- export { dequantizeGemma4VisionProjection as A, audioTokensToDacCodes as C, parseOuteTtsConfig as D, generateOuteTtsBackboneGraph as E, generateGemma4VisionGraph as M, patchGemma4VisionClips as N, KaniTTS as O, resolveGemma4VisionInfo as P, loadOuteSpeaker as S, generateDacSpeechDecoderGraph as T, smartResize as _, buildGemma4PosEmbeds as a, OuteTTS as b, buildMRoPECosSin as c, buildPositionIds as d, buildRotaryCosSin as f, preprocessImageGemma4 as g, preprocessImage as h, buildGemma4PoolMatrix as i, dequantizeMLXProjection as j, generateQwen3_5VisionGraph as k, buildMRoPEPositionIds as l, mropeFreqDims as m, GEMMA4_IMAGE_PROCESSOR as n, buildGemma4RotaryCosSin as o, buildVisionPositionTensors as p, QWEN3_5_IMAGE_PROCESSOR as r, buildGemma4VisionPositionTensors as s, WebGPUEngine as t, buildPosEmbeds as u, VisionExecutor as v, dacOutputLength as w, buildOutePromptString as x, ParlerTTS as y };
6784
- //# sourceMappingURL=gpu-CrzjQHv2.mjs.map
7464
+ export { KaniTTS as A, audioTokensToDacCodes as C, parseOuteTtsConfig as D, generateOuteTtsBackboneGraph as E, patchGemma4VisionClips as F, resolveGemma4VisionInfo as I, dequantizeGemma4VisionProjection as M, dequantizeMLXProjection as N, Swarm as O, generateGemma4VisionGraph as P, loadOuteSpeaker as S, generateDacSpeechDecoderGraph as T, smartResize as _, buildGemma4PosEmbeds as a, OuteTTS as b, buildMRoPECosSin as c, buildPositionIds as d, buildRotaryCosSin as f, preprocessImageGemma4 as g, preprocessImage as h, buildGemma4PoolMatrix as i, generateQwen3_5VisionGraph as j, RequestScheduler as k, buildMRoPEPositionIds as l, mropeFreqDims as m, GEMMA4_IMAGE_PROCESSOR as n, buildGemma4RotaryCosSin as o, buildVisionPositionTensors as p, QWEN3_5_IMAGE_PROCESSOR as r, buildGemma4VisionPositionTensors as s, WebGPUEngine as t, buildPosEmbeds as u, VisionExecutor as v, dacOutputLength as w, buildOutePromptString as x, ParlerTTS as y };
7465
+ //# sourceMappingURL=gpu-DRFhiv4R.mjs.map