@tryhamster/gerbil 1.11.4 → 1.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -0
- package/dist/{architectures-DmZMEFsA.mjs → architectures-BH_z6k9d.mjs} +14 -12
- package/dist/architectures-BH_z6k9d.mjs.map +1 -0
- package/dist/browser/index.d.ts.map +1 -1
- package/dist/browser/index.js +11 -0
- package/dist/browser/index.js.map +1 -1
- package/dist/cli.mjs +8 -8
- package/dist/cli.mjs.map +1 -1
- package/dist/frameworks/express.mjs +1 -1
- package/dist/frameworks/fastify.mjs +1 -1
- package/dist/frameworks/hono.mjs +1 -1
- package/dist/frameworks/next.d.mts +2 -2
- package/dist/frameworks/next.mjs +1 -1
- package/dist/frameworks/trpc.mjs +1 -1
- package/dist/gerbil-CZvoFo0T.mjs +4 -0
- package/dist/{gerbil-CYVmU8sQ.mjs → gerbil-CpS3P240.mjs} +13 -2
- package/dist/gerbil-CpS3P240.mjs.map +1 -0
- package/dist/{gerbil-5_80K0gA.d.mts → gerbil-DF009Waa.d.mts} +2 -2
- package/dist/{gerbil-5_80K0gA.d.mts.map → gerbil-DF009Waa.d.mts.map} +1 -1
- package/dist/gpu/hooks.d.mts +1 -1
- package/dist/gpu/index.d.mts +2 -2
- package/dist/gpu/index.mjs +4 -4
- package/dist/{gpu-CrzjQHv2.mjs → gpu-DRFhiv4R.mjs} +818 -137
- package/dist/gpu-DRFhiv4R.mjs.map +1 -0
- package/dist/{index-h8TDu1qm.d.mts → index-CNYoTRgr.d.mts} +1187 -550
- package/dist/index-CNYoTRgr.d.mts.map +1 -0
- package/dist/index.d.mts +3 -3
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +5 -5
- package/dist/index.mjs.map +1 -1
- package/dist/integrations/ai-sdk.mjs +1 -1
- package/dist/integrations/langchain.mjs +1 -1
- package/dist/integrations/llamaindex.mjs +1 -1
- package/dist/integrations/mcp.d.mts +2 -2
- package/dist/integrations/mcp.mjs +4 -4
- package/dist/{mcp-CAsD7eCj.mjs → mcp-DAbWO8VS.mjs} +3 -3
- package/dist/{mcp-CAsD7eCj.mjs.map → mcp-DAbWO8VS.mjs.map} +1 -1
- package/dist/{moonshine-stt-BXoZaHJE.mjs → moonshine-stt-B1kV5n1c.mjs} +5251 -1678
- package/dist/moonshine-stt-B1kV5n1c.mjs.map +1 -0
- package/dist/moonshine-stt-DA1WuiZb.mjs +4 -0
- package/dist/{one-liner-ppgw4jHH.mjs → one-liner-CmP9ktUn.mjs} +2 -2
- package/dist/{one-liner-ppgw4jHH.mjs.map → one-liner-CmP9ktUn.mjs.map} +1 -1
- package/dist/repl-17NVeIaJ.mjs +9 -0
- package/dist/skills/index.d.mts +4 -4
- package/dist/skills/index.mjs +3 -3
- package/dist/{skills-CQa1Gshd.mjs → skills-BDOEbHSx.mjs} +2 -2
- package/dist/{skills-CQa1Gshd.mjs.map → skills-BDOEbHSx.mjs.map} +1 -1
- package/dist/tune/index.mjs +1 -1
- package/package.json +1 -1
- package/dist/architectures-DmZMEFsA.mjs.map +0 -1
- package/dist/gerbil-CTefAwKp.mjs +0 -4
- package/dist/gerbil-CYVmU8sQ.mjs.map +0 -1
- package/dist/gpu-CrzjQHv2.mjs.map +0 -1
- package/dist/index-h8TDu1qm.d.mts.map +0 -1
- package/dist/moonshine-stt-BXoZaHJE.mjs.map +0 -1
- package/dist/moonshine-stt-DZVnKgPO.mjs +0 -4
- package/dist/repl-C0Ew7_Z-.mjs +0 -9
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { a as resolveDefaultRepo, i as isTTSRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "./defaults-DfGx4d1m.mjs";
|
|
2
|
-
import { A as kaniSinTensor, C as computeKaniPositions, D as kaniAttentionLayerIndices, E as generateNanoCodecDecoderGraph, F as CANONICAL_KEYS, I as DTYPE_BYTES, L as GEMMA4_VIS_KEYS, M as parseKaniConfig, N as DEFAULT_GROUP_SIZE, O as kaniCosTensor, S as buildKaniLayerCosSin, T as generateKaniTtsGraph, a as PARLER_DAC_LATENT_DIM, b as KANI_START_OF_HUMAN, c as PARLER_SAMPLE_RATE, d as generateParlerEncoderGraph, f as parseParlerConfig, i as PARLER_DAC_DECODER_DIM, k as kaniLayerAlpha, l as buildT5RelativeBias, o as PARLER_DECODER_RATES, p as revertDelayPattern, r as PARLER_BOS_TOKEN_ID, s as PARLER_EOS_TOKEN_ID, u as generateParlerDecoderGraph, x as audioTokensToCodes, y as KANI_END_OF_HUMAN } from "./architectures-
|
|
3
|
-
import { C as createUniformBuffer, D as verifyGPU, E as initGPU, S as createStorageBuffer, T as getOrCreatePipeline, a as loadModel, b as clearPipelineCache, c as loadParlerTTS, d as remapPrunedToken, g as fetchAdapter, h as buildLoRADeltas, i as loadKaniTTS, l as quantizeBackboneInt4, p as Executor, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as KERNEL_REGISTRY, w as destroyBuffers, x as createBindGroup, y as MATMUL_BIAS_F16C_SPEC } from "./moonshine-stt-
|
|
2
|
+
import { A as kaniSinTensor, C as computeKaniPositions, D as kaniAttentionLayerIndices, E as generateNanoCodecDecoderGraph, F as CANONICAL_KEYS, I as DTYPE_BYTES, L as GEMMA4_VIS_KEYS, M as parseKaniConfig, N as DEFAULT_GROUP_SIZE, O as kaniCosTensor, S as buildKaniLayerCosSin, T as generateKaniTtsGraph, a as PARLER_DAC_LATENT_DIM, b as KANI_START_OF_HUMAN, c as PARLER_SAMPLE_RATE, d as generateParlerEncoderGraph, f as parseParlerConfig, i as PARLER_DAC_DECODER_DIM, k as kaniLayerAlpha, l as buildT5RelativeBias, o as PARLER_DECODER_RATES, p as revertDelayPattern, r as PARLER_BOS_TOKEN_ID, s as PARLER_EOS_TOKEN_ID, u as generateParlerDecoderGraph, x as audioTokensToCodes, y as KANI_END_OF_HUMAN } from "./architectures-BH_z6k9d.mjs";
|
|
3
|
+
import { C as createUniformBuffer, D as verifyGPU, E as initGPU, S as createStorageBuffer, T as getOrCreatePipeline, a as loadModel, b as clearPipelineCache, c as loadParlerTTS, d as remapPrunedToken, g as fetchAdapter, h as buildLoRADeltas, i as loadKaniTTS, l as quantizeBackboneInt4, p as Executor, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as KERNEL_REGISTRY, w as destroyBuffers, x as createBindGroup, y as MATMUL_BIAS_F16C_SPEC } from "./moonshine-stt-B1kV5n1c.mjs";
|
|
4
4
|
|
|
5
5
|
//#region src/gpu/architectures/gemma4_vision.ts
|
|
6
6
|
/**
|
|
@@ -1225,6 +1225,571 @@ function cleanSuggestion(raw, typed, options = {}) {
|
|
|
1225
1225
|
return startsWithPunct || typedEndsWithSpace ? s : ` ${s}`;
|
|
1226
1226
|
}
|
|
1227
1227
|
|
|
1228
|
+
//#endregion
|
|
1229
|
+
//#region src/gpu/sampler.ts
|
|
1230
|
+
/** mulberry32 — tiny deterministic PRNG for seeded sampling. */
|
|
1231
|
+
function mulberry32(seed) {
|
|
1232
|
+
let a = seed >>> 0;
|
|
1233
|
+
return () => {
|
|
1234
|
+
a = a + 1831565813 | 0;
|
|
1235
|
+
let t = Math.imul(a ^ a >>> 15, 1 | a);
|
|
1236
|
+
t = t + Math.imul(t ^ t >>> 7, 61 | t) ^ t;
|
|
1237
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
1238
|
+
};
|
|
1239
|
+
}
|
|
1240
|
+
/** Per-params seeded RNG streams (one stream per options object identity). */
|
|
1241
|
+
const seededStreams = /* @__PURE__ */ new WeakMap();
|
|
1242
|
+
function nextRandom(params) {
|
|
1243
|
+
if (params.seed === void 0) return Math.random();
|
|
1244
|
+
let rng = seededStreams.get(params);
|
|
1245
|
+
if (!rng) {
|
|
1246
|
+
rng = mulberry32(params.seed);
|
|
1247
|
+
seededStreams.set(params, rng);
|
|
1248
|
+
}
|
|
1249
|
+
return rng();
|
|
1250
|
+
}
|
|
1251
|
+
let _heapIndices = null;
|
|
1252
|
+
let _heapValues = null;
|
|
1253
|
+
/**
|
|
1254
|
+
* Sample a token ID from logits.
|
|
1255
|
+
*
|
|
1256
|
+
* Pipeline: repetition penalty → temperature → top-k (min-heap) → softmax → top-p → sample.
|
|
1257
|
+
*/
|
|
1258
|
+
function sampleToken(logits, params = {}, previousTokens) {
|
|
1259
|
+
const temperature = params.temperature ?? .7;
|
|
1260
|
+
const topK = params.topK ?? 50;
|
|
1261
|
+
const topP = params.topP ?? .9;
|
|
1262
|
+
const repetitionPenalty = params.repetitionPenalty ?? 1;
|
|
1263
|
+
if (temperature < 1e-6) return argmax$2(logits);
|
|
1264
|
+
const N = logits.length;
|
|
1265
|
+
const K = Math.min(topK > 0 ? topK : N, N);
|
|
1266
|
+
if (!_heapIndices || _heapIndices.length < K) {
|
|
1267
|
+
_heapIndices = new Uint32Array(K);
|
|
1268
|
+
_heapValues = new Float32Array(K);
|
|
1269
|
+
}
|
|
1270
|
+
const hIdx = _heapIndices;
|
|
1271
|
+
const hVal = _heapValues;
|
|
1272
|
+
let penaltySet = null;
|
|
1273
|
+
if (repetitionPenalty !== 1 && previousTokens?.length) penaltySet = new Set(previousTokens);
|
|
1274
|
+
let heapSize = 0;
|
|
1275
|
+
for (let i = 0; i < N; i++) {
|
|
1276
|
+
let s = logits[i];
|
|
1277
|
+
if (penaltySet?.has(i)) s = s > 0 ? s / repetitionPenalty : s * repetitionPenalty;
|
|
1278
|
+
s /= temperature;
|
|
1279
|
+
if (heapSize < K) {
|
|
1280
|
+
hIdx[heapSize] = i;
|
|
1281
|
+
hVal[heapSize] = s;
|
|
1282
|
+
heapSize++;
|
|
1283
|
+
if (heapSize === K) for (let j = (K >> 1) - 1; j >= 0; j--) siftDown(hIdx, hVal, j, K);
|
|
1284
|
+
} else if (s > hVal[0]) {
|
|
1285
|
+
hIdx[0] = i;
|
|
1286
|
+
hVal[0] = s;
|
|
1287
|
+
siftDown(hIdx, hVal, 0, K);
|
|
1288
|
+
}
|
|
1289
|
+
}
|
|
1290
|
+
for (let i = 1; i < heapSize; i++) {
|
|
1291
|
+
const vi = hVal[i];
|
|
1292
|
+
const ii = hIdx[i];
|
|
1293
|
+
let j = i - 1;
|
|
1294
|
+
while (j >= 0 && hVal[j] < vi) {
|
|
1295
|
+
hVal[j + 1] = hVal[j];
|
|
1296
|
+
hIdx[j + 1] = hIdx[j];
|
|
1297
|
+
j--;
|
|
1298
|
+
}
|
|
1299
|
+
hVal[j + 1] = vi;
|
|
1300
|
+
hIdx[j + 1] = ii;
|
|
1301
|
+
}
|
|
1302
|
+
const maxScore = hVal[0];
|
|
1303
|
+
let sumExp = 0;
|
|
1304
|
+
for (let i = 0; i < heapSize; i++) {
|
|
1305
|
+
const p = Math.exp(hVal[i] - maxScore);
|
|
1306
|
+
hVal[i] = p;
|
|
1307
|
+
sumExp += p;
|
|
1308
|
+
}
|
|
1309
|
+
const invSum = 1 / sumExp;
|
|
1310
|
+
for (let i = 0; i < heapSize; i++) hVal[i] *= invSum;
|
|
1311
|
+
let candidateCount = heapSize;
|
|
1312
|
+
if (topP < 1) {
|
|
1313
|
+
let cumulative$1 = 0;
|
|
1314
|
+
for (let i = 0; i < heapSize; i++) {
|
|
1315
|
+
cumulative$1 += hVal[i];
|
|
1316
|
+
if (cumulative$1 >= topP) {
|
|
1317
|
+
candidateCount = i + 1;
|
|
1318
|
+
break;
|
|
1319
|
+
}
|
|
1320
|
+
}
|
|
1321
|
+
let sum = 0;
|
|
1322
|
+
for (let i = 0; i < candidateCount; i++) sum += hVal[i];
|
|
1323
|
+
const inv = 1 / sum;
|
|
1324
|
+
for (let i = 0; i < candidateCount; i++) hVal[i] *= inv;
|
|
1325
|
+
}
|
|
1326
|
+
const r = nextRandom(params);
|
|
1327
|
+
let cumulative = 0;
|
|
1328
|
+
for (let i = 0; i < candidateCount; i++) {
|
|
1329
|
+
cumulative += hVal[i];
|
|
1330
|
+
if (r <= cumulative) return hIdx[i];
|
|
1331
|
+
}
|
|
1332
|
+
return hIdx[candidateCount - 1];
|
|
1333
|
+
}
|
|
1334
|
+
/**
|
|
1335
|
+
* Return the index of the maximum value (greedy decoding).
|
|
1336
|
+
*/
|
|
1337
|
+
function argmax$2(arr) {
|
|
1338
|
+
let maxIdx = 0;
|
|
1339
|
+
let maxVal = arr[0];
|
|
1340
|
+
for (let i = 1; i < arr.length; i++) if (arr[i] > maxVal) {
|
|
1341
|
+
maxVal = arr[i];
|
|
1342
|
+
maxIdx = i;
|
|
1343
|
+
}
|
|
1344
|
+
return maxIdx;
|
|
1345
|
+
}
|
|
1346
|
+
/** Min-heap sift down on parallel index/value typed arrays. */
|
|
1347
|
+
function siftDown(indices, values, i, n) {
|
|
1348
|
+
while (true) {
|
|
1349
|
+
let smallest = i;
|
|
1350
|
+
const left = 2 * i + 1;
|
|
1351
|
+
const right = 2 * i + 2;
|
|
1352
|
+
if (left < n && values[left] < values[smallest]) smallest = left;
|
|
1353
|
+
if (right < n && values[right] < values[smallest]) smallest = right;
|
|
1354
|
+
if (smallest === i) break;
|
|
1355
|
+
const ti = indices[i];
|
|
1356
|
+
indices[i] = indices[smallest];
|
|
1357
|
+
indices[smallest] = ti;
|
|
1358
|
+
const tv = values[i];
|
|
1359
|
+
values[i] = values[smallest];
|
|
1360
|
+
values[smallest] = tv;
|
|
1361
|
+
i = smallest;
|
|
1362
|
+
}
|
|
1363
|
+
}
|
|
1364
|
+
|
|
1365
|
+
//#endregion
|
|
1366
|
+
//#region src/gpu/batch/scheduler.ts
|
|
1367
|
+
const PIPELINE_DEPTH = 2;
|
|
1368
|
+
var RequestScheduler = class {
|
|
1369
|
+
host;
|
|
1370
|
+
batchSize;
|
|
1371
|
+
maxSeqLen;
|
|
1372
|
+
blockSize;
|
|
1373
|
+
prefillChunk;
|
|
1374
|
+
/** Free KV blocks kept unreserved as safety headroom. */
|
|
1375
|
+
blockHeadroom;
|
|
1376
|
+
nextId = 1;
|
|
1377
|
+
queue = [];
|
|
1378
|
+
/** slot → occupying request (prefilling or decoding). */
|
|
1379
|
+
slots;
|
|
1380
|
+
/** Sum of blocksReserved across live (admitted, unfinished) requests. */
|
|
1381
|
+
reservedBlocks = 0;
|
|
1382
|
+
prefilling = null;
|
|
1383
|
+
inFlight = [];
|
|
1384
|
+
/**
|
|
1385
|
+
* The slot→adapter-id assignment last uploaded to the executor (null =
|
|
1386
|
+
* nothing uploaded this batch session). Steps only re-upload when the
|
|
1387
|
+
* assignment changes (admissions/retirements), and the very first upload is
|
|
1388
|
+
* skipped entirely while every lane is bare base — an adapter-free serving
|
|
1389
|
+
* session never touches the LoRA machinery.
|
|
1390
|
+
*/
|
|
1391
|
+
lastLaneAdapterIds = null;
|
|
1392
|
+
stepCounter = 0;
|
|
1393
|
+
loop = null;
|
|
1394
|
+
closed = false;
|
|
1395
|
+
stats = {
|
|
1396
|
+
submitted: 0,
|
|
1397
|
+
completed: 0,
|
|
1398
|
+
stepsSubmitted: 0,
|
|
1399
|
+
prefillChunks: 0,
|
|
1400
|
+
tokensGenerated: 0,
|
|
1401
|
+
busyMs: 0,
|
|
1402
|
+
peakActiveLanes: 0
|
|
1403
|
+
};
|
|
1404
|
+
constructor(host, options = {}) {
|
|
1405
|
+
this.host = host;
|
|
1406
|
+
this.batchSize = host.executor.batchSize;
|
|
1407
|
+
if (this.batchSize <= 0) throw new Error("RequestScheduler requires GERBIL_BATCH=N (N >= 2) on the Dawn path");
|
|
1408
|
+
this.maxSeqLen = host.executor.maxSequenceLength;
|
|
1409
|
+
this.blockSize = host.executor.kvBlockSizeTokens;
|
|
1410
|
+
this.slots = new Array(this.batchSize).fill(null);
|
|
1411
|
+
const envChunk = typeof process !== "undefined" ? Number(process.env?.GERBIL_PREFILL_CHUNK) : NaN;
|
|
1412
|
+
this.prefillChunk = options.prefillChunkTokens ?? (Number.isFinite(envChunk) && envChunk >= 8 ? envChunk : 128);
|
|
1413
|
+
this.blockHeadroom = 2;
|
|
1414
|
+
}
|
|
1415
|
+
/**
|
|
1416
|
+
* Submit a request; resolves with the full result when it finishes.
|
|
1417
|
+
* Tokens stream through options.onToken as they are decoded.
|
|
1418
|
+
*/
|
|
1419
|
+
submit(prompt, options = {}) {
|
|
1420
|
+
if (this.closed) return Promise.reject(/* @__PURE__ */ new Error("RequestScheduler is closed"));
|
|
1421
|
+
if ((options.sampling?.temperature ?? 0) >= 1e-6) return Promise.reject(/* @__PURE__ */ new Error("RequestScheduler: Phase 5 is greedy-only — pass sampling: { temperature: 0 }"));
|
|
1422
|
+
let adapterReg = null;
|
|
1423
|
+
if (options.adapter != null) {
|
|
1424
|
+
if (!this.host.executor.batchLoraEnabled) return Promise.reject(/* @__PURE__ */ new Error("RequestScheduler: per-request adapters require GERBIL_BATCH_LORA=1 (with GERBIL_BATCH=N)"));
|
|
1425
|
+
try {
|
|
1426
|
+
adapterReg = this.host.resolveAdapter(options.adapter);
|
|
1427
|
+
} catch (err) {
|
|
1428
|
+
return Promise.reject(err instanceof Error ? err : new Error(String(err)));
|
|
1429
|
+
}
|
|
1430
|
+
}
|
|
1431
|
+
const messages = typeof prompt === "string" ? [...options.systemPrompt ? [{
|
|
1432
|
+
role: "system",
|
|
1433
|
+
content: options.systemPrompt
|
|
1434
|
+
}] : [], {
|
|
1435
|
+
role: "user",
|
|
1436
|
+
content: prompt
|
|
1437
|
+
}] : options.systemPrompt ? [{
|
|
1438
|
+
role: "system",
|
|
1439
|
+
content: options.systemPrompt
|
|
1440
|
+
}, ...prompt] : prompt;
|
|
1441
|
+
const promptIds = new Uint32Array(this.host.encodeChat(messages));
|
|
1442
|
+
const roomFor = this.maxSeqLen - promptIds.length - PIPELINE_DEPTH - 1;
|
|
1443
|
+
if (roomFor < 1) return Promise.reject(/* @__PURE__ */ new Error(`RequestScheduler: prompt (${promptIds.length} tokens) leaves no room in maxSeqLen ${this.maxSeqLen}`));
|
|
1444
|
+
const maxTokens = Math.min(options.maxTokens ?? 128, roomFor);
|
|
1445
|
+
return new Promise((resolve, reject) => {
|
|
1446
|
+
const req = {
|
|
1447
|
+
id: this.nextId++,
|
|
1448
|
+
promptIds,
|
|
1449
|
+
maxTokens,
|
|
1450
|
+
stopSequences: options.stopSequences ?? [],
|
|
1451
|
+
adapterName: options.adapter ?? null,
|
|
1452
|
+
adapterId: adapterReg?.id ?? -1,
|
|
1453
|
+
adapterDeltas: adapterReg?.deltas ?? null,
|
|
1454
|
+
onToken: options.onToken,
|
|
1455
|
+
resolve,
|
|
1456
|
+
reject,
|
|
1457
|
+
state: "queued",
|
|
1458
|
+
slot: -1,
|
|
1459
|
+
prefillPos: 0,
|
|
1460
|
+
blocksReserved: 0,
|
|
1461
|
+
generatedIds: [],
|
|
1462
|
+
text: "",
|
|
1463
|
+
finishReason: "max_tokens",
|
|
1464
|
+
submittedAt: performance.now(),
|
|
1465
|
+
admittedAt: 0,
|
|
1466
|
+
firstTokenAt: 0
|
|
1467
|
+
};
|
|
1468
|
+
this.queue.push(req);
|
|
1469
|
+
this.stats.submitted++;
|
|
1470
|
+
this.pump();
|
|
1471
|
+
});
|
|
1472
|
+
}
|
|
1473
|
+
/** Resolves when every submitted request has completed. */
|
|
1474
|
+
async drain() {
|
|
1475
|
+
while (this.loop) await this.loop;
|
|
1476
|
+
}
|
|
1477
|
+
/** Stop accepting requests; resolves when in-flight work drains. */
|
|
1478
|
+
async close() {
|
|
1479
|
+
this.closed = true;
|
|
1480
|
+
await this.drain();
|
|
1481
|
+
}
|
|
1482
|
+
getStats() {
|
|
1483
|
+
return { ...this.stats };
|
|
1484
|
+
}
|
|
1485
|
+
pump() {
|
|
1486
|
+
if (this.loop) return;
|
|
1487
|
+
this.loop = this.run().finally(() => {
|
|
1488
|
+
this.loop = null;
|
|
1489
|
+
if (this.hasWork()) this.pump();
|
|
1490
|
+
});
|
|
1491
|
+
}
|
|
1492
|
+
hasWork() {
|
|
1493
|
+
return this.queue.length > 0 || this.prefilling !== null || this.inFlight.length > 0 || this.slots.some((s) => s !== null);
|
|
1494
|
+
}
|
|
1495
|
+
activeMask() {
|
|
1496
|
+
const mask = new Uint8Array(this.batchSize);
|
|
1497
|
+
const rows = [];
|
|
1498
|
+
for (let b = 0; b < this.batchSize; b++) {
|
|
1499
|
+
const req = this.slots[b];
|
|
1500
|
+
if (req && req.state === "decoding") {
|
|
1501
|
+
mask[b] = 1;
|
|
1502
|
+
rows.push({
|
|
1503
|
+
slot: b,
|
|
1504
|
+
req
|
|
1505
|
+
});
|
|
1506
|
+
}
|
|
1507
|
+
}
|
|
1508
|
+
return {
|
|
1509
|
+
mask,
|
|
1510
|
+
rows
|
|
1511
|
+
};
|
|
1512
|
+
}
|
|
1513
|
+
/** The serving loop. Holds the engine generation lock while live. */
|
|
1514
|
+
async run() {
|
|
1515
|
+
const release = await this.host.acquireLock();
|
|
1516
|
+
const ex = this.host.executor;
|
|
1517
|
+
const loopStart = performance.now();
|
|
1518
|
+
ex.resetBatch();
|
|
1519
|
+
this.stepCounter = 0;
|
|
1520
|
+
this.lastLaneAdapterIds = null;
|
|
1521
|
+
try {
|
|
1522
|
+
while (this.hasWork()) {
|
|
1523
|
+
this.admit();
|
|
1524
|
+
const { mask, rows } = this.activeMask();
|
|
1525
|
+
if (rows.length > 0 && this.inFlight.length < PIPELINE_DEPTH) {
|
|
1526
|
+
const readbackSlot = this.stepCounter % PIPELINE_DEPTH;
|
|
1527
|
+
this.syncLaneAdapters();
|
|
1528
|
+
ex.submitBatchDecodeStepMasked(mask, readbackSlot);
|
|
1529
|
+
this.stepCounter++;
|
|
1530
|
+
this.inFlight.push({
|
|
1531
|
+
readbackSlot,
|
|
1532
|
+
rows
|
|
1533
|
+
});
|
|
1534
|
+
this.stats.stepsSubmitted++;
|
|
1535
|
+
if (rows.length > this.stats.peakActiveLanes) this.stats.peakActiveLanes = rows.length;
|
|
1536
|
+
}
|
|
1537
|
+
if (this.prefilling) await this.prefillChunkStep();
|
|
1538
|
+
const pipelineFull = this.inFlight.length >= PIPELINE_DEPTH;
|
|
1539
|
+
const idleLanes = rows.length === 0 && this.prefilling === null;
|
|
1540
|
+
if (this.inFlight.length > 0 && (pipelineFull || idleLanes)) await this.readStep();
|
|
1541
|
+
}
|
|
1542
|
+
} catch (err) {
|
|
1543
|
+
const error = err instanceof Error ? err : new Error(String(err));
|
|
1544
|
+
if (this.prefilling?.adapterDeltas) try {
|
|
1545
|
+
ex.clearRuntimeLoRA();
|
|
1546
|
+
} catch {}
|
|
1547
|
+
for (const req of [...this.queue, ...this.slots.filter((s) => s !== null)]) if (req.state !== "done") {
|
|
1548
|
+
req.state = "done";
|
|
1549
|
+
req.reject(error);
|
|
1550
|
+
}
|
|
1551
|
+
this.queue = [];
|
|
1552
|
+
this.slots.fill(null);
|
|
1553
|
+
this.prefilling = null;
|
|
1554
|
+
this.inFlight = [];
|
|
1555
|
+
this.reservedBlocks = 0;
|
|
1556
|
+
throw error;
|
|
1557
|
+
} finally {
|
|
1558
|
+
this.stats.busyMs += performance.now() - loopStart;
|
|
1559
|
+
release();
|
|
1560
|
+
}
|
|
1561
|
+
}
|
|
1562
|
+
/**
|
|
1563
|
+
* Upload the slot→adapter assignment for the step about to be submitted,
|
|
1564
|
+
* when it differs from the last uploaded one. Called immediately before
|
|
1565
|
+
* every decode-step submit: `queue.writeBuffer` is queue-ordered, so steps
|
|
1566
|
+
* already in flight keep the assignment they were submitted under, and a
|
|
1567
|
+
* retiring lane's slot can never leak its adapter into the slot's next
|
|
1568
|
+
* occupant — the next occupant's assignment is re-derived from `slots`
|
|
1569
|
+
* before its first decode step. Masked (free/prefilling) lanes are pinned
|
|
1570
|
+
* to bare base; their output is discarded regardless.
|
|
1571
|
+
*/
|
|
1572
|
+
syncLaneAdapters() {
|
|
1573
|
+
if (!this.host.executor.batchLoraEnabled) return;
|
|
1574
|
+
const ids = new Int32Array(this.batchSize).fill(-1);
|
|
1575
|
+
let anyAdapter = false;
|
|
1576
|
+
for (let b = 0; b < this.batchSize; b++) {
|
|
1577
|
+
const req = this.slots[b];
|
|
1578
|
+
if (req && req.state === "decoding" && req.adapterId >= 0) {
|
|
1579
|
+
ids[b] = req.adapterId;
|
|
1580
|
+
anyAdapter = true;
|
|
1581
|
+
}
|
|
1582
|
+
}
|
|
1583
|
+
const last = this.lastLaneAdapterIds;
|
|
1584
|
+
if (last === null) {
|
|
1585
|
+
if (!anyAdapter) return;
|
|
1586
|
+
} else if (ids.every((id, b) => id === last[b])) return;
|
|
1587
|
+
this.host.executor.setBatchLaneAdapters(ids);
|
|
1588
|
+
this.lastLaneAdapterIds = ids;
|
|
1589
|
+
}
|
|
1590
|
+
/**
|
|
1591
|
+
* Admission: move the next queued request into a free slot when (a) a slot
|
|
1592
|
+
* is free, (b) no other prefill is staged (the single-sequence prefill path
|
|
1593
|
+
* stages SSM state in singleton buffers — one at a time), and (c) its
|
|
1594
|
+
* worst-case block need fits the unreserved KV pool with headroom.
|
|
1595
|
+
*/
|
|
1596
|
+
admit() {
|
|
1597
|
+
if (this.prefilling || this.queue.length === 0) return;
|
|
1598
|
+
const slot = this.slots.indexOf(null);
|
|
1599
|
+
if (slot === -1) return;
|
|
1600
|
+
const req = this.queue[0];
|
|
1601
|
+
const worstTokens = req.promptIds.length + req.maxTokens + PIPELINE_DEPTH;
|
|
1602
|
+
const blocksNeeded = Math.ceil(worstTokens / this.blockSize);
|
|
1603
|
+
const totalBlocks = this.host.executor.kvTotalBlocks;
|
|
1604
|
+
if (this.reservedBlocks + blocksNeeded + this.blockHeadroom > totalBlocks) return;
|
|
1605
|
+
this.queue.shift();
|
|
1606
|
+
req.state = "prefilling";
|
|
1607
|
+
req.slot = slot;
|
|
1608
|
+
req.blocksReserved = blocksNeeded;
|
|
1609
|
+
req.admittedAt = performance.now();
|
|
1610
|
+
this.reservedBlocks += blocksNeeded;
|
|
1611
|
+
this.slots[slot] = req;
|
|
1612
|
+
this.prefilling = req;
|
|
1613
|
+
if (req.adapterDeltas) {
|
|
1614
|
+
const { applied } = this.host.executor.applyRuntimeLoRA(req.adapterDeltas);
|
|
1615
|
+
if (applied === 0) throw new Error(`RequestScheduler: adapter "${req.adapterName}" resolved to 0 applicable prefill targets`);
|
|
1616
|
+
}
|
|
1617
|
+
this.host.executor.beginBatchSlot(slot);
|
|
1618
|
+
}
|
|
1619
|
+
/**
|
|
1620
|
+
* Run one bounded prefill chunk for the staged request on the
|
|
1621
|
+
* single-sequence path (into its slot's block-table row via tableBase).
|
|
1622
|
+
* Intermediate chunks are submit-only; the final chunk reads logits, adopts
|
|
1623
|
+
* the singleton SSM state into the slot's pool row, seeds the lane's first
|
|
1624
|
+
* token, and flips the lane to decoding.
|
|
1625
|
+
*/
|
|
1626
|
+
async prefillChunkStep() {
|
|
1627
|
+
const req = this.prefilling;
|
|
1628
|
+
if (!req) return;
|
|
1629
|
+
const ex = this.host.executor;
|
|
1630
|
+
const end = Math.min(req.prefillPos + this.prefillChunk, req.promptIds.length);
|
|
1631
|
+
const isLast = end === req.promptIds.length;
|
|
1632
|
+
const chunk = req.promptIds.subarray(req.prefillPos, end);
|
|
1633
|
+
const { logits } = await ex.forward(chunk, { readLogits: isLast });
|
|
1634
|
+
req.prefillPos = end;
|
|
1635
|
+
this.stats.prefillChunks++;
|
|
1636
|
+
if (!isLast) return;
|
|
1637
|
+
ex.finishBatchSlot(req.slot);
|
|
1638
|
+
if (req.adapterDeltas) ex.clearRuntimeLoRA();
|
|
1639
|
+
this.prefilling = null;
|
|
1640
|
+
const first = this.host.remapToken(sampleToken(logits, { temperature: 0 }));
|
|
1641
|
+
ex.injectBatchToken(req.slot, first);
|
|
1642
|
+
req.state = "decoding";
|
|
1643
|
+
req.firstTokenAt = performance.now();
|
|
1644
|
+
this.consumeToken(req, first);
|
|
1645
|
+
}
|
|
1646
|
+
/** Read the oldest in-flight step and deliver its tokens. */
|
|
1647
|
+
async readStep() {
|
|
1648
|
+
const step = this.inFlight.shift();
|
|
1649
|
+
if (!step) return;
|
|
1650
|
+
const tokens = await this.host.executor.readBatchTokens(step.readbackSlot);
|
|
1651
|
+
for (const { slot, req } of step.rows) {
|
|
1652
|
+
if (this.slots[slot] !== req || req.state !== "decoding") continue;
|
|
1653
|
+
this.consumeToken(req, tokens[slot]);
|
|
1654
|
+
}
|
|
1655
|
+
}
|
|
1656
|
+
/** Append one generated token; finish + retire the lane when terminal. */
|
|
1657
|
+
consumeToken(req, tokenId) {
|
|
1658
|
+
req.generatedIds.push(tokenId);
|
|
1659
|
+
this.stats.tokensGenerated++;
|
|
1660
|
+
if (req.firstTokenAt === 0) req.firstTokenAt = performance.now();
|
|
1661
|
+
const { eosTokenId, eotTokenId } = this.host;
|
|
1662
|
+
if (eosTokenId !== null && tokenId === eosTokenId || eotTokenId !== null && tokenId === eotTokenId) {
|
|
1663
|
+
req.finishReason = "eos";
|
|
1664
|
+
this.finish(req);
|
|
1665
|
+
return;
|
|
1666
|
+
}
|
|
1667
|
+
const piece = this.host.decodeToken(tokenId);
|
|
1668
|
+
req.text += piece;
|
|
1669
|
+
if (req.stopSequences.some((s) => req.text.includes(s))) {
|
|
1670
|
+
for (const s of req.stopSequences) {
|
|
1671
|
+
const idx = req.text.indexOf(s);
|
|
1672
|
+
if (idx !== -1) req.text = req.text.slice(0, idx);
|
|
1673
|
+
}
|
|
1674
|
+
req.finishReason = "stop_sequence";
|
|
1675
|
+
this.finish(req);
|
|
1676
|
+
return;
|
|
1677
|
+
}
|
|
1678
|
+
req.onToken?.(piece, {
|
|
1679
|
+
requestId: req.id,
|
|
1680
|
+
tokenId,
|
|
1681
|
+
generated: req.generatedIds.length
|
|
1682
|
+
});
|
|
1683
|
+
if (req.generatedIds.length >= req.maxTokens) {
|
|
1684
|
+
req.finishReason = "max_tokens";
|
|
1685
|
+
this.finish(req);
|
|
1686
|
+
}
|
|
1687
|
+
}
|
|
1688
|
+
/** Retire the lane immediately: free KV blocks, open the slot, resolve. */
|
|
1689
|
+
finish(req) {
|
|
1690
|
+
req.state = "done";
|
|
1691
|
+
if (req.slot >= 0 && this.slots[req.slot] === req) {
|
|
1692
|
+
this.host.executor.retireBatchSlot(req.slot);
|
|
1693
|
+
this.slots[req.slot] = null;
|
|
1694
|
+
}
|
|
1695
|
+
this.reservedBlocks -= req.blocksReserved;
|
|
1696
|
+
this.stats.completed++;
|
|
1697
|
+
const finishedAt = performance.now();
|
|
1698
|
+
req.resolve({
|
|
1699
|
+
requestId: req.id,
|
|
1700
|
+
text: req.text,
|
|
1701
|
+
tokenIds: [...req.generatedIds],
|
|
1702
|
+
tokensGenerated: req.generatedIds.length,
|
|
1703
|
+
finishReason: req.finishReason,
|
|
1704
|
+
adapter: req.adapterName ?? void 0,
|
|
1705
|
+
submittedAt: req.submittedAt,
|
|
1706
|
+
admittedAt: req.admittedAt,
|
|
1707
|
+
firstTokenAt: req.firstTokenAt,
|
|
1708
|
+
finishedAt,
|
|
1709
|
+
ttftMs: req.firstTokenAt - req.submittedAt,
|
|
1710
|
+
e2eMs: finishedAt - req.submittedAt
|
|
1711
|
+
});
|
|
1712
|
+
}
|
|
1713
|
+
};
|
|
1714
|
+
|
|
1715
|
+
//#endregion
|
|
1716
|
+
//#region src/gpu/batch/swarm.ts
|
|
1717
|
+
/** Default reduce: member-keyed map, disambiguating duplicated members. */
|
|
1718
|
+
function reduceToKeyedMap(results) {
|
|
1719
|
+
const out = {};
|
|
1720
|
+
const seen = /* @__PURE__ */ new Map();
|
|
1721
|
+
for (const { member, result } of results) {
|
|
1722
|
+
const n = (seen.get(member) ?? 0) + 1;
|
|
1723
|
+
seen.set(member, n);
|
|
1724
|
+
out[n === 1 ? member : `${member}#${n}`] = result;
|
|
1725
|
+
}
|
|
1726
|
+
return out;
|
|
1727
|
+
}
|
|
1728
|
+
/**
|
|
1729
|
+
* A swarm handle: registered members + plan/reduce over the engine's
|
|
1730
|
+
* scheduler. Create via {@link WebGPUEngine.createSwarm}.
|
|
1731
|
+
*/
|
|
1732
|
+
var Swarm = class {
|
|
1733
|
+
scheduler;
|
|
1734
|
+
memberSources;
|
|
1735
|
+
plan;
|
|
1736
|
+
reduce;
|
|
1737
|
+
constructor(scheduler, options) {
|
|
1738
|
+
if (Object.keys(options.members).length === 0) throw new Error("Swarm: members must contain at least one member");
|
|
1739
|
+
this.scheduler = scheduler;
|
|
1740
|
+
this.memberSources = { ...options.members };
|
|
1741
|
+
this.plan = options.plan;
|
|
1742
|
+
this.reduce = options.reduce;
|
|
1743
|
+
}
|
|
1744
|
+
/** Member names, declaration order. */
|
|
1745
|
+
get members() {
|
|
1746
|
+
return Object.keys(this.memberSources);
|
|
1747
|
+
}
|
|
1748
|
+
/**
|
|
1749
|
+
* Run one input through the swarm: plan → concurrent fan-out (one batched
|
|
1750
|
+
* pass across the scheduler's lanes) → reduce.
|
|
1751
|
+
*
|
|
1752
|
+
* @param input The task input handed to the plan (default plan: broadcast
|
|
1753
|
+
* it to every member as the prompt).
|
|
1754
|
+
* @param options Per-request options applied to every sub-task (a task's
|
|
1755
|
+
* own `options` win field-by-field).
|
|
1756
|
+
*/
|
|
1757
|
+
async run(input, options = {}) {
|
|
1758
|
+
const tasks = this.plan ? await this.plan(input, this.members) : this.members.map((member) => ({
|
|
1759
|
+
member,
|
|
1760
|
+
prompt: input
|
|
1761
|
+
}));
|
|
1762
|
+
if (tasks.length === 0) throw new Error("Swarm: plan produced no sub-tasks");
|
|
1763
|
+
const results = await Promise.all(tasks.map(async (task) => {
|
|
1764
|
+
const result = await this.submitFor(task.member, task.prompt, {
|
|
1765
|
+
...options,
|
|
1766
|
+
...task.options
|
|
1767
|
+
});
|
|
1768
|
+
return {
|
|
1769
|
+
member: task.member,
|
|
1770
|
+
result
|
|
1771
|
+
};
|
|
1772
|
+
}));
|
|
1773
|
+
if (this.reduce) return this.reduce(results);
|
|
1774
|
+
return reduceToKeyedMap(results);
|
|
1775
|
+
}
|
|
1776
|
+
/**
|
|
1777
|
+
* Mode-A access: decode one prompt with a single member (no plan/reduce).
|
|
1778
|
+
* Still rides the shared scheduler, so it batches with any concurrent work.
|
|
1779
|
+
*/
|
|
1780
|
+
generate(member, prompt, options = {}) {
|
|
1781
|
+
return this.submitFor(member, prompt, options);
|
|
1782
|
+
}
|
|
1783
|
+
submitFor(member, prompt, options) {
|
|
1784
|
+
if (!Object.hasOwn(this.memberSources, member)) throw new Error(`Swarm: unknown member "${member}" (members: ${this.members.join(", ")})`);
|
|
1785
|
+
const adapter = this.memberSources[member] == null ? void 0 : member;
|
|
1786
|
+
return this.scheduler.submit(prompt, {
|
|
1787
|
+
...options,
|
|
1788
|
+
adapter
|
|
1789
|
+
});
|
|
1790
|
+
}
|
|
1791
|
+
};
|
|
1792
|
+
|
|
1228
1793
|
//#endregion
|
|
1229
1794
|
//#region src/gpu/architectures/outetts.ts
|
|
1230
1795
|
const OUTETTS_C1_BASE = 151669;
|
|
@@ -2859,7 +3424,7 @@ function selectGraphWeights$1(graph, weights) {
|
|
|
2859
3424
|
return out;
|
|
2860
3425
|
}
|
|
2861
3426
|
/** Index of the maximum logit (greedy / argmax decode). */
|
|
2862
|
-
function argmax$
|
|
3427
|
+
function argmax$1(logits) {
|
|
2863
3428
|
let best = 0;
|
|
2864
3429
|
let bestV = logits[0];
|
|
2865
3430
|
for (let i = 1; i < logits.length; i++) if (logits[i] > bestV) {
|
|
@@ -3188,7 +3753,7 @@ var OuteTTS = class OuteTTS {
|
|
|
3188
3753
|
const generated = [];
|
|
3189
3754
|
let logits = prefillLogits;
|
|
3190
3755
|
for (let step = 0; step < p.maxNewTokens; step++) {
|
|
3191
|
-
const next = p.greedy ? argmax$
|
|
3756
|
+
const next = p.greedy ? argmax$1(logits) : this.sampleToken(logits, {
|
|
3192
3757
|
temperature: p.temperature,
|
|
3193
3758
|
topP: p.topP,
|
|
3194
3759
|
topK: p.topK,
|
|
@@ -3320,7 +3885,7 @@ function selectGraphWeights(graph, weights) {
|
|
|
3320
3885
|
return out;
|
|
3321
3886
|
}
|
|
3322
3887
|
/** Argmax over a logit row. */
|
|
3323
|
-
function argmax
|
|
3888
|
+
function argmax(logits) {
|
|
3324
3889
|
let best = 0;
|
|
3325
3890
|
let bestV = logits[0];
|
|
3326
3891
|
for (let i = 1; i < logits.length; i++) if (logits[i] > bestV) {
|
|
@@ -3671,7 +4236,7 @@ var ParlerTTS = class ParlerTTS {
|
|
|
3671
4236
|
continue;
|
|
3672
4237
|
}
|
|
3673
4238
|
if (cb > frontier) logitRows[cb][PARLER_EOS_TOKEN_ID] = Number.NEGATIVE_INFINITY;
|
|
3674
|
-
const code = opts.greedy ? argmax
|
|
4239
|
+
const code = opts.greedy ? argmax(logitRows[cb]) : sampleLogits(logitRows[cb], opts.temperature ?? 1, opts.topK ?? 0);
|
|
3675
4240
|
nextCodes[cb] = code;
|
|
3676
4241
|
grid[cb * maxSteps + step] = code < PARLER_CODEBOOK_SIZE ? code : 0;
|
|
3677
4242
|
}
|
|
@@ -3742,122 +4307,6 @@ function addPosition(row, posTable, pos, H) {
|
|
|
3742
4307
|
for (let i = 0; i < H; i++) row[i] += posTable[base + i];
|
|
3743
4308
|
}
|
|
3744
4309
|
|
|
3745
|
-
//#endregion
|
|
3746
|
-
//#region src/gpu/sampler.ts
|
|
3747
|
-
let _heapIndices = null;
|
|
3748
|
-
let _heapValues = null;
|
|
3749
|
-
/**
|
|
3750
|
-
* Sample a token ID from logits.
|
|
3751
|
-
*
|
|
3752
|
-
* Pipeline: repetition penalty → temperature → top-k (min-heap) → softmax → top-p → sample.
|
|
3753
|
-
*/
|
|
3754
|
-
function sampleToken(logits, params = {}, previousTokens) {
|
|
3755
|
-
const temperature = params.temperature ?? .7;
|
|
3756
|
-
const topK = params.topK ?? 50;
|
|
3757
|
-
const topP = params.topP ?? .9;
|
|
3758
|
-
const repetitionPenalty = params.repetitionPenalty ?? 1;
|
|
3759
|
-
if (temperature < 1e-6) return argmax(logits);
|
|
3760
|
-
const N = logits.length;
|
|
3761
|
-
const K = Math.min(topK > 0 ? topK : N, N);
|
|
3762
|
-
if (!_heapIndices || _heapIndices.length < K) {
|
|
3763
|
-
_heapIndices = new Uint32Array(K);
|
|
3764
|
-
_heapValues = new Float32Array(K);
|
|
3765
|
-
}
|
|
3766
|
-
const hIdx = _heapIndices;
|
|
3767
|
-
const hVal = _heapValues;
|
|
3768
|
-
let penaltySet = null;
|
|
3769
|
-
if (repetitionPenalty !== 1 && previousTokens?.length) penaltySet = new Set(previousTokens);
|
|
3770
|
-
let heapSize = 0;
|
|
3771
|
-
for (let i = 0; i < N; i++) {
|
|
3772
|
-
let s = logits[i];
|
|
3773
|
-
if (penaltySet?.has(i)) s = s > 0 ? s / repetitionPenalty : s * repetitionPenalty;
|
|
3774
|
-
s /= temperature;
|
|
3775
|
-
if (heapSize < K) {
|
|
3776
|
-
hIdx[heapSize] = i;
|
|
3777
|
-
hVal[heapSize] = s;
|
|
3778
|
-
heapSize++;
|
|
3779
|
-
if (heapSize === K) for (let j = (K >> 1) - 1; j >= 0; j--) siftDown(hIdx, hVal, j, K);
|
|
3780
|
-
} else if (s > hVal[0]) {
|
|
3781
|
-
hIdx[0] = i;
|
|
3782
|
-
hVal[0] = s;
|
|
3783
|
-
siftDown(hIdx, hVal, 0, K);
|
|
3784
|
-
}
|
|
3785
|
-
}
|
|
3786
|
-
for (let i = 1; i < heapSize; i++) {
|
|
3787
|
-
const vi = hVal[i];
|
|
3788
|
-
const ii = hIdx[i];
|
|
3789
|
-
let j = i - 1;
|
|
3790
|
-
while (j >= 0 && hVal[j] < vi) {
|
|
3791
|
-
hVal[j + 1] = hVal[j];
|
|
3792
|
-
hIdx[j + 1] = hIdx[j];
|
|
3793
|
-
j--;
|
|
3794
|
-
}
|
|
3795
|
-
hVal[j + 1] = vi;
|
|
3796
|
-
hIdx[j + 1] = ii;
|
|
3797
|
-
}
|
|
3798
|
-
const maxScore = hVal[0];
|
|
3799
|
-
let sumExp = 0;
|
|
3800
|
-
for (let i = 0; i < heapSize; i++) {
|
|
3801
|
-
const p = Math.exp(hVal[i] - maxScore);
|
|
3802
|
-
hVal[i] = p;
|
|
3803
|
-
sumExp += p;
|
|
3804
|
-
}
|
|
3805
|
-
const invSum = 1 / sumExp;
|
|
3806
|
-
for (let i = 0; i < heapSize; i++) hVal[i] *= invSum;
|
|
3807
|
-
let candidateCount = heapSize;
|
|
3808
|
-
if (topP < 1) {
|
|
3809
|
-
let cumulative$1 = 0;
|
|
3810
|
-
for (let i = 0; i < heapSize; i++) {
|
|
3811
|
-
cumulative$1 += hVal[i];
|
|
3812
|
-
if (cumulative$1 >= topP) {
|
|
3813
|
-
candidateCount = i + 1;
|
|
3814
|
-
break;
|
|
3815
|
-
}
|
|
3816
|
-
}
|
|
3817
|
-
let sum = 0;
|
|
3818
|
-
for (let i = 0; i < candidateCount; i++) sum += hVal[i];
|
|
3819
|
-
const inv = 1 / sum;
|
|
3820
|
-
for (let i = 0; i < candidateCount; i++) hVal[i] *= inv;
|
|
3821
|
-
}
|
|
3822
|
-
const r = Math.random();
|
|
3823
|
-
let cumulative = 0;
|
|
3824
|
-
for (let i = 0; i < candidateCount; i++) {
|
|
3825
|
-
cumulative += hVal[i];
|
|
3826
|
-
if (r <= cumulative) return hIdx[i];
|
|
3827
|
-
}
|
|
3828
|
-
return hIdx[candidateCount - 1];
|
|
3829
|
-
}
|
|
3830
|
-
/**
|
|
3831
|
-
* Return the index of the maximum value (greedy decoding).
|
|
3832
|
-
*/
|
|
3833
|
-
function argmax(arr) {
|
|
3834
|
-
let maxIdx = 0;
|
|
3835
|
-
let maxVal = arr[0];
|
|
3836
|
-
for (let i = 1; i < arr.length; i++) if (arr[i] > maxVal) {
|
|
3837
|
-
maxVal = arr[i];
|
|
3838
|
-
maxIdx = i;
|
|
3839
|
-
}
|
|
3840
|
-
return maxIdx;
|
|
3841
|
-
}
|
|
3842
|
-
/** Min-heap sift down on parallel index/value typed arrays. */
|
|
3843
|
-
function siftDown(indices, values, i, n) {
|
|
3844
|
-
while (true) {
|
|
3845
|
-
let smallest = i;
|
|
3846
|
-
const left = 2 * i + 1;
|
|
3847
|
-
const right = 2 * i + 2;
|
|
3848
|
-
if (left < n && values[left] < values[smallest]) smallest = left;
|
|
3849
|
-
if (right < n && values[right] < values[smallest]) smallest = right;
|
|
3850
|
-
if (smallest === i) break;
|
|
3851
|
-
const ti = indices[i];
|
|
3852
|
-
indices[i] = indices[smallest];
|
|
3853
|
-
indices[smallest] = ti;
|
|
3854
|
-
const tv = values[i];
|
|
3855
|
-
values[i] = values[smallest];
|
|
3856
|
-
values[smallest] = tv;
|
|
3857
|
-
i = smallest;
|
|
3858
|
-
}
|
|
3859
|
-
}
|
|
3860
|
-
|
|
3861
4310
|
//#endregion
|
|
3862
4311
|
//#region src/gpu/vision-executor.ts
|
|
3863
4312
|
const MAP_MODE_READ = 1;
|
|
@@ -5143,8 +5592,12 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5143
5592
|
_outeVoice = null;
|
|
5144
5593
|
/** Source of the runtime LoRA adapter currently applied on the static base, if any. */
|
|
5145
5594
|
_currentAdapter = null;
|
|
5595
|
+
/** Named adapters registered for per-lane batched decode (GERBIL_BATCH_LORA). */
|
|
5596
|
+
_batchAdapters = /* @__PURE__ */ new Map();
|
|
5146
5597
|
/** Lazily-created Parler-TTS engine (Flan-T5 encoder + decoder LM + dac_44khz). */
|
|
5147
5598
|
_parlerTTS = null;
|
|
5599
|
+
/** Lazily-created continuous-batching scheduler (GERBIL_BATCH, Phase 5). */
|
|
5600
|
+
_scheduler = null;
|
|
5148
5601
|
/**
|
|
5149
5602
|
* WebKit group-size probe state. When true, a candidate group size is being
|
|
5150
5603
|
* tried this page-load and must be promoted (or capped) after the FIRST
|
|
@@ -5250,6 +5703,35 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5250
5703
|
return this._currentAdapter;
|
|
5251
5704
|
}
|
|
5252
5705
|
/**
|
|
5706
|
+
* Register a named LoRA adapter for per-lane batched decode (SGMV phase 1,
|
|
5707
|
+
* `GERBIL_BATCH=N` + `GERBIL_BATCH_LORA=1`). The adapter's factors are
|
|
5708
|
+
* fetched and packed into the executor's shared factor buffers ONCE; any
|
|
5709
|
+
* number of batch lanes can then run it concurrently by passing its name in
|
|
5710
|
+
* `generateBatch`'s per-lane `adapters` option. Registering the same name
|
|
5711
|
+
* twice is a no-op.
|
|
5712
|
+
*/
|
|
5713
|
+
async registerAdapter(name, source) {
|
|
5714
|
+
this.checkDestroyed();
|
|
5715
|
+
if (!this.executor.batchLoraEnabled) throw new Error("registerAdapter requires GERBIL_BATCH=N (N >= 2) and GERBIL_BATCH_LORA=1 on the Dawn path");
|
|
5716
|
+
if (this._batchAdapters.has(name)) return;
|
|
5717
|
+
const adapter = await fetchAdapter(source, {
|
|
5718
|
+
hfToken: this._createOptions.hfToken,
|
|
5719
|
+
revision: this._createOptions.revision
|
|
5720
|
+
});
|
|
5721
|
+
if (!adapter) throw new Error(`No adapter found at ${source}.`);
|
|
5722
|
+
const deltas = buildLoRADeltas(adapter, createKeyMapperForArch(this._architecture));
|
|
5723
|
+
const id = this.executor.registerBatchAdapter(deltas);
|
|
5724
|
+
this._batchAdapters.set(name, {
|
|
5725
|
+
id,
|
|
5726
|
+
deltas
|
|
5727
|
+
});
|
|
5728
|
+
console.log(`[engine] batch adapter "${name}" (${source}) registered as id ${id}.`);
|
|
5729
|
+
}
|
|
5730
|
+
/** Names of the adapters registered for per-lane batched decode. */
|
|
5731
|
+
getRegisteredAdapters() {
|
|
5732
|
+
return [...this._batchAdapters.keys()];
|
|
5733
|
+
}
|
|
5734
|
+
/**
|
|
5253
5735
|
* Write a coarse crash-phase breadcrumb that survives a GPU-process kill / page
|
|
5254
5736
|
* reload. The iPad harness reads `localStorage["gerbil-crash-phase"]` after a
|
|
5255
5737
|
* crash; without these, a describe-time crash only shows the last load phase
|
|
@@ -5580,6 +6062,186 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5580
6062
|
}
|
|
5581
6063
|
}
|
|
5582
6064
|
/**
|
|
6065
|
+
* Batched lockstep generation (GERBIL_BATCH=N — Phase 2 of the
|
|
6066
|
+
* continuous-batching campaign, Dawn only).
|
|
6067
|
+
*
|
|
6068
|
+
* Prefills each prompt sequentially into its batch slot (KV into the slot's
|
|
6069
|
+
* block-table row, SSM/conv state adopted into the slot's state-pool rows),
|
|
6070
|
+
* then decodes all N sequences in lockstep: one batched dispatch stream per
|
|
6071
|
+
* step produces one token per row. Greedy-only in Phase 2 (per-row GPU
|
|
6072
|
+
* argmax); rows that hit EOS/stop keep decoding in lockstep with their
|
|
6073
|
+
* output discarded until every row has finished — rows never interact, so
|
|
6074
|
+
* per-row streams are token-exact vs single-sequence generation.
|
|
6075
|
+
*
|
|
6076
|
+
* Requires exactly `executor.batchSize` prompts (fixed lockstep batch; the
|
|
6077
|
+
* Phase 3 scheduler will lift this).
|
|
6078
|
+
*/
|
|
6079
|
+
async generateBatch(prompts, options = {}) {
|
|
6080
|
+
this.checkDestroyed();
|
|
6081
|
+
const B = this.executor.batchSize;
|
|
6082
|
+
if (B <= 0) throw new Error("generateBatch requires GERBIL_BATCH=N (N >= 2) on the Dawn path");
|
|
6083
|
+
if (prompts.length !== B) throw new Error(`generateBatch: got ${prompts.length} prompts for a fixed batch of ${B} (set GERBIL_BATCH=${prompts.length})`);
|
|
6084
|
+
if (this._multimodalGraph || this.executor.hasPleSource() || this.executor.hasRuntimeAdapter) throw new Error("generateBatch: multimodal, PLE, and runtime-LoRA models are Phase 2+ work");
|
|
6085
|
+
const { maxTokens = 512, stopSequences = [], sampling = {}, systemPrompt } = options;
|
|
6086
|
+
if ((sampling.temperature ?? .7) >= 1e-6) throw new Error("generateBatch: Phase 2 is greedy-only — pass sampling: { temperature: 0 }");
|
|
6087
|
+
const laneAdapterNames = options.adapters;
|
|
6088
|
+
if (laneAdapterNames && laneAdapterNames.length !== B) throw new Error(`generateBatch: got ${laneAdapterNames.length} lane adapters for a batch of ${B}`);
|
|
6089
|
+
if ((laneAdapterNames?.some((a) => a != null) ?? false) && !this.executor.batchLoraEnabled) throw new Error("generateBatch: per-lane adapters require GERBIL_BATCH_LORA=1");
|
|
6090
|
+
const laneAdapters = new Array(B).fill(null);
|
|
6091
|
+
if (laneAdapterNames) for (let i = 0; i < B; i++) {
|
|
6092
|
+
const name = laneAdapterNames[i];
|
|
6093
|
+
if (name == null) continue;
|
|
6094
|
+
const reg = this._batchAdapters.get(name);
|
|
6095
|
+
if (!reg) throw new Error(`generateBatch: adapter "${name}" not registered (call registerAdapter first)`);
|
|
6096
|
+
laneAdapters[i] = reg;
|
|
6097
|
+
}
|
|
6098
|
+
const release = await this._acquireGenLock();
|
|
6099
|
+
try {
|
|
6100
|
+
const startTime = performance.now();
|
|
6101
|
+
this.executor.resetBatch();
|
|
6102
|
+
const eosId = this.tokenizer.config.eosTokenId;
|
|
6103
|
+
const eotId = this.resolveEndOfTurnId();
|
|
6104
|
+
const keepPlan = this.config.vocabKeepPlan;
|
|
6105
|
+
const remapTok = keepPlan ? (i) => remapPrunedToken(i, keepPlan) : (i) => i;
|
|
6106
|
+
const rows = prompts.map((_p, i) => ({
|
|
6107
|
+
generatedIds: [],
|
|
6108
|
+
text: "",
|
|
6109
|
+
finished: false,
|
|
6110
|
+
finishReason: "max_tokens",
|
|
6111
|
+
limit: options.maxTokensPerRow?.[i] ?? maxTokens
|
|
6112
|
+
}));
|
|
6113
|
+
const firstTokens = new Uint32Array(B);
|
|
6114
|
+
for (let i = 0; i < B; i++) {
|
|
6115
|
+
const p = prompts[i];
|
|
6116
|
+
const messages = typeof p === "string" ? [...systemPrompt ? [{
|
|
6117
|
+
role: "system",
|
|
6118
|
+
content: systemPrompt
|
|
6119
|
+
}] : [], {
|
|
6120
|
+
role: "user",
|
|
6121
|
+
content: p
|
|
6122
|
+
}] : systemPrompt ? [{
|
|
6123
|
+
role: "system",
|
|
6124
|
+
content: systemPrompt
|
|
6125
|
+
}, ...p] : p;
|
|
6126
|
+
const inputIds = this.tokenizer.encodeChat(messages, { addGenerationPrompt: true });
|
|
6127
|
+
const laneReg = laneAdapters[i];
|
|
6128
|
+
if (laneReg) this.executor.applyRuntimeLoRA(laneReg.deltas);
|
|
6129
|
+
try {
|
|
6130
|
+
this.executor.beginBatchSlot(i);
|
|
6131
|
+
const { logits } = await this.executor.forward(new Uint32Array(inputIds));
|
|
6132
|
+
this.executor.finishBatchSlot(i);
|
|
6133
|
+
firstTokens[i] = remapTok(sampleToken(logits, { temperature: 0 }));
|
|
6134
|
+
} finally {
|
|
6135
|
+
if (laneReg) this.executor.clearRuntimeLoRA();
|
|
6136
|
+
}
|
|
6137
|
+
}
|
|
6138
|
+
if (this.executor.batchLoraEnabled) this.executor.setBatchLaneAdapters(laneAdapters.map((a) => a ? a.id : -1));
|
|
6139
|
+
const consume = (row, tok) => {
|
|
6140
|
+
if (row.finished) return;
|
|
6141
|
+
row.generatedIds.push(tok);
|
|
6142
|
+
if (eosId !== null && tok === eosId || eotId !== null && tok === eotId) {
|
|
6143
|
+
row.finished = true;
|
|
6144
|
+
row.finishReason = "eos";
|
|
6145
|
+
return;
|
|
6146
|
+
}
|
|
6147
|
+
row.text += this.tokenizer.decode([tok], true);
|
|
6148
|
+
if (stopSequences.some((s) => row.text.includes(s))) {
|
|
6149
|
+
for (const s of stopSequences) {
|
|
6150
|
+
const idx = row.text.indexOf(s);
|
|
6151
|
+
if (idx !== -1) row.text = row.text.slice(0, idx);
|
|
6152
|
+
}
|
|
6153
|
+
row.finished = true;
|
|
6154
|
+
row.finishReason = "stop_sequence";
|
|
6155
|
+
return;
|
|
6156
|
+
}
|
|
6157
|
+
if (row.generatedIds.length >= row.limit) row.finished = true;
|
|
6158
|
+
};
|
|
6159
|
+
const decodeStart = performance.now();
|
|
6160
|
+
for (let i = 0; i < B; i++) consume(rows[i], firstTokens[i]);
|
|
6161
|
+
const depth = Executor.PIPELINE_DEPTH;
|
|
6162
|
+
const lens = this.executor.batchSlotLengths;
|
|
6163
|
+
let capacity = Number.POSITIVE_INFINITY;
|
|
6164
|
+
for (let i = 0; i < B; i++) capacity = Math.min(capacity, this.maxSeqLen - lens[i] - 1);
|
|
6165
|
+
const maxRowLimit = rows.reduce((a, r) => Math.max(a, r.limit), 0);
|
|
6166
|
+
const stepsNeeded = Math.max(0, Math.min(maxRowLimit - 1, capacity));
|
|
6167
|
+
let submitted = 0;
|
|
6168
|
+
let consumedSteps = 0;
|
|
6169
|
+
while (consumedSteps < stepsNeeded && !rows.every((r) => r.finished)) {
|
|
6170
|
+
while (submitted < stepsNeeded && submitted < consumedSteps + depth) {
|
|
6171
|
+
this.executor.submitBatchDecodeStep(submitted === 0 ? firstTokens : null, submitted % depth);
|
|
6172
|
+
submitted++;
|
|
6173
|
+
}
|
|
6174
|
+
const toks = await this.executor.readBatchTokens(consumedSteps % depth);
|
|
6175
|
+
consumedSteps++;
|
|
6176
|
+
for (let i = 0; i < B; i++) consume(rows[i], toks[i]);
|
|
6177
|
+
}
|
|
6178
|
+
const decodeTime = performance.now() - decodeStart;
|
|
6179
|
+
const totalTime = performance.now() - startTime;
|
|
6180
|
+
return rows.map((r) => ({
|
|
6181
|
+
text: r.text,
|
|
6182
|
+
tokensGenerated: r.generatedIds.length,
|
|
6183
|
+
tokensPerSecond: r.generatedIds.length / (totalTime / 1e3),
|
|
6184
|
+
totalTime,
|
|
6185
|
+
decodeTime,
|
|
6186
|
+
finishReason: r.finishReason,
|
|
6187
|
+
tokenIds: [...r.generatedIds]
|
|
6188
|
+
}));
|
|
6189
|
+
} finally {
|
|
6190
|
+
release();
|
|
6191
|
+
}
|
|
6192
|
+
}
|
|
6193
|
+
/**
|
|
6194
|
+
* Continuous-batching scheduler (GERBIL_BATCH=N — Phase 5 of the batching
|
|
6195
|
+
* campaign, Dawn only). Returns the engine's request scheduler (created on
|
|
6196
|
+
* first call): submit many requests with independent prompts/limits and the
|
|
6197
|
+
* scheduler multiplexes them over the batch lanes with dynamic admission,
|
|
6198
|
+
* immediate lane retirement, and chunked prefill. Greedy-only; per-request
|
|
6199
|
+
* output is token-exact vs single-sequence generate().
|
|
6200
|
+
*/
|
|
6201
|
+
createScheduler() {
|
|
6202
|
+
this.checkDestroyed();
|
|
6203
|
+
if (this._scheduler) return this._scheduler;
|
|
6204
|
+
if (this.executor.batchSize <= 0) throw new Error("createScheduler requires GERBIL_BATCH=N (N >= 2) on the Dawn path");
|
|
6205
|
+
if (this._multimodalGraph || this.executor.hasPleSource() || this.executor.hasRuntimeAdapter) throw new Error("createScheduler: multimodal, PLE, and runtime-LoRA models are unsupported");
|
|
6206
|
+
const keepPlan = this.config.vocabKeepPlan;
|
|
6207
|
+
this._scheduler = new RequestScheduler({
|
|
6208
|
+
executor: this.executor,
|
|
6209
|
+
encodeChat: (messages) => this.tokenizer.encodeChat(messages, { addGenerationPrompt: true }),
|
|
6210
|
+
decodeToken: (id) => this.tokenizer.decode([id], true),
|
|
6211
|
+
eosTokenId: this.tokenizer.config.eosTokenId,
|
|
6212
|
+
eotTokenId: this.resolveEndOfTurnId(),
|
|
6213
|
+
remapToken: keepPlan ? (i) => remapPrunedToken(i, keepPlan) : (i) => i,
|
|
6214
|
+
resolveAdapter: (name) => {
|
|
6215
|
+
const reg = this._batchAdapters.get(name);
|
|
6216
|
+
if (!reg) throw new Error(`scheduler: adapter "${name}" not registered (call registerAdapter first)`);
|
|
6217
|
+
return reg;
|
|
6218
|
+
},
|
|
6219
|
+
acquireLock: () => this._acquireGenLock()
|
|
6220
|
+
});
|
|
6221
|
+
return this._scheduler;
|
|
6222
|
+
}
|
|
6223
|
+
/**
|
|
6224
|
+
* Create a swarm — the MoTA mode-B primitive (docs/research/slm-swarms.md):
|
|
6225
|
+
* named members (LoRA adapters on this engine's shared base, or null for the
|
|
6226
|
+
* bare base), an optional plan that fans an input out into per-member
|
|
6227
|
+
* sub-tasks, and an optional reduce that joins the results. `swarm.run()`
|
|
6228
|
+
* decodes every sub-task CONCURRENTLY through the continuous-batching
|
|
6229
|
+
* scheduler with per-request adapters — one batched pass, so the fan-out
|
|
6230
|
+
* costs roughly the slowest member, not the sum.
|
|
6231
|
+
*
|
|
6232
|
+
* Every non-null member source is registered via {@link registerAdapter}
|
|
6233
|
+
* (downloads in parallel; factors packed into shared GPU buffers once).
|
|
6234
|
+
* Requires `GERBIL_BATCH=N` (N >= 2) + `GERBIL_BATCH_LORA=1` on the Dawn
|
|
6235
|
+
* path; greedy-only like the scheduler underneath.
|
|
6236
|
+
*/
|
|
6237
|
+
async createSwarm(options) {
|
|
6238
|
+
this.checkDestroyed();
|
|
6239
|
+
if (!this.executor.batchLoraEnabled) throw new Error("createSwarm requires GERBIL_BATCH=N (N >= 2) and GERBIL_BATCH_LORA=1 on the Dawn path");
|
|
6240
|
+
const scheduler = this.createScheduler();
|
|
6241
|
+
await Promise.all(Object.entries(options.members).map(([name, source]) => source == null ? Promise.resolve() : this.registerAdapter(name, source)));
|
|
6242
|
+
return new Swarm(scheduler, options);
|
|
6243
|
+
}
|
|
6244
|
+
/**
|
|
5583
6245
|
* Resolve the end-of-turn stop token id, or null if the model has none.
|
|
5584
6246
|
*
|
|
5585
6247
|
* Chat models like Gemma 4 end an assistant turn with a dedicated end-of-turn
|
|
@@ -5665,18 +6327,34 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5665
6327
|
if (isGreedy && !this.executor.needsMultiEncoder && !mmDecode && !streamsPle) {
|
|
5666
6328
|
const firstToken = remapTok(sampleToken(logits, sampling, [...inputIds, ...generatedIds]));
|
|
5667
6329
|
if (!consumeToken(firstToken)) {
|
|
5668
|
-
const depth = Executor.PIPELINE_DEPTH;
|
|
5669
6330
|
const stepsNeeded = Math.min(maxTokens - 1, this.executor.decodeCapacityRemaining());
|
|
5670
|
-
|
|
5671
|
-
|
|
5672
|
-
|
|
5673
|
-
|
|
5674
|
-
|
|
5675
|
-
|
|
6331
|
+
const windowK = this.executor.decodeWindowK;
|
|
6332
|
+
if (windowK > 0) {
|
|
6333
|
+
let produced = 0;
|
|
6334
|
+
let stopped = false;
|
|
6335
|
+
while (!stopped && produced < stepsNeeded) {
|
|
6336
|
+
const steps = Math.min(windowK, stepsNeeded - produced);
|
|
6337
|
+
this.executor.submitGreedyDecodeWindow(produced === 0 ? firstToken : null, steps, 0);
|
|
6338
|
+
const tokens = await this.executor.readDecodeWindow(0, steps);
|
|
6339
|
+
produced += steps;
|
|
6340
|
+
for (const tok of tokens) if (consumeToken(tok)) {
|
|
6341
|
+
stopped = true;
|
|
6342
|
+
break;
|
|
6343
|
+
}
|
|
6344
|
+
}
|
|
6345
|
+
} else {
|
|
6346
|
+
const depth = Executor.PIPELINE_DEPTH;
|
|
6347
|
+
let submitted = 0;
|
|
6348
|
+
let consumed = 0;
|
|
6349
|
+
while (consumed < stepsNeeded) {
|
|
6350
|
+
while (submitted < stepsNeeded && submitted < consumed + depth) {
|
|
6351
|
+
this.executor.submitGreedyDecodeStep(submitted === 0 ? firstToken : null, submitted % depth);
|
|
6352
|
+
submitted++;
|
|
6353
|
+
}
|
|
6354
|
+
const tok = await this.executor.readDecodeToken(consumed % depth);
|
|
6355
|
+
consumed++;
|
|
6356
|
+
if (consumeToken(tok)) break;
|
|
5676
6357
|
}
|
|
5677
|
-
const tok = await this.executor.readDecodeToken(consumed % depth);
|
|
5678
|
-
consumed++;
|
|
5679
|
-
if (consumeToken(tok)) break;
|
|
5680
6358
|
}
|
|
5681
6359
|
}
|
|
5682
6360
|
} else {
|
|
@@ -5705,7 +6383,8 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5705
6383
|
tokensGenerated,
|
|
5706
6384
|
tokensPerSecond,
|
|
5707
6385
|
totalTime,
|
|
5708
|
-
finishReason
|
|
6386
|
+
finishReason,
|
|
6387
|
+
tokenIds: [...generatedIds]
|
|
5709
6388
|
};
|
|
5710
6389
|
}
|
|
5711
6390
|
/**
|
|
@@ -6214,7 +6893,8 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6214
6893
|
tokensGenerated,
|
|
6215
6894
|
tokensPerSecond: tokensGenerated / (totalTime / 1e3),
|
|
6216
6895
|
totalTime,
|
|
6217
|
-
finishReason
|
|
6896
|
+
finishReason,
|
|
6897
|
+
tokenIds: [...generatedIds]
|
|
6218
6898
|
};
|
|
6219
6899
|
}
|
|
6220
6900
|
/** Prepare + prefill + decode for a fully-specified multimodal token sequence. */
|
|
@@ -6299,7 +6979,8 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6299
6979
|
tokensGenerated,
|
|
6300
6980
|
tokensPerSecond: tokensGenerated / (totalTime / 1e3),
|
|
6301
6981
|
totalTime,
|
|
6302
|
-
finishReason
|
|
6982
|
+
finishReason,
|
|
6983
|
+
tokenIds: [...generatedIds]
|
|
6303
6984
|
};
|
|
6304
6985
|
}
|
|
6305
6986
|
/**
|
|
@@ -6780,5 +7461,5 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6780
7461
|
};
|
|
6781
7462
|
|
|
6782
7463
|
//#endregion
|
|
6783
|
-
export {
|
|
6784
|
-
//# sourceMappingURL=gpu-
|
|
7464
|
+
export { KaniTTS as A, audioTokensToDacCodes as C, parseOuteTtsConfig as D, generateOuteTtsBackboneGraph as E, patchGemma4VisionClips as F, resolveGemma4VisionInfo as I, dequantizeGemma4VisionProjection as M, dequantizeMLXProjection as N, Swarm as O, generateGemma4VisionGraph as P, loadOuteSpeaker as S, generateDacSpeechDecoderGraph as T, smartResize as _, buildGemma4PosEmbeds as a, OuteTTS as b, buildMRoPECosSin as c, buildPositionIds as d, buildRotaryCosSin as f, preprocessImageGemma4 as g, preprocessImage as h, buildGemma4PoolMatrix as i, generateQwen3_5VisionGraph as j, RequestScheduler as k, buildMRoPEPositionIds as l, mropeFreqDims as m, GEMMA4_IMAGE_PROCESSOR as n, buildGemma4RotaryCosSin as o, buildVisionPositionTensors as p, QWEN3_5_IMAGE_PROCESSOR as r, buildGemma4VisionPositionTensors as s, WebGPUEngine as t, buildPosEmbeds as u, VisionExecutor as v, dacOutputLength as w, buildOutePromptString as x, ParlerTTS as y };
|
|
7465
|
+
//# sourceMappingURL=gpu-DRFhiv4R.mjs.map
|