@aparte/provider-transformers 0.16.3 → 0.16.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,5 +1,5 @@
1
- import { uuid, contentToText } from "@aparte/core";
2
- const workerUrl = "" + new URL("assets/worker-Dz5bU2S7.js", import.meta.url).href;
1
+ import { uuid } from "@aparte/core";
2
+ const workerUrl = "" + new URL("assets/worker-Bm0eBd4l.js", import.meta.url).href;
3
3
  let _hardwareTiers = null;
4
4
  function setHardwareTierModels(tiers) {
5
5
  _hardwareTiers = tiers;
@@ -83,29 +83,20 @@ async function _refreshKnownModels() {
83
83
  } catch {
84
84
  }
85
85
  }
86
- let _warnedToolTurnsDropped = false;
87
- function toMessages(messages) {
88
- const result = [];
89
- let droppedToolTurns = 0;
90
- for (const m of messages) {
91
- if (m.role === "user" || m.role === "assistant" || m.role === "system") {
92
- const text = contentToText(m.content);
93
- if (text) result.push({ role: m.role, content: text });
94
- } else {
95
- droppedToolTurns++;
96
- }
97
- }
98
- if (droppedToolTurns > 0 && !_warnedToolTurnsDropped) {
99
- _warnedToolTurnsDropped = true;
100
- console.warn(
101
- `[transformers] Dropped ${droppedToolTurns} tool turn(s) from the prompt: this provider does not support tool calling (v1), so the model will not see the call or its result. Use an OpenAI-compatible endpoint for tools, or render the turns yourself before sending.`
102
- );
103
- }
104
- return result;
86
+ function _selection(modelId) {
87
+ const config = _registeredModels.get(modelId);
88
+ const runner = config?.runner;
89
+ return {
90
+ task: config?.task ?? "text-generation",
91
+ ...runner ? { runner: typeof location === "undefined" ? runner : new URL(runner, location.href).href } : {},
92
+ dtype: config?.dtype,
93
+ device: _computeDevice
94
+ };
105
95
  }
106
96
  let _worker = null;
107
97
  const _pendingPrepares = /* @__PURE__ */ new Map();
108
98
  const _pendingGenerates = /* @__PURE__ */ new Map();
99
+ const _pendingCommands = /* @__PURE__ */ new Map();
109
100
  let _generateChain = Promise.resolve();
110
101
  const _generateDoneResolvers = /* @__PURE__ */ new Map();
111
102
  const _queuedModelIds = /* @__PURE__ */ new Map();
@@ -140,7 +131,7 @@ function _spawnWorker() {
140
131
  const canMintBlob = typeof Blob === "function" && typeof URL.createObjectURL === "function";
141
132
  if (sameOrigin || !canMintBlob) return new Worker(new URL(
142
133
  /* @vite-ignore */
143
- "" + new URL("assets/worker-Dz5bU2S7.js", import.meta.url).href,
134
+ "" + new URL("assets/worker-Bm0eBd4l.js", import.meta.url).href,
144
135
  import.meta.url
145
136
  ), { type: "module" });
146
137
  _workerBlobUrl = URL.createObjectURL(
@@ -207,6 +198,8 @@ function _handleWorkerError(e) {
207
198
  }
208
199
  }
209
200
  _pendingGenerates.clear();
201
+ for (const c of _pendingCommands.values()) c.reject(new Error(message));
202
+ _pendingCommands.clear();
210
203
  for (const resolve of _generateDoneResolvers.values()) {
211
204
  try {
212
205
  resolve();
@@ -258,10 +251,23 @@ function _handleWorkerMessage(event) {
258
251
  void _enforceMaxCachedModels(msg.modelId).then(() => _refreshKnownModels());
259
252
  break;
260
253
  }
261
- case "gen-chunk": {
254
+ case "gen-event": {
262
255
  const ctrl = _pendingGenerates.get(msg.id);
263
256
  if (!ctrl) break;
264
- ctrl.enqueue({ type: msg.chunkType, delta: msg.delta });
257
+ ctrl.enqueue(msg.event);
258
+ break;
259
+ }
260
+ case "warning": {
261
+ console.warn(`[transformers] ${msg.message}`);
262
+ break;
263
+ }
264
+ case "command-result": {
265
+ _releaseGenerateSlot(msg.id);
266
+ const pending = _pendingCommands.get(msg.id);
267
+ if (!pending) break;
268
+ _pendingCommands.delete(msg.id);
269
+ if (msg.error !== void 0) pending.reject(new Error(msg.error));
270
+ else pending.resolve(msg.result);
265
271
  break;
266
272
  }
267
273
  case "gen-done": {
@@ -305,35 +311,59 @@ const TransformersProvider = {
305
311
  await _refreshKnownModels();
306
312
  return _knownModels;
307
313
  },
308
- async chat(request) {
309
- const messages = toMessages(request.messages);
314
+ async chat(request, _config, ctx) {
310
315
  const requestId = uuid();
311
316
  const options = {
312
317
  maxTokens: request.maxTokens,
313
318
  temperature: request.temperature,
314
319
  seed: request.seed
315
320
  };
316
- const task = _registeredModels.get(request.modelId)?.task ?? "text-generation";
321
+ const signal = ctx?.signal;
317
322
  _warnIfContended(request.modelId);
318
323
  _queuedModelIds.set(requestId, request.modelId);
319
324
  const prevGenerate = _generateChain;
320
325
  _generateChain = new Promise((resolveSlot) => {
321
326
  _generateDoneResolvers.set(requestId, resolveSlot);
322
327
  });
328
+ let posted = false;
329
+ let stopped = false;
330
+ const stop = () => {
331
+ if (stopped) return;
332
+ stopped = true;
333
+ signal?.removeEventListener("abort", stop);
334
+ if (posted) {
335
+ _getWorker().postMessage({ type: "cancel", id: requestId });
336
+ return;
337
+ }
338
+ const ctrl = _pendingGenerates.get(requestId);
339
+ _pendingGenerates.delete(requestId);
340
+ if (!ctrl) return;
341
+ try {
342
+ ctrl.enqueue({ type: "error", message: "Generation cancelled before it started" });
343
+ ctrl.close();
344
+ } catch {
345
+ }
346
+ };
323
347
  const postGenerate = () => {
348
+ if (stopped) {
349
+ _releaseGenerateSlot(requestId);
350
+ return;
351
+ }
352
+ posted = true;
324
353
  _getWorker().postMessage({
325
354
  type: "generate",
326
355
  id: requestId,
327
356
  modelId: request.modelId,
328
- messages,
357
+ // The conversation as it is, parts included: which parts a model can take
358
+ // is the runner's knowledge, not this thread's.
359
+ messages: request.messages,
329
360
  options,
330
- task,
331
- dtype: _registeredModels.get(request.modelId)?.dtype,
332
- device: _computeDevice
361
+ ..._selection(request.modelId)
333
362
  });
334
363
  };
364
+ let response;
335
365
  if (request.stream === false) {
336
- return new Promise((resolve, reject) => {
366
+ response = new Promise((resolve, reject) => {
337
367
  let result = "";
338
368
  const fakeCtrl = {
339
369
  enqueue: (chunk) => {
@@ -347,18 +377,22 @@ const TransformersProvider = {
347
377
  _pendingGenerates.set(requestId, fakeCtrl);
348
378
  void prevGenerate.then(postGenerate);
349
379
  });
380
+ } else {
381
+ response = new ReadableStream({
382
+ async start(controller) {
383
+ _pendingGenerates.set(requestId, controller);
384
+ await prevGenerate;
385
+ postGenerate();
386
+ },
387
+ cancel() {
388
+ _pendingGenerates.delete(requestId);
389
+ stop();
390
+ }
391
+ });
350
392
  }
351
- return new ReadableStream({
352
- async start(controller) {
353
- _pendingGenerates.set(requestId, controller);
354
- await prevGenerate;
355
- postGenerate();
356
- },
357
- cancel() {
358
- _pendingGenerates.delete(requestId);
359
- _getWorker().postMessage({ type: "cancel", id: requestId });
360
- }
361
- });
393
+ if (signal?.aborted) stop();
394
+ else signal?.addEventListener("abort", stop, { once: true });
395
+ return response;
362
396
  },
363
397
  async getModelStatus(modelId) {
364
398
  if (_loadedModelId === modelId) return "ready";
@@ -386,11 +420,9 @@ const TransformersProvider = {
386
420
  }
387
421
  const requestId = uuid();
388
422
  _preparingModelId = modelId;
389
- const task = _registeredModels.get(modelId)?.task ?? "text-generation";
390
- const dtype = _registeredModels.get(modelId)?.dtype;
391
423
  return new Promise((resolve, reject) => {
392
424
  _pendingPrepares.set(requestId, { modelId, onProgress, resolve, reject });
393
- _getWorker().postMessage({ type: "prepare", id: requestId, modelId, task, dtype, device: _computeDevice });
425
+ _getWorker().postMessage({ type: "prepare", id: requestId, modelId, ..._selection(modelId) });
394
426
  });
395
427
  },
396
428
  async deleteModel(modelId) {
@@ -400,6 +432,20 @@ const TransformersProvider = {
400
432
  function getLoadedModelId() {
401
433
  return _loadedModelId;
402
434
  }
435
+ function runnerCommand(modelId, name, payload) {
436
+ const requestId = uuid();
437
+ _queuedModelIds.set(requestId, modelId);
438
+ const previous = _generateChain;
439
+ _generateChain = new Promise((resolveSlot) => {
440
+ _generateDoneResolvers.set(requestId, resolveSlot);
441
+ });
442
+ return new Promise((resolve, reject) => {
443
+ _pendingCommands.set(requestId, { resolve, reject });
444
+ void previous.then(() => {
445
+ _getWorker().postMessage({ type: "command", id: requestId, modelId, name, payload, ..._selection(modelId) });
446
+ });
447
+ });
448
+ }
403
449
  function terminateWorker() {
404
450
  _worker?.terminate();
405
451
  _worker = null;
@@ -418,6 +464,8 @@ function terminateWorker() {
418
464
  }
419
465
  }
420
466
  _pendingGenerates.clear();
467
+ for (const c of _pendingCommands.values()) c.reject(new Error("Worker terminated"));
468
+ _pendingCommands.clear();
421
469
  for (const resolve of _generateDoneResolvers.values()) {
422
470
  try {
423
471
  resolve();
@@ -510,6 +558,7 @@ export {
510
558
  getMaxCachedModels,
511
559
  listCachedModels,
512
560
  registerModel,
561
+ runnerCommand,
513
562
  setComputeDevice,
514
563
  setHardwareTierModels,
515
564
  setMaxCachedModels,
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":["../src/index.ts"],"sourcesContent":["/**\n * @aparte/provider-transformers — run LLMs 100% in the browser via Transformers.js.\n *\n * A local, keyless `AparteAIProvider`: it owns its I/O (inference runs off the main\n * thread in a Web Worker) so `AparteDirectTransport` delegates to its `chat()`. Model\n * weights download once and persist in the Cache API.\n *\n * Scope (v1): generic **text-generation** streaming. Tool-calling for local models is\n * model-specific (every family has its own wire format) and is out of scope here — the\n * app registers models and streams plain replies. Vision / embeddings can follow on demand.\n *\n * ## This provider's state is TAB-scoped, on purpose\n *\n * Everything below the \"Worker bridge\" heading — the worker, the loaded model, the\n * generate chain — plus `setComputeDevice`, `setMaxCachedModels` and\n * `setHardwareTierModels`, is module-level and therefore shared by every chat on the\n * page. That is deliberate, and it is the opposite of what the rest of the suite does:\n * a plugin's providers scope to one chat, this one cannot.\n *\n * The reason is the resource, not the design. A local model is 1–2 GB of weights and one\n * WebGPU pipeline. Handing each chat its own worker would mean N copies resident in one\n * tab — which is the failure this package exists to avoid, not a capability. The\n * settings above describe the *machine* (which backend, how many models to keep\n * cached), so per-chat values would not mean anything either.\n *\n * What the constraint costs: two chats on the page driving DIFFERENT local models take\n * turns on one pipeline, so each turn may evict and reload gigabytes. That used to\n * happen silently — a multi-second stall with nothing to read. It now warns once, from\n * `chat()`, when a generate is queued for a model other than the one already in flight.\n * Same model in both chats is free and correct: they share the load.\n */\n\nimport type {\n AparteAIProvider,\n AparteAIModel,\n AparteChatRequest,\n AparteChatResponse,\n AparteChatMessage,\n ModelStatus,\n ModelLoadProgress,\n} from '@aparte/core';\nimport { contentToText, uuid } from '@aparte/core';\n\n// The worker's URL, not the worker itself: this package constructs it by hand because a\n// cross-origin copy has to go through a blob (see `_spawnWorker`). `?worker&url` is what\n// keeps Vite emitting the worker as its own chunk — the `new Worker(new URL(...))` form\n// it detects by pattern was the only other way, and moving the URL out of that call made\n// the build inline the worker's raw TypeScript as a data: URL instead. Caught by a\n// two-origin browser probe, not by any test.\nimport workerUrl from './worker.ts?worker&url';\n\n/** The minimal chat shape passed to the worker (the tokenizer applies the chat template). */\ntype SimpleMessage = { role: 'user' | 'assistant' | 'system'; content: string };\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Hardware detection\n// ─────────────────────────────────────────────────────────────────────────────\n\nexport interface HardwareProfile {\n hasGpu: boolean;\n ramGb: number;\n tier: 'low' | 'mid' | 'high';\n recommendedModelId: string;\n}\n\n/** Hardware-tier model overrides — set by the app via setHardwareTierModels(). */\nlet _hardwareTiers: { low: string; mid?: string; high: string } | null = null;\n\n/**\n * Set the model IDs to use per hardware tier. Call before detectHardware() is used\n * to pick a default model — the provider ships no model knowledge of its own.\n */\nexport function setHardwareTierModels(tiers: { low: string; mid?: string; high: string }): void {\n _hardwareTiers = tiers;\n}\n\nexport async function detectHardware(): Promise<HardwareProfile> {\n // navigator.deviceMemory: W3C API, Chromium only, capped at 8 GB for privacy\n // (1 | 2 | 4 | 8). Falls back to 4 on Firefox/Safari.\n const ramGb: number = (navigator as unknown as { deviceMemory?: number }).deviceMemory ?? 4;\n\n // Real WebGPU check: requestAdapter() returns null if no capable GPU is present.\n let hasGpu = false;\n if ('gpu' in navigator) {\n try {\n const adapter = await (navigator as unknown as { gpu: { requestAdapter(): Promise<unknown> } }).gpu.requestAdapter();\n hasGpu = adapter !== null;\n } catch {\n hasGpu = false;\n }\n }\n\n let tier: 'low' | 'mid' | 'high';\n if (!hasGpu || ramGb < 4) {\n tier = 'low';\n } else if (ramGb < 8) {\n tier = 'mid';\n } else {\n tier = 'high';\n }\n\n const recommendedModelId = _hardwareTiers\n ? (_hardwareTiers[tier] ?? _hardwareTiers.high ?? '')\n : '';\n\n return { hasGpu, ramGb, tier, recommendedModelId };\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Model catalog — all model knowledge lives in the app, not the provider.\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Configuration for a model registered with the provider. */\nexport interface TransformersModelConfig {\n id: string;\n name: string;\n description?: string;\n capabilities: AparteAIModel['capabilities'];\n /** Transformers.js pipeline task — determines the model architecture / load path. */\n task: 'text-generation';\n /** ONNX dtype or per-part dtype map (e.g. `'q4'` or `{ decoder_model_merged: 'q4' }`). */\n dtype?: string | Record<string, string>;\n /** Preferred device. Defaults to WebGPU when available, else WASM. */\n device?: 'webgpu' | 'wasm' | 'auto';\n metadata?: Record<string, unknown>;\n}\n\n/** Models registered by the app via registerModel(). */\nconst _registeredModels = new Map<string, TransformersModelConfig>();\n\n/** Mutable model list — populated by registerModel() and cache discovery. */\nlet _knownModels: AparteAIModel[] = [];\n\n/**\n * Register a model with the provider. Call before the model is used for inference.\n */\nexport function registerModel(config: TransformersModelConfig): void {\n _registeredModels.set(config.id, config);\n if (!_knownModels.find(m => m.id === config.id)) {\n _knownModels = [..._knownModels, {\n id: config.id,\n name: config.name,\n description: config.description,\n capabilities: config.capabilities,\n }];\n }\n}\n\n/** Build an AparteAIModel entry from a cache-discovered modelId not in the registry. */\nfunction _modelFromCacheEntry(modelId: string): AparteAIModel {\n const config = _registeredModels.get(modelId);\n if (config) return { id: config.id, name: config.name, description: config.description, capabilities: config.capabilities };\n const name = (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n return { id: modelId, name, capabilities: ['streaming'] };\n}\n\n/** Max number of models to keep in cache. 0 = unlimited. Default: 1. */\nlet _maxCachedModels = 1;\n\n/**\n * Set the maximum number of models to keep in cache. When exceeded after a new\n * model is ready, the oldest models are evicted. 0 = unlimited.\n */\nexport function setMaxCachedModels(max: number): void {\n _maxCachedModels = max;\n}\n\n/** Returns the current max-cached-models setting. */\nexport function getMaxCachedModels(): number {\n return _maxCachedModels;\n}\n\n/**\n * User's preferred compute backend for local inference.\n * 'auto' → WebGPU when available, else WASM (default)\n * 'webgpu' → force WebGPU\n * 'wasm' → force WASM CPU\n */\nexport type ComputeDevice = 'auto' | 'webgpu' | 'wasm';\nlet _computeDevice: ComputeDevice = 'auto';\n\nexport function setComputeDevice(d: ComputeDevice): void {\n _computeDevice = d;\n}\n\nexport function getComputeDevice(): ComputeDevice {\n return _computeDevice;\n}\n\n/** Evict models from cache until count <= _maxCachedModels; `keepModelId` is never evicted. */\nasync function _enforceMaxCachedModels(keepModelId: string): Promise<void> {\n if (_maxCachedModels === 0) return; // unlimited\n try {\n const cached = await listCachedModels();\n const others = cached.filter(e => e.modelId !== keepModelId);\n const excess = cached.length - _maxCachedModels;\n if (excess <= 0) return;\n // Delete the excess models (oldest first — they appear first in cache scan order).\n for (let i = 0; i < excess && i < others.length; i++) {\n await deleteCachedModel(others[i]!.modelId);\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Merge cached models into _knownModels (idempotent). Called by fetchModels(). */\nasync function _refreshKnownModels(): Promise<void> {\n try {\n const cached = await listCachedModels();\n for (const entry of cached) {\n if (!_knownModels.find(m => m.id === entry.modelId)) {\n _knownModels = [..._knownModels, _modelFromCacheEntry(entry.modelId)];\n }\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Warn at most once per session that tool turns were left out of the prompt. */\nlet _warnedToolTurnsDropped = false;\n\n/** AparteChatMessage[] → plain chat turns (the tokenizer's chat template does the rest). */\nfunction toMessages(messages: AparteChatMessage[]): SimpleMessage[] {\n const result: SimpleMessage[] = [];\n let droppedToolTurns = 0;\n for (const m of messages) {\n if (m.role === 'user' || m.role === 'assistant' || m.role === 'system') {\n const text = contentToText(m.content);\n if (text) result.push({ role: m.role, content: text });\n } else {\n // tool_call / tool_result are not supported by this generic provider (v1):\n // rendering them needs a model-specific tool syntax. Dropping them\n // silently meant an app with registered tools got a model that never saw\n // the call or its result, with nothing to explain the behaviour.\n droppedToolTurns++;\n }\n }\n if (droppedToolTurns > 0 && !_warnedToolTurnsDropped) {\n _warnedToolTurnsDropped = true;\n console.warn(\n `[transformers] Dropped ${droppedToolTurns} tool turn(s) from the prompt: this provider ` +\n 'does not support tool calling (v1), so the model will not see the call or its result. ' +\n 'Use an OpenAI-compatible endpoint for tools, or render the turns yourself before sending.',\n );\n }\n return result;\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Worker bridge\n// ─────────────────────────────────────────────────────────────────────────────\n\nlet _worker: Worker | null = null;\n\ninterface PendingPrepare {\n modelId: string;\n onProgress: (p: ModelLoadProgress) => void;\n resolve: () => void;\n reject: (err: Error) => void;\n}\nconst _pendingPrepares = new Map<string, PendingPrepare>();\nconst _pendingGenerates = new Map<string, ReadableStreamDefaultController>();\n\n// ── Generate serialization ──────────────────────────────────────────────────\n// The worker holds ONE pipeline: two concurrent generates would corrupt each\n// other. Each chat() chains its `generate` behind the previous generate's\n// completion (gen-done / gen-error).\nlet _generateChain: Promise<void> = Promise.resolve();\nconst _generateDoneResolvers = new Map<string, () => void>();\n\n// ── Contention on the one pipeline ──────────────────────────────────────────\n// Serialization is correct but invisible: two chats driving DIFFERENT local\n// models take turns, and with `maxCachedModels` at its default of 1 each turn\n// can evict and reload gigabytes. The user sees a stall; the developer sees\n// nothing. These two track just enough to say so, once.\nconst _queuedModelIds = new Map<string, string>();\nlet _warnedModelContention = false;\n\n/** Model ids of generates currently queued or running on the single pipeline. */\nfunction _contendingModelId(requested: string): string | undefined {\n for (const id of _queuedModelIds.values()) if (id !== requested) return id;\n return undefined;\n}\n\n/**\n * Warn once when a generate has to queue behind another chat's DIFFERENT model.\n * Not a warning about switching models in one chat — that is a deliberate act\n * with visible feedback. This fires only when two are in flight at once.\n */\nfunction _warnIfContended(requested: string): void {\n if (_warnedModelContention) return;\n const other = _contendingModelId(requested);\n if (!other) return;\n _warnedModelContention = true;\n console.warn(\n `[Aparte] Two chats are driving different local models at once (\"${requested}\" behind `\n + `\"${other}\"). Transformers.js runs one pipeline per tab, so these generates are `\n + `serialized, and with a cache budget of ${_maxCachedModels} each switch can evict and `\n + `reload gigabytes of weights. Point both chats at one model, or raise the budget with `\n + `setMaxCachedModels(2) if the machine has the memory. This warns once.`,\n );\n}\n\n/** Settle the serialization slot for a finished generate. */\nfunction _releaseGenerateSlot(id: string): void {\n _queuedModelIds.delete(id);\n const resolve = _generateDoneResolvers.get(id);\n if (resolve) {\n _generateDoneResolvers.delete(id);\n resolve();\n }\n}\n\n/** Model known to be loaded (main-thread view). */\nlet _loadedModelId: string | null = null;\n/** Model currently being prepared (for the getModelStatus 'cached' path). */\nlet _preparingModelId: string | null = null;\n\n/** The blob URL the worker was built from, if it needed one. Revoked with the worker. */\nlet _workerBlobUrl: string | null = null;\n\n/**\n * Build the worker — including when this package is served from another origin.\n *\n * `new Worker()` refuses a cross-origin script outright, and that is not an exotic\n * case: it is every deploy whose JavaScript lives on a CDN or an asset host while the\n * page lives somewhere else, with or without a bundler. Reproduced with the package on\n * one port and the page on another: `SecurityError: Script at '…/assets/worker-*.js'\n * cannot be accessed from origin '…'`.\n *\n * A blob inherits the ORIGIN OF THE DOCUMENT THAT CREATES IT, so a one-line blob whose\n * body imports the real worker by absolute URL is same-origin by construction, and the\n * import inside it is a normal cross-origin module fetch, which is allowed. It is the\n * shim ffmpeg.wasm and tesseract.js use for the same reason.\n *\n * Same-origin keeps the direct path: no blob, nothing to revoke, and a stack trace that\n * names the real file.\n */\nfunction _spawnWorker(): Worker {\n const url = new URL(workerUrl, import.meta.url);\n const sameOrigin = typeof location === 'undefined' || url.origin === location.origin;\n // A blob is the only way across an origin, so an environment that cannot mint one has\n // nothing to gain from trying: construct directly and let the platform say what it\n // thinks. jsdom is that environment — it has `Blob` and no `URL.createObjectURL` — and\n // every test in this package went through the blob path and threw before this line\n // existed.\n const canMintBlob = typeof Blob === 'function' && typeof URL.createObjectURL === 'function';\n // The literal below is not style. `new Worker(new URL('./worker.ts', import.meta.url))`\n // is the exact shape Vite's worker detection and webpack's WorkerPlugin match on, and\n // matching it is what makes a CONSUMER's bundler process the worker as a module — which\n // is how `@huggingface/transformers` gets resolved inside it today. Behind a variable\n // the chunk is copied as an opaque asset and its imports are never touched, so hoisting\n // this line to reuse it for the blob would fix a CDN page by breaking every bundled app.\n if (sameOrigin || !canMintBlob) return new Worker(new URL('./worker.ts', import.meta.url), { type: 'module' });\n\n _workerBlobUrl = URL.createObjectURL(\n new Blob([`import ${JSON.stringify(url.href)};`], { type: 'text/javascript' }),\n );\n try {\n return new Worker(_workerBlobUrl, { type: 'module' });\n } catch (error) {\n // A page with `worker-src 'self'` (or `script-src` without `blob:`) blocks the\n // shim, and the direct URL was already refused for its origin — so there is\n // nothing left to try. Say which of the two walls was hit, because the browser's\n // own message does not distinguish them.\n URL.revokeObjectURL(_workerBlobUrl);\n _workerBlobUrl = null;\n throw new Error(\n `@aparte/provider-transformers is served from ${url.origin}, which is not this page's origin, `\n + 'so its worker has to be started through a blob: URL — and this page\\'s Content-Security-Policy '\n + 'refuses that. Allow `blob:` in `worker-src` (or `script-src`), or serve the package from your '\n + `own origin. Original error: ${String(error)}`,\n );\n }\n}\n\nfunction _releaseWorkerBlob(): void {\n if (_workerBlobUrl) {\n URL.revokeObjectURL(_workerBlobUrl);\n _workerBlobUrl = null;\n }\n}\n\n/**\n * Where the page says Transformers.js lives, if it says so at all.\n *\n * The worker cannot ask: an import map is the DOCUMENT's, and by spec it does not reach\n * a worker. The main thread can, and does it the platform's way — `import.meta.resolve`\n * consults that same map — so a page that already maps `@huggingface/transformers` (it\n * has to, to import this package by name at all) is telling us where its copy is. That\n * map is the CDN consumer's manifest: the version pin stays with the consumer, which is\n * the whole point of a peer dependency, and this package invents no second place to say\n * it.\n *\n * `undefined` under a bundler, where the specifier is resolved at build time and the\n * worker's own `import('@huggingface/transformers')` is the path that runs.\n */\nfunction _peerModuleUrl(): string | undefined {\n const resolve = (import.meta as unknown as { resolve?: (specifier: string) => string }).resolve;\n if (typeof resolve === 'function') {\n try {\n const href = resolve('@huggingface/transformers');\n if (href && /^https?:/i.test(href)) return href;\n } catch { /* not in the map — fall through */ }\n }\n // Older engines have no `import.meta.resolve`; read the map they do have.\n try {\n const el = document.querySelector('script[type=\"importmap\"]');\n const map = el?.textContent ? JSON.parse(el.textContent) as { imports?: Record<string, string> } : null;\n const href = map?.imports?.['@huggingface/transformers'];\n if (href) return new URL(href, location.href).href;\n } catch { /* no document, or a map that is not JSON */ }\n return undefined;\n}\n\nfunction _getWorker(): Worker {\n if (!_worker) {\n _worker = _spawnWorker();\n _worker.addEventListener('message', _handleWorkerMessage);\n _worker.addEventListener('error', _handleWorkerError);\n _worker.addEventListener('messageerror', _handleWorkerError);\n // First message, before any work: postMessage keeps order, so the worker has it\n // by the time a prepare or a generate needs the module.\n _worker.postMessage({ type: 'init', transformersUrl: _peerModuleUrl() });\n }\n return _worker;\n}\n\n/**\n * Worker crashed (uncaught error / WASM init failure / OOM). Reject every in-flight\n * prepare and close every open generate stream so the UI doesn't hang. Subsequent\n * calls rebuild the worker.\n */\nfunction _handleWorkerError(e: Event): void {\n const message = (e as ErrorEvent)?.message || 'Worker crashed unexpectedly';\n\n for (const p of _pendingPrepares.values()) {\n try { p.reject(new Error(message)); } catch { /* ignore */ }\n }\n _pendingPrepares.clear();\n\n for (const ctrl of _pendingGenerates.values()) {\n try { ctrl.enqueue({ type: 'error' as const, message }); ctrl.close(); }\n catch { /* ignore */ }\n }\n _pendingGenerates.clear();\n\n // Release every serialization slot so the generate chain doesn't deadlock.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n\n _loadedModelId = null;\n _preparingModelId = null;\n try { _worker?.terminate(); } catch { /* ignore */ }\n _worker = null;\n _releaseWorkerBlob();\n}\n\nfunction _handleWorkerMessage(event: MessageEvent): void {\n const msg = event.data;\n\n switch (msg.type) {\n case 'progress': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n if (msg.status === 'ready') {\n pending.onProgress({ status: 'ready' });\n pending.resolve();\n _pendingPrepares.delete(msg.id);\n } else if (msg.status === 'loading') {\n pending.onProgress({ status: 'loading' });\n } else if (msg.status === 'cached') {\n pending.onProgress({ status: 'cached', file: msg.file, progress: msg.progress });\n } else {\n pending.onProgress({ status: 'downloading', file: msg.file, progress: msg.progress });\n }\n break;\n }\n case 'prepare-error': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n pending.reject(new Error(msg.message));\n _pendingPrepares.delete(msg.id);\n if (_preparingModelId === pending.modelId) _preparingModelId = null;\n break;\n }\n case 'pipeline-ready': {\n _loadedModelId = msg.modelId;\n _preparingModelId = null;\n // Evict models over the cache limit, then refresh the known list.\n void _enforceMaxCachedModels(msg.modelId).then(() => _refreshKnownModels());\n break;\n }\n case 'gen-chunk': {\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: msg.chunkType as 'text' | 'thinking', delta: msg.delta });\n break;\n }\n case 'gen-done': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'done' as const, ...(msg.usage ? { usage: msg.usage } : {}) });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n case 'gen-error': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'error' as const, message: msg.message });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n }\n}\n\n/**\n * Narrowed so the two members the docs tell you to CALL are not optional.\n *\n * `AparteAIProvider` declares `prepareModel` and `getModelStatus` optional (most\n * providers have nothing to download), and widening to it made both\n * possibly-undefined — so the documented `TransformersProvider.prepareModel(...)`\n * needed a `!` or a guard in every strict consumer. Same technique openai-compat\n * already used for its own always-present members.\n *\n * `chat` joined the list once `AparteAIProvider` became a union: it is optional on\n * the format-adapter arm, and this provider IS its `chat()` — running inference\n * locally is the whole package. Narrowing it here says so once, instead of every\n * caller writing `provider.chat!(...)`.\n */\nexport const TransformersProvider: AparteAIProvider\n & Required<Pick<AparteAIProvider, 'prepareModel' | 'getModelStatus' | 'chat'>> = {\n id: 'transformers',\n\n getMetadata() {\n return {\n id: 'transformers',\n name: 'Transformers.js',\n icon: `<svg viewBox=\"0 0 24 24\" fill=\"none\" xmlns=\"http://www.w3.org/2000/svg\"><path d=\"M12 2L2 7l10 5 10-5-10-5z\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 17l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 12l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/></svg>`,\n color: '#f59e0b',\n description: 'Run LLMs directly in your browser via WebGPU or WASM — no API, no key',\n hasFreeModels: true,\n isLocal: true,\n helpUrl: 'https://huggingface.co/docs/transformers.js',\n };\n },\n\n getModels(): AparteAIModel[] {\n return _knownModels;\n },\n\n async fetchModels(): Promise<AparteAIModel[]> {\n await _refreshKnownModels();\n return _knownModels;\n },\n\n async chat(request: AparteChatRequest): Promise<AparteChatResponse> {\n const messages = toMessages(request.messages);\n const requestId = uuid();\n const options = {\n maxTokens: request.maxTokens,\n temperature: request.temperature,\n seed: request.seed,\n };\n const task = _registeredModels.get(request.modelId)?.task ?? 'text-generation';\n\n // ── Reserve a serialization slot ─────────────────────────────────────\n // Chain this generate behind the previous one; the worker has a single\n // pipeline, so generates MUST NOT overlap.\n _warnIfContended(request.modelId);\n _queuedModelIds.set(requestId, request.modelId);\n const prevGenerate = _generateChain;\n _generateChain = new Promise<void>((resolveSlot) => {\n _generateDoneResolvers.set(requestId, resolveSlot);\n });\n const postGenerate = (): void => {\n _getWorker().postMessage({\n type: 'generate',\n id: requestId,\n modelId: request.modelId,\n messages,\n options,\n task,\n dtype: _registeredModels.get(request.modelId)?.dtype,\n device: _computeDevice,\n });\n };\n\n if (request.stream === false) {\n return new Promise<string>((resolve, reject) => {\n let result = '';\n const fakeCtrl = {\n enqueue: (chunk: { type: string; delta?: string; message?: string }) => {\n if (chunk.type === 'text') result += chunk.delta ?? '';\n else if (chunk.type === 'done') resolve(result);\n else if (chunk.type === 'error') reject(new Error(chunk.message));\n },\n close: () => { /* no-op */ },\n } as unknown as ReadableStreamDefaultController;\n _pendingGenerates.set(requestId, fakeCtrl);\n void prevGenerate.then(postGenerate);\n });\n }\n\n return new ReadableStream({\n async start(controller) {\n _pendingGenerates.set(requestId, controller);\n await prevGenerate;\n postGenerate();\n },\n cancel() {\n _pendingGenerates.delete(requestId);\n // Actually STOP the model (not just detach the reader): tell the worker\n // to interrupt this generate. The serialization slot is still released\n // by the resulting gen-done/gen-error, so a queued generate can't start\n // before the worker has stopped this one.\n _getWorker().postMessage({ type: 'cancel', id: requestId });\n },\n });\n },\n\n async getModelStatus(modelId: string): Promise<ModelStatus> {\n if (_loadedModelId === modelId) return 'ready';\n if (_preparingModelId === modelId) return 'cached';\n if ('caches' in globalThis) {\n try {\n const encodedId = encodeURIComponent(modelId);\n const names = await caches.keys();\n for (const name of names) {\n const cache = await caches.open(name);\n const keys = await cache.keys();\n if (keys.some(r => r.url.includes(encodedId) || r.url.includes(modelId + '/'))) {\n return 'cached';\n }\n }\n } catch {\n // Cache API unavailable\n }\n }\n return 'not-downloaded';\n },\n\n async prepareModel(modelId: string, onProgress: (p: ModelLoadProgress) => void): Promise<void> {\n if (_loadedModelId === modelId) {\n onProgress({ status: 'ready' });\n return;\n }\n\n const requestId = uuid();\n _preparingModelId = modelId;\n\n const task = _registeredModels.get(modelId)?.task ?? 'text-generation';\n const dtype = _registeredModels.get(modelId)?.dtype;\n return new Promise<void>((resolve, reject) => {\n _pendingPrepares.set(requestId, { modelId, onProgress, resolve, reject });\n _getWorker().postMessage({ type: 'prepare', id: requestId, modelId, task, dtype, device: _computeDevice });\n });\n },\n\n async deleteModel(modelId: string): Promise<void> {\n await deleteCachedModel(modelId);\n },\n};\n\nexport default TransformersProvider;\nexport type { AparteAIProvider, AparteAIModel, ModelStatus, ModelLoadProgress } from '@aparte/core';\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Cache utilities (settings panels, etc.)\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Returns the modelId currently loaded in the worker's pipeline, or null. */\nexport function getLoadedModelId(): string | null {\n return _loadedModelId;\n}\n\n/** Terminate the shared worker and reset in-memory state. Safe to call any time. */\nexport function terminateWorker(): void {\n _worker?.terminate();\n _worker = null;\n _releaseWorkerBlob();\n _loadedModelId = null;\n _preparingModelId = null;\n for (const [, p] of _pendingPrepares) {\n p.reject(new Error('Worker terminated'));\n }\n _pendingPrepares.clear();\n for (const [, ctrl] of _pendingGenerates) {\n try { ctrl.enqueue({ type: 'error' as const, message: 'Worker terminated' }); ctrl.close(); } catch { /* already closed */ }\n }\n _pendingGenerates.clear();\n\n // Release every serialization slot and reset the chain — the same three lines\n // the worker-error handler above already carried, with the same reason. Without\n // them, terminating mid-generate left `_generateChain` pending on a resolver\n // that had just been dropped, so the NEXT chat() awaited a promise that could\n // never settle: no error, no rejection, the stream simply never started again\n // for the life of the page.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n // A terminated worker is a fresh situation; let the contention warning speak again.\n _warnedModelContention = false;\n}\n\nexport interface CachedModelEntry {\n modelId: string;\n name: string;\n /** Total size in bytes of all cached files for this model. -1 if unknown. */\n sizeBytes: number;\n /** True if the model is currently loaded in the worker. */\n loaded: boolean;\n}\n\n/**\n * Scan the Cache API to find which Transformers.js models have been downloaded,\n * by matching cache entry URLs against the Hugging Face resolve path.\n */\nexport async function listCachedModels(): Promise<CachedModelEntry[]> {\n if (!('caches' in globalThis)) return [];\n\n const found = new Map<string, { name: string; sizeBytes: number }>();\n\n // e.g. https://huggingface.co/onnx-community/Qwen2.5-0.5B/resolve/main/config.json\n // → onnx-community/Qwen2.5-0.5B\n function extractModelId(url: string): string | null {\n const m = url.match(/huggingface\\.co\\/([^/]+\\/[^/]+)\\/resolve\\//);\n return m ? decodeURIComponent(m[1]!) : null;\n }\n\n function modelName(modelId: string): string {\n const config = _registeredModels.get(modelId);\n if (config) return config.name;\n return (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n }\n\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n for (const req of requests) {\n const modelId = extractModelId(req.url);\n if (!modelId) continue;\n if (!found.has(modelId)) {\n found.set(modelId, { name: modelName(modelId), sizeBytes: 0 });\n }\n const response = await cache.match(req);\n if (!response) continue;\n const contentLength = response.headers.get('content-length');\n if (contentLength) {\n found.get(modelId)!.sizeBytes += parseInt(contentLength, 10);\n } else {\n try {\n const blob = await response.clone().blob();\n found.get(modelId)!.sizeBytes += blob.size;\n } catch { /* skip */ }\n }\n }\n } catch { /* skip inaccessible cache */ }\n }));\n } catch {\n return [];\n }\n\n return Array.from(found.entries()).map(([modelId, { name, sizeBytes }]) => ({\n modelId,\n name,\n sizeBytes,\n loaded: _loadedModelId === modelId,\n }));\n}\n\n/**\n * Delete all cached files for a modelId from the Cache API, terminating the worker\n * first if that model is currently loaded.\n */\nexport async function deleteCachedModel(modelId: string): Promise<void> {\n if (_loadedModelId === modelId || _preparingModelId === modelId) {\n terminateWorker();\n }\n if (!('caches' in globalThis)) return;\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n const encoded = encodeURIComponent(modelId);\n await Promise.all(\n requests\n .filter(r => r.url.includes(modelId) || r.url.includes(encoded))\n .map(r => cache.delete(r)),\n );\n } catch { /* skip */ }\n }));\n } catch { /* Cache API unavailable */ }\n}\n"],"names":[],"mappings":";;AAkEA,IAAI,iBAAqE;AAMlE,SAAS,sBAAsB,OAA0D;AAC5F,mBAAiB;AACrB;AAEA,eAAsB,iBAA2C;AAG7D,QAAM,QAAiB,UAAmD,gBAAgB;AAG1F,MAAI,SAAS;AACb,MAAI,SAAS,WAAW;AACpB,QAAI;AACA,YAAM,UAAU,MAAO,UAAyE,IAAI,eAAA;AACpG,eAAS,YAAY;AAAA,IACzB,QAAQ;AACJ,eAAS;AAAA,IACb;AAAA,EACJ;AAEA,MAAI;AACJ,MAAI,CAAC,UAAU,QAAQ,GAAG;AACtB,WAAO;AAAA,EACX,WAAW,QAAQ,GAAG;AAClB,WAAO;AAAA,EACX,OAAO;AACH,WAAO;AAAA,EACX;AAEA,QAAM,qBAAqB,iBACpB,eAAe,IAAI,KAAK,eAAe,QAAQ,KAChD;AAEN,SAAO,EAAE,QAAQ,OAAO,MAAM,mBAAA;AAClC;AAsBA,MAAM,wCAAwB,IAAA;AAG9B,IAAI,eAAgC,CAAA;AAK7B,SAAS,cAAc,QAAuC;AACjE,oBAAkB,IAAI,OAAO,IAAI,MAAM;AACvC,MAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,OAAO,EAAE,GAAG;AAC7C,mBAAe,CAAC,GAAG,cAAc;AAAA,MAC7B,IAAI,OAAO;AAAA,MACX,MAAM,OAAO;AAAA,MACb,aAAa,OAAO;AAAA,MACpB,cAAc,OAAO;AAAA,IAAA,CACxB;AAAA,EACL;AACJ;AAGA,SAAS,qBAAqB,SAAgC;AAC1D,QAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,MAAI,OAAQ,QAAO,EAAE,IAAI,OAAO,IAAI,MAAM,OAAO,MAAM,aAAa,OAAO,aAAa,cAAc,OAAO,aAAA;AAC7G,QAAM,QAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AACpE,SAAO,EAAE,IAAI,SAAS,MAAM,cAAc,CAAC,WAAW,EAAA;AAC1D;AAGA,IAAI,mBAAmB;AAMhB,SAAS,mBAAmB,KAAmB;AAClD,qBAAmB;AACvB;AAGO,SAAS,qBAA6B;AACzC,SAAO;AACX;AASA,IAAI,iBAAgC;AAE7B,SAAS,iBAAiB,GAAwB;AACrD,mBAAiB;AACrB;AAEO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGA,eAAe,wBAAwB,aAAoC;AACvE,MAAI,qBAAqB,EAAG;AAC5B,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,UAAM,SAAS,OAAO,OAAO,CAAA,MAAK,EAAE,YAAY,WAAW;AAC3D,UAAM,SAAS,OAAO,SAAS;AAC/B,QAAI,UAAU,EAAG;AAEjB,aAAS,IAAI,GAAG,IAAI,UAAU,IAAI,OAAO,QAAQ,KAAK;AAClD,YAAM,kBAAkB,OAAO,CAAC,EAAG,OAAO;AAAA,IAC9C;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,eAAe,sBAAqC;AAChD,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,eAAW,SAAS,QAAQ;AACxB,UAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,MAAM,OAAO,GAAG;AACjD,uBAAe,CAAC,GAAG,cAAc,qBAAqB,MAAM,OAAO,CAAC;AAAA,MACxE;AAAA,IACJ;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,IAAI,0BAA0B;AAG9B,SAAS,WAAW,UAAgD;AAChE,QAAM,SAA0B,CAAA;AAChC,MAAI,mBAAmB;AACvB,aAAW,KAAK,UAAU;AACtB,QAAI,EAAE,SAAS,UAAU,EAAE,SAAS,eAAe,EAAE,SAAS,UAAU;AACpE,YAAM,OAAO,cAAc,EAAE,OAAO;AACpC,UAAI,aAAa,KAAK,EAAE,MAAM,EAAE,MAAM,SAAS,MAAM;AAAA,IACzD,OAAO;AAKH;AAAA,IACJ;AAAA,EACJ;AACA,MAAI,mBAAmB,KAAK,CAAC,yBAAyB;AAClD,8BAA0B;AAC1B,YAAQ;AAAA,MACJ,0BAA0B,gBAAgB;AAAA,IAAA;AAAA,EAIlD;AACA,SAAO;AACX;AAMA,IAAI,UAAyB;AAQ7B,MAAM,uCAAuB,IAAA;AAC7B,MAAM,wCAAwB,IAAA;AAM9B,IAAI,iBAAgC,QAAQ,QAAA;AAC5C,MAAM,6CAA6B,IAAA;AAOnC,MAAM,sCAAsB,IAAA;AAC5B,IAAI,yBAAyB;AAG7B,SAAS,mBAAmB,WAAuC;AAC/D,aAAW,MAAM,gBAAgB,OAAA,EAAU,KAAI,OAAO,UAAW,QAAO;AACxE,SAAO;AACX;AAOA,SAAS,iBAAiB,WAAyB;AAC/C,MAAI,uBAAwB;AAC5B,QAAM,QAAQ,mBAAmB,SAAS;AAC1C,MAAI,CAAC,MAAO;AACZ,2BAAyB;AACzB,UAAQ;AAAA,IACJ,mEAAmE,SAAS,aACtE,KAAK,gHACiC,gBAAgB;AAAA,EAAA;AAIpE;AAGA,SAAS,qBAAqB,IAAkB;AAC5C,kBAAgB,OAAO,EAAE;AACzB,QAAM,UAAU,uBAAuB,IAAI,EAAE;AAC7C,MAAI,SAAS;AACT,2BAAuB,OAAO,EAAE;AAChC,YAAA;AAAA,EACJ;AACJ;AAGA,IAAI,iBAAgC;AAEpC,IAAI,oBAAmC;AAGvC,IAAI,iBAAgC;AAmBpC,SAAS,eAAuB;AAC5B,QAAM,MAAM,IAAI,IAAI,WAAW,YAAY,GAAG;AAC9C,QAAM,aAAa,OAAO,aAAa,eAAe,IAAI,WAAW,SAAS;AAM9E,QAAM,cAAc,OAAO,SAAS,cAAc,OAAO,IAAI,oBAAoB;AAOjF,MAAI,cAAc,CAAC,YAAa,QAAO,IAAI,OAAO,IAAA;AAAA;AAAA,IAAA,KAAA,IAAA,IAAA,6BAAA,YAAA,GAAA,EAAA;AAAA,IAAA,YAAA;AAAA,EAAA,GAAyC,EAAE,MAAM,UAAU;AAE7G,mBAAiB,IAAI;AAAA,IACjB,IAAI,KAAK,CAAC,UAAU,KAAK,UAAU,IAAI,IAAI,CAAC,GAAG,GAAG,EAAE,MAAM,mBAAmB;AAAA,EAAA;AAEjF,MAAI;AACA,WAAO,IAAI,OAAO,gBAAgB,EAAE,MAAM,UAAU;AAAA,EACxD,SAAS,OAAO;AAKZ,QAAI,gBAAgB,cAAc;AAClC,qBAAiB;AACjB,UAAM,IAAI;AAAA,MACN,gDAAgD,IAAI,MAAM,oQAGzB,OAAO,KAAK,CAAC;AAAA,IAAA;AAAA,EAEtD;AACJ;AAEA,SAAS,qBAA2B;AAChC,MAAI,gBAAgB;AAChB,QAAI,gBAAgB,cAAc;AAClC,qBAAiB;AAAA,EACrB;AACJ;AAgBA,SAAS,iBAAqC;AAC1C,QAAM,UAAW,YAAuE;AACxF,MAAI,OAAO,YAAY,YAAY;AAC/B,QAAI;AACA,YAAM,OAAO,QAAQ,2BAA2B;AAChD,UAAI,QAAQ,YAAY,KAAK,IAAI,EAAG,QAAO;AAAA,IAC/C,QAAQ;AAAA,IAAsC;AAAA,EAClD;AAEA,MAAI;AACA,UAAM,KAAK,SAAS,cAAc,0BAA0B;AAC5D,UAAM,MAAM,IAAI,cAAc,KAAK,MAAM,GAAG,WAAW,IAA4C;AACnG,UAAM,OAAO,KAAK,UAAU,2BAA2B;AACvD,QAAI,KAAM,QAAO,IAAI,IAAI,MAAM,SAAS,IAAI,EAAE;AAAA,EAClD,QAAQ;AAAA,EAA+C;AACvD,SAAO;AACX;AAEA,SAAS,aAAqB;AAC1B,MAAI,CAAC,SAAS;AACV,cAAU,aAAA;AACV,YAAQ,iBAAiB,WAAW,oBAAoB;AACxD,YAAQ,iBAAiB,SAAS,kBAAkB;AACpD,YAAQ,iBAAiB,gBAAgB,kBAAkB;AAG3D,YAAQ,YAAY,EAAE,MAAM,QAAQ,iBAAiB,eAAA,GAAkB;AAAA,EAC3E;AACA,SAAO;AACX;AAOA,SAAS,mBAAmB,GAAgB;AACxC,QAAM,UAAW,GAAkB,WAAW;AAE9C,aAAW,KAAK,iBAAiB,UAAU;AACvC,QAAI;AAAE,QAAE,OAAO,IAAI,MAAM,OAAO,CAAC;AAAA,IAAG,QAAQ;AAAA,IAAe;AAAA,EAC/D;AACA,mBAAiB,MAAA;AAEjB,aAAW,QAAQ,kBAAkB,UAAU;AAC3C,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS;AAAG,WAAK,MAAA;AAAA,IAAS,QACjE;AAAA,IAAe;AAAA,EACzB;AACA,oBAAkB,MAAA;AAGlB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,mBAAiB;AACjB,sBAAoB;AACpB,MAAI;AAAE,aAAS,UAAA;AAAA,EAAa,QAAQ;AAAA,EAAe;AACnD,YAAU;AACV,qBAAA;AACJ;AAEA,SAAS,qBAAqB,OAA2B;AACrD,QAAM,MAAM,MAAM;AAElB,UAAQ,IAAI,MAAA;AAAA,IACR,KAAK,YAAY;AACb,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,UAAI,IAAI,WAAW,SAAS;AACxB,gBAAQ,WAAW,EAAE,QAAQ,QAAA,CAAS;AACtC,gBAAQ,QAAA;AACR,yBAAiB,OAAO,IAAI,EAAE;AAAA,MAClC,WAAW,IAAI,WAAW,WAAW;AACjC,gBAAQ,WAAW,EAAE,QAAQ,UAAA,CAAW;AAAA,MAC5C,WAAW,IAAI,WAAW,UAAU;AAChC,gBAAQ,WAAW,EAAE,QAAQ,UAAU,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACnF,OAAO;AACH,gBAAQ,WAAW,EAAE,QAAQ,eAAe,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACxF;AACA;AAAA,IACJ;AAAA,IACA,KAAK,iBAAiB;AAClB,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,cAAQ,OAAO,IAAI,MAAM,IAAI,OAAO,CAAC;AACrC,uBAAiB,OAAO,IAAI,EAAE;AAC9B,UAAI,sBAAsB,QAAQ,QAAS,qBAAoB;AAC/D;AAAA,IACJ;AAAA,IACA,KAAK,kBAAkB;AACnB,uBAAiB,IAAI;AACrB,0BAAoB;AAEpB,WAAK,wBAAwB,IAAI,OAAO,EAAE,KAAK,MAAM,qBAAqB;AAC1E;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,IAAI,WAAkC,OAAO,IAAI,OAAO;AAC7E;AAAA,IACJ;AAAA,IACA,KAAK,YAAY;AACb,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,QAAiB,GAAI,IAAI,QAAQ,EAAE,OAAO,IAAI,MAAA,IAAU,CAAA,GAAK;AAClF,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,IAAI,SAAS;AAC7D,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,EAAA;AAER;AAgBO,MAAM,uBACwE;AAAA,EACjF,IAAI;AAAA,EAEJ,cAAc;AACV,WAAO;AAAA,MACH,IAAI;AAAA,MACJ,MAAM;AAAA,MACN,MAAM;AAAA,MACN,OAAO;AAAA,MACP,aAAa;AAAA,MACb,eAAe;AAAA,MACf,SAAS;AAAA,MACT,SAAS;AAAA,IAAA;AAAA,EAEjB;AAAA,EAEA,YAA6B;AACzB,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,cAAwC;AAC1C,UAAM,oBAAA;AACN,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,KAAK,SAAyD;AAChE,UAAM,WAAW,WAAW,QAAQ,QAAQ;AAC5C,UAAM,YAAY,KAAA;AAClB,UAAM,UAAU;AAAA,MACZ,WAAW,QAAQ;AAAA,MACnB,aAAa,QAAQ;AAAA,MACrB,MAAM,QAAQ;AAAA,IAAA;AAElB,UAAM,OAAO,kBAAkB,IAAI,QAAQ,OAAO,GAAG,QAAQ;AAK7D,qBAAiB,QAAQ,OAAO;AAChC,oBAAgB,IAAI,WAAW,QAAQ,OAAO;AAC9C,UAAM,eAAe;AACrB,qBAAiB,IAAI,QAAc,CAAC,gBAAgB;AAChD,6BAAuB,IAAI,WAAW,WAAW;AAAA,IACrD,CAAC;AACD,UAAM,eAAe,MAAY;AAC7B,iBAAA,EAAa,YAAY;AAAA,QACrB,MAAM;AAAA,QACN,IAAI;AAAA,QACJ,SAAS,QAAQ;AAAA,QACjB;AAAA,QACA;AAAA,QACA;AAAA,QACA,OAAO,kBAAkB,IAAI,QAAQ,OAAO,GAAG;AAAA,QAC/C,QAAQ;AAAA,MAAA,CACX;AAAA,IACL;AAEA,QAAI,QAAQ,WAAW,OAAO;AAC1B,aAAO,IAAI,QAAgB,CAAC,SAAS,WAAW;AAC5C,YAAI,SAAS;AACb,cAAM,WAAW;AAAA,UACb,SAAS,CAAC,UAA8D;AACpE,gBAAI,MAAM,SAAS,OAAQ,WAAU,MAAM,SAAS;AAAA,qBAC3C,MAAM,SAAS,OAAQ,SAAQ,MAAM;AAAA,qBACrC,MAAM,SAAS,QAAS,QAAO,IAAI,MAAM,MAAM,OAAO,CAAC;AAAA,UACpE;AAAA,UACA,OAAO,MAAM;AAAA,UAAc;AAAA,QAAA;AAE/B,0BAAkB,IAAI,WAAW,QAAQ;AACzC,aAAK,aAAa,KAAK,YAAY;AAAA,MACvC,CAAC;AAAA,IACL;AAEA,WAAO,IAAI,eAAe;AAAA,MACtB,MAAM,MAAM,YAAY;AACpB,0BAAkB,IAAI,WAAW,UAAU;AAC3C,cAAM;AACN,qBAAA;AAAA,MACJ;AAAA,MACA,SAAS;AACL,0BAAkB,OAAO,SAAS;AAKlC,mBAAA,EAAa,YAAY,EAAE,MAAM,UAAU,IAAI,WAAW;AAAA,MAC9D;AAAA,IAAA,CACH;AAAA,EACL;AAAA,EAEA,MAAM,eAAe,SAAuC;AACxD,QAAI,mBAAmB,QAAS,QAAO;AACvC,QAAI,sBAAsB,QAAS,QAAO;AAC1C,QAAI,YAAY,YAAY;AACxB,UAAI;AACA,cAAM,YAAY,mBAAmB,OAAO;AAC5C,cAAM,QAAQ,MAAM,OAAO,KAAA;AAC3B,mBAAW,QAAQ,OAAO;AACtB,gBAAM,QAAQ,MAAM,OAAO,KAAK,IAAI;AACpC,gBAAM,OAAO,MAAM,MAAM,KAAA;AACzB,cAAI,KAAK,KAAK,CAAA,MAAK,EAAE,IAAI,SAAS,SAAS,KAAK,EAAE,IAAI,SAAS,UAAU,GAAG,CAAC,GAAG;AAC5E,mBAAO;AAAA,UACX;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAER;AAAA,IACJ;AACA,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,aAAa,SAAiB,YAA2D;AAC3F,QAAI,mBAAmB,SAAS;AAC5B,iBAAW,EAAE,QAAQ,SAAS;AAC9B;AAAA,IACJ;AAEA,UAAM,YAAY,KAAA;AAClB,wBAAoB;AAEpB,UAAM,OAAO,kBAAkB,IAAI,OAAO,GAAG,QAAQ;AACrD,UAAM,QAAQ,kBAAkB,IAAI,OAAO,GAAG;AAC9C,WAAO,IAAI,QAAc,CAAC,SAAS,WAAW;AAC1C,uBAAiB,IAAI,WAAW,EAAE,SAAS,YAAY,SAAS,QAAQ;AACxE,iBAAA,EAAa,YAAY,EAAE,MAAM,WAAW,IAAI,WAAW,SAAS,MAAM,OAAO,QAAQ,eAAA,CAAgB;AAAA,IAC7G,CAAC;AAAA,EACL;AAAA,EAEA,MAAM,YAAY,SAAgC;AAC9C,UAAM,kBAAkB,OAAO;AAAA,EACnC;AACJ;AAUO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGO,SAAS,kBAAwB;AACpC,WAAS,UAAA;AACT,YAAU;AACV,qBAAA;AACA,mBAAiB;AACjB,sBAAoB;AACpB,aAAW,CAAA,EAAG,CAAC,KAAK,kBAAkB;AAClC,MAAE,OAAO,IAAI,MAAM,mBAAmB,CAAC;AAAA,EAC3C;AACA,mBAAiB,MAAA;AACjB,aAAW,CAAA,EAAG,IAAI,KAAK,mBAAmB;AACtC,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,qBAAqB;AAAG,WAAK,MAAA;AAAA,IAAS,QAAQ;AAAA,IAAuB;AAAA,EAC/H;AACA,oBAAkB,MAAA;AAQlB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,2BAAyB;AAC7B;AAeA,eAAsB,mBAAgD;AAClE,MAAI,EAAE,YAAY,YAAa,QAAO,CAAA;AAEtC,QAAM,4BAAY,IAAA;AAIlB,WAAS,eAAe,KAA4B;AAChD,UAAM,IAAI,IAAI,MAAM,4CAA4C;AAChE,WAAO,IAAI,mBAAmB,EAAE,CAAC,CAAE,IAAI;AAAA,EAC3C;AAEA,WAAS,UAAU,SAAyB;AACxC,UAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,QAAI,eAAe,OAAO;AAC1B,YAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AAAA,EAClE;AAEA,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,mBAAW,OAAO,UAAU;AACxB,gBAAM,UAAU,eAAe,IAAI,GAAG;AACtC,cAAI,CAAC,QAAS;AACd,cAAI,CAAC,MAAM,IAAI,OAAO,GAAG;AACrB,kBAAM,IAAI,SAAS,EAAE,MAAM,UAAU,OAAO,GAAG,WAAW,GAAG;AAAA,UACjE;AACA,gBAAM,WAAW,MAAM,MAAM,MAAM,GAAG;AACtC,cAAI,CAAC,SAAU;AACf,gBAAM,gBAAgB,SAAS,QAAQ,IAAI,gBAAgB;AAC3D,cAAI,eAAe;AACf,kBAAM,IAAI,OAAO,EAAG,aAAa,SAAS,eAAe,EAAE;AAAA,UAC/D,OAAO;AACH,gBAAI;AACA,oBAAM,OAAO,MAAM,SAAS,MAAA,EAAQ,KAAA;AACpC,oBAAM,IAAI,OAAO,EAAG,aAAa,KAAK;AAAA,YAC1C,QAAQ;AAAA,YAAa;AAAA,UACzB;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAAgC;AAAA,IAC5C,CAAC,CAAC;AAAA,EACN,QAAQ;AACJ,WAAO,CAAA;AAAA,EACX;AAEA,SAAO,MAAM,KAAK,MAAM,QAAA,CAAS,EAAE,IAAI,CAAC,CAAC,SAAS,EAAE,MAAM,UAAA,CAAW,OAAO;AAAA,IACxE;AAAA,IACA;AAAA,IACA;AAAA,IACA,QAAQ,mBAAmB;AAAA,EAAA,EAC7B;AACN;AAMA,eAAsB,kBAAkB,SAAgC;AACpE,MAAI,mBAAmB,WAAW,sBAAsB,SAAS;AAC7D,oBAAA;AAAA,EACJ;AACA,MAAI,EAAE,YAAY,YAAa;AAC/B,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,cAAM,UAAU,mBAAmB,OAAO;AAC1C,cAAM,QAAQ;AAAA,UACV,SACK,OAAO,CAAA,MAAK,EAAE,IAAI,SAAS,OAAO,KAAK,EAAE,IAAI,SAAS,OAAO,CAAC,EAC9D,IAAI,OAAK,MAAM,OAAO,CAAC,CAAC;AAAA,QAAA;AAAA,MAErC,QAAQ;AAAA,MAAa;AAAA,IACzB,CAAC,CAAC;AAAA,EACN,QAAQ;AAAA,EAA8B;AAC1C;"}
1
+ {"version":3,"file":"index.js","sources":["../src/index.ts"],"sourcesContent":["/**\n * @aparte/provider-transformers — run LLMs 100% in the browser via Transformers.js.\n *\n * A local, keyless `AparteAIProvider`: it owns its I/O (inference runs off the main\n * thread in a Web Worker) so `AparteDirectTransport` delegates to its `chat()`. Model\n * weights download once and persist in the Cache API.\n *\n * Scope: the worker runs a **runner** — the built-in `text-generation` (any chat model\n * behind Transformers.js' `pipeline()`), or a module of the app's own named by\n * `TransformersModelConfig.runner` (see `runners/types.ts` for the contract). Tool-calling\n * for local models is model-specific (every family has its own wire format), so the\n * built-in drops tool turns and says so; a custom runner may render them.\n *\n * ## This provider's state is TAB-scoped, on purpose\n *\n * Everything below the \"Worker bridge\" heading — the worker, the loaded model, the\n * generate chain — plus `setComputeDevice`, `setMaxCachedModels` and\n * `setHardwareTierModels`, is module-level and therefore shared by every chat on the\n * page. That is deliberate, and it is the opposite of what the rest of the suite does:\n * a plugin's providers scope to one chat, this one cannot.\n *\n * The reason is the resource, not the design. A local model is 1–2 GB of weights and one\n * WebGPU pipeline. Handing each chat its own worker would mean N copies resident in one\n * tab — which is the failure this package exists to avoid, not a capability. The\n * settings above describe the *machine* (which backend, how many models to keep\n * cached), so per-chat values would not mean anything either.\n *\n * What the constraint costs: two chats on the page driving DIFFERENT local models take\n * turns on one pipeline, so each turn may evict and reload gigabytes. That used to\n * happen silently — a multi-second stall with nothing to read. It now warns once, from\n * `chat()`, when a generate is queued for a model other than the one already in flight.\n * Same model in both chats is free and correct: they share the load.\n */\n\nimport type {\n AparteAIProvider,\n AparteAIModel,\n AparteChatRequest,\n AparteChatResponse,\n ModelStatus,\n ModelLoadProgress,\n} from '@aparte/core';\nimport { uuid } from '@aparte/core';\nimport type { BuiltInRunner, Device, Dtype } from './runners/types.js';\n\n// The worker's URL, not the worker itself: this package constructs it by hand because a\n// cross-origin copy has to go through a blob (see `_spawnWorker`). `?worker&url` is what\n// keeps Vite emitting the worker as its own chunk — the `new Worker(new URL(...))` form\n// it detects by pattern was the only other way, and moving the URL out of that call made\n// the build inline the worker's raw TypeScript as a data: URL instead. Caught by a\n// two-origin browser probe, not by any test.\nimport workerUrl from './worker.ts?worker&url';\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Hardware detection\n// ─────────────────────────────────────────────────────────────────────────────\n\nexport interface HardwareProfile {\n hasGpu: boolean;\n ramGb: number;\n tier: 'low' | 'mid' | 'high';\n recommendedModelId: string;\n}\n\n/** Hardware-tier model overrides — set by the app via setHardwareTierModels(). */\nlet _hardwareTiers: { low: string; mid?: string; high: string } | null = null;\n\n/**\n * Set the model IDs to use per hardware tier. Call before detectHardware() is used\n * to pick a default model — the provider ships no model knowledge of its own.\n */\nexport function setHardwareTierModels(tiers: { low: string; mid?: string; high: string }): void {\n _hardwareTiers = tiers;\n}\n\nexport async function detectHardware(): Promise<HardwareProfile> {\n // navigator.deviceMemory: W3C API, Chromium only, capped at 8 GB for privacy\n // (1 | 2 | 4 | 8). Falls back to 4 on Firefox/Safari.\n const ramGb: number = (navigator as unknown as { deviceMemory?: number }).deviceMemory ?? 4;\n\n // Real WebGPU check: requestAdapter() returns null if no capable GPU is present.\n let hasGpu = false;\n if ('gpu' in navigator) {\n try {\n const adapter = await (navigator as unknown as { gpu: { requestAdapter(): Promise<unknown> } }).gpu.requestAdapter();\n hasGpu = adapter !== null;\n } catch {\n hasGpu = false;\n }\n }\n\n let tier: 'low' | 'mid' | 'high';\n if (!hasGpu || ramGb < 4) {\n tier = 'low';\n } else if (ramGb < 8) {\n tier = 'mid';\n } else {\n tier = 'high';\n }\n\n const recommendedModelId = _hardwareTiers\n ? (_hardwareTiers[tier] ?? _hardwareTiers.high ?? '')\n : '';\n\n return { hasGpu, ramGb, tier, recommendedModelId };\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Model catalog — all model knowledge lives in the app, not the provider.\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Configuration for a model registered with the provider. */\nexport interface TransformersModelConfig {\n id: string;\n name: string;\n description?: string;\n capabilities: AparteAIModel['capabilities'];\n /**\n * Which built-in runner loads and drives the model. `'text-generation'` (the default)\n * is any chat model behind Transformers.js' `pipeline()`. Ignored when `runner` is set.\n */\n task?: BuiltInRunner;\n /**\n * A runner of your own: the URL of an ES module exporting `createRunner` (see\n * `TransformersRunner`). Resolved against the page, imported by the worker, and handed\n * the same Transformers.js instance the built-ins use. Wins over `task`.\n */\n runner?: string;\n /** ONNX dtype or per-part dtype map (e.g. `'q4'` or `{ decoder_model_merged: 'q4' }`). */\n dtype?: Dtype;\n /** Preferred device. Defaults to WebGPU when available, else WASM. */\n device?: Device;\n metadata?: Record<string, unknown>;\n}\n\n/** Models registered by the app via registerModel(). */\nconst _registeredModels = new Map<string, TransformersModelConfig>();\n\n/** Mutable model list — populated by registerModel() and cache discovery. */\nlet _knownModels: AparteAIModel[] = [];\n\n/**\n * Register a model with the provider. Call before the model is used for inference.\n */\nexport function registerModel(config: TransformersModelConfig): void {\n _registeredModels.set(config.id, config);\n if (!_knownModels.find(m => m.id === config.id)) {\n _knownModels = [..._knownModels, {\n id: config.id,\n name: config.name,\n description: config.description,\n capabilities: config.capabilities,\n }];\n }\n}\n\n/** Build an AparteAIModel entry from a cache-discovered modelId not in the registry. */\nfunction _modelFromCacheEntry(modelId: string): AparteAIModel {\n const config = _registeredModels.get(modelId);\n if (config) return { id: config.id, name: config.name, description: config.description, capabilities: config.capabilities };\n const name = (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n return { id: modelId, name, capabilities: ['streaming'] };\n}\n\n/** Max number of models to keep in cache. 0 = unlimited. Default: 1. */\nlet _maxCachedModels = 1;\n\n/**\n * Set the maximum number of models to keep in cache. When exceeded after a new\n * model is ready, the oldest models are evicted. 0 = unlimited.\n */\nexport function setMaxCachedModels(max: number): void {\n _maxCachedModels = max;\n}\n\n/** Returns the current max-cached-models setting. */\nexport function getMaxCachedModels(): number {\n return _maxCachedModels;\n}\n\n/**\n * User's preferred compute backend for local inference.\n * 'auto' → WebGPU when available, else WASM (default)\n * 'webgpu' → force WebGPU\n * 'wasm' → force WASM CPU\n */\nexport type ComputeDevice = 'auto' | 'webgpu' | 'wasm';\nlet _computeDevice: ComputeDevice = 'auto';\n\nexport function setComputeDevice(d: ComputeDevice): void {\n _computeDevice = d;\n}\n\nexport function getComputeDevice(): ComputeDevice {\n return _computeDevice;\n}\n\n/** Evict models from cache until count <= _maxCachedModels; `keepModelId` is never evicted. */\nasync function _enforceMaxCachedModels(keepModelId: string): Promise<void> {\n if (_maxCachedModels === 0) return; // unlimited\n try {\n const cached = await listCachedModels();\n const others = cached.filter(e => e.modelId !== keepModelId);\n const excess = cached.length - _maxCachedModels;\n if (excess <= 0) return;\n // Delete the excess models (oldest first — they appear first in cache scan order).\n for (let i = 0; i < excess && i < others.length; i++) {\n await deleteCachedModel(others[i]!.modelId);\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Merge cached models into _knownModels (idempotent). Called by fetchModels(). */\nasync function _refreshKnownModels(): Promise<void> {\n try {\n const cached = await listCachedModels();\n for (const entry of cached) {\n if (!_knownModels.find(m => m.id === entry.modelId)) {\n _knownModels = [..._knownModels, _modelFromCacheEntry(entry.modelId)];\n }\n }\n } catch { /* cache unavailable */ }\n}\n\n/**\n * How the worker should load `modelId`: which runner, which weights, which device.\n *\n * A custom `runner` is made absolute HERE, not in the worker: a worker's base URL is its\n * own script's, not the page's, and the blob shim `_spawnWorker` may build has no\n * meaningful base at all — so a relative path would resolve against the wrong place or\n * fail outright. The page is the one place that knows what the app meant.\n */\nfunction _selection(modelId: string): { task: BuiltInRunner; runner?: string; dtype?: Dtype; device: ComputeDevice } {\n const config = _registeredModels.get(modelId);\n const runner = config?.runner;\n return {\n task: config?.task ?? 'text-generation',\n ...(runner ? { runner: typeof location === 'undefined' ? runner : new URL(runner, location.href).href } : {}),\n dtype: config?.dtype,\n device: _computeDevice,\n };\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Worker bridge\n// ─────────────────────────────────────────────────────────────────────────────\n\nlet _worker: Worker | null = null;\n\ninterface PendingPrepare {\n modelId: string;\n onProgress: (p: ModelLoadProgress) => void;\n resolve: () => void;\n reject: (err: Error) => void;\n}\nconst _pendingPrepares = new Map<string, PendingPrepare>();\nconst _pendingGenerates = new Map<string, ReadableStreamDefaultController>();\nconst _pendingCommands = new Map<string, { resolve: (value: unknown) => void; reject: (err: Error) => void }>();\n\n// ── Generate serialization ──────────────────────────────────────────────────\n// The worker holds ONE pipeline: two concurrent generates would corrupt each\n// other. Each chat() chains its `generate` behind the previous generate's\n// completion (gen-done / gen-error).\nlet _generateChain: Promise<void> = Promise.resolve();\nconst _generateDoneResolvers = new Map<string, () => void>();\n\n// ── Contention on the one pipeline ──────────────────────────────────────────\n// Serialization is correct but invisible: two chats driving DIFFERENT local\n// models take turns, and with `maxCachedModels` at its default of 1 each turn\n// can evict and reload gigabytes. The user sees a stall; the developer sees\n// nothing. These two track just enough to say so, once.\nconst _queuedModelIds = new Map<string, string>();\nlet _warnedModelContention = false;\n\n/** Model ids of generates currently queued or running on the single pipeline. */\nfunction _contendingModelId(requested: string): string | undefined {\n for (const id of _queuedModelIds.values()) if (id !== requested) return id;\n return undefined;\n}\n\n/**\n * Warn once when a generate has to queue behind another chat's DIFFERENT model.\n * Not a warning about switching models in one chat — that is a deliberate act\n * with visible feedback. This fires only when two are in flight at once.\n */\nfunction _warnIfContended(requested: string): void {\n if (_warnedModelContention) return;\n const other = _contendingModelId(requested);\n if (!other) return;\n _warnedModelContention = true;\n console.warn(\n `[Aparte] Two chats are driving different local models at once (\"${requested}\" behind `\n + `\"${other}\"). Transformers.js runs one pipeline per tab, so these generates are `\n + `serialized, and with a cache budget of ${_maxCachedModels} each switch can evict and `\n + `reload gigabytes of weights. Point both chats at one model, or raise the budget with `\n + `setMaxCachedModels(2) if the machine has the memory. This warns once.`,\n );\n}\n\n/** Settle the serialization slot for a finished generate. */\nfunction _releaseGenerateSlot(id: string): void {\n _queuedModelIds.delete(id);\n const resolve = _generateDoneResolvers.get(id);\n if (resolve) {\n _generateDoneResolvers.delete(id);\n resolve();\n }\n}\n\n/** Model known to be loaded (main-thread view). */\nlet _loadedModelId: string | null = null;\n/** Model currently being prepared (for the getModelStatus 'cached' path). */\nlet _preparingModelId: string | null = null;\n\n/** The blob URL the worker was built from, if it needed one. Revoked with the worker. */\nlet _workerBlobUrl: string | null = null;\n\n/**\n * Build the worker — including when this package is served from another origin.\n *\n * `new Worker()` refuses a cross-origin script outright, and that is not an exotic\n * case: it is every deploy whose JavaScript lives on a CDN or an asset host while the\n * page lives somewhere else, with or without a bundler. Reproduced with the package on\n * one port and the page on another: `SecurityError: Script at '…/assets/worker-*.js'\n * cannot be accessed from origin '…'`.\n *\n * A blob inherits the ORIGIN OF THE DOCUMENT THAT CREATES IT, so a one-line blob whose\n * body imports the real worker by absolute URL is same-origin by construction, and the\n * import inside it is a normal cross-origin module fetch, which is allowed. It is the\n * shim ffmpeg.wasm and tesseract.js use for the same reason.\n *\n * Same-origin keeps the direct path: no blob, nothing to revoke, and a stack trace that\n * names the real file.\n */\nfunction _spawnWorker(): Worker {\n const url = new URL(workerUrl, import.meta.url);\n const sameOrigin = typeof location === 'undefined' || url.origin === location.origin;\n // A blob is the only way across an origin, so an environment that cannot mint one has\n // nothing to gain from trying: construct directly and let the platform say what it\n // thinks. jsdom is that environment — it has `Blob` and no `URL.createObjectURL` — and\n // every test in this package went through the blob path and threw before this line\n // existed.\n const canMintBlob = typeof Blob === 'function' && typeof URL.createObjectURL === 'function';\n // The literal below is not style. `new Worker(new URL('./worker.ts', import.meta.url))`\n // is the exact shape Vite's worker detection and webpack's WorkerPlugin match on, and\n // matching it is what makes a CONSUMER's bundler process the worker as a module — which\n // is how `@huggingface/transformers` gets resolved inside it today. Behind a variable\n // the chunk is copied as an opaque asset and its imports are never touched, so hoisting\n // this line to reuse it for the blob would fix a CDN page by breaking every bundled app.\n if (sameOrigin || !canMintBlob) return new Worker(new URL('./worker.ts', import.meta.url), { type: 'module' });\n\n _workerBlobUrl = URL.createObjectURL(\n new Blob([`import ${JSON.stringify(url.href)};`], { type: 'text/javascript' }),\n );\n try {\n return new Worker(_workerBlobUrl, { type: 'module' });\n } catch (error) {\n // A page with `worker-src 'self'` (or `script-src` without `blob:`) blocks the\n // shim, and the direct URL was already refused for its origin — so there is\n // nothing left to try. Say which of the two walls was hit, because the browser's\n // own message does not distinguish them.\n URL.revokeObjectURL(_workerBlobUrl);\n _workerBlobUrl = null;\n throw new Error(\n `@aparte/provider-transformers is served from ${url.origin}, which is not this page's origin, `\n + 'so its worker has to be started through a blob: URL — and this page\\'s Content-Security-Policy '\n + 'refuses that. Allow `blob:` in `worker-src` (or `script-src`), or serve the package from your '\n + `own origin. Original error: ${String(error)}`,\n );\n }\n}\n\nfunction _releaseWorkerBlob(): void {\n if (_workerBlobUrl) {\n URL.revokeObjectURL(_workerBlobUrl);\n _workerBlobUrl = null;\n }\n}\n\n/**\n * Where the page says Transformers.js lives, if it says so at all.\n *\n * The worker cannot ask: an import map is the DOCUMENT's, and by spec it does not reach\n * a worker. The main thread can, and does it the platform's way — `import.meta.resolve`\n * consults that same map — so a page that already maps `@huggingface/transformers` (it\n * has to, to import this package by name at all) is telling us where its copy is. That\n * map is the CDN consumer's manifest: the version pin stays with the consumer, which is\n * the whole point of a peer dependency, and this package invents no second place to say\n * it.\n *\n * `undefined` under a bundler, where the specifier is resolved at build time and the\n * worker's own `import('@huggingface/transformers')` is the path that runs.\n */\nfunction _peerModuleUrl(): string | undefined {\n const resolve = (import.meta as unknown as { resolve?: (specifier: string) => string }).resolve;\n if (typeof resolve === 'function') {\n try {\n const href = resolve('@huggingface/transformers');\n if (href && /^https?:/i.test(href)) return href;\n } catch { /* not in the map — fall through */ }\n }\n // Older engines have no `import.meta.resolve`; read the map they do have.\n try {\n const el = document.querySelector('script[type=\"importmap\"]');\n const map = el?.textContent ? JSON.parse(el.textContent) as { imports?: Record<string, string> } : null;\n const href = map?.imports?.['@huggingface/transformers'];\n if (href) return new URL(href, location.href).href;\n } catch { /* no document, or a map that is not JSON */ }\n return undefined;\n}\n\nfunction _getWorker(): Worker {\n if (!_worker) {\n _worker = _spawnWorker();\n _worker.addEventListener('message', _handleWorkerMessage);\n _worker.addEventListener('error', _handleWorkerError);\n _worker.addEventListener('messageerror', _handleWorkerError);\n // First message, before any work: postMessage keeps order, so the worker has it\n // by the time a prepare or a generate needs the module.\n _worker.postMessage({ type: 'init', transformersUrl: _peerModuleUrl() });\n }\n return _worker;\n}\n\n/**\n * Worker crashed (uncaught error / WASM init failure / OOM). Reject every in-flight\n * prepare and close every open generate stream so the UI doesn't hang. Subsequent\n * calls rebuild the worker.\n */\nfunction _handleWorkerError(e: Event): void {\n const message = (e as ErrorEvent)?.message || 'Worker crashed unexpectedly';\n\n for (const p of _pendingPrepares.values()) {\n try { p.reject(new Error(message)); } catch { /* ignore */ }\n }\n _pendingPrepares.clear();\n\n for (const ctrl of _pendingGenerates.values()) {\n try { ctrl.enqueue({ type: 'error' as const, message }); ctrl.close(); }\n catch { /* ignore */ }\n }\n _pendingGenerates.clear();\n for (const c of _pendingCommands.values()) c.reject(new Error(message));\n _pendingCommands.clear();\n\n // Release every serialization slot so the generate chain doesn't deadlock.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n\n _loadedModelId = null;\n _preparingModelId = null;\n try { _worker?.terminate(); } catch { /* ignore */ }\n _worker = null;\n _releaseWorkerBlob();\n}\n\nfunction _handleWorkerMessage(event: MessageEvent): void {\n const msg = event.data;\n\n switch (msg.type) {\n case 'progress': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n if (msg.status === 'ready') {\n pending.onProgress({ status: 'ready' });\n pending.resolve();\n _pendingPrepares.delete(msg.id);\n } else if (msg.status === 'loading') {\n pending.onProgress({ status: 'loading' });\n } else if (msg.status === 'cached') {\n pending.onProgress({ status: 'cached', file: msg.file, progress: msg.progress });\n } else {\n pending.onProgress({ status: 'downloading', file: msg.file, progress: msg.progress });\n }\n break;\n }\n case 'prepare-error': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n pending.reject(new Error(msg.message));\n _pendingPrepares.delete(msg.id);\n if (_preparingModelId === pending.modelId) _preparingModelId = null;\n break;\n }\n case 'pipeline-ready': {\n _loadedModelId = msg.modelId;\n _preparingModelId = null;\n // Evict models over the cache limit, then refresh the known list.\n void _enforceMaxCachedModels(msg.modelId).then(() => _refreshKnownModels());\n break;\n }\n case 'gen-event': {\n // The runner speaks the stream vocabulary itself; nothing to translate.\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue(msg.event);\n break;\n }\n case 'warning': {\n // Already said once per text by the worker; the page just carries the voice.\n console.warn(`[transformers] ${msg.message}`);\n break;\n }\n case 'command-result': {\n _releaseGenerateSlot(msg.id);\n const pending = _pendingCommands.get(msg.id);\n if (!pending) break;\n _pendingCommands.delete(msg.id);\n if (msg.error !== undefined) pending.reject(new Error(msg.error));\n else pending.resolve(msg.result);\n break;\n }\n case 'gen-done': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'done' as const, ...(msg.usage ? { usage: msg.usage } : {}) });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n case 'gen-error': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'error' as const, message: msg.message });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n }\n}\n\n/**\n * Narrowed so the two members the docs tell you to CALL are not optional.\n *\n * `AparteAIProvider` declares `prepareModel` and `getModelStatus` optional (most\n * providers have nothing to download), and widening to it made both\n * possibly-undefined — so the documented `TransformersProvider.prepareModel(...)`\n * needed a `!` or a guard in every strict consumer. Same technique openai-compat\n * already used for its own always-present members.\n *\n * `chat` joined the list once `AparteAIProvider` became a union: it is optional on\n * the format-adapter arm, and this provider IS its `chat()` — running inference\n * locally is the whole package. Narrowing it here says so once, instead of every\n * caller writing `provider.chat!(...)`.\n */\nexport const TransformersProvider: AparteAIProvider\n & Required<Pick<AparteAIProvider, 'prepareModel' | 'getModelStatus' | 'chat'>> = {\n id: 'transformers',\n\n getMetadata() {\n return {\n id: 'transformers',\n name: 'Transformers.js',\n icon: `<svg viewBox=\"0 0 24 24\" fill=\"none\" xmlns=\"http://www.w3.org/2000/svg\"><path d=\"M12 2L2 7l10 5 10-5-10-5z\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 17l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 12l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/></svg>`,\n color: '#f59e0b',\n description: 'Run LLMs directly in your browser via WebGPU or WASM — no API, no key',\n hasFreeModels: true,\n isLocal: true,\n helpUrl: 'https://huggingface.co/docs/transformers.js',\n };\n },\n\n getModels(): AparteAIModel[] {\n return _knownModels;\n },\n\n async fetchModels(): Promise<AparteAIModel[]> {\n await _refreshKnownModels();\n return _knownModels;\n },\n\n async chat(\n request: AparteChatRequest,\n _config?: string | Record<string, string>,\n ctx?: { providerId: string; signal?: AbortSignal },\n ): Promise<AparteChatResponse> {\n const requestId = uuid();\n const options = {\n maxTokens: request.maxTokens,\n temperature: request.temperature,\n seed: request.seed,\n };\n const signal = ctx?.signal;\n\n // ── Reserve a serialization slot ─────────────────────────────────────\n // Chain this generate behind the previous one; the worker has a single\n // pipeline, so generates MUST NOT overlap.\n _warnIfContended(request.modelId);\n _queuedModelIds.set(requestId, request.modelId);\n const prevGenerate = _generateChain;\n _generateChain = new Promise<void>((resolveSlot) => {\n _generateDoneResolvers.set(requestId, resolveSlot);\n });\n // ── Stop, from either side ───────────────────────────────────────────\n // The transport's `ctx.signal` (the user's Stop, which the provider contract\n // says a bridge MUST honour — this one read it nowhere) and the stream's own\n // `cancel()` say the same thing, and the worker hears it once. Before the\n // generate has been posted there is nothing to interrupt: the stream is\n // settled here, and the slot is released when its turn in the chain comes —\n // not earlier, or the next generate would start over the one still running.\n let posted = false;\n let stopped = false;\n const stop = (): void => {\n if (stopped) return;\n stopped = true;\n signal?.removeEventListener('abort', stop);\n if (posted) {\n _getWorker().postMessage({ type: 'cancel', id: requestId });\n return;\n }\n const ctrl = _pendingGenerates.get(requestId);\n _pendingGenerates.delete(requestId);\n if (!ctrl) return;\n try { ctrl.enqueue({ type: 'error' as const, message: 'Generation cancelled before it started' }); ctrl.close(); }\n catch { /* already closed */ }\n };\n const postGenerate = (): void => {\n if (stopped) { _releaseGenerateSlot(requestId); return; }\n posted = true;\n _getWorker().postMessage({\n type: 'generate',\n id: requestId,\n modelId: request.modelId,\n // The conversation as it is, parts included: which parts a model can take\n // is the runner's knowledge, not this thread's.\n messages: request.messages,\n options,\n ..._selection(request.modelId),\n });\n };\n\n let response: AparteChatResponse | Promise<string>;\n if (request.stream === false) {\n response = new Promise<string>((resolve, reject) => {\n let result = '';\n const fakeCtrl = {\n enqueue: (chunk: { type: string; delta?: string; message?: string }) => {\n if (chunk.type === 'text') result += chunk.delta ?? '';\n else if (chunk.type === 'done') resolve(result);\n else if (chunk.type === 'error') reject(new Error(chunk.message));\n },\n close: () => { /* no-op */ },\n } as unknown as ReadableStreamDefaultController;\n _pendingGenerates.set(requestId, fakeCtrl);\n void prevGenerate.then(postGenerate);\n });\n } else {\n response = new ReadableStream({\n async start(controller) {\n _pendingGenerates.set(requestId, controller);\n await prevGenerate;\n postGenerate();\n },\n cancel() {\n // The reader is gone, so nothing may be enqueued for it again — and the\n // model actually STOPS (not just the read): the worker interrupts this\n // generate, and the slot is still released by the resulting\n // gen-done/gen-error, so a queued generate cannot start before that.\n _pendingGenerates.delete(requestId);\n stop();\n },\n });\n }\n\n // `start` has run by now, so the controller is registered and a stop settles it.\n if (signal?.aborted) stop();\n else signal?.addEventListener('abort', stop, { once: true });\n return response;\n },\n\n async getModelStatus(modelId: string): Promise<ModelStatus> {\n if (_loadedModelId === modelId) return 'ready';\n if (_preparingModelId === modelId) return 'cached';\n if ('caches' in globalThis) {\n try {\n const encodedId = encodeURIComponent(modelId);\n const names = await caches.keys();\n for (const name of names) {\n const cache = await caches.open(name);\n const keys = await cache.keys();\n if (keys.some(r => r.url.includes(encodedId) || r.url.includes(modelId + '/'))) {\n return 'cached';\n }\n }\n } catch {\n // Cache API unavailable\n }\n }\n return 'not-downloaded';\n },\n\n async prepareModel(modelId: string, onProgress: (p: ModelLoadProgress) => void): Promise<void> {\n if (_loadedModelId === modelId) {\n onProgress({ status: 'ready' });\n return;\n }\n\n const requestId = uuid();\n _preparingModelId = modelId;\n\n return new Promise<void>((resolve, reject) => {\n _pendingPrepares.set(requestId, { modelId, onProgress, resolve, reject });\n _getWorker().postMessage({ type: 'prepare', id: requestId, modelId, ..._selection(modelId) });\n });\n },\n\n async deleteModel(modelId: string): Promise<void> {\n await deleteCachedModel(modelId);\n },\n};\n\nexport default TransformersProvider;\nexport type { AparteAIProvider, AparteAIModel, ModelStatus, ModelLoadProgress } from '@aparte/core';\nexport type {\n TransformersRunner,\n RunnerContext,\n RunnerGenerateInput,\n RunnerProgress,\n RunnerModule,\n CreateRunner,\n BuiltInRunner,\n TransformersModule,\n} from './runners/types.js';\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Cache utilities (settings panels, etc.)\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Returns the modelId currently loaded in the worker's pipeline, or null. */\nexport function getLoadedModelId(): string | null {\n return _loadedModelId;\n}\n\n/**\n * Send a runner something that is not a generation — swap an adapter, warm a cache, ask\n * a capability — and get its answer. The name and payload are the runner's vocabulary\n * (the built-in runners answer none). Queued behind the generates in flight: the worker\n * holds one runner, and a command on it mid-stream would race the stream.\n */\nexport function runnerCommand(modelId: string, name: string, payload: unknown): Promise<unknown> {\n const requestId = uuid();\n _queuedModelIds.set(requestId, modelId);\n const previous = _generateChain;\n _generateChain = new Promise<void>((resolveSlot) => {\n _generateDoneResolvers.set(requestId, resolveSlot);\n });\n return new Promise<unknown>((resolve, reject) => {\n _pendingCommands.set(requestId, { resolve, reject });\n void previous.then(() => {\n _getWorker().postMessage({ type: 'command', id: requestId, modelId, name, payload, ..._selection(modelId) });\n });\n });\n}\n\n/** Terminate the shared worker and reset in-memory state. Safe to call any time. */\nexport function terminateWorker(): void {\n _worker?.terminate();\n _worker = null;\n _releaseWorkerBlob();\n _loadedModelId = null;\n _preparingModelId = null;\n for (const [, p] of _pendingPrepares) {\n p.reject(new Error('Worker terminated'));\n }\n _pendingPrepares.clear();\n for (const [, ctrl] of _pendingGenerates) {\n try { ctrl.enqueue({ type: 'error' as const, message: 'Worker terminated' }); ctrl.close(); } catch { /* already closed */ }\n }\n _pendingGenerates.clear();\n for (const c of _pendingCommands.values()) c.reject(new Error('Worker terminated'));\n _pendingCommands.clear();\n\n // Release every serialization slot and reset the chain — the same three lines\n // the worker-error handler above already carried, with the same reason. Without\n // them, terminating mid-generate left `_generateChain` pending on a resolver\n // that had just been dropped, so the NEXT chat() awaited a promise that could\n // never settle: no error, no rejection, the stream simply never started again\n // for the life of the page.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n // A terminated worker is a fresh situation; let the contention warning speak again.\n _warnedModelContention = false;\n}\n\nexport interface CachedModelEntry {\n modelId: string;\n name: string;\n /** Total size in bytes of all cached files for this model. -1 if unknown. */\n sizeBytes: number;\n /** True if the model is currently loaded in the worker. */\n loaded: boolean;\n}\n\n/**\n * Scan the Cache API to find which Transformers.js models have been downloaded,\n * by matching cache entry URLs against the Hugging Face resolve path.\n */\nexport async function listCachedModels(): Promise<CachedModelEntry[]> {\n if (!('caches' in globalThis)) return [];\n\n const found = new Map<string, { name: string; sizeBytes: number }>();\n\n // e.g. https://huggingface.co/onnx-community/Qwen2.5-0.5B/resolve/main/config.json\n // → onnx-community/Qwen2.5-0.5B\n function extractModelId(url: string): string | null {\n const m = url.match(/huggingface\\.co\\/([^/]+\\/[^/]+)\\/resolve\\//);\n return m ? decodeURIComponent(m[1]!) : null;\n }\n\n function modelName(modelId: string): string {\n const config = _registeredModels.get(modelId);\n if (config) return config.name;\n return (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n }\n\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n for (const req of requests) {\n const modelId = extractModelId(req.url);\n if (!modelId) continue;\n if (!found.has(modelId)) {\n found.set(modelId, { name: modelName(modelId), sizeBytes: 0 });\n }\n const response = await cache.match(req);\n if (!response) continue;\n const contentLength = response.headers.get('content-length');\n if (contentLength) {\n found.get(modelId)!.sizeBytes += parseInt(contentLength, 10);\n } else {\n try {\n const blob = await response.clone().blob();\n found.get(modelId)!.sizeBytes += blob.size;\n } catch { /* skip */ }\n }\n }\n } catch { /* skip inaccessible cache */ }\n }));\n } catch {\n return [];\n }\n\n return Array.from(found.entries()).map(([modelId, { name, sizeBytes }]) => ({\n modelId,\n name,\n sizeBytes,\n loaded: _loadedModelId === modelId,\n }));\n}\n\n/**\n * Delete all cached files for a modelId from the Cache API, terminating the worker\n * first if that model is currently loaded.\n */\nexport async function deleteCachedModel(modelId: string): Promise<void> {\n if (_loadedModelId === modelId || _preparingModelId === modelId) {\n terminateWorker();\n }\n if (!('caches' in globalThis)) return;\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n const encoded = encodeURIComponent(modelId);\n await Promise.all(\n requests\n .filter(r => r.url.includes(modelId) || r.url.includes(encoded))\n .map(r => cache.delete(r)),\n );\n } catch { /* skip */ }\n }));\n } catch { /* Cache API unavailable */ }\n}\n"],"names":[],"mappings":";;AAiEA,IAAI,iBAAqE;AAMlE,SAAS,sBAAsB,OAA0D;AAC5F,mBAAiB;AACrB;AAEA,eAAsB,iBAA2C;AAG7D,QAAM,QAAiB,UAAmD,gBAAgB;AAG1F,MAAI,SAAS;AACb,MAAI,SAAS,WAAW;AACpB,QAAI;AACA,YAAM,UAAU,MAAO,UAAyE,IAAI,eAAA;AACpG,eAAS,YAAY;AAAA,IACzB,QAAQ;AACJ,eAAS;AAAA,IACb;AAAA,EACJ;AAEA,MAAI;AACJ,MAAI,CAAC,UAAU,QAAQ,GAAG;AACtB,WAAO;AAAA,EACX,WAAW,QAAQ,GAAG;AAClB,WAAO;AAAA,EACX,OAAO;AACH,WAAO;AAAA,EACX;AAEA,QAAM,qBAAqB,iBACpB,eAAe,IAAI,KAAK,eAAe,QAAQ,KAChD;AAEN,SAAO,EAAE,QAAQ,OAAO,MAAM,mBAAA;AAClC;AA+BA,MAAM,wCAAwB,IAAA;AAG9B,IAAI,eAAgC,CAAA;AAK7B,SAAS,cAAc,QAAuC;AACjE,oBAAkB,IAAI,OAAO,IAAI,MAAM;AACvC,MAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,OAAO,EAAE,GAAG;AAC7C,mBAAe,CAAC,GAAG,cAAc;AAAA,MAC7B,IAAI,OAAO;AAAA,MACX,MAAM,OAAO;AAAA,MACb,aAAa,OAAO;AAAA,MACpB,cAAc,OAAO;AAAA,IAAA,CACxB;AAAA,EACL;AACJ;AAGA,SAAS,qBAAqB,SAAgC;AAC1D,QAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,MAAI,OAAQ,QAAO,EAAE,IAAI,OAAO,IAAI,MAAM,OAAO,MAAM,aAAa,OAAO,aAAa,cAAc,OAAO,aAAA;AAC7G,QAAM,QAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AACpE,SAAO,EAAE,IAAI,SAAS,MAAM,cAAc,CAAC,WAAW,EAAA;AAC1D;AAGA,IAAI,mBAAmB;AAMhB,SAAS,mBAAmB,KAAmB;AAClD,qBAAmB;AACvB;AAGO,SAAS,qBAA6B;AACzC,SAAO;AACX;AASA,IAAI,iBAAgC;AAE7B,SAAS,iBAAiB,GAAwB;AACrD,mBAAiB;AACrB;AAEO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGA,eAAe,wBAAwB,aAAoC;AACvE,MAAI,qBAAqB,EAAG;AAC5B,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,UAAM,SAAS,OAAO,OAAO,CAAA,MAAK,EAAE,YAAY,WAAW;AAC3D,UAAM,SAAS,OAAO,SAAS;AAC/B,QAAI,UAAU,EAAG;AAEjB,aAAS,IAAI,GAAG,IAAI,UAAU,IAAI,OAAO,QAAQ,KAAK;AAClD,YAAM,kBAAkB,OAAO,CAAC,EAAG,OAAO;AAAA,IAC9C;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,eAAe,sBAAqC;AAChD,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,eAAW,SAAS,QAAQ;AACxB,UAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,MAAM,OAAO,GAAG;AACjD,uBAAe,CAAC,GAAG,cAAc,qBAAqB,MAAM,OAAO,CAAC;AAAA,MACxE;AAAA,IACJ;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAUA,SAAS,WAAW,SAAiG;AACjH,QAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,QAAM,SAAS,QAAQ;AACvB,SAAO;AAAA,IACH,MAAM,QAAQ,QAAQ;AAAA,IACtB,GAAI,SAAS,EAAE,QAAQ,OAAO,aAAa,cAAc,SAAS,IAAI,IAAI,QAAQ,SAAS,IAAI,EAAE,KAAA,IAAS,CAAA;AAAA,IAC1G,OAAO,QAAQ;AAAA,IACf,QAAQ;AAAA,EAAA;AAEhB;AAMA,IAAI,UAAyB;AAQ7B,MAAM,uCAAuB,IAAA;AAC7B,MAAM,wCAAwB,IAAA;AAC9B,MAAM,uCAAuB,IAAA;AAM7B,IAAI,iBAAgC,QAAQ,QAAA;AAC5C,MAAM,6CAA6B,IAAA;AAOnC,MAAM,sCAAsB,IAAA;AAC5B,IAAI,yBAAyB;AAG7B,SAAS,mBAAmB,WAAuC;AAC/D,aAAW,MAAM,gBAAgB,OAAA,EAAU,KAAI,OAAO,UAAW,QAAO;AACxE,SAAO;AACX;AAOA,SAAS,iBAAiB,WAAyB;AAC/C,MAAI,uBAAwB;AAC5B,QAAM,QAAQ,mBAAmB,SAAS;AAC1C,MAAI,CAAC,MAAO;AACZ,2BAAyB;AACzB,UAAQ;AAAA,IACJ,mEAAmE,SAAS,aACtE,KAAK,gHACiC,gBAAgB;AAAA,EAAA;AAIpE;AAGA,SAAS,qBAAqB,IAAkB;AAC5C,kBAAgB,OAAO,EAAE;AACzB,QAAM,UAAU,uBAAuB,IAAI,EAAE;AAC7C,MAAI,SAAS;AACT,2BAAuB,OAAO,EAAE;AAChC,YAAA;AAAA,EACJ;AACJ;AAGA,IAAI,iBAAgC;AAEpC,IAAI,oBAAmC;AAGvC,IAAI,iBAAgC;AAmBpC,SAAS,eAAuB;AAC5B,QAAM,MAAM,IAAI,IAAI,WAAW,YAAY,GAAG;AAC9C,QAAM,aAAa,OAAO,aAAa,eAAe,IAAI,WAAW,SAAS;AAM9E,QAAM,cAAc,OAAO,SAAS,cAAc,OAAO,IAAI,oBAAoB;AAOjF,MAAI,cAAc,CAAC,YAAa,QAAO,IAAI,OAAO,IAAA;AAAA;AAAA,IAAA,KAAA,IAAA,IAAA,6BAAA,YAAA,GAAA,EAAA;AAAA,IAAA,YAAA;AAAA,EAAA,GAAyC,EAAE,MAAM,UAAU;AAE7G,mBAAiB,IAAI;AAAA,IACjB,IAAI,KAAK,CAAC,UAAU,KAAK,UAAU,IAAI,IAAI,CAAC,GAAG,GAAG,EAAE,MAAM,mBAAmB;AAAA,EAAA;AAEjF,MAAI;AACA,WAAO,IAAI,OAAO,gBAAgB,EAAE,MAAM,UAAU;AAAA,EACxD,SAAS,OAAO;AAKZ,QAAI,gBAAgB,cAAc;AAClC,qBAAiB;AACjB,UAAM,IAAI;AAAA,MACN,gDAAgD,IAAI,MAAM,oQAGzB,OAAO,KAAK,CAAC;AAAA,IAAA;AAAA,EAEtD;AACJ;AAEA,SAAS,qBAA2B;AAChC,MAAI,gBAAgB;AAChB,QAAI,gBAAgB,cAAc;AAClC,qBAAiB;AAAA,EACrB;AACJ;AAgBA,SAAS,iBAAqC;AAC1C,QAAM,UAAW,YAAuE;AACxF,MAAI,OAAO,YAAY,YAAY;AAC/B,QAAI;AACA,YAAM,OAAO,QAAQ,2BAA2B;AAChD,UAAI,QAAQ,YAAY,KAAK,IAAI,EAAG,QAAO;AAAA,IAC/C,QAAQ;AAAA,IAAsC;AAAA,EAClD;AAEA,MAAI;AACA,UAAM,KAAK,SAAS,cAAc,0BAA0B;AAC5D,UAAM,MAAM,IAAI,cAAc,KAAK,MAAM,GAAG,WAAW,IAA4C;AACnG,UAAM,OAAO,KAAK,UAAU,2BAA2B;AACvD,QAAI,KAAM,QAAO,IAAI,IAAI,MAAM,SAAS,IAAI,EAAE;AAAA,EAClD,QAAQ;AAAA,EAA+C;AACvD,SAAO;AACX;AAEA,SAAS,aAAqB;AAC1B,MAAI,CAAC,SAAS;AACV,cAAU,aAAA;AACV,YAAQ,iBAAiB,WAAW,oBAAoB;AACxD,YAAQ,iBAAiB,SAAS,kBAAkB;AACpD,YAAQ,iBAAiB,gBAAgB,kBAAkB;AAG3D,YAAQ,YAAY,EAAE,MAAM,QAAQ,iBAAiB,eAAA,GAAkB;AAAA,EAC3E;AACA,SAAO;AACX;AAOA,SAAS,mBAAmB,GAAgB;AACxC,QAAM,UAAW,GAAkB,WAAW;AAE9C,aAAW,KAAK,iBAAiB,UAAU;AACvC,QAAI;AAAE,QAAE,OAAO,IAAI,MAAM,OAAO,CAAC;AAAA,IAAG,QAAQ;AAAA,IAAe;AAAA,EAC/D;AACA,mBAAiB,MAAA;AAEjB,aAAW,QAAQ,kBAAkB,UAAU;AAC3C,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS;AAAG,WAAK,MAAA;AAAA,IAAS,QACjE;AAAA,IAAe;AAAA,EACzB;AACA,oBAAkB,MAAA;AAClB,aAAW,KAAK,iBAAiB,OAAA,KAAY,OAAO,IAAI,MAAM,OAAO,CAAC;AACtE,mBAAiB,MAAA;AAGjB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,mBAAiB;AACjB,sBAAoB;AACpB,MAAI;AAAE,aAAS,UAAA;AAAA,EAAa,QAAQ;AAAA,EAAe;AACnD,YAAU;AACV,qBAAA;AACJ;AAEA,SAAS,qBAAqB,OAA2B;AACrD,QAAM,MAAM,MAAM;AAElB,UAAQ,IAAI,MAAA;AAAA,IACR,KAAK,YAAY;AACb,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,UAAI,IAAI,WAAW,SAAS;AACxB,gBAAQ,WAAW,EAAE,QAAQ,QAAA,CAAS;AACtC,gBAAQ,QAAA;AACR,yBAAiB,OAAO,IAAI,EAAE;AAAA,MAClC,WAAW,IAAI,WAAW,WAAW;AACjC,gBAAQ,WAAW,EAAE,QAAQ,UAAA,CAAW;AAAA,MAC5C,WAAW,IAAI,WAAW,UAAU;AAChC,gBAAQ,WAAW,EAAE,QAAQ,UAAU,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACnF,OAAO;AACH,gBAAQ,WAAW,EAAE,QAAQ,eAAe,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACxF;AACA;AAAA,IACJ;AAAA,IACA,KAAK,iBAAiB;AAClB,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,cAAQ,OAAO,IAAI,MAAM,IAAI,OAAO,CAAC;AACrC,uBAAiB,OAAO,IAAI,EAAE;AAC9B,UAAI,sBAAsB,QAAQ,QAAS,qBAAoB;AAC/D;AAAA,IACJ;AAAA,IACA,KAAK,kBAAkB;AACnB,uBAAiB,IAAI;AACrB,0BAAoB;AAEpB,WAAK,wBAAwB,IAAI,OAAO,EAAE,KAAK,MAAM,qBAAqB;AAC1E;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AAEd,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,IAAI,KAAK;AACtB;AAAA,IACJ;AAAA,IACA,KAAK,WAAW;AAEZ,cAAQ,KAAK,kBAAkB,IAAI,OAAO,EAAE;AAC5C;AAAA,IACJ;AAAA,IACA,KAAK,kBAAkB;AACnB,2BAAqB,IAAI,EAAE;AAC3B,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,uBAAiB,OAAO,IAAI,EAAE;AAC9B,UAAI,IAAI,UAAU,OAAW,SAAQ,OAAO,IAAI,MAAM,IAAI,KAAK,CAAC;AAAA,UAC3D,SAAQ,QAAQ,IAAI,MAAM;AAC/B;AAAA,IACJ;AAAA,IACA,KAAK,YAAY;AACb,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,QAAiB,GAAI,IAAI,QAAQ,EAAE,OAAO,IAAI,MAAA,IAAU,CAAA,GAAK;AAClF,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,IAAI,SAAS;AAC7D,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,EAAA;AAER;AAgBO,MAAM,uBACwE;AAAA,EACjF,IAAI;AAAA,EAEJ,cAAc;AACV,WAAO;AAAA,MACH,IAAI;AAAA,MACJ,MAAM;AAAA,MACN,MAAM;AAAA,MACN,OAAO;AAAA,MACP,aAAa;AAAA,MACb,eAAe;AAAA,MACf,SAAS;AAAA,MACT,SAAS;AAAA,IAAA;AAAA,EAEjB;AAAA,EAEA,YAA6B;AACzB,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,cAAwC;AAC1C,UAAM,oBAAA;AACN,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,KACF,SACA,SACA,KAC2B;AAC3B,UAAM,YAAY,KAAA;AAClB,UAAM,UAAU;AAAA,MACZ,WAAW,QAAQ;AAAA,MACnB,aAAa,QAAQ;AAAA,MACrB,MAAM,QAAQ;AAAA,IAAA;AAElB,UAAM,SAAS,KAAK;AAKpB,qBAAiB,QAAQ,OAAO;AAChC,oBAAgB,IAAI,WAAW,QAAQ,OAAO;AAC9C,UAAM,eAAe;AACrB,qBAAiB,IAAI,QAAc,CAAC,gBAAgB;AAChD,6BAAuB,IAAI,WAAW,WAAW;AAAA,IACrD,CAAC;AAQD,QAAI,SAAS;AACb,QAAI,UAAU;AACd,UAAM,OAAO,MAAY;AACrB,UAAI,QAAS;AACb,gBAAU;AACV,cAAQ,oBAAoB,SAAS,IAAI;AACzC,UAAI,QAAQ;AACR,mBAAA,EAAa,YAAY,EAAE,MAAM,UAAU,IAAI,WAAW;AAC1D;AAAA,MACJ;AACA,YAAM,OAAO,kBAAkB,IAAI,SAAS;AAC5C,wBAAkB,OAAO,SAAS;AAClC,UAAI,CAAC,KAAM;AACX,UAAI;AAAE,aAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,0CAA0C;AAAG,aAAK,MAAA;AAAA,MAAS,QAC3G;AAAA,MAAuB;AAAA,IACjC;AACA,UAAM,eAAe,MAAY;AAC7B,UAAI,SAAS;AAAE,6BAAqB,SAAS;AAAG;AAAA,MAAQ;AACxD,eAAS;AACT,iBAAA,EAAa,YAAY;AAAA,QACrB,MAAM;AAAA,QACN,IAAI;AAAA,QACJ,SAAS,QAAQ;AAAA;AAAA;AAAA,QAGjB,UAAU,QAAQ;AAAA,QAClB;AAAA,QACA,GAAG,WAAW,QAAQ,OAAO;AAAA,MAAA,CAChC;AAAA,IACL;AAEA,QAAI;AACJ,QAAI,QAAQ,WAAW,OAAO;AAC1B,iBAAW,IAAI,QAAgB,CAAC,SAAS,WAAW;AAChD,YAAI,SAAS;AACb,cAAM,WAAW;AAAA,UACb,SAAS,CAAC,UAA8D;AACpE,gBAAI,MAAM,SAAS,OAAQ,WAAU,MAAM,SAAS;AAAA,qBAC3C,MAAM,SAAS,OAAQ,SAAQ,MAAM;AAAA,qBACrC,MAAM,SAAS,QAAS,QAAO,IAAI,MAAM,MAAM,OAAO,CAAC;AAAA,UACpE;AAAA,UACA,OAAO,MAAM;AAAA,UAAc;AAAA,QAAA;AAE/B,0BAAkB,IAAI,WAAW,QAAQ;AACzC,aAAK,aAAa,KAAK,YAAY;AAAA,MACvC,CAAC;AAAA,IACL,OAAO;AACH,iBAAW,IAAI,eAAe;AAAA,QAC1B,MAAM,MAAM,YAAY;AACpB,4BAAkB,IAAI,WAAW,UAAU;AAC3C,gBAAM;AACN,uBAAA;AAAA,QACJ;AAAA,QACA,SAAS;AAKL,4BAAkB,OAAO,SAAS;AAClC,eAAA;AAAA,QACJ;AAAA,MAAA,CACH;AAAA,IACL;AAGA,QAAI,QAAQ,QAAS,MAAA;AAAA,iBACR,iBAAiB,SAAS,MAAM,EAAE,MAAM,MAAM;AAC3D,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,eAAe,SAAuC;AACxD,QAAI,mBAAmB,QAAS,QAAO;AACvC,QAAI,sBAAsB,QAAS,QAAO;AAC1C,QAAI,YAAY,YAAY;AACxB,UAAI;AACA,cAAM,YAAY,mBAAmB,OAAO;AAC5C,cAAM,QAAQ,MAAM,OAAO,KAAA;AAC3B,mBAAW,QAAQ,OAAO;AACtB,gBAAM,QAAQ,MAAM,OAAO,KAAK,IAAI;AACpC,gBAAM,OAAO,MAAM,MAAM,KAAA;AACzB,cAAI,KAAK,KAAK,CAAA,MAAK,EAAE,IAAI,SAAS,SAAS,KAAK,EAAE,IAAI,SAAS,UAAU,GAAG,CAAC,GAAG;AAC5E,mBAAO;AAAA,UACX;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAER;AAAA,IACJ;AACA,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,aAAa,SAAiB,YAA2D;AAC3F,QAAI,mBAAmB,SAAS;AAC5B,iBAAW,EAAE,QAAQ,SAAS;AAC9B;AAAA,IACJ;AAEA,UAAM,YAAY,KAAA;AAClB,wBAAoB;AAEpB,WAAO,IAAI,QAAc,CAAC,SAAS,WAAW;AAC1C,uBAAiB,IAAI,WAAW,EAAE,SAAS,YAAY,SAAS,QAAQ;AACxE,mBAAa,YAAY,EAAE,MAAM,WAAW,IAAI,WAAW,SAAS,GAAG,WAAW,OAAO,EAAA,CAAG;AAAA,IAChG,CAAC;AAAA,EACL;AAAA,EAEA,MAAM,YAAY,SAAgC;AAC9C,UAAM,kBAAkB,OAAO;AAAA,EACnC;AACJ;AAoBO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAQO,SAAS,cAAc,SAAiB,MAAc,SAAoC;AAC7F,QAAM,YAAY,KAAA;AAClB,kBAAgB,IAAI,WAAW,OAAO;AACtC,QAAM,WAAW;AACjB,mBAAiB,IAAI,QAAc,CAAC,gBAAgB;AAChD,2BAAuB,IAAI,WAAW,WAAW;AAAA,EACrD,CAAC;AACD,SAAO,IAAI,QAAiB,CAAC,SAAS,WAAW;AAC7C,qBAAiB,IAAI,WAAW,EAAE,SAAS,QAAQ;AACnD,SAAK,SAAS,KAAK,MAAM;AACrB,iBAAA,EAAa,YAAY,EAAE,MAAM,WAAW,IAAI,WAAW,SAAS,MAAM,SAAS,GAAG,WAAW,OAAO,GAAG;AAAA,IAC/G,CAAC;AAAA,EACL,CAAC;AACL;AAGO,SAAS,kBAAwB;AACpC,WAAS,UAAA;AACT,YAAU;AACV,qBAAA;AACA,mBAAiB;AACjB,sBAAoB;AACpB,aAAW,CAAA,EAAG,CAAC,KAAK,kBAAkB;AAClC,MAAE,OAAO,IAAI,MAAM,mBAAmB,CAAC;AAAA,EAC3C;AACA,mBAAiB,MAAA;AACjB,aAAW,CAAA,EAAG,IAAI,KAAK,mBAAmB;AACtC,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,qBAAqB;AAAG,WAAK,MAAA;AAAA,IAAS,QAAQ;AAAA,IAAuB;AAAA,EAC/H;AACA,oBAAkB,MAAA;AAClB,aAAW,KAAK,iBAAiB,OAAA,KAAY,OAAO,IAAI,MAAM,mBAAmB,CAAC;AAClF,mBAAiB,MAAA;AAQjB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,2BAAyB;AAC7B;AAeA,eAAsB,mBAAgD;AAClE,MAAI,EAAE,YAAY,YAAa,QAAO,CAAA;AAEtC,QAAM,4BAAY,IAAA;AAIlB,WAAS,eAAe,KAA4B;AAChD,UAAM,IAAI,IAAI,MAAM,4CAA4C;AAChE,WAAO,IAAI,mBAAmB,EAAE,CAAC,CAAE,IAAI;AAAA,EAC3C;AAEA,WAAS,UAAU,SAAyB;AACxC,UAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,QAAI,eAAe,OAAO;AAC1B,YAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AAAA,EAClE;AAEA,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,mBAAW,OAAO,UAAU;AACxB,gBAAM,UAAU,eAAe,IAAI,GAAG;AACtC,cAAI,CAAC,QAAS;AACd,cAAI,CAAC,MAAM,IAAI,OAAO,GAAG;AACrB,kBAAM,IAAI,SAAS,EAAE,MAAM,UAAU,OAAO,GAAG,WAAW,GAAG;AAAA,UACjE;AACA,gBAAM,WAAW,MAAM,MAAM,MAAM,GAAG;AACtC,cAAI,CAAC,SAAU;AACf,gBAAM,gBAAgB,SAAS,QAAQ,IAAI,gBAAgB;AAC3D,cAAI,eAAe;AACf,kBAAM,IAAI,OAAO,EAAG,aAAa,SAAS,eAAe,EAAE;AAAA,UAC/D,OAAO;AACH,gBAAI;AACA,oBAAM,OAAO,MAAM,SAAS,MAAA,EAAQ,KAAA;AACpC,oBAAM,IAAI,OAAO,EAAG,aAAa,KAAK;AAAA,YAC1C,QAAQ;AAAA,YAAa;AAAA,UACzB;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAAgC;AAAA,IAC5C,CAAC,CAAC;AAAA,EACN,QAAQ;AACJ,WAAO,CAAA;AAAA,EACX;AAEA,SAAO,MAAM,KAAK,MAAM,QAAA,CAAS,EAAE,IAAI,CAAC,CAAC,SAAS,EAAE,MAAM,UAAA,CAAW,OAAO;AAAA,IACxE;AAAA,IACA;AAAA,IACA;AAAA,IACA,QAAQ,mBAAmB;AAAA,EAAA,EAC7B;AACN;AAMA,eAAsB,kBAAkB,SAAgC;AACpE,MAAI,mBAAmB,WAAW,sBAAsB,SAAS;AAC7D,oBAAA;AAAA,EACJ;AACA,MAAI,EAAE,YAAY,YAAa;AAC/B,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,cAAM,UAAU,mBAAmB,OAAO;AAC1C,cAAM,QAAQ;AAAA,UACV,SACK,OAAO,CAAA,MAAK,EAAE,IAAI,SAAS,OAAO,KAAK,EAAE,IAAI,SAAS,OAAO,CAAC,EAC9D,IAAI,OAAK,MAAM,OAAO,CAAC,CAAC;AAAA,QAAA;AAAA,MAErC,QAAQ;AAAA,MAAa;AAAA,IACzB,CAAC,CAAC;AAAA,EACN,QAAQ;AAAA,EAA8B;AAC1C;"}
@@ -0,0 +1,36 @@
1
+ /**
2
+ * The built-in vision runner — a model that reads images and text and writes text
3
+ * (SmolVLM, Qwen2-VL, LFM2-VL, Gemma 3, LLaVA…: everything `AutoModelForImageTextToText`
4
+ * resolves).
5
+ *
6
+ * Transformers.js 4.x has no `image-text-to-text` PIPELINE, so this runner goes through
7
+ * the model classes themselves, the way the SmolVLM examples do: `AutoProcessor` renders
8
+ * the chat template with `{ type: 'image' }` placeholders, the images are decoded beside
9
+ * the prompt in the same order, the processor turns both into tensors, and `generate()`
10
+ * streams through a `TextStreamer`. Tool turns are dropped with the shared warning.
11
+ */
12
+ import type { AparteChatMessage } from '@aparte/core';
13
+ import type { CreateRunner } from './types.js';
14
+ type HFPart = {
15
+ type: 'image';
16
+ } | {
17
+ type: 'text';
18
+ text: string;
19
+ };
20
+ type HFMessage = {
21
+ role: 'user' | 'assistant' | 'system';
22
+ content: HFPart[];
23
+ };
24
+ export declare const UNSUPPORTED_PARTS_DROPPED = "Dropped content part(s) this vision runner cannot carry (only text and image parts reach the model).";
25
+ /**
26
+ * The conversation in the HF chat shape the processor's template expects — every turn's
27
+ * content as parts, an `{ type: 'image' }` placeholder where a picture goes — plus the
28
+ * pictures themselves, in order of appearance, for the processor to pair with them.
29
+ */
30
+ export declare function toChatTemplate(messages: AparteChatMessage[], warn: (message: string) => void): {
31
+ chat: HFMessage[];
32
+ images: string[];
33
+ };
34
+ export declare const createRunner: CreateRunner;
35
+ export {};
36
+ //# sourceMappingURL=image-text-to-text.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"image-text-to-text.d.ts","sourceRoot":"","sources":["../../src/runners/image-text-to-text.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,cAAc,CAAC;AACtD,OAAO,KAAK,EAAE,YAAY,EAAuB,MAAM,YAAY,CAAC;AAGpE,KAAK,MAAM,GAAG;IAAE,IAAI,EAAE,OAAO,CAAA;CAAE,GAAG;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AACjE,KAAK,SAAS,GAAG;IAAE,IAAI,EAAE,MAAM,GAAG,WAAW,GAAG,QAAQ,CAAC;IAAC,OAAO,EAAE,MAAM,EAAE,CAAA;CAAE,CAAC;AAE9E,eAAO,MAAM,yBAAyB,yGACoE,CAAC;AAE3G;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,QAAQ,EAAE,iBAAiB,EAAE,EAAE,IAAI,EAAE,CAAC,OAAO,EAAE,MAAM,KAAK,IAAI,GAAG;IAAE,IAAI,EAAE,SAAS,EAAE,CAAC;IAAC,MAAM,EAAE,MAAM,EAAE,CAAA;CAAE,CAsBtI;AAcD,eAAO,MAAM,YAAY,EAAE,YAsC1B,CAAC"}
@@ -0,0 +1,30 @@
1
+ /**
2
+ * What the two built-in runners share — kept in one place so the two cannot drift on
3
+ * how progress is reported, how a stop reaches the model, or what a dropped tool turn
4
+ * says. Types only from core (see `text-generation.ts` for why).
5
+ */
6
+ import type { AparteStreamEvent } from '@aparte/core';
7
+ import type { RunnerContext, TransformersModule } from './types.js';
8
+ export declare const TOOL_TURNS_DROPPED: string;
9
+ /**
10
+ * The options a `from_pretrained` / `pipeline()` call takes from the context: download
11
+ * progress forwarded to the page (percentages, rounded), dtype and device when set.
12
+ */
13
+ export declare function loadOptions(ctx: RunnerContext): Record<string, unknown>;
14
+ /**
15
+ * A stopping criteria the signal interrupts — so a Stop actually STOPS the model, not just
16
+ * the read; otherwise generation runs to `max_new_tokens` off-thread, spending exactly the
17
+ * CPU/GPU/battery this provider exists to save. Call `release()` in a `finally`.
18
+ */
19
+ export declare function interruptOn(signal: AbortSignal, transformers: TransformersModule): {
20
+ stopping: unknown;
21
+ release(): void;
22
+ };
23
+ /** A `TextStreamer` that emits each decoded token as a `text` event, prompt skipped. */
24
+ export declare function textStreamer(transformers: TransformersModule, tokenizer: unknown, emit: (event: AparteStreamEvent) => void): unknown;
25
+ /** Sampling options in Transformers.js' vocabulary, from the request's. */
26
+ export declare function generationOptions(options: {
27
+ maxTokens?: number;
28
+ temperature?: number;
29
+ }): Record<string, unknown>;
30
+ //# sourceMappingURL=shared.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"shared.d.ts","sourceRoot":"","sources":["../../src/runners/shared.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAEH,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,cAAc,CAAC;AACtD,OAAO,KAAK,EAAE,aAAa,EAAE,kBAAkB,EAAE,MAAM,YAAY,CAAC;AAEpE,eAAO,MAAM,kBAAkB,QAGO,CAAC;AAEvC;;;GAGG;AACH,wBAAgB,WAAW,CAAC,GAAG,EAAE,aAAa,GAAG,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAUvE;AAED;;;;GAIG;AACH,wBAAgB,WAAW,CAAC,MAAM,EAAE,WAAW,EAAE,YAAY,EAAE,kBAAkB,GAAG;IAAE,QAAQ,EAAE,OAAO,CAAC;IAAC,OAAO,IAAI,IAAI,CAAA;CAAE,CAMzH;AAED,wFAAwF;AACxF,wBAAgB,YAAY,CAAC,YAAY,EAAE,kBAAkB,EAAE,SAAS,EAAE,OAAO,EAAE,IAAI,EAAE,CAAC,KAAK,EAAE,iBAAiB,KAAK,IAAI,GAAG,OAAO,CAOpI;AAED,2EAA2E;AAC3E,wBAAgB,iBAAiB,CAAC,OAAO,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAC;IAAC,WAAW,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAOhH"}
@@ -0,0 +1,22 @@
1
+ /**
2
+ * The built-in text runner — the generic `pipeline('text-generation')` path this
3
+ * provider has always run, extracted from the worker so it is one runner among others.
4
+ *
5
+ * It flattens the conversation to `{ role, content: string }` turns (the tokenizer applies
6
+ * the chat template). Two things it cannot carry, it SAYS: tool turns (their wire syntax is
7
+ * model-specific) and image parts (a text model has no eyes). The second used to vanish
8
+ * silently — a photo attached to a text model produced an answer that pretended — and
9
+ * that silence, not the limitation, was the defect.
10
+ */
11
+ import type { AparteChatMessage } from '@aparte/core';
12
+ import type { CreateRunner, RunnerContext } from './types.js';
13
+ type SimpleMessage = {
14
+ role: 'user' | 'assistant' | 'system';
15
+ content: string;
16
+ };
17
+ export declare const IMAGES_DROPPED: string;
18
+ /** Flatten to what the chat template takes; say what was left out. */
19
+ export declare function flattenForChatTemplate(messages: AparteChatMessage[], warn: RunnerContext['warn']): SimpleMessage[];
20
+ export declare const createRunner: CreateRunner;
21
+ export {};
22
+ //# sourceMappingURL=text-generation.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"text-generation.d.ts","sourceRoot":"","sources":["../../src/runners/text-generation.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAMH,OAAO,KAAK,EAAE,iBAAiB,EAAqB,MAAM,cAAc,CAAC;AACzE,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAuB,MAAM,YAAY,CAAC;AAGnF,KAAK,aAAa,GAAG;IAAE,IAAI,EAAE,MAAM,GAAG,WAAW,GAAG,QAAQ,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC;AAEhF,eAAO,MAAM,cAAc,QAGoB,CAAC;AAWhD,sEAAsE;AACtE,wBAAgB,sBAAsB,CAAC,QAAQ,EAAE,iBAAiB,EAAE,EAAE,IAAI,EAAE,aAAa,CAAC,MAAM,CAAC,GAAG,aAAa,EAAE,CAgBlH;AASD,eAAO,MAAM,YAAY,EAAE,YAqB1B,CAAC"}