@aparte/provider-transformers 0.16.0 → 0.16.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,23 @@
1
- import { env, InterruptableStoppingCriteria, TextStreamer, pipeline } from "@huggingface/transformers";
2
- env.allowLocalModels = false;
3
- env.useBrowserCache = true;
1
+ let _moduleUrl;
2
+ let _tf = null;
3
+ function transformers() {
4
+ _tf ??= (async () => {
5
+ let mod;
6
+ try {
7
+ mod = await import("@huggingface/transformers");
8
+ } catch (bundlerPathFailed) {
9
+ if (!_moduleUrl) throw bundlerPathFailed;
10
+ mod = await import(
11
+ /* @vite-ignore */
12
+ _moduleUrl
13
+ );
14
+ }
15
+ mod.env.allowLocalModels = false;
16
+ mod.env.useBrowserCache = true;
17
+ return mod;
18
+ })();
19
+ return _tf;
20
+ }
4
21
  const ctx = self;
5
22
  function post(message) {
6
23
  ctx.postMessage(message);
@@ -21,6 +38,7 @@ async function ensurePipeline(modelId, dtype, device, id) {
21
38
  };
22
39
  if (dtype) opts["dtype"] = dtype;
23
40
  if (device && device !== "auto") opts["device"] = device;
41
+ const { pipeline } = await transformers();
24
42
  const pipe = await pipeline("text-generation", modelId, opts);
25
43
  _current = { modelId, pipe };
26
44
  post({ type: "pipeline-ready", modelId });
@@ -35,6 +53,7 @@ async function handlePrepare(msg) {
35
53
  }
36
54
  }
37
55
  async function handleGenerate(msg) {
56
+ const { TextStreamer, InterruptableStoppingCriteria } = await transformers();
38
57
  const stoppingCriteria = new InterruptableStoppingCriteria();
39
58
  _activeStops.set(msg.id, stoppingCriteria);
40
59
  try {
@@ -63,8 +82,12 @@ async function handleGenerate(msg) {
63
82
  }
64
83
  ctx.addEventListener("message", (event) => {
65
84
  const msg = event.data;
85
+ if (msg.type === "init") {
86
+ _moduleUrl = msg.transformersUrl;
87
+ return;
88
+ }
66
89
  if (msg.type === "prepare") void handlePrepare(msg);
67
90
  else if (msg.type === "generate") void handleGenerate(msg);
68
91
  else if (msg.type === "cancel") _activeStops.get(msg.id)?.interrupt();
69
92
  });
70
- //# sourceMappingURL=worker-Bk-8pt3W.js.map
93
+ //# sourceMappingURL=worker-Dz5bU2S7.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"worker-Dz5bU2S7.js","sources":["../src/worker.ts"],"sourcesContent":["/**\n * Generic Transformers.js inference worker.\n *\n * Runs entirely off the main thread. It holds ONE text-generation pipeline at a\n * time and speaks a tiny postMessage protocol with the provider on the main\n * thread (see `index.ts`):\n *\n * main → worker : { type: 'prepare', id, modelId, dtype?, device? }\n * { type: 'generate', id, modelId, messages, options, dtype?, device? }\n * worker → main : { type: 'progress', id, status, file?, progress? }\n * { type: 'prepare-error', id, message }\n * { type: 'pipeline-ready', modelId }\n * { type: 'gen-chunk', id, chunkType: 'text', delta }\n * { type: 'gen-done', id }\n * { type: 'gen-error', id, message }\n *\n * Deliberately generic: no vision, no low-level ORT session management, no\n * model-family specifics — just the high-level `pipeline()` + `TextStreamer`.\n */\n\nimport type { pipeline as Pipeline, TextStreamer as TextStreamerClass, InterruptableStoppingCriteria as StoppingCriteriaClass, TextGenerationPipeline } from '@huggingface/transformers';\n\n/**\n * Where Transformers.js comes from, resolved once, from whichever path has it.\n *\n * A static `import … from '@huggingface/transformers'` is unresolvable in a worker\n * served without a bundler: an import map lives on the DOCUMENT and, by spec, does not\n * reach a worker — so the page can map the specifier for itself and the worker still\n * cannot. The two paths, in order:\n *\n * 1. `import('@huggingface/transformers')` — a bare specifier, statically visible, so a\n * consumer's bundler resolves and bundles the peer exactly as it did before.\n * 2. the absolute URL the main thread read from the page's own import map and sent in\n * the first message — the CDN path, where that map is the consumer's manifest.\n *\n * The order matters: a bundled app must never reach for the network copy.\n */\ntype TransformersModule = {\n pipeline: typeof Pipeline;\n TextStreamer: typeof TextStreamerClass;\n InterruptableStoppingCriteria: typeof StoppingCriteriaClass;\n env: { allowLocalModels: boolean; useBrowserCache: boolean };\n};\n\nlet _moduleUrl: string | undefined;\nlet _tf: Promise<TransformersModule> | null = null;\n\nfunction transformers(): Promise<TransformersModule> {\n _tf ??= (async () => {\n let mod: TransformersModule;\n try {\n mod = await import('@huggingface/transformers') as unknown as TransformersModule;\n } catch (bundlerPathFailed) {\n if (!_moduleUrl) throw bundlerPathFailed;\n mod = await import(/* @vite-ignore */ _moduleUrl) as unknown as TransformersModule;\n }\n // Fetch weights from the Hugging Face hub (not local paths) and cache them in the\n // browser Cache API — this is what `listCachedModels()` scans on the main thread.\n mod.env.allowLocalModels = false;\n mod.env.useBrowserCache = true;\n return mod;\n })();\n return _tf;\n}\n\n// DOM's `Worker` interface types `postMessage` + typed `addEventListener('message')`,\n// which is enough for the worker scope — avoids pulling the WebWorker lib (it clashes\n// with DOM's global `postMessage`).\nconst ctx = self as unknown as Worker;\n\ntype Dtype = string | Record<string, string>;\ntype Device = 'webgpu' | 'wasm' | 'auto';\ninterface GenOptions { maxTokens?: number; temperature?: number; seed?: number }\ntype SimpleMessage = { role: 'user' | 'assistant' | 'system'; content: string };\n\ntype InMessage =\n | { type: 'prepare'; id: string; modelId: string; dtype?: Dtype; device?: Device }\n | { type: 'generate'; id: string; modelId: string; messages: SimpleMessage[]; options: GenOptions; dtype?: Dtype; device?: Device }\n | { type: 'cancel'; id: string }\n // Sent once, before anything else, when the main thread could read the peer's URL out\n // of the page's import map. Absent under a bundler, which has already resolved it.\n | { type: 'init'; transformersUrl?: string };\n\nfunction post(message: unknown): void {\n ctx.postMessage(message);\n}\n\nlet _current: { modelId: string; pipe: TextGenerationPipeline } | null = null;\n// Per-generate interrupts, so a consumer's stream-cancel actually STOPS the model\n// (not just detaches the reader) — otherwise generation runs to max_new_tokens\n// off-thread, wasting exactly the CPU/GPU/battery this provider exists to save.\nconst _activeStops = new Map<string, InstanceType<typeof StoppingCriteriaClass>>();\n\n/**\n * Ensure the pipeline for `modelId` is loaded, reusing the current one when it\n * matches. On a fresh load it forwards download progress (when `id` is given) and\n * announces `pipeline-ready`.\n */\nasync function ensurePipeline(modelId: string, dtype: Dtype | undefined, device: Device | undefined, id?: string): Promise<TextGenerationPipeline> {\n if (_current?.modelId === modelId) return _current.pipe;\n\n // eslint-disable-next-line @typescript-eslint/no-explicit-any\n const opts: Record<string, any> = {\n progress_callback: (p: { status?: string; file?: string; progress?: number }) => {\n if (!id) return;\n if (p.status === 'progress') {\n post({ type: 'progress', id, status: 'downloading', file: p.file, progress: Math.round(p.progress ?? 0) });\n } else if (p.status === 'done') {\n post({ type: 'progress', id, status: 'loading', file: p.file });\n }\n },\n };\n if (dtype) opts['dtype'] = dtype;\n if (device && device !== 'auto') opts['device'] = device;\n\n const { pipeline } = await transformers();\n const pipe = await pipeline('text-generation', modelId, opts) as TextGenerationPipeline;\n _current = { modelId, pipe };\n post({ type: 'pipeline-ready', modelId });\n return pipe;\n}\n\nasync function handlePrepare(msg: Extract<InMessage, { type: 'prepare' }>): Promise<void> {\n try {\n await ensurePipeline(msg.modelId, msg.dtype, msg.device, msg.id);\n post({ type: 'progress', id: msg.id, status: 'ready' });\n } catch (err) {\n post({ type: 'prepare-error', id: msg.id, message: (err as Error)?.message ?? 'Failed to load model' });\n }\n}\n\nasync function handleGenerate(msg: Extract<InMessage, { type: 'generate' }>): Promise<void> {\n const { TextStreamer, InterruptableStoppingCriteria } = await transformers();\n const stoppingCriteria = new InterruptableStoppingCriteria();\n _activeStops.set(msg.id, stoppingCriteria);\n try {\n const pipe = await ensurePipeline(msg.modelId, msg.dtype, msg.device, msg.id);\n\n const streamer = new TextStreamer(pipe.tokenizer, {\n skip_prompt: true,\n skip_special_tokens: true,\n callback_function: (text: string) => {\n if (text) post({ type: 'gen-chunk', id: msg.id, chunkType: 'text', delta: text });\n },\n });\n\n const temperature = msg.options.temperature ?? 0;\n await pipe(msg.messages, {\n max_new_tokens: msg.options.maxTokens ?? 512,\n do_sample: temperature > 0,\n temperature: temperature > 0 ? temperature : undefined,\n streamer,\n stopping_criteria: stoppingCriteria,\n });\n\n post({ type: 'gen-done', id: msg.id });\n } catch (err) {\n post({ type: 'gen-error', id: msg.id, message: (err as Error)?.message ?? 'Generation failed' });\n } finally {\n _activeStops.delete(msg.id);\n }\n}\n\nctx.addEventListener('message', (event: MessageEvent<InMessage>) => {\n const msg = event.data;\n if (msg.type === 'init') { _moduleUrl = msg.transformersUrl; return; }\n if (msg.type === 'prepare') void handlePrepare(msg);\n else if (msg.type === 'generate') void handleGenerate(msg);\n else if (msg.type === 'cancel') _activeStops.get(msg.id)?.interrupt();\n});\n"],"names":[],"mappings":"AA4CA,IAAI;AACJ,IAAI,MAA0C;AAE9C,SAAS,eAA4C;AACjD,WAAS,YAAY;AACjB,QAAI;AACJ,QAAI;AACA,YAAM,MAAM,OAAO,2BAA2B;AAAA,IAClD,SAAS,mBAAmB;AACxB,UAAI,CAAC,WAAY,OAAM;AACvB,YAAM,MAAM;AAAA;AAAA,QAA0B;AAAA;AAAA,IAC1C;AAGA,QAAI,IAAI,mBAAmB;AAC3B,QAAI,IAAI,kBAAkB;AAC1B,WAAO;AAAA,EACX,GAAA;AACA,SAAO;AACX;AAKA,MAAM,MAAM;AAeZ,SAAS,KAAK,SAAwB;AAClC,MAAI,YAAY,OAAO;AAC3B;AAEA,IAAI,WAAqE;AAIzE,MAAM,mCAAmB,IAAA;AAOzB,eAAe,eAAe,SAAiB,OAA0B,QAA4B,IAA8C;AAC/I,MAAI,UAAU,YAAY,QAAS,QAAO,SAAS;AAGnD,QAAM,OAA4B;AAAA,IAC9B,mBAAmB,CAAC,MAA6D;AAC7E,UAAI,CAAC,GAAI;AACT,UAAI,EAAE,WAAW,YAAY;AACzB,aAAK,EAAE,MAAM,YAAY,IAAI,QAAQ,eAAe,MAAM,EAAE,MAAM,UAAU,KAAK,MAAM,EAAE,YAAY,CAAC,GAAG;AAAA,MAC7G,WAAW,EAAE,WAAW,QAAQ;AAC5B,aAAK,EAAE,MAAM,YAAY,IAAI,QAAQ,WAAW,MAAM,EAAE,MAAM;AAAA,MAClE;AAAA,IACJ;AAAA,EAAA;AAEJ,MAAI,MAAO,MAAK,OAAO,IAAI;AAC3B,MAAI,UAAU,WAAW,OAAQ,MAAK,QAAQ,IAAI;AAElD,QAAM,EAAE,aAAa,MAAM,aAAA;AAC3B,QAAM,OAAO,MAAM,SAAS,mBAAmB,SAAS,IAAI;AAC5D,aAAW,EAAE,SAAS,KAAA;AACtB,OAAK,EAAE,MAAM,kBAAkB,QAAA,CAAS;AACxC,SAAO;AACX;AAEA,eAAe,cAAc,KAA6D;AACtF,MAAI;AACA,UAAM,eAAe,IAAI,SAAS,IAAI,OAAO,IAAI,QAAQ,IAAI,EAAE;AAC/D,SAAK,EAAE,MAAM,YAAY,IAAI,IAAI,IAAI,QAAQ,SAAS;AAAA,EAC1D,SAAS,KAAK;AACV,SAAK,EAAE,MAAM,iBAAiB,IAAI,IAAI,IAAI,SAAU,KAAe,WAAW,uBAAA,CAAwB;AAAA,EAC1G;AACJ;AAEA,eAAe,eAAe,KAA8D;AACxF,QAAM,EAAE,cAAc,8BAAA,IAAkC,MAAM,aAAA;AAC9D,QAAM,mBAAmB,IAAI,8BAAA;AAC7B,eAAa,IAAI,IAAI,IAAI,gBAAgB;AACzC,MAAI;AACA,UAAM,OAAO,MAAM,eAAe,IAAI,SAAS,IAAI,OAAO,IAAI,QAAQ,IAAI,EAAE;AAE5E,UAAM,WAAW,IAAI,aAAa,KAAK,WAAW;AAAA,MAC9C,aAAa;AAAA,MACb,qBAAqB;AAAA,MACrB,mBAAmB,CAAC,SAAiB;AACjC,YAAI,KAAM,MAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,WAAW,QAAQ,OAAO,KAAA,CAAM;AAAA,MACpF;AAAA,IAAA,CACH;AAED,UAAM,cAAc,IAAI,QAAQ,eAAe;AAC/C,UAAM,KAAK,IAAI,UAAU;AAAA,MACrB,gBAAgB,IAAI,QAAQ,aAAa;AAAA,MACzC,WAAW,cAAc;AAAA,MACzB,aAAa,cAAc,IAAI,cAAc;AAAA,MAC7C;AAAA,MACA,mBAAmB;AAAA,IAAA,CACtB;AAED,SAAK,EAAE,MAAM,YAAY,IAAI,IAAI,IAAI;AAAA,EACzC,SAAS,KAAK;AACV,SAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,SAAU,KAAe,WAAW,oBAAA,CAAqB;AAAA,EACnG,UAAA;AACI,iBAAa,OAAO,IAAI,EAAE;AAAA,EAC9B;AACJ;AAEA,IAAI,iBAAiB,WAAW,CAAC,UAAmC;AAChE,QAAM,MAAM,MAAM;AAClB,MAAI,IAAI,SAAS,QAAQ;AAAE,iBAAa,IAAI;AAAiB;AAAA,EAAQ;AACrE,MAAI,IAAI,SAAS,UAAW,MAAK,cAAc,GAAG;AAAA,WACzC,IAAI,SAAS,WAAY,MAAK,eAAe,GAAG;AAAA,WAChD,IAAI,SAAS,SAAU,cAAa,IAAI,IAAI,EAAE,GAAG,UAAA;AAC9D,CAAC;"}
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8BG;AAEH,OAAO,KAAK,EACR,gBAAgB,EAChB,aAAa,EAMhB,MAAM,cAAc,CAAC;AAUtB,MAAM,WAAW,eAAe;IAC5B,MAAM,EAAE,OAAO,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;IACd,IAAI,EAAE,KAAK,GAAG,KAAK,GAAG,MAAM,CAAC;IAC7B,kBAAkB,EAAE,MAAM,CAAC;CAC9B;AAKD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,KAAK,EAAE;IAAE,GAAG,EAAE,MAAM,CAAC;IAAC,GAAG,CAAC,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAAG,IAAI,CAE9F;AAED,wBAAsB,cAAc,IAAI,OAAO,CAAC,eAAe,CAAC,CA8B/D;AAMD,8DAA8D;AAC9D,MAAM,WAAW,uBAAuB;IACpC,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,aAAa,CAAC,cAAc,CAAC,CAAC;IAC5C,qFAAqF;IACrF,IAAI,EAAE,iBAAiB,CAAC;IACxB,0FAA0F;IAC1F,KAAK,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACxC,sEAAsE;IACtE,MAAM,CAAC,EAAE,QAAQ,GAAG,MAAM,GAAG,MAAM,CAAC;IACpC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACtC;AAQD;;GAEG;AACH,wBAAgB,aAAa,CAAC,MAAM,EAAE,uBAAuB,GAAG,IAAI,CAUnE;AAaD;;;GAGG;AACH,wBAAgB,kBAAkB,CAAC,GAAG,EAAE,MAAM,GAAG,IAAI,CAEpD;AAED,qDAAqD;AACrD,wBAAgB,kBAAkB,IAAI,MAAM,CAE3C;AAED;;;;;GAKG;AACH,MAAM,MAAM,aAAa,GAAG,MAAM,GAAG,QAAQ,GAAG,MAAM,CAAC;AAGvD,wBAAgB,gBAAgB,CAAC,CAAC,EAAE,aAAa,GAAG,IAAI,CAEvD;AAED,wBAAgB,gBAAgB,IAAI,aAAa,CAEhD;AA0OD;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,oBAAoB,EAAE,gBAAgB,GAC7C,QAAQ,CAAC,IAAI,CAAC,gBAAgB,EAAE,cAAc,GAAG,gBAAgB,GAAG,MAAM,CAAC,CAmIhF,CAAC;AAEF,eAAe,oBAAoB,CAAC;AACpC,YAAY,EAAE,gBAAgB,EAAE,aAAa,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,cAAc,CAAC;AAMpG,8EAA8E;AAC9E,wBAAgB,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAEhD;AAED,oFAAoF;AACpF,wBAAgB,eAAe,IAAI,IAAI,CA4BtC;AAED,MAAM,WAAW,gBAAgB;IAC7B,OAAO,EAAE,MAAM,CAAC;IAChB,IAAI,EAAE,MAAM,CAAC;IACb,6EAA6E;IAC7E,SAAS,EAAE,MAAM,CAAC;IAClB,2DAA2D;IAC3D,MAAM,EAAE,OAAO,CAAC;CACnB;AAED;;;GAGG;AACH,wBAAsB,gBAAgB,IAAI,OAAO,CAAC,gBAAgB,EAAE,CAAC,CAsDpE;AAED;;;GAGG;AACH,wBAAsB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAoBtE"}
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8BG;AAEH,OAAO,KAAK,EACR,gBAAgB,EAChB,aAAa,EAMhB,MAAM,cAAc,CAAC;AAkBtB,MAAM,WAAW,eAAe;IAC5B,MAAM,EAAE,OAAO,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;IACd,IAAI,EAAE,KAAK,GAAG,KAAK,GAAG,MAAM,CAAC;IAC7B,kBAAkB,EAAE,MAAM,CAAC;CAC9B;AAKD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,KAAK,EAAE;IAAE,GAAG,EAAE,MAAM,CAAC;IAAC,GAAG,CAAC,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAAG,IAAI,CAE9F;AAED,wBAAsB,cAAc,IAAI,OAAO,CAAC,eAAe,CAAC,CA8B/D;AAMD,8DAA8D;AAC9D,MAAM,WAAW,uBAAuB;IACpC,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,aAAa,CAAC,cAAc,CAAC,CAAC;IAC5C,qFAAqF;IACrF,IAAI,EAAE,iBAAiB,CAAC;IACxB,0FAA0F;IAC1F,KAAK,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACxC,sEAAsE;IACtE,MAAM,CAAC,EAAE,QAAQ,GAAG,MAAM,GAAG,MAAM,CAAC;IACpC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACtC;AAQD;;GAEG;AACH,wBAAgB,aAAa,CAAC,MAAM,EAAE,uBAAuB,GAAG,IAAI,CAUnE;AAaD;;;GAGG;AACH,wBAAgB,kBAAkB,CAAC,GAAG,EAAE,MAAM,GAAG,IAAI,CAEpD;AAED,qDAAqD;AACrD,wBAAgB,kBAAkB,IAAI,MAAM,CAE3C;AAED;;;;;GAKG;AACH,MAAM,MAAM,aAAa,GAAG,MAAM,GAAG,QAAQ,GAAG,MAAM,CAAC;AAGvD,wBAAgB,gBAAgB,CAAC,CAAC,EAAE,aAAa,GAAG,IAAI,CAEvD;AAED,wBAAgB,gBAAgB,IAAI,aAAa,CAEhD;AA+UD;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,oBAAoB,EAAE,gBAAgB,GAC7C,QAAQ,CAAC,IAAI,CAAC,gBAAgB,EAAE,cAAc,GAAG,gBAAgB,GAAG,MAAM,CAAC,CAmIhF,CAAC;AAEF,eAAe,oBAAoB,CAAC;AACpC,YAAY,EAAE,gBAAgB,EAAE,aAAa,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,cAAc,CAAC;AAMpG,8EAA8E;AAC9E,wBAAgB,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAEhD;AAED,oFAAoF;AACpF,wBAAgB,eAAe,IAAI,IAAI,CA6BtC;AAED,MAAM,WAAW,gBAAgB;IAC7B,OAAO,EAAE,MAAM,CAAC;IAChB,IAAI,EAAE,MAAM,CAAC;IACb,6EAA6E;IAC7E,SAAS,EAAE,MAAM,CAAC;IAClB,2DAA2D;IAC3D,MAAM,EAAE,OAAO,CAAC;CACnB;AAED;;;GAGG;AACH,wBAAsB,gBAAgB,IAAI,OAAO,CAAC,gBAAgB,EAAE,CAAC,CAsDpE;AAED;;;GAGG;AACH,wBAAsB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAoBtE"}
package/dist/index.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { uuid, contentToText } from "@aparte/core";
2
+ const workerUrl = "" + new URL("assets/worker-Dz5bU2S7.js", import.meta.url).href;
2
3
  let _hardwareTiers = null;
3
4
  function setHardwareTierModels(tiers) {
4
5
  _hardwareTiers = tiers;
@@ -132,16 +133,60 @@ function _releaseGenerateSlot(id) {
132
133
  }
133
134
  let _loadedModelId = null;
134
135
  let _preparingModelId = null;
136
+ let _workerBlobUrl = null;
137
+ function _spawnWorker() {
138
+ const url = new URL(workerUrl, import.meta.url);
139
+ const sameOrigin = typeof location === "undefined" || url.origin === location.origin;
140
+ const canMintBlob = typeof Blob === "function" && typeof URL.createObjectURL === "function";
141
+ if (sameOrigin || !canMintBlob) return new Worker(new URL(
142
+ /* @vite-ignore */
143
+ "" + new URL("assets/worker-Dz5bU2S7.js", import.meta.url).href,
144
+ import.meta.url
145
+ ), { type: "module" });
146
+ _workerBlobUrl = URL.createObjectURL(
147
+ new Blob([`import ${JSON.stringify(url.href)};`], { type: "text/javascript" })
148
+ );
149
+ try {
150
+ return new Worker(_workerBlobUrl, { type: "module" });
151
+ } catch (error) {
152
+ URL.revokeObjectURL(_workerBlobUrl);
153
+ _workerBlobUrl = null;
154
+ throw new Error(
155
+ `@aparte/provider-transformers is served from ${url.origin}, which is not this page's origin, so its worker has to be started through a blob: URL — and this page's Content-Security-Policy refuses that. Allow \`blob:\` in \`worker-src\` (or \`script-src\`), or serve the package from your own origin. Original error: ${String(error)}`
156
+ );
157
+ }
158
+ }
159
+ function _releaseWorkerBlob() {
160
+ if (_workerBlobUrl) {
161
+ URL.revokeObjectURL(_workerBlobUrl);
162
+ _workerBlobUrl = null;
163
+ }
164
+ }
165
+ function _peerModuleUrl() {
166
+ const resolve = import.meta.resolve;
167
+ if (typeof resolve === "function") {
168
+ try {
169
+ const href = resolve("@huggingface/transformers");
170
+ if (href && /^https?:/i.test(href)) return href;
171
+ } catch {
172
+ }
173
+ }
174
+ try {
175
+ const el = document.querySelector('script[type="importmap"]');
176
+ const map = el?.textContent ? JSON.parse(el.textContent) : null;
177
+ const href = map?.imports?.["@huggingface/transformers"];
178
+ if (href) return new URL(href, location.href).href;
179
+ } catch {
180
+ }
181
+ return void 0;
182
+ }
135
183
  function _getWorker() {
136
184
  if (!_worker) {
137
- _worker = new Worker(new URL(
138
- /* @vite-ignore */
139
- "" + new URL("assets/worker-Bk-8pt3W.js", import.meta.url).href,
140
- import.meta.url
141
- ), { type: "module" });
185
+ _worker = _spawnWorker();
142
186
  _worker.addEventListener("message", _handleWorkerMessage);
143
187
  _worker.addEventListener("error", _handleWorkerError);
144
188
  _worker.addEventListener("messageerror", _handleWorkerError);
189
+ _worker.postMessage({ type: "init", transformersUrl: _peerModuleUrl() });
145
190
  }
146
191
  return _worker;
147
192
  }
@@ -178,6 +223,7 @@ function _handleWorkerError(e) {
178
223
  } catch {
179
224
  }
180
225
  _worker = null;
226
+ _releaseWorkerBlob();
181
227
  }
182
228
  function _handleWorkerMessage(event) {
183
229
  const msg = event.data;
@@ -357,6 +403,7 @@ function getLoadedModelId() {
357
403
  function terminateWorker() {
358
404
  _worker?.terminate();
359
405
  _worker = null;
406
+ _releaseWorkerBlob();
360
407
  _loadedModelId = null;
361
408
  _preparingModelId = null;
362
409
  for (const [, p] of _pendingPrepares) {
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":["../src/index.ts"],"sourcesContent":["/**\n * @aparte/provider-transformers — run LLMs 100% in the browser via Transformers.js.\n *\n * A local, keyless `AparteAIProvider`: it owns its I/O (inference runs off the main\n * thread in a Web Worker) so `AparteDirectTransport` delegates to its `chat()`. Model\n * weights download once and persist in the Cache API.\n *\n * Scope (v1): generic **text-generation** streaming. Tool-calling for local models is\n * model-specific (every family has its own wire format) and is out of scope here — the\n * app registers models and streams plain replies. Vision / embeddings can follow on demand.\n *\n * ## This provider's state is TAB-scoped, on purpose\n *\n * Everything below the \"Worker bridge\" heading — the worker, the loaded model, the\n * generate chain — plus `setComputeDevice`, `setMaxCachedModels` and\n * `setHardwareTierModels`, is module-level and therefore shared by every chat on the\n * page. That is deliberate, and it is the opposite of what the rest of the suite does:\n * a plugin's providers scope to one chat, this one cannot.\n *\n * The reason is the resource, not the design. A local model is 1–2 GB of weights and one\n * WebGPU pipeline. Handing each chat its own worker would mean N copies resident in one\n * tab — which is the failure this package exists to avoid, not a capability. The\n * settings above describe the *machine* (which backend, how many models to keep\n * cached), so per-chat values would not mean anything either.\n *\n * What the constraint costs: two chats on the page driving DIFFERENT local models take\n * turns on one pipeline, so each turn may evict and reload gigabytes. That used to\n * happen silently — a multi-second stall with nothing to read. It now warns once, from\n * `chat()`, when a generate is queued for a model other than the one already in flight.\n * Same model in both chats is free and correct: they share the load.\n */\n\nimport type {\n AparteAIProvider,\n AparteAIModel,\n AparteChatRequest,\n AparteChatResponse,\n AparteChatMessage,\n ModelStatus,\n ModelLoadProgress,\n} from '@aparte/core';\nimport { contentToText, uuid } from '@aparte/core';\n\n/** The minimal chat shape passed to the worker (the tokenizer applies the chat template). */\ntype SimpleMessage = { role: 'user' | 'assistant' | 'system'; content: string };\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Hardware detection\n// ─────────────────────────────────────────────────────────────────────────────\n\nexport interface HardwareProfile {\n hasGpu: boolean;\n ramGb: number;\n tier: 'low' | 'mid' | 'high';\n recommendedModelId: string;\n}\n\n/** Hardware-tier model overrides — set by the app via setHardwareTierModels(). */\nlet _hardwareTiers: { low: string; mid?: string; high: string } | null = null;\n\n/**\n * Set the model IDs to use per hardware tier. Call before detectHardware() is used\n * to pick a default model — the provider ships no model knowledge of its own.\n */\nexport function setHardwareTierModels(tiers: { low: string; mid?: string; high: string }): void {\n _hardwareTiers = tiers;\n}\n\nexport async function detectHardware(): Promise<HardwareProfile> {\n // navigator.deviceMemory: W3C API, Chromium only, capped at 8 GB for privacy\n // (1 | 2 | 4 | 8). Falls back to 4 on Firefox/Safari.\n const ramGb: number = (navigator as unknown as { deviceMemory?: number }).deviceMemory ?? 4;\n\n // Real WebGPU check: requestAdapter() returns null if no capable GPU is present.\n let hasGpu = false;\n if ('gpu' in navigator) {\n try {\n const adapter = await (navigator as unknown as { gpu: { requestAdapter(): Promise<unknown> } }).gpu.requestAdapter();\n hasGpu = adapter !== null;\n } catch {\n hasGpu = false;\n }\n }\n\n let tier: 'low' | 'mid' | 'high';\n if (!hasGpu || ramGb < 4) {\n tier = 'low';\n } else if (ramGb < 8) {\n tier = 'mid';\n } else {\n tier = 'high';\n }\n\n const recommendedModelId = _hardwareTiers\n ? (_hardwareTiers[tier] ?? _hardwareTiers.high ?? '')\n : '';\n\n return { hasGpu, ramGb, tier, recommendedModelId };\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Model catalog — all model knowledge lives in the app, not the provider.\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Configuration for a model registered with the provider. */\nexport interface TransformersModelConfig {\n id: string;\n name: string;\n description?: string;\n capabilities: AparteAIModel['capabilities'];\n /** Transformers.js pipeline task — determines the model architecture / load path. */\n task: 'text-generation';\n /** ONNX dtype or per-part dtype map (e.g. `'q4'` or `{ decoder_model_merged: 'q4' }`). */\n dtype?: string | Record<string, string>;\n /** Preferred device. Defaults to WebGPU when available, else WASM. */\n device?: 'webgpu' | 'wasm' | 'auto';\n metadata?: Record<string, unknown>;\n}\n\n/** Models registered by the app via registerModel(). */\nconst _registeredModels = new Map<string, TransformersModelConfig>();\n\n/** Mutable model list — populated by registerModel() and cache discovery. */\nlet _knownModels: AparteAIModel[] = [];\n\n/**\n * Register a model with the provider. Call before the model is used for inference.\n */\nexport function registerModel(config: TransformersModelConfig): void {\n _registeredModels.set(config.id, config);\n if (!_knownModels.find(m => m.id === config.id)) {\n _knownModels = [..._knownModels, {\n id: config.id,\n name: config.name,\n description: config.description,\n capabilities: config.capabilities,\n }];\n }\n}\n\n/** Build an AparteAIModel entry from a cache-discovered modelId not in the registry. */\nfunction _modelFromCacheEntry(modelId: string): AparteAIModel {\n const config = _registeredModels.get(modelId);\n if (config) return { id: config.id, name: config.name, description: config.description, capabilities: config.capabilities };\n const name = (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n return { id: modelId, name, capabilities: ['streaming'] };\n}\n\n/** Max number of models to keep in cache. 0 = unlimited. Default: 1. */\nlet _maxCachedModels = 1;\n\n/**\n * Set the maximum number of models to keep in cache. When exceeded after a new\n * model is ready, the oldest models are evicted. 0 = unlimited.\n */\nexport function setMaxCachedModels(max: number): void {\n _maxCachedModels = max;\n}\n\n/** Returns the current max-cached-models setting. */\nexport function getMaxCachedModels(): number {\n return _maxCachedModels;\n}\n\n/**\n * User's preferred compute backend for local inference.\n * 'auto' → WebGPU when available, else WASM (default)\n * 'webgpu' → force WebGPU\n * 'wasm' → force WASM CPU\n */\nexport type ComputeDevice = 'auto' | 'webgpu' | 'wasm';\nlet _computeDevice: ComputeDevice = 'auto';\n\nexport function setComputeDevice(d: ComputeDevice): void {\n _computeDevice = d;\n}\n\nexport function getComputeDevice(): ComputeDevice {\n return _computeDevice;\n}\n\n/** Evict models from cache until count <= _maxCachedModels; `keepModelId` is never evicted. */\nasync function _enforceMaxCachedModels(keepModelId: string): Promise<void> {\n if (_maxCachedModels === 0) return; // unlimited\n try {\n const cached = await listCachedModels();\n const others = cached.filter(e => e.modelId !== keepModelId);\n const excess = cached.length - _maxCachedModels;\n if (excess <= 0) return;\n // Delete the excess models (oldest first — they appear first in cache scan order).\n for (let i = 0; i < excess && i < others.length; i++) {\n await deleteCachedModel(others[i]!.modelId);\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Merge cached models into _knownModels (idempotent). Called by fetchModels(). */\nasync function _refreshKnownModels(): Promise<void> {\n try {\n const cached = await listCachedModels();\n for (const entry of cached) {\n if (!_knownModels.find(m => m.id === entry.modelId)) {\n _knownModels = [..._knownModels, _modelFromCacheEntry(entry.modelId)];\n }\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Warn at most once per session that tool turns were left out of the prompt. */\nlet _warnedToolTurnsDropped = false;\n\n/** AparteChatMessage[] → plain chat turns (the tokenizer's chat template does the rest). */\nfunction toMessages(messages: AparteChatMessage[]): SimpleMessage[] {\n const result: SimpleMessage[] = [];\n let droppedToolTurns = 0;\n for (const m of messages) {\n if (m.role === 'user' || m.role === 'assistant' || m.role === 'system') {\n const text = contentToText(m.content);\n if (text) result.push({ role: m.role, content: text });\n } else {\n // tool_call / tool_result are not supported by this generic provider (v1):\n // rendering them needs a model-specific tool syntax. Dropping them\n // silently meant an app with registered tools got a model that never saw\n // the call or its result, with nothing to explain the behaviour.\n droppedToolTurns++;\n }\n }\n if (droppedToolTurns > 0 && !_warnedToolTurnsDropped) {\n _warnedToolTurnsDropped = true;\n console.warn(\n `[transformers] Dropped ${droppedToolTurns} tool turn(s) from the prompt: this provider ` +\n 'does not support tool calling (v1), so the model will not see the call or its result. ' +\n 'Use an OpenAI-compatible endpoint for tools, or render the turns yourself before sending.',\n );\n }\n return result;\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Worker bridge\n// ─────────────────────────────────────────────────────────────────────────────\n\nlet _worker: Worker | null = null;\n\ninterface PendingPrepare {\n modelId: string;\n onProgress: (p: ModelLoadProgress) => void;\n resolve: () => void;\n reject: (err: Error) => void;\n}\nconst _pendingPrepares = new Map<string, PendingPrepare>();\nconst _pendingGenerates = new Map<string, ReadableStreamDefaultController>();\n\n// ── Generate serialization ──────────────────────────────────────────────────\n// The worker holds ONE pipeline: two concurrent generates would corrupt each\n// other. Each chat() chains its `generate` behind the previous generate's\n// completion (gen-done / gen-error).\nlet _generateChain: Promise<void> = Promise.resolve();\nconst _generateDoneResolvers = new Map<string, () => void>();\n\n// ── Contention on the one pipeline ──────────────────────────────────────────\n// Serialization is correct but invisible: two chats driving DIFFERENT local\n// models take turns, and with `maxCachedModels` at its default of 1 each turn\n// can evict and reload gigabytes. The user sees a stall; the developer sees\n// nothing. These two track just enough to say so, once.\nconst _queuedModelIds = new Map<string, string>();\nlet _warnedModelContention = false;\n\n/** Model ids of generates currently queued or running on the single pipeline. */\nfunction _contendingModelId(requested: string): string | undefined {\n for (const id of _queuedModelIds.values()) if (id !== requested) return id;\n return undefined;\n}\n\n/**\n * Warn once when a generate has to queue behind another chat's DIFFERENT model.\n * Not a warning about switching models in one chat — that is a deliberate act\n * with visible feedback. This fires only when two are in flight at once.\n */\nfunction _warnIfContended(requested: string): void {\n if (_warnedModelContention) return;\n const other = _contendingModelId(requested);\n if (!other) return;\n _warnedModelContention = true;\n console.warn(\n `[Aparte] Two chats are driving different local models at once (\"${requested}\" behind `\n + `\"${other}\"). Transformers.js runs one pipeline per tab, so these generates are `\n + `serialized, and with a cache budget of ${_maxCachedModels} each switch can evict and `\n + `reload gigabytes of weights. Point both chats at one model, or raise the budget with `\n + `setMaxCachedModels(2) if the machine has the memory. This warns once.`,\n );\n}\n\n/** Settle the serialization slot for a finished generate. */\nfunction _releaseGenerateSlot(id: string): void {\n _queuedModelIds.delete(id);\n const resolve = _generateDoneResolvers.get(id);\n if (resolve) {\n _generateDoneResolvers.delete(id);\n resolve();\n }\n}\n\n/** Model known to be loaded (main-thread view). */\nlet _loadedModelId: string | null = null;\n/** Model currently being prepared (for the getModelStatus 'cached' path). */\nlet _preparingModelId: string | null = null;\n\nfunction _getWorker(): Worker {\n if (!_worker) {\n _worker = new Worker(new URL('./worker.ts', import.meta.url), { type: 'module' });\n _worker.addEventListener('message', _handleWorkerMessage);\n _worker.addEventListener('error', _handleWorkerError);\n _worker.addEventListener('messageerror', _handleWorkerError);\n }\n return _worker;\n}\n\n/**\n * Worker crashed (uncaught error / WASM init failure / OOM). Reject every in-flight\n * prepare and close every open generate stream so the UI doesn't hang. Subsequent\n * calls rebuild the worker.\n */\nfunction _handleWorkerError(e: Event): void {\n const message = (e as ErrorEvent)?.message || 'Worker crashed unexpectedly';\n\n for (const p of _pendingPrepares.values()) {\n try { p.reject(new Error(message)); } catch { /* ignore */ }\n }\n _pendingPrepares.clear();\n\n for (const ctrl of _pendingGenerates.values()) {\n try { ctrl.enqueue({ type: 'error' as const, message }); ctrl.close(); }\n catch { /* ignore */ }\n }\n _pendingGenerates.clear();\n\n // Release every serialization slot so the generate chain doesn't deadlock.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n\n _loadedModelId = null;\n _preparingModelId = null;\n try { _worker?.terminate(); } catch { /* ignore */ }\n _worker = null;\n}\n\nfunction _handleWorkerMessage(event: MessageEvent): void {\n const msg = event.data;\n\n switch (msg.type) {\n case 'progress': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n if (msg.status === 'ready') {\n pending.onProgress({ status: 'ready' });\n pending.resolve();\n _pendingPrepares.delete(msg.id);\n } else if (msg.status === 'loading') {\n pending.onProgress({ status: 'loading' });\n } else if (msg.status === 'cached') {\n pending.onProgress({ status: 'cached', file: msg.file, progress: msg.progress });\n } else {\n pending.onProgress({ status: 'downloading', file: msg.file, progress: msg.progress });\n }\n break;\n }\n case 'prepare-error': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n pending.reject(new Error(msg.message));\n _pendingPrepares.delete(msg.id);\n if (_preparingModelId === pending.modelId) _preparingModelId = null;\n break;\n }\n case 'pipeline-ready': {\n _loadedModelId = msg.modelId;\n _preparingModelId = null;\n // Evict models over the cache limit, then refresh the known list.\n void _enforceMaxCachedModels(msg.modelId).then(() => _refreshKnownModels());\n break;\n }\n case 'gen-chunk': {\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: msg.chunkType as 'text' | 'thinking', delta: msg.delta });\n break;\n }\n case 'gen-done': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'done' as const, ...(msg.usage ? { usage: msg.usage } : {}) });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n case 'gen-error': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'error' as const, message: msg.message });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n }\n}\n\n/**\n * Narrowed so the two members the docs tell you to CALL are not optional.\n *\n * `AparteAIProvider` declares `prepareModel` and `getModelStatus` optional (most\n * providers have nothing to download), and widening to it made both\n * possibly-undefined — so the documented `TransformersProvider.prepareModel(...)`\n * needed a `!` or a guard in every strict consumer. Same technique openai-compat\n * already used for its own always-present members.\n *\n * `chat` joined the list once `AparteAIProvider` became a union: it is optional on\n * the format-adapter arm, and this provider IS its `chat()` — running inference\n * locally is the whole package. Narrowing it here says so once, instead of every\n * caller writing `provider.chat!(...)`.\n */\nexport const TransformersProvider: AparteAIProvider\n & Required<Pick<AparteAIProvider, 'prepareModel' | 'getModelStatus' | 'chat'>> = {\n id: 'transformers',\n\n getMetadata() {\n return {\n id: 'transformers',\n name: 'Transformers.js',\n icon: `<svg viewBox=\"0 0 24 24\" fill=\"none\" xmlns=\"http://www.w3.org/2000/svg\"><path d=\"M12 2L2 7l10 5 10-5-10-5z\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 17l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 12l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/></svg>`,\n color: '#f59e0b',\n description: 'Run LLMs directly in your browser via WebGPU or WASM — no API, no key',\n hasFreeModels: true,\n isLocal: true,\n helpUrl: 'https://huggingface.co/docs/transformers.js',\n };\n },\n\n getModels(): AparteAIModel[] {\n return _knownModels;\n },\n\n async fetchModels(): Promise<AparteAIModel[]> {\n await _refreshKnownModels();\n return _knownModels;\n },\n\n async chat(request: AparteChatRequest): Promise<AparteChatResponse> {\n const messages = toMessages(request.messages);\n const requestId = uuid();\n const options = {\n maxTokens: request.maxTokens,\n temperature: request.temperature,\n seed: request.seed,\n };\n const task = _registeredModels.get(request.modelId)?.task ?? 'text-generation';\n\n // ── Reserve a serialization slot ─────────────────────────────────────\n // Chain this generate behind the previous one; the worker has a single\n // pipeline, so generates MUST NOT overlap.\n _warnIfContended(request.modelId);\n _queuedModelIds.set(requestId, request.modelId);\n const prevGenerate = _generateChain;\n _generateChain = new Promise<void>((resolveSlot) => {\n _generateDoneResolvers.set(requestId, resolveSlot);\n });\n const postGenerate = (): void => {\n _getWorker().postMessage({\n type: 'generate',\n id: requestId,\n modelId: request.modelId,\n messages,\n options,\n task,\n dtype: _registeredModels.get(request.modelId)?.dtype,\n device: _computeDevice,\n });\n };\n\n if (request.stream === false) {\n return new Promise<string>((resolve, reject) => {\n let result = '';\n const fakeCtrl = {\n enqueue: (chunk: { type: string; delta?: string; message?: string }) => {\n if (chunk.type === 'text') result += chunk.delta ?? '';\n else if (chunk.type === 'done') resolve(result);\n else if (chunk.type === 'error') reject(new Error(chunk.message));\n },\n close: () => { /* no-op */ },\n } as unknown as ReadableStreamDefaultController;\n _pendingGenerates.set(requestId, fakeCtrl);\n void prevGenerate.then(postGenerate);\n });\n }\n\n return new ReadableStream({\n async start(controller) {\n _pendingGenerates.set(requestId, controller);\n await prevGenerate;\n postGenerate();\n },\n cancel() {\n _pendingGenerates.delete(requestId);\n // Actually STOP the model (not just detach the reader): tell the worker\n // to interrupt this generate. The serialization slot is still released\n // by the resulting gen-done/gen-error, so a queued generate can't start\n // before the worker has stopped this one.\n _getWorker().postMessage({ type: 'cancel', id: requestId });\n },\n });\n },\n\n async getModelStatus(modelId: string): Promise<ModelStatus> {\n if (_loadedModelId === modelId) return 'ready';\n if (_preparingModelId === modelId) return 'cached';\n if ('caches' in globalThis) {\n try {\n const encodedId = encodeURIComponent(modelId);\n const names = await caches.keys();\n for (const name of names) {\n const cache = await caches.open(name);\n const keys = await cache.keys();\n if (keys.some(r => r.url.includes(encodedId) || r.url.includes(modelId + '/'))) {\n return 'cached';\n }\n }\n } catch {\n // Cache API unavailable\n }\n }\n return 'not-downloaded';\n },\n\n async prepareModel(modelId: string, onProgress: (p: ModelLoadProgress) => void): Promise<void> {\n if (_loadedModelId === modelId) {\n onProgress({ status: 'ready' });\n return;\n }\n\n const requestId = uuid();\n _preparingModelId = modelId;\n\n const task = _registeredModels.get(modelId)?.task ?? 'text-generation';\n const dtype = _registeredModels.get(modelId)?.dtype;\n return new Promise<void>((resolve, reject) => {\n _pendingPrepares.set(requestId, { modelId, onProgress, resolve, reject });\n _getWorker().postMessage({ type: 'prepare', id: requestId, modelId, task, dtype, device: _computeDevice });\n });\n },\n\n async deleteModel(modelId: string): Promise<void> {\n await deleteCachedModel(modelId);\n },\n};\n\nexport default TransformersProvider;\nexport type { AparteAIProvider, AparteAIModel, ModelStatus, ModelLoadProgress } from '@aparte/core';\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Cache utilities (settings panels, etc.)\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Returns the modelId currently loaded in the worker's pipeline, or null. */\nexport function getLoadedModelId(): string | null {\n return _loadedModelId;\n}\n\n/** Terminate the shared worker and reset in-memory state. Safe to call any time. */\nexport function terminateWorker(): void {\n _worker?.terminate();\n _worker = null;\n _loadedModelId = null;\n _preparingModelId = null;\n for (const [, p] of _pendingPrepares) {\n p.reject(new Error('Worker terminated'));\n }\n _pendingPrepares.clear();\n for (const [, ctrl] of _pendingGenerates) {\n try { ctrl.enqueue({ type: 'error' as const, message: 'Worker terminated' }); ctrl.close(); } catch { /* already closed */ }\n }\n _pendingGenerates.clear();\n\n // Release every serialization slot and reset the chain — the same three lines\n // the worker-error handler above already carried, with the same reason. Without\n // them, terminating mid-generate left `_generateChain` pending on a resolver\n // that had just been dropped, so the NEXT chat() awaited a promise that could\n // never settle: no error, no rejection, the stream simply never started again\n // for the life of the page.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n // A terminated worker is a fresh situation; let the contention warning speak again.\n _warnedModelContention = false;\n}\n\nexport interface CachedModelEntry {\n modelId: string;\n name: string;\n /** Total size in bytes of all cached files for this model. -1 if unknown. */\n sizeBytes: number;\n /** True if the model is currently loaded in the worker. */\n loaded: boolean;\n}\n\n/**\n * Scan the Cache API to find which Transformers.js models have been downloaded,\n * by matching cache entry URLs against the Hugging Face resolve path.\n */\nexport async function listCachedModels(): Promise<CachedModelEntry[]> {\n if (!('caches' in globalThis)) return [];\n\n const found = new Map<string, { name: string; sizeBytes: number }>();\n\n // e.g. https://huggingface.co/onnx-community/Qwen2.5-0.5B/resolve/main/config.json\n // → onnx-community/Qwen2.5-0.5B\n function extractModelId(url: string): string | null {\n const m = url.match(/huggingface\\.co\\/([^/]+\\/[^/]+)\\/resolve\\//);\n return m ? decodeURIComponent(m[1]!) : null;\n }\n\n function modelName(modelId: string): string {\n const config = _registeredModels.get(modelId);\n if (config) return config.name;\n return (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n }\n\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n for (const req of requests) {\n const modelId = extractModelId(req.url);\n if (!modelId) continue;\n if (!found.has(modelId)) {\n found.set(modelId, { name: modelName(modelId), sizeBytes: 0 });\n }\n const response = await cache.match(req);\n if (!response) continue;\n const contentLength = response.headers.get('content-length');\n if (contentLength) {\n found.get(modelId)!.sizeBytes += parseInt(contentLength, 10);\n } else {\n try {\n const blob = await response.clone().blob();\n found.get(modelId)!.sizeBytes += blob.size;\n } catch { /* skip */ }\n }\n }\n } catch { /* skip inaccessible cache */ }\n }));\n } catch {\n return [];\n }\n\n return Array.from(found.entries()).map(([modelId, { name, sizeBytes }]) => ({\n modelId,\n name,\n sizeBytes,\n loaded: _loadedModelId === modelId,\n }));\n}\n\n/**\n * Delete all cached files for a modelId from the Cache API, terminating the worker\n * first if that model is currently loaded.\n */\nexport async function deleteCachedModel(modelId: string): Promise<void> {\n if (_loadedModelId === modelId || _preparingModelId === modelId) {\n terminateWorker();\n }\n if (!('caches' in globalThis)) return;\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n const encoded = encodeURIComponent(modelId);\n await Promise.all(\n requests\n .filter(r => r.url.includes(modelId) || r.url.includes(encoded))\n .map(r => cache.delete(r)),\n );\n } catch { /* skip */ }\n }));\n } catch { /* Cache API unavailable */ }\n}\n"],"names":[],"mappings":";AA0DA,IAAI,iBAAqE;AAMlE,SAAS,sBAAsB,OAA0D;AAC5F,mBAAiB;AACrB;AAEA,eAAsB,iBAA2C;AAG7D,QAAM,QAAiB,UAAmD,gBAAgB;AAG1F,MAAI,SAAS;AACb,MAAI,SAAS,WAAW;AACpB,QAAI;AACA,YAAM,UAAU,MAAO,UAAyE,IAAI,eAAA;AACpG,eAAS,YAAY;AAAA,IACzB,QAAQ;AACJ,eAAS;AAAA,IACb;AAAA,EACJ;AAEA,MAAI;AACJ,MAAI,CAAC,UAAU,QAAQ,GAAG;AACtB,WAAO;AAAA,EACX,WAAW,QAAQ,GAAG;AAClB,WAAO;AAAA,EACX,OAAO;AACH,WAAO;AAAA,EACX;AAEA,QAAM,qBAAqB,iBACpB,eAAe,IAAI,KAAK,eAAe,QAAQ,KAChD;AAEN,SAAO,EAAE,QAAQ,OAAO,MAAM,mBAAA;AAClC;AAsBA,MAAM,wCAAwB,IAAA;AAG9B,IAAI,eAAgC,CAAA;AAK7B,SAAS,cAAc,QAAuC;AACjE,oBAAkB,IAAI,OAAO,IAAI,MAAM;AACvC,MAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,OAAO,EAAE,GAAG;AAC7C,mBAAe,CAAC,GAAG,cAAc;AAAA,MAC7B,IAAI,OAAO;AAAA,MACX,MAAM,OAAO;AAAA,MACb,aAAa,OAAO;AAAA,MACpB,cAAc,OAAO;AAAA,IAAA,CACxB;AAAA,EACL;AACJ;AAGA,SAAS,qBAAqB,SAAgC;AAC1D,QAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,MAAI,OAAQ,QAAO,EAAE,IAAI,OAAO,IAAI,MAAM,OAAO,MAAM,aAAa,OAAO,aAAa,cAAc,OAAO,aAAA;AAC7G,QAAM,QAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AACpE,SAAO,EAAE,IAAI,SAAS,MAAM,cAAc,CAAC,WAAW,EAAA;AAC1D;AAGA,IAAI,mBAAmB;AAMhB,SAAS,mBAAmB,KAAmB;AAClD,qBAAmB;AACvB;AAGO,SAAS,qBAA6B;AACzC,SAAO;AACX;AASA,IAAI,iBAAgC;AAE7B,SAAS,iBAAiB,GAAwB;AACrD,mBAAiB;AACrB;AAEO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGA,eAAe,wBAAwB,aAAoC;AACvE,MAAI,qBAAqB,EAAG;AAC5B,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,UAAM,SAAS,OAAO,OAAO,CAAA,MAAK,EAAE,YAAY,WAAW;AAC3D,UAAM,SAAS,OAAO,SAAS;AAC/B,QAAI,UAAU,EAAG;AAEjB,aAAS,IAAI,GAAG,IAAI,UAAU,IAAI,OAAO,QAAQ,KAAK;AAClD,YAAM,kBAAkB,OAAO,CAAC,EAAG,OAAO;AAAA,IAC9C;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,eAAe,sBAAqC;AAChD,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,eAAW,SAAS,QAAQ;AACxB,UAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,MAAM,OAAO,GAAG;AACjD,uBAAe,CAAC,GAAG,cAAc,qBAAqB,MAAM,OAAO,CAAC;AAAA,MACxE;AAAA,IACJ;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,IAAI,0BAA0B;AAG9B,SAAS,WAAW,UAAgD;AAChE,QAAM,SAA0B,CAAA;AAChC,MAAI,mBAAmB;AACvB,aAAW,KAAK,UAAU;AACtB,QAAI,EAAE,SAAS,UAAU,EAAE,SAAS,eAAe,EAAE,SAAS,UAAU;AACpE,YAAM,OAAO,cAAc,EAAE,OAAO;AACpC,UAAI,aAAa,KAAK,EAAE,MAAM,EAAE,MAAM,SAAS,MAAM;AAAA,IACzD,OAAO;AAKH;AAAA,IACJ;AAAA,EACJ;AACA,MAAI,mBAAmB,KAAK,CAAC,yBAAyB;AAClD,8BAA0B;AAC1B,YAAQ;AAAA,MACJ,0BAA0B,gBAAgB;AAAA,IAAA;AAAA,EAIlD;AACA,SAAO;AACX;AAMA,IAAI,UAAyB;AAQ7B,MAAM,uCAAuB,IAAA;AAC7B,MAAM,wCAAwB,IAAA;AAM9B,IAAI,iBAAgC,QAAQ,QAAA;AAC5C,MAAM,6CAA6B,IAAA;AAOnC,MAAM,sCAAsB,IAAA;AAC5B,IAAI,yBAAyB;AAG7B,SAAS,mBAAmB,WAAuC;AAC/D,aAAW,MAAM,gBAAgB,OAAA,EAAU,KAAI,OAAO,UAAW,QAAO;AACxE,SAAO;AACX;AAOA,SAAS,iBAAiB,WAAyB;AAC/C,MAAI,uBAAwB;AAC5B,QAAM,QAAQ,mBAAmB,SAAS;AAC1C,MAAI,CAAC,MAAO;AACZ,2BAAyB;AACzB,UAAQ;AAAA,IACJ,mEAAmE,SAAS,aACtE,KAAK,gHACiC,gBAAgB;AAAA,EAAA;AAIpE;AAGA,SAAS,qBAAqB,IAAkB;AAC5C,kBAAgB,OAAO,EAAE;AACzB,QAAM,UAAU,uBAAuB,IAAI,EAAE;AAC7C,MAAI,SAAS;AACT,2BAAuB,OAAO,EAAE;AAChC,YAAA;AAAA,EACJ;AACJ;AAGA,IAAI,iBAAgC;AAEpC,IAAI,oBAAmC;AAEvC,SAAS,aAAqB;AAC1B,MAAI,CAAC,SAAS;AACV,cAAU,IAAI,OAAO,IAAA;AAAA;AAAA,MAAA;MAAA,YAAA;AAAA,IAAA,GAAyC,EAAE,MAAM,SAAA,CAAU;AAChF,YAAQ,iBAAiB,WAAW,oBAAoB;AACxD,YAAQ,iBAAiB,SAAS,kBAAkB;AACpD,YAAQ,iBAAiB,gBAAgB,kBAAkB;AAAA,EAC/D;AACA,SAAO;AACX;AAOA,SAAS,mBAAmB,GAAgB;AACxC,QAAM,UAAW,GAAkB,WAAW;AAE9C,aAAW,KAAK,iBAAiB,UAAU;AACvC,QAAI;AAAE,QAAE,OAAO,IAAI,MAAM,OAAO,CAAC;AAAA,IAAG,QAAQ;AAAA,IAAe;AAAA,EAC/D;AACA,mBAAiB,MAAA;AAEjB,aAAW,QAAQ,kBAAkB,UAAU;AAC3C,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS;AAAG,WAAK,MAAA;AAAA,IAAS,QACjE;AAAA,IAAe;AAAA,EACzB;AACA,oBAAkB,MAAA;AAGlB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,mBAAiB;AACjB,sBAAoB;AACpB,MAAI;AAAE,aAAS,UAAA;AAAA,EAAa,QAAQ;AAAA,EAAe;AACnD,YAAU;AACd;AAEA,SAAS,qBAAqB,OAA2B;AACrD,QAAM,MAAM,MAAM;AAElB,UAAQ,IAAI,MAAA;AAAA,IACR,KAAK,YAAY;AACb,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,UAAI,IAAI,WAAW,SAAS;AACxB,gBAAQ,WAAW,EAAE,QAAQ,QAAA,CAAS;AACtC,gBAAQ,QAAA;AACR,yBAAiB,OAAO,IAAI,EAAE;AAAA,MAClC,WAAW,IAAI,WAAW,WAAW;AACjC,gBAAQ,WAAW,EAAE,QAAQ,UAAA,CAAW;AAAA,MAC5C,WAAW,IAAI,WAAW,UAAU;AAChC,gBAAQ,WAAW,EAAE,QAAQ,UAAU,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACnF,OAAO;AACH,gBAAQ,WAAW,EAAE,QAAQ,eAAe,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACxF;AACA;AAAA,IACJ;AAAA,IACA,KAAK,iBAAiB;AAClB,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,cAAQ,OAAO,IAAI,MAAM,IAAI,OAAO,CAAC;AACrC,uBAAiB,OAAO,IAAI,EAAE;AAC9B,UAAI,sBAAsB,QAAQ,QAAS,qBAAoB;AAC/D;AAAA,IACJ;AAAA,IACA,KAAK,kBAAkB;AACnB,uBAAiB,IAAI;AACrB,0BAAoB;AAEpB,WAAK,wBAAwB,IAAI,OAAO,EAAE,KAAK,MAAM,qBAAqB;AAC1E;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,IAAI,WAAkC,OAAO,IAAI,OAAO;AAC7E;AAAA,IACJ;AAAA,IACA,KAAK,YAAY;AACb,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,QAAiB,GAAI,IAAI,QAAQ,EAAE,OAAO,IAAI,MAAA,IAAU,CAAA,GAAK;AAClF,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,IAAI,SAAS;AAC7D,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,EAAA;AAER;AAgBO,MAAM,uBACwE;AAAA,EACjF,IAAI;AAAA,EAEJ,cAAc;AACV,WAAO;AAAA,MACH,IAAI;AAAA,MACJ,MAAM;AAAA,MACN,MAAM;AAAA,MACN,OAAO;AAAA,MACP,aAAa;AAAA,MACb,eAAe;AAAA,MACf,SAAS;AAAA,MACT,SAAS;AAAA,IAAA;AAAA,EAEjB;AAAA,EAEA,YAA6B;AACzB,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,cAAwC;AAC1C,UAAM,oBAAA;AACN,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,KAAK,SAAyD;AAChE,UAAM,WAAW,WAAW,QAAQ,QAAQ;AAC5C,UAAM,YAAY,KAAA;AAClB,UAAM,UAAU;AAAA,MACZ,WAAW,QAAQ;AAAA,MACnB,aAAa,QAAQ;AAAA,MACrB,MAAM,QAAQ;AAAA,IAAA;AAElB,UAAM,OAAO,kBAAkB,IAAI,QAAQ,OAAO,GAAG,QAAQ;AAK7D,qBAAiB,QAAQ,OAAO;AAChC,oBAAgB,IAAI,WAAW,QAAQ,OAAO;AAC9C,UAAM,eAAe;AACrB,qBAAiB,IAAI,QAAc,CAAC,gBAAgB;AAChD,6BAAuB,IAAI,WAAW,WAAW;AAAA,IACrD,CAAC;AACD,UAAM,eAAe,MAAY;AAC7B,iBAAA,EAAa,YAAY;AAAA,QACrB,MAAM;AAAA,QACN,IAAI;AAAA,QACJ,SAAS,QAAQ;AAAA,QACjB;AAAA,QACA;AAAA,QACA;AAAA,QACA,OAAO,kBAAkB,IAAI,QAAQ,OAAO,GAAG;AAAA,QAC/C,QAAQ;AAAA,MAAA,CACX;AAAA,IACL;AAEA,QAAI,QAAQ,WAAW,OAAO;AAC1B,aAAO,IAAI,QAAgB,CAAC,SAAS,WAAW;AAC5C,YAAI,SAAS;AACb,cAAM,WAAW;AAAA,UACb,SAAS,CAAC,UAA8D;AACpE,gBAAI,MAAM,SAAS,OAAQ,WAAU,MAAM,SAAS;AAAA,qBAC3C,MAAM,SAAS,OAAQ,SAAQ,MAAM;AAAA,qBACrC,MAAM,SAAS,QAAS,QAAO,IAAI,MAAM,MAAM,OAAO,CAAC;AAAA,UACpE;AAAA,UACA,OAAO,MAAM;AAAA,UAAc;AAAA,QAAA;AAE/B,0BAAkB,IAAI,WAAW,QAAQ;AACzC,aAAK,aAAa,KAAK,YAAY;AAAA,MACvC,CAAC;AAAA,IACL;AAEA,WAAO,IAAI,eAAe;AAAA,MACtB,MAAM,MAAM,YAAY;AACpB,0BAAkB,IAAI,WAAW,UAAU;AAC3C,cAAM;AACN,qBAAA;AAAA,MACJ;AAAA,MACA,SAAS;AACL,0BAAkB,OAAO,SAAS;AAKlC,mBAAA,EAAa,YAAY,EAAE,MAAM,UAAU,IAAI,WAAW;AAAA,MAC9D;AAAA,IAAA,CACH;AAAA,EACL;AAAA,EAEA,MAAM,eAAe,SAAuC;AACxD,QAAI,mBAAmB,QAAS,QAAO;AACvC,QAAI,sBAAsB,QAAS,QAAO;AAC1C,QAAI,YAAY,YAAY;AACxB,UAAI;AACA,cAAM,YAAY,mBAAmB,OAAO;AAC5C,cAAM,QAAQ,MAAM,OAAO,KAAA;AAC3B,mBAAW,QAAQ,OAAO;AACtB,gBAAM,QAAQ,MAAM,OAAO,KAAK,IAAI;AACpC,gBAAM,OAAO,MAAM,MAAM,KAAA;AACzB,cAAI,KAAK,KAAK,CAAA,MAAK,EAAE,IAAI,SAAS,SAAS,KAAK,EAAE,IAAI,SAAS,UAAU,GAAG,CAAC,GAAG;AAC5E,mBAAO;AAAA,UACX;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAER;AAAA,IACJ;AACA,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,aAAa,SAAiB,YAA2D;AAC3F,QAAI,mBAAmB,SAAS;AAC5B,iBAAW,EAAE,QAAQ,SAAS;AAC9B;AAAA,IACJ;AAEA,UAAM,YAAY,KAAA;AAClB,wBAAoB;AAEpB,UAAM,OAAO,kBAAkB,IAAI,OAAO,GAAG,QAAQ;AACrD,UAAM,QAAQ,kBAAkB,IAAI,OAAO,GAAG;AAC9C,WAAO,IAAI,QAAc,CAAC,SAAS,WAAW;AAC1C,uBAAiB,IAAI,WAAW,EAAE,SAAS,YAAY,SAAS,QAAQ;AACxE,iBAAA,EAAa,YAAY,EAAE,MAAM,WAAW,IAAI,WAAW,SAAS,MAAM,OAAO,QAAQ,eAAA,CAAgB;AAAA,IAC7G,CAAC;AAAA,EACL;AAAA,EAEA,MAAM,YAAY,SAAgC;AAC9C,UAAM,kBAAkB,OAAO;AAAA,EACnC;AACJ;AAUO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGO,SAAS,kBAAwB;AACpC,WAAS,UAAA;AACT,YAAU;AACV,mBAAiB;AACjB,sBAAoB;AACpB,aAAW,CAAA,EAAG,CAAC,KAAK,kBAAkB;AAClC,MAAE,OAAO,IAAI,MAAM,mBAAmB,CAAC;AAAA,EAC3C;AACA,mBAAiB,MAAA;AACjB,aAAW,CAAA,EAAG,IAAI,KAAK,mBAAmB;AACtC,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,qBAAqB;AAAG,WAAK,MAAA;AAAA,IAAS,QAAQ;AAAA,IAAuB;AAAA,EAC/H;AACA,oBAAkB,MAAA;AAQlB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,2BAAyB;AAC7B;AAeA,eAAsB,mBAAgD;AAClE,MAAI,EAAE,YAAY,YAAa,QAAO,CAAA;AAEtC,QAAM,4BAAY,IAAA;AAIlB,WAAS,eAAe,KAA4B;AAChD,UAAM,IAAI,IAAI,MAAM,4CAA4C;AAChE,WAAO,IAAI,mBAAmB,EAAE,CAAC,CAAE,IAAI;AAAA,EAC3C;AAEA,WAAS,UAAU,SAAyB;AACxC,UAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,QAAI,eAAe,OAAO;AAC1B,YAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AAAA,EAClE;AAEA,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,mBAAW,OAAO,UAAU;AACxB,gBAAM,UAAU,eAAe,IAAI,GAAG;AACtC,cAAI,CAAC,QAAS;AACd,cAAI,CAAC,MAAM,IAAI,OAAO,GAAG;AACrB,kBAAM,IAAI,SAAS,EAAE,MAAM,UAAU,OAAO,GAAG,WAAW,GAAG;AAAA,UACjE;AACA,gBAAM,WAAW,MAAM,MAAM,MAAM,GAAG;AACtC,cAAI,CAAC,SAAU;AACf,gBAAM,gBAAgB,SAAS,QAAQ,IAAI,gBAAgB;AAC3D,cAAI,eAAe;AACf,kBAAM,IAAI,OAAO,EAAG,aAAa,SAAS,eAAe,EAAE;AAAA,UAC/D,OAAO;AACH,gBAAI;AACA,oBAAM,OAAO,MAAM,SAAS,MAAA,EAAQ,KAAA;AACpC,oBAAM,IAAI,OAAO,EAAG,aAAa,KAAK;AAAA,YAC1C,QAAQ;AAAA,YAAa;AAAA,UACzB;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAAgC;AAAA,IAC5C,CAAC,CAAC;AAAA,EACN,QAAQ;AACJ,WAAO,CAAA;AAAA,EACX;AAEA,SAAO,MAAM,KAAK,MAAM,QAAA,CAAS,EAAE,IAAI,CAAC,CAAC,SAAS,EAAE,MAAM,UAAA,CAAW,OAAO;AAAA,IACxE;AAAA,IACA;AAAA,IACA;AAAA,IACA,QAAQ,mBAAmB;AAAA,EAAA,EAC7B;AACN;AAMA,eAAsB,kBAAkB,SAAgC;AACpE,MAAI,mBAAmB,WAAW,sBAAsB,SAAS;AAC7D,oBAAA;AAAA,EACJ;AACA,MAAI,EAAE,YAAY,YAAa;AAC/B,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,cAAM,UAAU,mBAAmB,OAAO;AAC1C,cAAM,QAAQ;AAAA,UACV,SACK,OAAO,CAAA,MAAK,EAAE,IAAI,SAAS,OAAO,KAAK,EAAE,IAAI,SAAS,OAAO,CAAC,EAC9D,IAAI,OAAK,MAAM,OAAO,CAAC,CAAC;AAAA,QAAA;AAAA,MAErC,QAAQ;AAAA,MAAa;AAAA,IACzB,CAAC,CAAC;AAAA,EACN,QAAQ;AAAA,EAA8B;AAC1C;"}
1
+ {"version":3,"file":"index.js","sources":["../src/index.ts"],"sourcesContent":["/**\n * @aparte/provider-transformers — run LLMs 100% in the browser via Transformers.js.\n *\n * A local, keyless `AparteAIProvider`: it owns its I/O (inference runs off the main\n * thread in a Web Worker) so `AparteDirectTransport` delegates to its `chat()`. Model\n * weights download once and persist in the Cache API.\n *\n * Scope (v1): generic **text-generation** streaming. Tool-calling for local models is\n * model-specific (every family has its own wire format) and is out of scope here — the\n * app registers models and streams plain replies. Vision / embeddings can follow on demand.\n *\n * ## This provider's state is TAB-scoped, on purpose\n *\n * Everything below the \"Worker bridge\" heading — the worker, the loaded model, the\n * generate chain — plus `setComputeDevice`, `setMaxCachedModels` and\n * `setHardwareTierModels`, is module-level and therefore shared by every chat on the\n * page. That is deliberate, and it is the opposite of what the rest of the suite does:\n * a plugin's providers scope to one chat, this one cannot.\n *\n * The reason is the resource, not the design. A local model is 1–2 GB of weights and one\n * WebGPU pipeline. Handing each chat its own worker would mean N copies resident in one\n * tab — which is the failure this package exists to avoid, not a capability. The\n * settings above describe the *machine* (which backend, how many models to keep\n * cached), so per-chat values would not mean anything either.\n *\n * What the constraint costs: two chats on the page driving DIFFERENT local models take\n * turns on one pipeline, so each turn may evict and reload gigabytes. That used to\n * happen silently — a multi-second stall with nothing to read. It now warns once, from\n * `chat()`, when a generate is queued for a model other than the one already in flight.\n * Same model in both chats is free and correct: they share the load.\n */\n\nimport type {\n AparteAIProvider,\n AparteAIModel,\n AparteChatRequest,\n AparteChatResponse,\n AparteChatMessage,\n ModelStatus,\n ModelLoadProgress,\n} from '@aparte/core';\nimport { contentToText, uuid } from '@aparte/core';\n\n// The worker's URL, not the worker itself: this package constructs it by hand because a\n// cross-origin copy has to go through a blob (see `_spawnWorker`). `?worker&url` is what\n// keeps Vite emitting the worker as its own chunk — the `new Worker(new URL(...))` form\n// it detects by pattern was the only other way, and moving the URL out of that call made\n// the build inline the worker's raw TypeScript as a data: URL instead. Caught by a\n// two-origin browser probe, not by any test.\nimport workerUrl from './worker.ts?worker&url';\n\n/** The minimal chat shape passed to the worker (the tokenizer applies the chat template). */\ntype SimpleMessage = { role: 'user' | 'assistant' | 'system'; content: string };\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Hardware detection\n// ─────────────────────────────────────────────────────────────────────────────\n\nexport interface HardwareProfile {\n hasGpu: boolean;\n ramGb: number;\n tier: 'low' | 'mid' | 'high';\n recommendedModelId: string;\n}\n\n/** Hardware-tier model overrides — set by the app via setHardwareTierModels(). */\nlet _hardwareTiers: { low: string; mid?: string; high: string } | null = null;\n\n/**\n * Set the model IDs to use per hardware tier. Call before detectHardware() is used\n * to pick a default model — the provider ships no model knowledge of its own.\n */\nexport function setHardwareTierModels(tiers: { low: string; mid?: string; high: string }): void {\n _hardwareTiers = tiers;\n}\n\nexport async function detectHardware(): Promise<HardwareProfile> {\n // navigator.deviceMemory: W3C API, Chromium only, capped at 8 GB for privacy\n // (1 | 2 | 4 | 8). Falls back to 4 on Firefox/Safari.\n const ramGb: number = (navigator as unknown as { deviceMemory?: number }).deviceMemory ?? 4;\n\n // Real WebGPU check: requestAdapter() returns null if no capable GPU is present.\n let hasGpu = false;\n if ('gpu' in navigator) {\n try {\n const adapter = await (navigator as unknown as { gpu: { requestAdapter(): Promise<unknown> } }).gpu.requestAdapter();\n hasGpu = adapter !== null;\n } catch {\n hasGpu = false;\n }\n }\n\n let tier: 'low' | 'mid' | 'high';\n if (!hasGpu || ramGb < 4) {\n tier = 'low';\n } else if (ramGb < 8) {\n tier = 'mid';\n } else {\n tier = 'high';\n }\n\n const recommendedModelId = _hardwareTiers\n ? (_hardwareTiers[tier] ?? _hardwareTiers.high ?? '')\n : '';\n\n return { hasGpu, ramGb, tier, recommendedModelId };\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Model catalog — all model knowledge lives in the app, not the provider.\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Configuration for a model registered with the provider. */\nexport interface TransformersModelConfig {\n id: string;\n name: string;\n description?: string;\n capabilities: AparteAIModel['capabilities'];\n /** Transformers.js pipeline task — determines the model architecture / load path. */\n task: 'text-generation';\n /** ONNX dtype or per-part dtype map (e.g. `'q4'` or `{ decoder_model_merged: 'q4' }`). */\n dtype?: string | Record<string, string>;\n /** Preferred device. Defaults to WebGPU when available, else WASM. */\n device?: 'webgpu' | 'wasm' | 'auto';\n metadata?: Record<string, unknown>;\n}\n\n/** Models registered by the app via registerModel(). */\nconst _registeredModels = new Map<string, TransformersModelConfig>();\n\n/** Mutable model list — populated by registerModel() and cache discovery. */\nlet _knownModels: AparteAIModel[] = [];\n\n/**\n * Register a model with the provider. Call before the model is used for inference.\n */\nexport function registerModel(config: TransformersModelConfig): void {\n _registeredModels.set(config.id, config);\n if (!_knownModels.find(m => m.id === config.id)) {\n _knownModels = [..._knownModels, {\n id: config.id,\n name: config.name,\n description: config.description,\n capabilities: config.capabilities,\n }];\n }\n}\n\n/** Build an AparteAIModel entry from a cache-discovered modelId not in the registry. */\nfunction _modelFromCacheEntry(modelId: string): AparteAIModel {\n const config = _registeredModels.get(modelId);\n if (config) return { id: config.id, name: config.name, description: config.description, capabilities: config.capabilities };\n const name = (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n return { id: modelId, name, capabilities: ['streaming'] };\n}\n\n/** Max number of models to keep in cache. 0 = unlimited. Default: 1. */\nlet _maxCachedModels = 1;\n\n/**\n * Set the maximum number of models to keep in cache. When exceeded after a new\n * model is ready, the oldest models are evicted. 0 = unlimited.\n */\nexport function setMaxCachedModels(max: number): void {\n _maxCachedModels = max;\n}\n\n/** Returns the current max-cached-models setting. */\nexport function getMaxCachedModels(): number {\n return _maxCachedModels;\n}\n\n/**\n * User's preferred compute backend for local inference.\n * 'auto' → WebGPU when available, else WASM (default)\n * 'webgpu' → force WebGPU\n * 'wasm' → force WASM CPU\n */\nexport type ComputeDevice = 'auto' | 'webgpu' | 'wasm';\nlet _computeDevice: ComputeDevice = 'auto';\n\nexport function setComputeDevice(d: ComputeDevice): void {\n _computeDevice = d;\n}\n\nexport function getComputeDevice(): ComputeDevice {\n return _computeDevice;\n}\n\n/** Evict models from cache until count <= _maxCachedModels; `keepModelId` is never evicted. */\nasync function _enforceMaxCachedModels(keepModelId: string): Promise<void> {\n if (_maxCachedModels === 0) return; // unlimited\n try {\n const cached = await listCachedModels();\n const others = cached.filter(e => e.modelId !== keepModelId);\n const excess = cached.length - _maxCachedModels;\n if (excess <= 0) return;\n // Delete the excess models (oldest first — they appear first in cache scan order).\n for (let i = 0; i < excess && i < others.length; i++) {\n await deleteCachedModel(others[i]!.modelId);\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Merge cached models into _knownModels (idempotent). Called by fetchModels(). */\nasync function _refreshKnownModels(): Promise<void> {\n try {\n const cached = await listCachedModels();\n for (const entry of cached) {\n if (!_knownModels.find(m => m.id === entry.modelId)) {\n _knownModels = [..._knownModels, _modelFromCacheEntry(entry.modelId)];\n }\n }\n } catch { /* cache unavailable */ }\n}\n\n/** Warn at most once per session that tool turns were left out of the prompt. */\nlet _warnedToolTurnsDropped = false;\n\n/** AparteChatMessage[] → plain chat turns (the tokenizer's chat template does the rest). */\nfunction toMessages(messages: AparteChatMessage[]): SimpleMessage[] {\n const result: SimpleMessage[] = [];\n let droppedToolTurns = 0;\n for (const m of messages) {\n if (m.role === 'user' || m.role === 'assistant' || m.role === 'system') {\n const text = contentToText(m.content);\n if (text) result.push({ role: m.role, content: text });\n } else {\n // tool_call / tool_result are not supported by this generic provider (v1):\n // rendering them needs a model-specific tool syntax. Dropping them\n // silently meant an app with registered tools got a model that never saw\n // the call or its result, with nothing to explain the behaviour.\n droppedToolTurns++;\n }\n }\n if (droppedToolTurns > 0 && !_warnedToolTurnsDropped) {\n _warnedToolTurnsDropped = true;\n console.warn(\n `[transformers] Dropped ${droppedToolTurns} tool turn(s) from the prompt: this provider ` +\n 'does not support tool calling (v1), so the model will not see the call or its result. ' +\n 'Use an OpenAI-compatible endpoint for tools, or render the turns yourself before sending.',\n );\n }\n return result;\n}\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Worker bridge\n// ─────────────────────────────────────────────────────────────────────────────\n\nlet _worker: Worker | null = null;\n\ninterface PendingPrepare {\n modelId: string;\n onProgress: (p: ModelLoadProgress) => void;\n resolve: () => void;\n reject: (err: Error) => void;\n}\nconst _pendingPrepares = new Map<string, PendingPrepare>();\nconst _pendingGenerates = new Map<string, ReadableStreamDefaultController>();\n\n// ── Generate serialization ──────────────────────────────────────────────────\n// The worker holds ONE pipeline: two concurrent generates would corrupt each\n// other. Each chat() chains its `generate` behind the previous generate's\n// completion (gen-done / gen-error).\nlet _generateChain: Promise<void> = Promise.resolve();\nconst _generateDoneResolvers = new Map<string, () => void>();\n\n// ── Contention on the one pipeline ──────────────────────────────────────────\n// Serialization is correct but invisible: two chats driving DIFFERENT local\n// models take turns, and with `maxCachedModels` at its default of 1 each turn\n// can evict and reload gigabytes. The user sees a stall; the developer sees\n// nothing. These two track just enough to say so, once.\nconst _queuedModelIds = new Map<string, string>();\nlet _warnedModelContention = false;\n\n/** Model ids of generates currently queued or running on the single pipeline. */\nfunction _contendingModelId(requested: string): string | undefined {\n for (const id of _queuedModelIds.values()) if (id !== requested) return id;\n return undefined;\n}\n\n/**\n * Warn once when a generate has to queue behind another chat's DIFFERENT model.\n * Not a warning about switching models in one chat — that is a deliberate act\n * with visible feedback. This fires only when two are in flight at once.\n */\nfunction _warnIfContended(requested: string): void {\n if (_warnedModelContention) return;\n const other = _contendingModelId(requested);\n if (!other) return;\n _warnedModelContention = true;\n console.warn(\n `[Aparte] Two chats are driving different local models at once (\"${requested}\" behind `\n + `\"${other}\"). Transformers.js runs one pipeline per tab, so these generates are `\n + `serialized, and with a cache budget of ${_maxCachedModels} each switch can evict and `\n + `reload gigabytes of weights. Point both chats at one model, or raise the budget with `\n + `setMaxCachedModels(2) if the machine has the memory. This warns once.`,\n );\n}\n\n/** Settle the serialization slot for a finished generate. */\nfunction _releaseGenerateSlot(id: string): void {\n _queuedModelIds.delete(id);\n const resolve = _generateDoneResolvers.get(id);\n if (resolve) {\n _generateDoneResolvers.delete(id);\n resolve();\n }\n}\n\n/** Model known to be loaded (main-thread view). */\nlet _loadedModelId: string | null = null;\n/** Model currently being prepared (for the getModelStatus 'cached' path). */\nlet _preparingModelId: string | null = null;\n\n/** The blob URL the worker was built from, if it needed one. Revoked with the worker. */\nlet _workerBlobUrl: string | null = null;\n\n/**\n * Build the worker — including when this package is served from another origin.\n *\n * `new Worker()` refuses a cross-origin script outright, and that is not an exotic\n * case: it is every deploy whose JavaScript lives on a CDN or an asset host while the\n * page lives somewhere else, with or without a bundler. Reproduced with the package on\n * one port and the page on another: `SecurityError: Script at '…/assets/worker-*.js'\n * cannot be accessed from origin '…'`.\n *\n * A blob inherits the ORIGIN OF THE DOCUMENT THAT CREATES IT, so a one-line blob whose\n * body imports the real worker by absolute URL is same-origin by construction, and the\n * import inside it is a normal cross-origin module fetch, which is allowed. It is the\n * shim ffmpeg.wasm and tesseract.js use for the same reason.\n *\n * Same-origin keeps the direct path: no blob, nothing to revoke, and a stack trace that\n * names the real file.\n */\nfunction _spawnWorker(): Worker {\n const url = new URL(workerUrl, import.meta.url);\n const sameOrigin = typeof location === 'undefined' || url.origin === location.origin;\n // A blob is the only way across an origin, so an environment that cannot mint one has\n // nothing to gain from trying: construct directly and let the platform say what it\n // thinks. jsdom is that environment — it has `Blob` and no `URL.createObjectURL` — and\n // every test in this package went through the blob path and threw before this line\n // existed.\n const canMintBlob = typeof Blob === 'function' && typeof URL.createObjectURL === 'function';\n // The literal below is not style. `new Worker(new URL('./worker.ts', import.meta.url))`\n // is the exact shape Vite's worker detection and webpack's WorkerPlugin match on, and\n // matching it is what makes a CONSUMER's bundler process the worker as a module — which\n // is how `@huggingface/transformers` gets resolved inside it today. Behind a variable\n // the chunk is copied as an opaque asset and its imports are never touched, so hoisting\n // this line to reuse it for the blob would fix a CDN page by breaking every bundled app.\n if (sameOrigin || !canMintBlob) return new Worker(new URL('./worker.ts', import.meta.url), { type: 'module' });\n\n _workerBlobUrl = URL.createObjectURL(\n new Blob([`import ${JSON.stringify(url.href)};`], { type: 'text/javascript' }),\n );\n try {\n return new Worker(_workerBlobUrl, { type: 'module' });\n } catch (error) {\n // A page with `worker-src 'self'` (or `script-src` without `blob:`) blocks the\n // shim, and the direct URL was already refused for its origin — so there is\n // nothing left to try. Say which of the two walls was hit, because the browser's\n // own message does not distinguish them.\n URL.revokeObjectURL(_workerBlobUrl);\n _workerBlobUrl = null;\n throw new Error(\n `@aparte/provider-transformers is served from ${url.origin}, which is not this page's origin, `\n + 'so its worker has to be started through a blob: URL — and this page\\'s Content-Security-Policy '\n + 'refuses that. Allow `blob:` in `worker-src` (or `script-src`), or serve the package from your '\n + `own origin. Original error: ${String(error)}`,\n );\n }\n}\n\nfunction _releaseWorkerBlob(): void {\n if (_workerBlobUrl) {\n URL.revokeObjectURL(_workerBlobUrl);\n _workerBlobUrl = null;\n }\n}\n\n/**\n * Where the page says Transformers.js lives, if it says so at all.\n *\n * The worker cannot ask: an import map is the DOCUMENT's, and by spec it does not reach\n * a worker. The main thread can, and does it the platform's way — `import.meta.resolve`\n * consults that same map — so a page that already maps `@huggingface/transformers` (it\n * has to, to import this package by name at all) is telling us where its copy is. That\n * map is the CDN consumer's manifest: the version pin stays with the consumer, which is\n * the whole point of a peer dependency, and this package invents no second place to say\n * it.\n *\n * `undefined` under a bundler, where the specifier is resolved at build time and the\n * worker's own `import('@huggingface/transformers')` is the path that runs.\n */\nfunction _peerModuleUrl(): string | undefined {\n const resolve = (import.meta as unknown as { resolve?: (specifier: string) => string }).resolve;\n if (typeof resolve === 'function') {\n try {\n const href = resolve('@huggingface/transformers');\n if (href && /^https?:/i.test(href)) return href;\n } catch { /* not in the map — fall through */ }\n }\n // Older engines have no `import.meta.resolve`; read the map they do have.\n try {\n const el = document.querySelector('script[type=\"importmap\"]');\n const map = el?.textContent ? JSON.parse(el.textContent) as { imports?: Record<string, string> } : null;\n const href = map?.imports?.['@huggingface/transformers'];\n if (href) return new URL(href, location.href).href;\n } catch { /* no document, or a map that is not JSON */ }\n return undefined;\n}\n\nfunction _getWorker(): Worker {\n if (!_worker) {\n _worker = _spawnWorker();\n _worker.addEventListener('message', _handleWorkerMessage);\n _worker.addEventListener('error', _handleWorkerError);\n _worker.addEventListener('messageerror', _handleWorkerError);\n // First message, before any work: postMessage keeps order, so the worker has it\n // by the time a prepare or a generate needs the module.\n _worker.postMessage({ type: 'init', transformersUrl: _peerModuleUrl() });\n }\n return _worker;\n}\n\n/**\n * Worker crashed (uncaught error / WASM init failure / OOM). Reject every in-flight\n * prepare and close every open generate stream so the UI doesn't hang. Subsequent\n * calls rebuild the worker.\n */\nfunction _handleWorkerError(e: Event): void {\n const message = (e as ErrorEvent)?.message || 'Worker crashed unexpectedly';\n\n for (const p of _pendingPrepares.values()) {\n try { p.reject(new Error(message)); } catch { /* ignore */ }\n }\n _pendingPrepares.clear();\n\n for (const ctrl of _pendingGenerates.values()) {\n try { ctrl.enqueue({ type: 'error' as const, message }); ctrl.close(); }\n catch { /* ignore */ }\n }\n _pendingGenerates.clear();\n\n // Release every serialization slot so the generate chain doesn't deadlock.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n\n _loadedModelId = null;\n _preparingModelId = null;\n try { _worker?.terminate(); } catch { /* ignore */ }\n _worker = null;\n _releaseWorkerBlob();\n}\n\nfunction _handleWorkerMessage(event: MessageEvent): void {\n const msg = event.data;\n\n switch (msg.type) {\n case 'progress': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n if (msg.status === 'ready') {\n pending.onProgress({ status: 'ready' });\n pending.resolve();\n _pendingPrepares.delete(msg.id);\n } else if (msg.status === 'loading') {\n pending.onProgress({ status: 'loading' });\n } else if (msg.status === 'cached') {\n pending.onProgress({ status: 'cached', file: msg.file, progress: msg.progress });\n } else {\n pending.onProgress({ status: 'downloading', file: msg.file, progress: msg.progress });\n }\n break;\n }\n case 'prepare-error': {\n const pending = _pendingPrepares.get(msg.id);\n if (!pending) break;\n pending.reject(new Error(msg.message));\n _pendingPrepares.delete(msg.id);\n if (_preparingModelId === pending.modelId) _preparingModelId = null;\n break;\n }\n case 'pipeline-ready': {\n _loadedModelId = msg.modelId;\n _preparingModelId = null;\n // Evict models over the cache limit, then refresh the known list.\n void _enforceMaxCachedModels(msg.modelId).then(() => _refreshKnownModels());\n break;\n }\n case 'gen-chunk': {\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: msg.chunkType as 'text' | 'thinking', delta: msg.delta });\n break;\n }\n case 'gen-done': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'done' as const, ...(msg.usage ? { usage: msg.usage } : {}) });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n case 'gen-error': {\n _releaseGenerateSlot(msg.id);\n const ctrl = _pendingGenerates.get(msg.id);\n if (!ctrl) break;\n ctrl.enqueue({ type: 'error' as const, message: msg.message });\n ctrl.close();\n _pendingGenerates.delete(msg.id);\n break;\n }\n }\n}\n\n/**\n * Narrowed so the two members the docs tell you to CALL are not optional.\n *\n * `AparteAIProvider` declares `prepareModel` and `getModelStatus` optional (most\n * providers have nothing to download), and widening to it made both\n * possibly-undefined — so the documented `TransformersProvider.prepareModel(...)`\n * needed a `!` or a guard in every strict consumer. Same technique openai-compat\n * already used for its own always-present members.\n *\n * `chat` joined the list once `AparteAIProvider` became a union: it is optional on\n * the format-adapter arm, and this provider IS its `chat()` — running inference\n * locally is the whole package. Narrowing it here says so once, instead of every\n * caller writing `provider.chat!(...)`.\n */\nexport const TransformersProvider: AparteAIProvider\n & Required<Pick<AparteAIProvider, 'prepareModel' | 'getModelStatus' | 'chat'>> = {\n id: 'transformers',\n\n getMetadata() {\n return {\n id: 'transformers',\n name: 'Transformers.js',\n icon: `<svg viewBox=\"0 0 24 24\" fill=\"none\" xmlns=\"http://www.w3.org/2000/svg\"><path d=\"M12 2L2 7l10 5 10-5-10-5z\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 17l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/><path d=\"M2 12l10 5 10-5\" stroke=\"currentColor\" stroke-width=\"2\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/></svg>`,\n color: '#f59e0b',\n description: 'Run LLMs directly in your browser via WebGPU or WASM — no API, no key',\n hasFreeModels: true,\n isLocal: true,\n helpUrl: 'https://huggingface.co/docs/transformers.js',\n };\n },\n\n getModels(): AparteAIModel[] {\n return _knownModels;\n },\n\n async fetchModels(): Promise<AparteAIModel[]> {\n await _refreshKnownModels();\n return _knownModels;\n },\n\n async chat(request: AparteChatRequest): Promise<AparteChatResponse> {\n const messages = toMessages(request.messages);\n const requestId = uuid();\n const options = {\n maxTokens: request.maxTokens,\n temperature: request.temperature,\n seed: request.seed,\n };\n const task = _registeredModels.get(request.modelId)?.task ?? 'text-generation';\n\n // ── Reserve a serialization slot ─────────────────────────────────────\n // Chain this generate behind the previous one; the worker has a single\n // pipeline, so generates MUST NOT overlap.\n _warnIfContended(request.modelId);\n _queuedModelIds.set(requestId, request.modelId);\n const prevGenerate = _generateChain;\n _generateChain = new Promise<void>((resolveSlot) => {\n _generateDoneResolvers.set(requestId, resolveSlot);\n });\n const postGenerate = (): void => {\n _getWorker().postMessage({\n type: 'generate',\n id: requestId,\n modelId: request.modelId,\n messages,\n options,\n task,\n dtype: _registeredModels.get(request.modelId)?.dtype,\n device: _computeDevice,\n });\n };\n\n if (request.stream === false) {\n return new Promise<string>((resolve, reject) => {\n let result = '';\n const fakeCtrl = {\n enqueue: (chunk: { type: string; delta?: string; message?: string }) => {\n if (chunk.type === 'text') result += chunk.delta ?? '';\n else if (chunk.type === 'done') resolve(result);\n else if (chunk.type === 'error') reject(new Error(chunk.message));\n },\n close: () => { /* no-op */ },\n } as unknown as ReadableStreamDefaultController;\n _pendingGenerates.set(requestId, fakeCtrl);\n void prevGenerate.then(postGenerate);\n });\n }\n\n return new ReadableStream({\n async start(controller) {\n _pendingGenerates.set(requestId, controller);\n await prevGenerate;\n postGenerate();\n },\n cancel() {\n _pendingGenerates.delete(requestId);\n // Actually STOP the model (not just detach the reader): tell the worker\n // to interrupt this generate. The serialization slot is still released\n // by the resulting gen-done/gen-error, so a queued generate can't start\n // before the worker has stopped this one.\n _getWorker().postMessage({ type: 'cancel', id: requestId });\n },\n });\n },\n\n async getModelStatus(modelId: string): Promise<ModelStatus> {\n if (_loadedModelId === modelId) return 'ready';\n if (_preparingModelId === modelId) return 'cached';\n if ('caches' in globalThis) {\n try {\n const encodedId = encodeURIComponent(modelId);\n const names = await caches.keys();\n for (const name of names) {\n const cache = await caches.open(name);\n const keys = await cache.keys();\n if (keys.some(r => r.url.includes(encodedId) || r.url.includes(modelId + '/'))) {\n return 'cached';\n }\n }\n } catch {\n // Cache API unavailable\n }\n }\n return 'not-downloaded';\n },\n\n async prepareModel(modelId: string, onProgress: (p: ModelLoadProgress) => void): Promise<void> {\n if (_loadedModelId === modelId) {\n onProgress({ status: 'ready' });\n return;\n }\n\n const requestId = uuid();\n _preparingModelId = modelId;\n\n const task = _registeredModels.get(modelId)?.task ?? 'text-generation';\n const dtype = _registeredModels.get(modelId)?.dtype;\n return new Promise<void>((resolve, reject) => {\n _pendingPrepares.set(requestId, { modelId, onProgress, resolve, reject });\n _getWorker().postMessage({ type: 'prepare', id: requestId, modelId, task, dtype, device: _computeDevice });\n });\n },\n\n async deleteModel(modelId: string): Promise<void> {\n await deleteCachedModel(modelId);\n },\n};\n\nexport default TransformersProvider;\nexport type { AparteAIProvider, AparteAIModel, ModelStatus, ModelLoadProgress } from '@aparte/core';\n\n// ─────────────────────────────────────────────────────────────────────────────\n// Cache utilities (settings panels, etc.)\n// ─────────────────────────────────────────────────────────────────────────────\n\n/** Returns the modelId currently loaded in the worker's pipeline, or null. */\nexport function getLoadedModelId(): string | null {\n return _loadedModelId;\n}\n\n/** Terminate the shared worker and reset in-memory state. Safe to call any time. */\nexport function terminateWorker(): void {\n _worker?.terminate();\n _worker = null;\n _releaseWorkerBlob();\n _loadedModelId = null;\n _preparingModelId = null;\n for (const [, p] of _pendingPrepares) {\n p.reject(new Error('Worker terminated'));\n }\n _pendingPrepares.clear();\n for (const [, ctrl] of _pendingGenerates) {\n try { ctrl.enqueue({ type: 'error' as const, message: 'Worker terminated' }); ctrl.close(); } catch { /* already closed */ }\n }\n _pendingGenerates.clear();\n\n // Release every serialization slot and reset the chain — the same three lines\n // the worker-error handler above already carried, with the same reason. Without\n // them, terminating mid-generate left `_generateChain` pending on a resolver\n // that had just been dropped, so the NEXT chat() awaited a promise that could\n // never settle: no error, no rejection, the stream simply never started again\n // for the life of the page.\n for (const resolve of _generateDoneResolvers.values()) {\n try { resolve(); } catch { /* ignore */ }\n }\n _generateDoneResolvers.clear();\n _generateChain = Promise.resolve();\n _queuedModelIds.clear();\n // A terminated worker is a fresh situation; let the contention warning speak again.\n _warnedModelContention = false;\n}\n\nexport interface CachedModelEntry {\n modelId: string;\n name: string;\n /** Total size in bytes of all cached files for this model. -1 if unknown. */\n sizeBytes: number;\n /** True if the model is currently loaded in the worker. */\n loaded: boolean;\n}\n\n/**\n * Scan the Cache API to find which Transformers.js models have been downloaded,\n * by matching cache entry URLs against the Hugging Face resolve path.\n */\nexport async function listCachedModels(): Promise<CachedModelEntry[]> {\n if (!('caches' in globalThis)) return [];\n\n const found = new Map<string, { name: string; sizeBytes: number }>();\n\n // e.g. https://huggingface.co/onnx-community/Qwen2.5-0.5B/resolve/main/config.json\n // → onnx-community/Qwen2.5-0.5B\n function extractModelId(url: string): string | null {\n const m = url.match(/huggingface\\.co\\/([^/]+\\/[^/]+)\\/resolve\\//);\n return m ? decodeURIComponent(m[1]!) : null;\n }\n\n function modelName(modelId: string): string {\n const config = _registeredModels.get(modelId);\n if (config) return config.name;\n return (modelId.split('/').pop() ?? modelId).replace(/-/g, ' ');\n }\n\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n for (const req of requests) {\n const modelId = extractModelId(req.url);\n if (!modelId) continue;\n if (!found.has(modelId)) {\n found.set(modelId, { name: modelName(modelId), sizeBytes: 0 });\n }\n const response = await cache.match(req);\n if (!response) continue;\n const contentLength = response.headers.get('content-length');\n if (contentLength) {\n found.get(modelId)!.sizeBytes += parseInt(contentLength, 10);\n } else {\n try {\n const blob = await response.clone().blob();\n found.get(modelId)!.sizeBytes += blob.size;\n } catch { /* skip */ }\n }\n }\n } catch { /* skip inaccessible cache */ }\n }));\n } catch {\n return [];\n }\n\n return Array.from(found.entries()).map(([modelId, { name, sizeBytes }]) => ({\n modelId,\n name,\n sizeBytes,\n loaded: _loadedModelId === modelId,\n }));\n}\n\n/**\n * Delete all cached files for a modelId from the Cache API, terminating the worker\n * first if that model is currently loaded.\n */\nexport async function deleteCachedModel(modelId: string): Promise<void> {\n if (_loadedModelId === modelId || _preparingModelId === modelId) {\n terminateWorker();\n }\n if (!('caches' in globalThis)) return;\n try {\n const cacheNames = await caches.keys();\n await Promise.all(cacheNames.map(async (cacheName) => {\n try {\n const cache = await caches.open(cacheName);\n const requests = await cache.keys();\n const encoded = encodeURIComponent(modelId);\n await Promise.all(\n requests\n .filter(r => r.url.includes(modelId) || r.url.includes(encoded))\n .map(r => cache.delete(r)),\n );\n } catch { /* skip */ }\n }));\n } catch { /* Cache API unavailable */ }\n}\n"],"names":[],"mappings":";;AAkEA,IAAI,iBAAqE;AAMlE,SAAS,sBAAsB,OAA0D;AAC5F,mBAAiB;AACrB;AAEA,eAAsB,iBAA2C;AAG7D,QAAM,QAAiB,UAAmD,gBAAgB;AAG1F,MAAI,SAAS;AACb,MAAI,SAAS,WAAW;AACpB,QAAI;AACA,YAAM,UAAU,MAAO,UAAyE,IAAI,eAAA;AACpG,eAAS,YAAY;AAAA,IACzB,QAAQ;AACJ,eAAS;AAAA,IACb;AAAA,EACJ;AAEA,MAAI;AACJ,MAAI,CAAC,UAAU,QAAQ,GAAG;AACtB,WAAO;AAAA,EACX,WAAW,QAAQ,GAAG;AAClB,WAAO;AAAA,EACX,OAAO;AACH,WAAO;AAAA,EACX;AAEA,QAAM,qBAAqB,iBACpB,eAAe,IAAI,KAAK,eAAe,QAAQ,KAChD;AAEN,SAAO,EAAE,QAAQ,OAAO,MAAM,mBAAA;AAClC;AAsBA,MAAM,wCAAwB,IAAA;AAG9B,IAAI,eAAgC,CAAA;AAK7B,SAAS,cAAc,QAAuC;AACjE,oBAAkB,IAAI,OAAO,IAAI,MAAM;AACvC,MAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,OAAO,EAAE,GAAG;AAC7C,mBAAe,CAAC,GAAG,cAAc;AAAA,MAC7B,IAAI,OAAO;AAAA,MACX,MAAM,OAAO;AAAA,MACb,aAAa,OAAO;AAAA,MACpB,cAAc,OAAO;AAAA,IAAA,CACxB;AAAA,EACL;AACJ;AAGA,SAAS,qBAAqB,SAAgC;AAC1D,QAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,MAAI,OAAQ,QAAO,EAAE,IAAI,OAAO,IAAI,MAAM,OAAO,MAAM,aAAa,OAAO,aAAa,cAAc,OAAO,aAAA;AAC7G,QAAM,QAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AACpE,SAAO,EAAE,IAAI,SAAS,MAAM,cAAc,CAAC,WAAW,EAAA;AAC1D;AAGA,IAAI,mBAAmB;AAMhB,SAAS,mBAAmB,KAAmB;AAClD,qBAAmB;AACvB;AAGO,SAAS,qBAA6B;AACzC,SAAO;AACX;AASA,IAAI,iBAAgC;AAE7B,SAAS,iBAAiB,GAAwB;AACrD,mBAAiB;AACrB;AAEO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGA,eAAe,wBAAwB,aAAoC;AACvE,MAAI,qBAAqB,EAAG;AAC5B,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,UAAM,SAAS,OAAO,OAAO,CAAA,MAAK,EAAE,YAAY,WAAW;AAC3D,UAAM,SAAS,OAAO,SAAS;AAC/B,QAAI,UAAU,EAAG;AAEjB,aAAS,IAAI,GAAG,IAAI,UAAU,IAAI,OAAO,QAAQ,KAAK;AAClD,YAAM,kBAAkB,OAAO,CAAC,EAAG,OAAO;AAAA,IAC9C;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,eAAe,sBAAqC;AAChD,MAAI;AACA,UAAM,SAAS,MAAM,iBAAA;AACrB,eAAW,SAAS,QAAQ;AACxB,UAAI,CAAC,aAAa,KAAK,CAAA,MAAK,EAAE,OAAO,MAAM,OAAO,GAAG;AACjD,uBAAe,CAAC,GAAG,cAAc,qBAAqB,MAAM,OAAO,CAAC;AAAA,MACxE;AAAA,IACJ;AAAA,EACJ,QAAQ;AAAA,EAA0B;AACtC;AAGA,IAAI,0BAA0B;AAG9B,SAAS,WAAW,UAAgD;AAChE,QAAM,SAA0B,CAAA;AAChC,MAAI,mBAAmB;AACvB,aAAW,KAAK,UAAU;AACtB,QAAI,EAAE,SAAS,UAAU,EAAE,SAAS,eAAe,EAAE,SAAS,UAAU;AACpE,YAAM,OAAO,cAAc,EAAE,OAAO;AACpC,UAAI,aAAa,KAAK,EAAE,MAAM,EAAE,MAAM,SAAS,MAAM;AAAA,IACzD,OAAO;AAKH;AAAA,IACJ;AAAA,EACJ;AACA,MAAI,mBAAmB,KAAK,CAAC,yBAAyB;AAClD,8BAA0B;AAC1B,YAAQ;AAAA,MACJ,0BAA0B,gBAAgB;AAAA,IAAA;AAAA,EAIlD;AACA,SAAO;AACX;AAMA,IAAI,UAAyB;AAQ7B,MAAM,uCAAuB,IAAA;AAC7B,MAAM,wCAAwB,IAAA;AAM9B,IAAI,iBAAgC,QAAQ,QAAA;AAC5C,MAAM,6CAA6B,IAAA;AAOnC,MAAM,sCAAsB,IAAA;AAC5B,IAAI,yBAAyB;AAG7B,SAAS,mBAAmB,WAAuC;AAC/D,aAAW,MAAM,gBAAgB,OAAA,EAAU,KAAI,OAAO,UAAW,QAAO;AACxE,SAAO;AACX;AAOA,SAAS,iBAAiB,WAAyB;AAC/C,MAAI,uBAAwB;AAC5B,QAAM,QAAQ,mBAAmB,SAAS;AAC1C,MAAI,CAAC,MAAO;AACZ,2BAAyB;AACzB,UAAQ;AAAA,IACJ,mEAAmE,SAAS,aACtE,KAAK,gHACiC,gBAAgB;AAAA,EAAA;AAIpE;AAGA,SAAS,qBAAqB,IAAkB;AAC5C,kBAAgB,OAAO,EAAE;AACzB,QAAM,UAAU,uBAAuB,IAAI,EAAE;AAC7C,MAAI,SAAS;AACT,2BAAuB,OAAO,EAAE;AAChC,YAAA;AAAA,EACJ;AACJ;AAGA,IAAI,iBAAgC;AAEpC,IAAI,oBAAmC;AAGvC,IAAI,iBAAgC;AAmBpC,SAAS,eAAuB;AAC5B,QAAM,MAAM,IAAI,IAAI,WAAW,YAAY,GAAG;AAC9C,QAAM,aAAa,OAAO,aAAa,eAAe,IAAI,WAAW,SAAS;AAM9E,QAAM,cAAc,OAAO,SAAS,cAAc,OAAO,IAAI,oBAAoB;AAOjF,MAAI,cAAc,CAAC,YAAa,QAAO,IAAI,OAAO,IAAA;AAAA;AAAA,IAAA,KAAA,IAAA,IAAA,6BAAA,YAAA,GAAA,EAAA;AAAA,IAAA,YAAA;AAAA,EAAA,GAAyC,EAAE,MAAM,UAAU;AAE7G,mBAAiB,IAAI;AAAA,IACjB,IAAI,KAAK,CAAC,UAAU,KAAK,UAAU,IAAI,IAAI,CAAC,GAAG,GAAG,EAAE,MAAM,mBAAmB;AAAA,EAAA;AAEjF,MAAI;AACA,WAAO,IAAI,OAAO,gBAAgB,EAAE,MAAM,UAAU;AAAA,EACxD,SAAS,OAAO;AAKZ,QAAI,gBAAgB,cAAc;AAClC,qBAAiB;AACjB,UAAM,IAAI;AAAA,MACN,gDAAgD,IAAI,MAAM,oQAGzB,OAAO,KAAK,CAAC;AAAA,IAAA;AAAA,EAEtD;AACJ;AAEA,SAAS,qBAA2B;AAChC,MAAI,gBAAgB;AAChB,QAAI,gBAAgB,cAAc;AAClC,qBAAiB;AAAA,EACrB;AACJ;AAgBA,SAAS,iBAAqC;AAC1C,QAAM,UAAW,YAAuE;AACxF,MAAI,OAAO,YAAY,YAAY;AAC/B,QAAI;AACA,YAAM,OAAO,QAAQ,2BAA2B;AAChD,UAAI,QAAQ,YAAY,KAAK,IAAI,EAAG,QAAO;AAAA,IAC/C,QAAQ;AAAA,IAAsC;AAAA,EAClD;AAEA,MAAI;AACA,UAAM,KAAK,SAAS,cAAc,0BAA0B;AAC5D,UAAM,MAAM,IAAI,cAAc,KAAK,MAAM,GAAG,WAAW,IAA4C;AACnG,UAAM,OAAO,KAAK,UAAU,2BAA2B;AACvD,QAAI,KAAM,QAAO,IAAI,IAAI,MAAM,SAAS,IAAI,EAAE;AAAA,EAClD,QAAQ;AAAA,EAA+C;AACvD,SAAO;AACX;AAEA,SAAS,aAAqB;AAC1B,MAAI,CAAC,SAAS;AACV,cAAU,aAAA;AACV,YAAQ,iBAAiB,WAAW,oBAAoB;AACxD,YAAQ,iBAAiB,SAAS,kBAAkB;AACpD,YAAQ,iBAAiB,gBAAgB,kBAAkB;AAG3D,YAAQ,YAAY,EAAE,MAAM,QAAQ,iBAAiB,eAAA,GAAkB;AAAA,EAC3E;AACA,SAAO;AACX;AAOA,SAAS,mBAAmB,GAAgB;AACxC,QAAM,UAAW,GAAkB,WAAW;AAE9C,aAAW,KAAK,iBAAiB,UAAU;AACvC,QAAI;AAAE,QAAE,OAAO,IAAI,MAAM,OAAO,CAAC;AAAA,IAAG,QAAQ;AAAA,IAAe;AAAA,EAC/D;AACA,mBAAiB,MAAA;AAEjB,aAAW,QAAQ,kBAAkB,UAAU;AAC3C,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS;AAAG,WAAK,MAAA;AAAA,IAAS,QACjE;AAAA,IAAe;AAAA,EACzB;AACA,oBAAkB,MAAA;AAGlB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,mBAAiB;AACjB,sBAAoB;AACpB,MAAI;AAAE,aAAS,UAAA;AAAA,EAAa,QAAQ;AAAA,EAAe;AACnD,YAAU;AACV,qBAAA;AACJ;AAEA,SAAS,qBAAqB,OAA2B;AACrD,QAAM,MAAM,MAAM;AAElB,UAAQ,IAAI,MAAA;AAAA,IACR,KAAK,YAAY;AACb,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,UAAI,IAAI,WAAW,SAAS;AACxB,gBAAQ,WAAW,EAAE,QAAQ,QAAA,CAAS;AACtC,gBAAQ,QAAA;AACR,yBAAiB,OAAO,IAAI,EAAE;AAAA,MAClC,WAAW,IAAI,WAAW,WAAW;AACjC,gBAAQ,WAAW,EAAE,QAAQ,UAAA,CAAW;AAAA,MAC5C,WAAW,IAAI,WAAW,UAAU;AAChC,gBAAQ,WAAW,EAAE,QAAQ,UAAU,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACnF,OAAO;AACH,gBAAQ,WAAW,EAAE,QAAQ,eAAe,MAAM,IAAI,MAAM,UAAU,IAAI,SAAA,CAAU;AAAA,MACxF;AACA;AAAA,IACJ;AAAA,IACA,KAAK,iBAAiB;AAClB,YAAM,UAAU,iBAAiB,IAAI,IAAI,EAAE;AAC3C,UAAI,CAAC,QAAS;AACd,cAAQ,OAAO,IAAI,MAAM,IAAI,OAAO,CAAC;AACrC,uBAAiB,OAAO,IAAI,EAAE;AAC9B,UAAI,sBAAsB,QAAQ,QAAS,qBAAoB;AAC/D;AAAA,IACJ;AAAA,IACA,KAAK,kBAAkB;AACnB,uBAAiB,IAAI;AACrB,0BAAoB;AAEpB,WAAK,wBAAwB,IAAI,OAAO,EAAE,KAAK,MAAM,qBAAqB;AAC1E;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,IAAI,WAAkC,OAAO,IAAI,OAAO;AAC7E;AAAA,IACJ;AAAA,IACA,KAAK,YAAY;AACb,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,QAAiB,GAAI,IAAI,QAAQ,EAAE,OAAO,IAAI,MAAA,IAAU,CAAA,GAAK;AAClF,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,IACA,KAAK,aAAa;AACd,2BAAqB,IAAI,EAAE;AAC3B,YAAM,OAAO,kBAAkB,IAAI,IAAI,EAAE;AACzC,UAAI,CAAC,KAAM;AACX,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,IAAI,SAAS;AAC7D,WAAK,MAAA;AACL,wBAAkB,OAAO,IAAI,EAAE;AAC/B;AAAA,IACJ;AAAA,EAAA;AAER;AAgBO,MAAM,uBACwE;AAAA,EACjF,IAAI;AAAA,EAEJ,cAAc;AACV,WAAO;AAAA,MACH,IAAI;AAAA,MACJ,MAAM;AAAA,MACN,MAAM;AAAA,MACN,OAAO;AAAA,MACP,aAAa;AAAA,MACb,eAAe;AAAA,MACf,SAAS;AAAA,MACT,SAAS;AAAA,IAAA;AAAA,EAEjB;AAAA,EAEA,YAA6B;AACzB,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,cAAwC;AAC1C,UAAM,oBAAA;AACN,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,KAAK,SAAyD;AAChE,UAAM,WAAW,WAAW,QAAQ,QAAQ;AAC5C,UAAM,YAAY,KAAA;AAClB,UAAM,UAAU;AAAA,MACZ,WAAW,QAAQ;AAAA,MACnB,aAAa,QAAQ;AAAA,MACrB,MAAM,QAAQ;AAAA,IAAA;AAElB,UAAM,OAAO,kBAAkB,IAAI,QAAQ,OAAO,GAAG,QAAQ;AAK7D,qBAAiB,QAAQ,OAAO;AAChC,oBAAgB,IAAI,WAAW,QAAQ,OAAO;AAC9C,UAAM,eAAe;AACrB,qBAAiB,IAAI,QAAc,CAAC,gBAAgB;AAChD,6BAAuB,IAAI,WAAW,WAAW;AAAA,IACrD,CAAC;AACD,UAAM,eAAe,MAAY;AAC7B,iBAAA,EAAa,YAAY;AAAA,QACrB,MAAM;AAAA,QACN,IAAI;AAAA,QACJ,SAAS,QAAQ;AAAA,QACjB;AAAA,QACA;AAAA,QACA;AAAA,QACA,OAAO,kBAAkB,IAAI,QAAQ,OAAO,GAAG;AAAA,QAC/C,QAAQ;AAAA,MAAA,CACX;AAAA,IACL;AAEA,QAAI,QAAQ,WAAW,OAAO;AAC1B,aAAO,IAAI,QAAgB,CAAC,SAAS,WAAW;AAC5C,YAAI,SAAS;AACb,cAAM,WAAW;AAAA,UACb,SAAS,CAAC,UAA8D;AACpE,gBAAI,MAAM,SAAS,OAAQ,WAAU,MAAM,SAAS;AAAA,qBAC3C,MAAM,SAAS,OAAQ,SAAQ,MAAM;AAAA,qBACrC,MAAM,SAAS,QAAS,QAAO,IAAI,MAAM,MAAM,OAAO,CAAC;AAAA,UACpE;AAAA,UACA,OAAO,MAAM;AAAA,UAAc;AAAA,QAAA;AAE/B,0BAAkB,IAAI,WAAW,QAAQ;AACzC,aAAK,aAAa,KAAK,YAAY;AAAA,MACvC,CAAC;AAAA,IACL;AAEA,WAAO,IAAI,eAAe;AAAA,MACtB,MAAM,MAAM,YAAY;AACpB,0BAAkB,IAAI,WAAW,UAAU;AAC3C,cAAM;AACN,qBAAA;AAAA,MACJ;AAAA,MACA,SAAS;AACL,0BAAkB,OAAO,SAAS;AAKlC,mBAAA,EAAa,YAAY,EAAE,MAAM,UAAU,IAAI,WAAW;AAAA,MAC9D;AAAA,IAAA,CACH;AAAA,EACL;AAAA,EAEA,MAAM,eAAe,SAAuC;AACxD,QAAI,mBAAmB,QAAS,QAAO;AACvC,QAAI,sBAAsB,QAAS,QAAO;AAC1C,QAAI,YAAY,YAAY;AACxB,UAAI;AACA,cAAM,YAAY,mBAAmB,OAAO;AAC5C,cAAM,QAAQ,MAAM,OAAO,KAAA;AAC3B,mBAAW,QAAQ,OAAO;AACtB,gBAAM,QAAQ,MAAM,OAAO,KAAK,IAAI;AACpC,gBAAM,OAAO,MAAM,MAAM,KAAA;AACzB,cAAI,KAAK,KAAK,CAAA,MAAK,EAAE,IAAI,SAAS,SAAS,KAAK,EAAE,IAAI,SAAS,UAAU,GAAG,CAAC,GAAG;AAC5E,mBAAO;AAAA,UACX;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAER;AAAA,IACJ;AACA,WAAO;AAAA,EACX;AAAA,EAEA,MAAM,aAAa,SAAiB,YAA2D;AAC3F,QAAI,mBAAmB,SAAS;AAC5B,iBAAW,EAAE,QAAQ,SAAS;AAC9B;AAAA,IACJ;AAEA,UAAM,YAAY,KAAA;AAClB,wBAAoB;AAEpB,UAAM,OAAO,kBAAkB,IAAI,OAAO,GAAG,QAAQ;AACrD,UAAM,QAAQ,kBAAkB,IAAI,OAAO,GAAG;AAC9C,WAAO,IAAI,QAAc,CAAC,SAAS,WAAW;AAC1C,uBAAiB,IAAI,WAAW,EAAE,SAAS,YAAY,SAAS,QAAQ;AACxE,iBAAA,EAAa,YAAY,EAAE,MAAM,WAAW,IAAI,WAAW,SAAS,MAAM,OAAO,QAAQ,eAAA,CAAgB;AAAA,IAC7G,CAAC;AAAA,EACL;AAAA,EAEA,MAAM,YAAY,SAAgC;AAC9C,UAAM,kBAAkB,OAAO;AAAA,EACnC;AACJ;AAUO,SAAS,mBAAkC;AAC9C,SAAO;AACX;AAGO,SAAS,kBAAwB;AACpC,WAAS,UAAA;AACT,YAAU;AACV,qBAAA;AACA,mBAAiB;AACjB,sBAAoB;AACpB,aAAW,CAAA,EAAG,CAAC,KAAK,kBAAkB;AAClC,MAAE,OAAO,IAAI,MAAM,mBAAmB,CAAC;AAAA,EAC3C;AACA,mBAAiB,MAAA;AACjB,aAAW,CAAA,EAAG,IAAI,KAAK,mBAAmB;AACtC,QAAI;AAAE,WAAK,QAAQ,EAAE,MAAM,SAAkB,SAAS,qBAAqB;AAAG,WAAK,MAAA;AAAA,IAAS,QAAQ;AAAA,IAAuB;AAAA,EAC/H;AACA,oBAAkB,MAAA;AAQlB,aAAW,WAAW,uBAAuB,UAAU;AACnD,QAAI;AAAE,cAAA;AAAA,IAAW,QAAQ;AAAA,IAAe;AAAA,EAC5C;AACA,yBAAuB,MAAA;AACvB,mBAAiB,QAAQ,QAAA;AACzB,kBAAgB,MAAA;AAEhB,2BAAyB;AAC7B;AAeA,eAAsB,mBAAgD;AAClE,MAAI,EAAE,YAAY,YAAa,QAAO,CAAA;AAEtC,QAAM,4BAAY,IAAA;AAIlB,WAAS,eAAe,KAA4B;AAChD,UAAM,IAAI,IAAI,MAAM,4CAA4C;AAChE,WAAO,IAAI,mBAAmB,EAAE,CAAC,CAAE,IAAI;AAAA,EAC3C;AAEA,WAAS,UAAU,SAAyB;AACxC,UAAM,SAAS,kBAAkB,IAAI,OAAO;AAC5C,QAAI,eAAe,OAAO;AAC1B,YAAQ,QAAQ,MAAM,GAAG,EAAE,SAAS,SAAS,QAAQ,MAAM,GAAG;AAAA,EAClE;AAEA,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,mBAAW,OAAO,UAAU;AACxB,gBAAM,UAAU,eAAe,IAAI,GAAG;AACtC,cAAI,CAAC,QAAS;AACd,cAAI,CAAC,MAAM,IAAI,OAAO,GAAG;AACrB,kBAAM,IAAI,SAAS,EAAE,MAAM,UAAU,OAAO,GAAG,WAAW,GAAG;AAAA,UACjE;AACA,gBAAM,WAAW,MAAM,MAAM,MAAM,GAAG;AACtC,cAAI,CAAC,SAAU;AACf,gBAAM,gBAAgB,SAAS,QAAQ,IAAI,gBAAgB;AAC3D,cAAI,eAAe;AACf,kBAAM,IAAI,OAAO,EAAG,aAAa,SAAS,eAAe,EAAE;AAAA,UAC/D,OAAO;AACH,gBAAI;AACA,oBAAM,OAAO,MAAM,SAAS,MAAA,EAAQ,KAAA;AACpC,oBAAM,IAAI,OAAO,EAAG,aAAa,KAAK;AAAA,YAC1C,QAAQ;AAAA,YAAa;AAAA,UACzB;AAAA,QACJ;AAAA,MACJ,QAAQ;AAAA,MAAgC;AAAA,IAC5C,CAAC,CAAC;AAAA,EACN,QAAQ;AACJ,WAAO,CAAA;AAAA,EACX;AAEA,SAAO,MAAM,KAAK,MAAM,QAAA,CAAS,EAAE,IAAI,CAAC,CAAC,SAAS,EAAE,MAAM,UAAA,CAAW,OAAO;AAAA,IACxE;AAAA,IACA;AAAA,IACA;AAAA,IACA,QAAQ,mBAAmB;AAAA,EAAA,EAC7B;AACN;AAMA,eAAsB,kBAAkB,SAAgC;AACpE,MAAI,mBAAmB,WAAW,sBAAsB,SAAS;AAC7D,oBAAA;AAAA,EACJ;AACA,MAAI,EAAE,YAAY,YAAa;AAC/B,MAAI;AACA,UAAM,aAAa,MAAM,OAAO,KAAA;AAChC,UAAM,QAAQ,IAAI,WAAW,IAAI,OAAO,cAAc;AAClD,UAAI;AACA,cAAM,QAAQ,MAAM,OAAO,KAAK,SAAS;AACzC,cAAM,WAAW,MAAM,MAAM,KAAA;AAC7B,cAAM,UAAU,mBAAmB,OAAO;AAC1C,cAAM,QAAQ;AAAA,UACV,SACK,OAAO,CAAA,MAAK,EAAE,IAAI,SAAS,OAAO,KAAK,EAAE,IAAI,SAAS,OAAO,CAAC,EAC9D,IAAI,OAAK,MAAM,OAAO,CAAC,CAAC;AAAA,QAAA;AAAA,MAErC,QAAQ;AAAA,MAAa;AAAA,IACzB,CAAC,CAAC;AAAA,EACN,QAAQ;AAAA,EAA8B;AAC1C;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aparte/provider-transformers",
3
- "version": "0.16.0",
3
+ "version": "0.16.1",
4
4
  "description": "Run LLMs 100% in the browser via Transformers.js (WebGPU/WASM) — a local, keyless AI provider for aparté. Streams tokens off the main thread in a Web Worker.",
5
5
  "type": "module",
6
6
  "sideEffects": false,
@@ -24,7 +24,7 @@
24
24
  "node": ">=18"
25
25
  },
26
26
  "peerDependencies": {
27
- "@aparte/core": ">=0.16.0 <1.0.0",
27
+ "@aparte/core": ">=0.16.1 <1.0.0",
28
28
  "@huggingface/transformers": "^4.2.0"
29
29
  },
30
30
  "devDependencies": {
@@ -32,7 +32,7 @@
32
32
  "@types/node": "^22.0.0",
33
33
  "typescript": "^5.4.0",
34
34
  "vite": "^6.0.0",
35
- "@aparte/core": "0.16.0"
35
+ "@aparte/core": "0.16.1"
36
36
  },
37
37
  "keywords": [
38
38
  "ai",
@@ -1 +0,0 @@
1
- {"version":3,"file":"worker-Bk-8pt3W.js","sources":["../src/worker.ts"],"sourcesContent":["/**\n * Generic Transformers.js inference worker.\n *\n * Runs entirely off the main thread. It holds ONE text-generation pipeline at a\n * time and speaks a tiny postMessage protocol with the provider on the main\n * thread (see `index.ts`):\n *\n * main → worker : { type: 'prepare', id, modelId, dtype?, device? }\n * { type: 'generate', id, modelId, messages, options, dtype?, device? }\n * worker → main : { type: 'progress', id, status, file?, progress? }\n * { type: 'prepare-error', id, message }\n * { type: 'pipeline-ready', modelId }\n * { type: 'gen-chunk', id, chunkType: 'text', delta }\n * { type: 'gen-done', id }\n * { type: 'gen-error', id, message }\n *\n * Deliberately generic: no vision, no low-level ORT session management, no\n * model-family specifics — just the high-level `pipeline()` + `TextStreamer`.\n */\n\nimport { pipeline, TextStreamer, InterruptableStoppingCriteria, env, type TextGenerationPipeline } from '@huggingface/transformers';\n\n// Fetch weights from the Hugging Face hub (not local paths) and cache them in the\n// browser Cache API — this is what `listCachedModels()` scans on the main thread.\nenv.allowLocalModels = false;\nenv.useBrowserCache = true;\n\n// DOM's `Worker` interface types `postMessage` + typed `addEventListener('message')`,\n// which is enough for the worker scope — avoids pulling the WebWorker lib (it clashes\n// with DOM's global `postMessage`).\nconst ctx = self as unknown as Worker;\n\ntype Dtype = string | Record<string, string>;\ntype Device = 'webgpu' | 'wasm' | 'auto';\ninterface GenOptions { maxTokens?: number; temperature?: number; seed?: number }\ntype SimpleMessage = { role: 'user' | 'assistant' | 'system'; content: string };\n\ntype InMessage =\n | { type: 'prepare'; id: string; modelId: string; dtype?: Dtype; device?: Device }\n | { type: 'generate'; id: string; modelId: string; messages: SimpleMessage[]; options: GenOptions; dtype?: Dtype; device?: Device }\n | { type: 'cancel'; id: string };\n\nfunction post(message: unknown): void {\n ctx.postMessage(message);\n}\n\nlet _current: { modelId: string; pipe: TextGenerationPipeline } | null = null;\n// Per-generate interrupts, so a consumer's stream-cancel actually STOPS the model\n// (not just detaches the reader) — otherwise generation runs to max_new_tokens\n// off-thread, wasting exactly the CPU/GPU/battery this provider exists to save.\nconst _activeStops = new Map<string, InterruptableStoppingCriteria>();\n\n/**\n * Ensure the pipeline for `modelId` is loaded, reusing the current one when it\n * matches. On a fresh load it forwards download progress (when `id` is given) and\n * announces `pipeline-ready`.\n */\nasync function ensurePipeline(modelId: string, dtype: Dtype | undefined, device: Device | undefined, id?: string): Promise<TextGenerationPipeline> {\n if (_current?.modelId === modelId) return _current.pipe;\n\n // eslint-disable-next-line @typescript-eslint/no-explicit-any\n const opts: Record<string, any> = {\n progress_callback: (p: { status?: string; file?: string; progress?: number }) => {\n if (!id) return;\n if (p.status === 'progress') {\n post({ type: 'progress', id, status: 'downloading', file: p.file, progress: Math.round(p.progress ?? 0) });\n } else if (p.status === 'done') {\n post({ type: 'progress', id, status: 'loading', file: p.file });\n }\n },\n };\n if (dtype) opts['dtype'] = dtype;\n if (device && device !== 'auto') opts['device'] = device;\n\n const pipe = await pipeline('text-generation', modelId, opts) as TextGenerationPipeline;\n _current = { modelId, pipe };\n post({ type: 'pipeline-ready', modelId });\n return pipe;\n}\n\nasync function handlePrepare(msg: Extract<InMessage, { type: 'prepare' }>): Promise<void> {\n try {\n await ensurePipeline(msg.modelId, msg.dtype, msg.device, msg.id);\n post({ type: 'progress', id: msg.id, status: 'ready' });\n } catch (err) {\n post({ type: 'prepare-error', id: msg.id, message: (err as Error)?.message ?? 'Failed to load model' });\n }\n}\n\nasync function handleGenerate(msg: Extract<InMessage, { type: 'generate' }>): Promise<void> {\n const stoppingCriteria = new InterruptableStoppingCriteria();\n _activeStops.set(msg.id, stoppingCriteria);\n try {\n const pipe = await ensurePipeline(msg.modelId, msg.dtype, msg.device, msg.id);\n\n const streamer = new TextStreamer(pipe.tokenizer, {\n skip_prompt: true,\n skip_special_tokens: true,\n callback_function: (text: string) => {\n if (text) post({ type: 'gen-chunk', id: msg.id, chunkType: 'text', delta: text });\n },\n });\n\n const temperature = msg.options.temperature ?? 0;\n await pipe(msg.messages, {\n max_new_tokens: msg.options.maxTokens ?? 512,\n do_sample: temperature > 0,\n temperature: temperature > 0 ? temperature : undefined,\n streamer,\n stopping_criteria: stoppingCriteria,\n });\n\n post({ type: 'gen-done', id: msg.id });\n } catch (err) {\n post({ type: 'gen-error', id: msg.id, message: (err as Error)?.message ?? 'Generation failed' });\n } finally {\n _activeStops.delete(msg.id);\n }\n}\n\nctx.addEventListener('message', (event: MessageEvent<InMessage>) => {\n const msg = event.data;\n if (msg.type === 'prepare') void handlePrepare(msg);\n else if (msg.type === 'generate') void handleGenerate(msg);\n else if (msg.type === 'cancel') _activeStops.get(msg.id)?.interrupt();\n});\n"],"names":[],"mappings":";AAwBA,IAAI,mBAAmB;AACvB,IAAI,kBAAkB;AAKtB,MAAM,MAAM;AAYZ,SAAS,KAAK,SAAwB;AAClC,MAAI,YAAY,OAAO;AAC3B;AAEA,IAAI,WAAqE;AAIzE,MAAM,mCAAmB,IAAA;AAOzB,eAAe,eAAe,SAAiB,OAA0B,QAA4B,IAA8C;AAC/I,MAAI,UAAU,YAAY,QAAS,QAAO,SAAS;AAGnD,QAAM,OAA4B;AAAA,IAC9B,mBAAmB,CAAC,MAA6D;AAC7E,UAAI,CAAC,GAAI;AACT,UAAI,EAAE,WAAW,YAAY;AACzB,aAAK,EAAE,MAAM,YAAY,IAAI,QAAQ,eAAe,MAAM,EAAE,MAAM,UAAU,KAAK,MAAM,EAAE,YAAY,CAAC,GAAG;AAAA,MAC7G,WAAW,EAAE,WAAW,QAAQ;AAC5B,aAAK,EAAE,MAAM,YAAY,IAAI,QAAQ,WAAW,MAAM,EAAE,MAAM;AAAA,MAClE;AAAA,IACJ;AAAA,EAAA;AAEJ,MAAI,MAAO,MAAK,OAAO,IAAI;AAC3B,MAAI,UAAU,WAAW,OAAQ,MAAK,QAAQ,IAAI;AAElD,QAAM,OAAO,MAAM,SAAS,mBAAmB,SAAS,IAAI;AAC5D,aAAW,EAAE,SAAS,KAAA;AACtB,OAAK,EAAE,MAAM,kBAAkB,QAAA,CAAS;AACxC,SAAO;AACX;AAEA,eAAe,cAAc,KAA6D;AACtF,MAAI;AACA,UAAM,eAAe,IAAI,SAAS,IAAI,OAAO,IAAI,QAAQ,IAAI,EAAE;AAC/D,SAAK,EAAE,MAAM,YAAY,IAAI,IAAI,IAAI,QAAQ,SAAS;AAAA,EAC1D,SAAS,KAAK;AACV,SAAK,EAAE,MAAM,iBAAiB,IAAI,IAAI,IAAI,SAAU,KAAe,WAAW,uBAAA,CAAwB;AAAA,EAC1G;AACJ;AAEA,eAAe,eAAe,KAA8D;AACxF,QAAM,mBAAmB,IAAI,8BAAA;AAC7B,eAAa,IAAI,IAAI,IAAI,gBAAgB;AACzC,MAAI;AACA,UAAM,OAAO,MAAM,eAAe,IAAI,SAAS,IAAI,OAAO,IAAI,QAAQ,IAAI,EAAE;AAE5E,UAAM,WAAW,IAAI,aAAa,KAAK,WAAW;AAAA,MAC9C,aAAa;AAAA,MACb,qBAAqB;AAAA,MACrB,mBAAmB,CAAC,SAAiB;AACjC,YAAI,KAAM,MAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,WAAW,QAAQ,OAAO,KAAA,CAAM;AAAA,MACpF;AAAA,IAAA,CACH;AAED,UAAM,cAAc,IAAI,QAAQ,eAAe;AAC/C,UAAM,KAAK,IAAI,UAAU;AAAA,MACrB,gBAAgB,IAAI,QAAQ,aAAa;AAAA,MACzC,WAAW,cAAc;AAAA,MACzB,aAAa,cAAc,IAAI,cAAc;AAAA,MAC7C;AAAA,MACA,mBAAmB;AAAA,IAAA,CACtB;AAED,SAAK,EAAE,MAAM,YAAY,IAAI,IAAI,IAAI;AAAA,EACzC,SAAS,KAAK;AACV,SAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,SAAU,KAAe,WAAW,oBAAA,CAAqB;AAAA,EACnG,UAAA;AACI,iBAAa,OAAO,IAAI,EAAE;AAAA,EAC9B;AACJ;AAEA,IAAI,iBAAiB,WAAW,CAAC,UAAmC;AAChE,QAAM,MAAM,MAAM;AAClB,MAAI,IAAI,SAAS,UAAW,MAAK,cAAc,GAAG;AAAA,WACzC,IAAI,SAAS,WAAY,MAAK,eAAe,GAAG;AAAA,WAChD,IAAI,SAAS,SAAU,cAAa,IAAI,IAAI,EAAE,GAAG,UAAA;AAC9D,CAAC;"}