@aparte/provider-transformers 0.16.2 → 0.16.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -30,6 +30,84 @@ The provider owns its I/O (it runs inference locally), so `AparteDirectTransport
30
30
  Model weights download once and persist in the Cache API; `prepareModel` reports progress, and
31
31
  `listCachedModels` / `deleteCachedModel` manage the on-disk cache.
32
32
 
33
+ ## Runners — what loads the model
34
+
35
+ The worker runs a **runner**: the piece that loads a model and drives it. Two ship with the
36
+ package, picked with `task`; a third option is a module of your own.
37
+
38
+ | `task` | loads | for |
39
+ |---|---|---|
40
+ | `'text-generation'` (default) | `pipeline('text-generation')` | any chat model — Qwen, SmolLM, Llama, Phi… |
41
+ | `'image-text-to-text'` | `AutoProcessor` + `AutoModelForImageTextToText` | vision models — whatever `AutoModelForImageTextToText` resolves (SmolVLM, Qwen2-VL, LFM2-VL, Gemma 3…); measured on SmolVLM |
42
+
43
+ A vision model reads the image parts of a message (`{ type: 'image', image: dataUrl }`) exactly
44
+ as the composer attaches them:
45
+
46
+ ```ts
47
+ registerModel({
48
+ id: 'HuggingFaceTB/SmolVLM-256M-Instruct',
49
+ name: 'SmolVLM 256M',
50
+ task: 'image-text-to-text',
51
+ capabilities: ['streaming', 'vision'],
52
+ // Three ONNX parts, three dtypes — the shape the model card recommends for WebGPU.
53
+ dtype: { embed_tokens: 'fp16', vision_encoder: 'q4', decoder_model_merged: 'q4' },
54
+ });
55
+ ```
56
+
57
+ Measured: SmolVLM-256M on WebGPU (Chromium, AMD Radeon 8060S): first load 7 s (download included), first token 3.7 s cold; Stop interrupts the model, not just the read.
58
+
59
+ Each runner is its own chunk, loaded only when a model asks for it. Both drop `tool_call` /
60
+ `tool_result` turns with one warning (tool syntax is per model family), and the text runner
61
+ **says so when it drops an image** — it never answers a photo it could not see as if it had.
62
+
63
+ ### A runner of your own
64
+
65
+ Point `runner` at an ES module that exports `createRunner`; it wins over `task`. The worker
66
+ imports it and hands it the same Transformers.js instance the built-ins use, so a research-grade
67
+ model — a custom vision tower, an adapter you swap at runtime — fits without giving up the
68
+ provider's worker, queue, progress, cancel and cache:
69
+
70
+ ```ts
71
+ registerModel({ id: 'my-org/my-model', name: 'Mine', capabilities: ['streaming'], runner: './runners/mine.js' });
72
+ ```
73
+
74
+ ```js
75
+ // runners/mine.js — served by your app; the worker imports it by URL
76
+ export async function createRunner(ctx) {
77
+ // ctx.transformers is the Transformers.js the page installed or mapped — never import your own
78
+ const { AutoTokenizer, AutoModelForCausalLM, TextStreamer } = ctx.transformers;
79
+ const tokenizer = await AutoTokenizer.from_pretrained(ctx.modelId);
80
+ const model = await AutoModelForCausalLM.from_pretrained(ctx.modelId, { dtype: ctx.dtype, device: ctx.device });
81
+ return {
82
+ async generate({ messages, options, emit }) {
83
+ // messages arrive WITH their content parts — render them the way this model wants;
84
+ // this one reads text, so the parts are flattened and tool turns left out
85
+ const turns = messages
86
+ .filter((m) => m.role === 'user' || m.role === 'assistant' || m.role === 'system')
87
+ .map((m) => ({ role: m.role, content: typeof m.content === 'string' ? m.content : m.content.filter((p) => p.type === 'text').map((p) => p.text).join('') }));
88
+ const inputs = tokenizer.apply_chat_template(turns, { add_generation_prompt: true, return_dict: true });
89
+ const streamer = new TextStreamer(tokenizer, {
90
+ skip_prompt: true, skip_special_tokens: true,
91
+ callback_function: (text) => emit({ type: 'text', delta: text }),
92
+ });
93
+ await model.generate({ ...inputs, max_new_tokens: options.maxTokens ?? 512, streamer });
94
+ // `done` is optional — the provider closes the stream when generate() resolves
95
+ },
96
+ // anything that is not a generation: the provider passes it through, untouched
97
+ async command(name, payload) { if (name === 'adapter') { /* swap a LoRA, warm a cache… */ } },
98
+ dispose() { model.dispose(); },
99
+ };
100
+ }
101
+ ```
102
+
103
+ `emit` speaks aparté's stream vocabulary (`text`, `thinking`, `tool_use`, `done`, `error`),
104
+ `signal` fires on Stop, `ctx.progress` (a `RunnerProgress`) / `ctx.warn` reach the page. The contract is
105
+ exported: `TransformersRunner`, `RunnerContext`, `RunnerGenerateInput`, `CreateRunner`, `RunnerModule` (what
106
+ the module exports), `BuiltInRunner` (the two `task` names) and `TransformersModule` (the type of
107
+ `ctx.transformers`).
108
+ `runnerCommand(modelId, name, payload)` reaches a runner's `command()` from the page, queued
109
+ behind the generates in flight.
110
+
33
111
  ## One pipeline per tab
34
112
 
35
113
  Unlike every other aparté provider, this one's state is **tab-scoped, not chat-scoped**: one
@@ -52,10 +130,10 @@ setMaxCachedModels(0); // no limit — you are managing memory yourse
52
130
  getMaxCachedModels(); // the current budget
53
131
  ```
54
132
 
55
- > **Scope (v1):** generic text-generation streaming, **browser-only** (unlike the other
56
- > providers — it needs WebGPU/WASM, Workers and the Cache API, so it is the one adapter that
57
- > does not run in Node). Tool-calling for local models is model-specific and out of scope for
58
- > now: `tool_call` / `tool_result` turns are dropped from the prompt, with a one-time console
59
- > warning. Part of the
133
+ > **Scope:** text and vision models through the built-in runners, anything else through a
134
+ > runner of your own; **browser-only** (unlike the other providers — it needs WebGPU/WASM,
135
+ > Workers and the Cache API, so it is the one adapter that does not run in Node). Tool-calling
136
+ > for local models is model-specific: the built-in runners drop `tool_call` / `tool_result`
137
+ > turns with one console warning; a custom runner may render them. Part of the
60
138
  > [aparté](https://github.com/apartejs/aparte) monorepo. ESM-only.
61
139
  > See the **Providers** guide in the docs for the full usage.
@@ -0,0 +1,65 @@
1
+ import { l as loadOptions, i as interruptOn, t as textStreamer, g as generationOptions, T as TOOL_TURNS_DROPPED } from "./shared-ClkqaoeM.js";
2
+ const UNSUPPORTED_PARTS_DROPPED = "Dropped content part(s) this vision runner cannot carry (only text and image parts reach the model).";
3
+ function toChatTemplate(messages, warn) {
4
+ const chat = [];
5
+ const images = [];
6
+ let droppedToolTurns = 0;
7
+ let droppedParts = 0;
8
+ for (const m of messages) {
9
+ if (m.role !== "user" && m.role !== "assistant" && m.role !== "system") {
10
+ droppedToolTurns++;
11
+ continue;
12
+ }
13
+ const parts = [];
14
+ if (typeof m.content === "string") {
15
+ if (m.content) parts.push({ type: "text", text: m.content });
16
+ } else {
17
+ for (const p of m.content) {
18
+ if (p.type === "text") {
19
+ if (p.text) parts.push({ type: "text", text: p.text });
20
+ } else if (p.type === "image") {
21
+ images.push(p.image);
22
+ parts.push({ type: "image" });
23
+ } else droppedParts++;
24
+ }
25
+ }
26
+ if (parts.length > 0) chat.push({ role: m.role, content: parts });
27
+ }
28
+ if (droppedToolTurns > 0) warn(TOOL_TURNS_DROPPED);
29
+ if (droppedParts > 0) warn(UNSUPPORTED_PARTS_DROPPED);
30
+ return { chat, images };
31
+ }
32
+ const createRunner = async (ctx) => {
33
+ const tf = ctx.transformers;
34
+ const [processor, model] = await Promise.all([
35
+ tf.AutoProcessor.from_pretrained(ctx.modelId),
36
+ tf.AutoModelForImageTextToText.from_pretrained(ctx.modelId, loadOptions(ctx))
37
+ ]);
38
+ return {
39
+ async generate({ messages, options, emit, signal }) {
40
+ const { stopping, release } = interruptOn(signal, ctx.transformers);
41
+ try {
42
+ const { chat, images } = toChatTemplate(messages, ctx.warn);
43
+ const prompt = processor.apply_chat_template(chat, { add_generation_prompt: true, tokenize: false });
44
+ const inputs = images.length > 0 ? await processor(prompt, await Promise.all(images.map((src) => tf.load_image(src)))) : await processor.tokenizer(prompt);
45
+ await model.generate({
46
+ ...inputs,
47
+ ...generationOptions(options),
48
+ streamer: textStreamer(ctx.transformers, processor.tokenizer, emit),
49
+ stopping_criteria: stopping
50
+ });
51
+ } finally {
52
+ release();
53
+ }
54
+ },
55
+ async dispose() {
56
+ await model.dispose();
57
+ }
58
+ };
59
+ };
60
+ export {
61
+ UNSUPPORTED_PARTS_DROPPED,
62
+ createRunner,
63
+ toChatTemplate
64
+ };
65
+ //# sourceMappingURL=image-text-to-text-BR39-mrM.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"image-text-to-text-BR39-mrM.js","sources":["../src/runners/image-text-to-text.ts"],"sourcesContent":["/**\n * The built-in vision runner — a model that reads images and text and writes text\n * (SmolVLM, Qwen2-VL, LFM2-VL, Gemma 3, LLaVA…: everything `AutoModelForImageTextToText`\n * resolves).\n *\n * Transformers.js 4.x has no `image-text-to-text` PIPELINE, so this runner goes through\n * the model classes themselves, the way the SmolVLM examples do: `AutoProcessor` renders\n * the chat template with `{ type: 'image' }` placeholders, the images are decoded beside\n * the prompt in the same order, the processor turns both into tensors, and `generate()`\n * streams through a `TextStreamer`. Tool turns are dropped with the shared warning.\n */\n\nimport type { AparteChatMessage } from '@aparte/core';\nimport type { CreateRunner, RunnerGenerateInput } from './types.js';\nimport { TOOL_TURNS_DROPPED, generationOptions, interruptOn, loadOptions, textStreamer } from './shared.js';\n\ntype HFPart = { type: 'image' } | { type: 'text'; text: string };\ntype HFMessage = { role: 'user' | 'assistant' | 'system'; content: HFPart[] };\n\nexport const UNSUPPORTED_PARTS_DROPPED =\n 'Dropped content part(s) this vision runner cannot carry (only text and image parts reach the model).';\n\n/**\n * The conversation in the HF chat shape the processor's template expects — every turn's\n * content as parts, an `{ type: 'image' }` placeholder where a picture goes — plus the\n * pictures themselves, in order of appearance, for the processor to pair with them.\n */\nexport function toChatTemplate(messages: AparteChatMessage[], warn: (message: string) => void): { chat: HFMessage[]; images: string[] } {\n const chat: HFMessage[] = [];\n const images: string[] = [];\n let droppedToolTurns = 0;\n let droppedParts = 0;\n for (const m of messages) {\n if (m.role !== 'user' && m.role !== 'assistant' && m.role !== 'system') { droppedToolTurns++; continue; }\n const parts: HFPart[] = [];\n if (typeof m.content === 'string') {\n if (m.content) parts.push({ type: 'text', text: m.content });\n } else {\n for (const p of m.content) {\n if (p.type === 'text') { if (p.text) parts.push({ type: 'text', text: p.text }); }\n else if (p.type === 'image') { images.push(p.image); parts.push({ type: 'image' }); }\n else droppedParts++;\n }\n }\n if (parts.length > 0) chat.push({ role: m.role, content: parts });\n }\n if (droppedToolTurns > 0) warn(TOOL_TURNS_DROPPED);\n if (droppedParts > 0) warn(UNSUPPORTED_PARTS_DROPPED);\n return { chat, images };\n}\n\n/** The processor and model, as this runner calls them (Transformers.js types them through `Callable`). */\ninterface VisionProcessor {\n (text: string, images: unknown[]): Promise<Record<string, unknown>>;\n /** The text half alone — callable, for a turn that carries no picture. */\n tokenizer: (text: string) => Record<string, unknown> | Promise<Record<string, unknown>>;\n apply_chat_template(messages: HFMessage[], options: Record<string, unknown>): unknown;\n}\ninterface VisionModel {\n generate(args: Record<string, unknown>): Promise<unknown>;\n dispose(): Promise<unknown>;\n}\n\nexport const createRunner: CreateRunner = async (ctx) => {\n const tf = ctx.transformers as unknown as {\n AutoProcessor: { from_pretrained(model: string, options?: unknown): Promise<VisionProcessor> };\n AutoModelForImageTextToText: { from_pretrained(model: string, options?: unknown): Promise<VisionModel> };\n load_image(source: string): Promise<unknown>;\n };\n // The processor carries no weights: dtype and device are the model's alone.\n const [processor, model] = await Promise.all([\n tf.AutoProcessor.from_pretrained(ctx.modelId),\n tf.AutoModelForImageTextToText.from_pretrained(ctx.modelId, loadOptions(ctx)),\n ]);\n\n return {\n async generate({ messages, options, emit, signal }: RunnerGenerateInput): Promise<void> {\n const { stopping, release } = interruptOn(signal, ctx.transformers);\n try {\n const { chat, images } = toChatTemplate(messages, ctx.warn);\n const prompt = processor.apply_chat_template(chat, { add_generation_prompt: true, tokenize: false }) as string;\n // The processor pairs the prompt with pictures and wants at least one (Idefics3's\n // reads `images.rows` — \"hello\" as a first message crashed it on a real SmolVLM).\n // A turn with no image is text: the tokenizer alone takes it.\n const inputs = images.length > 0\n ? await processor(prompt, await Promise.all(images.map((src) => tf.load_image(src))))\n : await processor.tokenizer(prompt);\n await model.generate({\n ...inputs,\n ...generationOptions(options),\n streamer: textStreamer(ctx.transformers, processor.tokenizer, emit),\n stopping_criteria: stopping,\n });\n } finally {\n release();\n }\n },\n async dispose() {\n await model.dispose();\n },\n };\n};\n"],"names":[],"mappings":";AAmBO,MAAM,4BACT;AAOG,SAAS,eAAe,UAA+B,MAA0E;AACpI,QAAM,OAAoB,CAAA;AAC1B,QAAM,SAAmB,CAAA;AACzB,MAAI,mBAAmB;AACvB,MAAI,eAAe;AACnB,aAAW,KAAK,UAAU;AACtB,QAAI,EAAE,SAAS,UAAU,EAAE,SAAS,eAAe,EAAE,SAAS,UAAU;AAAE;AAAoB;AAAA,IAAU;AACxG,UAAM,QAAkB,CAAA;AACxB,QAAI,OAAO,EAAE,YAAY,UAAU;AAC/B,UAAI,EAAE,QAAS,OAAM,KAAK,EAAE,MAAM,QAAQ,MAAM,EAAE,SAAS;AAAA,IAC/D,OAAO;AACH,iBAAW,KAAK,EAAE,SAAS;AACvB,YAAI,EAAE,SAAS,QAAQ;AAAE,cAAI,EAAE,KAAM,OAAM,KAAK,EAAE,MAAM,QAAQ,MAAM,EAAE,MAAM;AAAA,QAAG,WACxE,EAAE,SAAS,SAAS;AAAE,iBAAO,KAAK,EAAE,KAAK;AAAG,gBAAM,KAAK,EAAE,MAAM,QAAA,CAAS;AAAA,QAAG,MAC/E;AAAA,MACT;AAAA,IACJ;AACA,QAAI,MAAM,SAAS,EAAG,MAAK,KAAK,EAAE,MAAM,EAAE,MAAM,SAAS,MAAA,CAAO;AAAA,EACpE;AACA,MAAI,mBAAmB,EAAG,MAAK,kBAAkB;AACjD,MAAI,eAAe,EAAG,MAAK,yBAAyB;AACpD,SAAO,EAAE,MAAM,OAAA;AACnB;AAcO,MAAM,eAA6B,OAAO,QAAQ;AACrD,QAAM,KAAK,IAAI;AAMf,QAAM,CAAC,WAAW,KAAK,IAAI,MAAM,QAAQ,IAAI;AAAA,IACzC,GAAG,cAAc,gBAAgB,IAAI,OAAO;AAAA,IAC5C,GAAG,4BAA4B,gBAAgB,IAAI,SAAS,YAAY,GAAG,CAAC;AAAA,EAAA,CAC/E;AAED,SAAO;AAAA,IACH,MAAM,SAAS,EAAE,UAAU,SAAS,MAAM,UAA8C;AACpF,YAAM,EAAE,UAAU,QAAA,IAAY,YAAY,QAAQ,IAAI,YAAY;AAClE,UAAI;AACA,cAAM,EAAE,MAAM,OAAA,IAAW,eAAe,UAAU,IAAI,IAAI;AAC1D,cAAM,SAAS,UAAU,oBAAoB,MAAM,EAAE,uBAAuB,MAAM,UAAU,OAAO;AAInG,cAAM,SAAS,OAAO,SAAS,IACzB,MAAM,UAAU,QAAQ,MAAM,QAAQ,IAAI,OAAO,IAAI,CAAC,QAAQ,GAAG,WAAW,GAAG,CAAC,CAAC,CAAC,IAClF,MAAM,UAAU,UAAU,MAAM;AACtC,cAAM,MAAM,SAAS;AAAA,UACjB,GAAG;AAAA,UACH,GAAG,kBAAkB,OAAO;AAAA,UAC5B,UAAU,aAAa,IAAI,cAAc,UAAU,WAAW,IAAI;AAAA,UAClE,mBAAmB;AAAA,QAAA,CACtB;AAAA,MACL,UAAA;AACI,gBAAA;AAAA,MACJ;AAAA,IACJ;AAAA,IACA,MAAM,UAAU;AACZ,YAAM,MAAM,QAAA;AAAA,IAChB;AAAA,EAAA;AAER;"}
@@ -0,0 +1,49 @@
1
+ const TOOL_TURNS_DROPPED = "Dropped tool turn(s) from the prompt: this runner does not support tool calling, so the model will not see the call or its result. Use an OpenAI-compatible endpoint for tools, or a runner that renders them.";
2
+ function loadOptions(ctx) {
3
+ const opts = {
4
+ progress_callback: (p) => {
5
+ if (p.status === "progress") ctx.progress({ status: "downloading", file: p.file, progress: Math.round(p.progress ?? 0) });
6
+ else if (p.status === "done") ctx.progress({ status: "loading", file: p.file });
7
+ }
8
+ };
9
+ if (ctx.dtype) opts["dtype"] = ctx.dtype;
10
+ if (ctx.device && ctx.device !== "auto") opts["device"] = ctx.device;
11
+ return opts;
12
+ }
13
+ function interruptOn(signal, transformers) {
14
+ const stopping = new transformers.InterruptableStoppingCriteria();
15
+ const onAbort = () => {
16
+ stopping.interrupt();
17
+ };
18
+ if (signal.aborted) onAbort();
19
+ else signal.addEventListener("abort", onAbort, { once: true });
20
+ return { stopping, release: () => {
21
+ signal.removeEventListener("abort", onAbort);
22
+ } };
23
+ }
24
+ function textStreamer(transformers, tokenizer, emit) {
25
+ const TextStreamer = transformers.TextStreamer;
26
+ return new TextStreamer(tokenizer, {
27
+ skip_prompt: true,
28
+ skip_special_tokens: true,
29
+ callback_function: (text) => {
30
+ if (text) emit({ type: "text", delta: text });
31
+ }
32
+ });
33
+ }
34
+ function generationOptions(options) {
35
+ const temperature = options.temperature ?? 0;
36
+ return {
37
+ max_new_tokens: options.maxTokens ?? 512,
38
+ do_sample: temperature > 0,
39
+ temperature: temperature > 0 ? temperature : void 0
40
+ };
41
+ }
42
+ export {
43
+ TOOL_TURNS_DROPPED as T,
44
+ generationOptions as g,
45
+ interruptOn as i,
46
+ loadOptions as l,
47
+ textStreamer as t
48
+ };
49
+ //# sourceMappingURL=shared-ClkqaoeM.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"shared-ClkqaoeM.js","sources":["../src/runners/shared.ts"],"sourcesContent":["/**\n * What the two built-in runners share — kept in one place so the two cannot drift on\n * how progress is reported, how a stop reaches the model, or what a dropped tool turn\n * says. Types only from core (see `text-generation.ts` for why).\n */\n\nimport type { AparteStreamEvent } from '@aparte/core';\nimport type { RunnerContext, TransformersModule } from './types.js';\n\nexport const TOOL_TURNS_DROPPED =\n 'Dropped tool turn(s) from the prompt: this runner does not support tool calling, so the '\n + 'model will not see the call or its result. Use an OpenAI-compatible endpoint for tools, '\n + 'or a runner that renders them.';\n\n/**\n * The options a `from_pretrained` / `pipeline()` call takes from the context: download\n * progress forwarded to the page (percentages, rounded), dtype and device when set.\n */\nexport function loadOptions(ctx: RunnerContext): Record<string, unknown> {\n const opts: Record<string, unknown> = {\n progress_callback: (p: { status?: string; file?: string; progress?: number }) => {\n if (p.status === 'progress') ctx.progress({ status: 'downloading', file: p.file, progress: Math.round(p.progress ?? 0) });\n else if (p.status === 'done') ctx.progress({ status: 'loading', file: p.file });\n },\n };\n if (ctx.dtype) opts['dtype'] = ctx.dtype;\n if (ctx.device && ctx.device !== 'auto') opts['device'] = ctx.device;\n return opts;\n}\n\n/**\n * A stopping criteria the signal interrupts — so a Stop actually STOPS the model, not just\n * the read; otherwise generation runs to `max_new_tokens` off-thread, spending exactly the\n * CPU/GPU/battery this provider exists to save. Call `release()` in a `finally`.\n */\nexport function interruptOn(signal: AbortSignal, transformers: TransformersModule): { stopping: unknown; release(): void } {\n const stopping = new transformers.InterruptableStoppingCriteria();\n const onAbort = (): void => { stopping.interrupt(); };\n if (signal.aborted) onAbort();\n else signal.addEventListener('abort', onAbort, { once: true });\n return { stopping, release: () => { signal.removeEventListener('abort', onAbort); } };\n}\n\n/** A `TextStreamer` that emits each decoded token as a `text` event, prompt skipped. */\nexport function textStreamer(transformers: TransformersModule, tokenizer: unknown, emit: (event: AparteStreamEvent) => void): unknown {\n const TextStreamer = transformers.TextStreamer as unknown as new (tokenizer: unknown, options: Record<string, unknown>) => unknown;\n return new TextStreamer(tokenizer, {\n skip_prompt: true,\n skip_special_tokens: true,\n callback_function: (text: string) => { if (text) emit({ type: 'text', delta: text }); },\n });\n}\n\n/** Sampling options in Transformers.js' vocabulary, from the request's. */\nexport function generationOptions(options: { maxTokens?: number; temperature?: number }): Record<string, unknown> {\n const temperature = options.temperature ?? 0;\n return {\n max_new_tokens: options.maxTokens ?? 512,\n do_sample: temperature > 0,\n temperature: temperature > 0 ? temperature : undefined,\n };\n}\n"],"names":[],"mappings":"AASO,MAAM,qBACT;AAQG,SAAS,YAAY,KAA6C;AACrE,QAAM,OAAgC;AAAA,IAClC,mBAAmB,CAAC,MAA6D;AAC7E,UAAI,EAAE,WAAW,gBAAgB,SAAS,EAAE,QAAQ,eAAe,MAAM,EAAE,MAAM,UAAU,KAAK,MAAM,EAAE,YAAY,CAAC,GAAG;AAAA,eAC/G,EAAE,WAAW,OAAQ,KAAI,SAAS,EAAE,QAAQ,WAAW,MAAM,EAAE,KAAA,CAAM;AAAA,IAClF;AAAA,EAAA;AAEJ,MAAI,IAAI,MAAO,MAAK,OAAO,IAAI,IAAI;AACnC,MAAI,IAAI,UAAU,IAAI,WAAW,OAAQ,MAAK,QAAQ,IAAI,IAAI;AAC9D,SAAO;AACX;AAOO,SAAS,YAAY,QAAqB,cAA0E;AACvH,QAAM,WAAW,IAAI,aAAa,8BAAA;AAClC,QAAM,UAAU,MAAY;AAAE,aAAS,UAAA;AAAA,EAAa;AACpD,MAAI,OAAO,QAAS,SAAA;AAAA,cACR,iBAAiB,SAAS,SAAS,EAAE,MAAM,MAAM;AAC7D,SAAO,EAAE,UAAU,SAAS,MAAM;AAAE,WAAO,oBAAoB,SAAS,OAAO;AAAA,EAAG,EAAA;AACtF;AAGO,SAAS,aAAa,cAAkC,WAAoB,MAAmD;AAClI,QAAM,eAAe,aAAa;AAClC,SAAO,IAAI,aAAa,WAAW;AAAA,IAC/B,aAAa;AAAA,IACb,qBAAqB;AAAA,IACrB,mBAAmB,CAAC,SAAiB;AAAE,UAAI,KAAM,MAAK,EAAE,MAAM,QAAQ,OAAO,MAAM;AAAA,IAAG;AAAA,EAAA,CACzF;AACL;AAGO,SAAS,kBAAkB,SAAgF;AAC9G,QAAM,cAAc,QAAQ,eAAe;AAC3C,SAAO;AAAA,IACH,gBAAgB,QAAQ,aAAa;AAAA,IACrC,WAAW,cAAc;AAAA,IACzB,aAAa,cAAc,IAAI,cAAc;AAAA,EAAA;AAErD;"}
@@ -0,0 +1,50 @@
1
+ import { l as loadOptions, i as interruptOn, t as textStreamer, g as generationOptions, T as TOOL_TURNS_DROPPED } from "./shared-ClkqaoeM.js";
2
+ const IMAGES_DROPPED = 'This model has no vision runner: image parts were dropped from the prompt, so the model answers as if there were none. Register the model with task: "image-text-to-text", or point `runner` at a module of your own.';
3
+ function textOf(content) {
4
+ if (typeof content === "string") return content;
5
+ return content.filter((p) => p.type === "text").map((p) => p.text).join("");
6
+ }
7
+ function flattenForChatTemplate(messages, warn) {
8
+ const result = [];
9
+ let droppedImages = 0;
10
+ let droppedToolTurns = 0;
11
+ for (const m of messages) {
12
+ if (m.role === "user" || m.role === "assistant" || m.role === "system") {
13
+ if (Array.isArray(m.content)) droppedImages += m.content.filter((p) => p.type === "image").length;
14
+ const text = textOf(m.content);
15
+ if (text) result.push({ role: m.role, content: text });
16
+ } else {
17
+ droppedToolTurns++;
18
+ }
19
+ }
20
+ if (droppedImages > 0) warn(IMAGES_DROPPED);
21
+ if (droppedToolTurns > 0) warn(TOOL_TURNS_DROPPED);
22
+ return result;
23
+ }
24
+ const createRunner = async (ctx) => {
25
+ const pipeline = ctx.transformers.pipeline;
26
+ const pipe = await pipeline("text-generation", ctx.modelId, loadOptions(ctx));
27
+ return {
28
+ async generate({ messages, options, emit, signal }) {
29
+ const { stopping, release } = interruptOn(signal, ctx.transformers);
30
+ try {
31
+ await pipe(flattenForChatTemplate(messages, ctx.warn), {
32
+ ...generationOptions(options),
33
+ streamer: textStreamer(ctx.transformers, pipe.tokenizer, emit),
34
+ stopping_criteria: stopping
35
+ });
36
+ } finally {
37
+ release();
38
+ }
39
+ },
40
+ dispose() {
41
+ return pipe.dispose?.();
42
+ }
43
+ };
44
+ };
45
+ export {
46
+ IMAGES_DROPPED,
47
+ createRunner,
48
+ flattenForChatTemplate
49
+ };
50
+ //# sourceMappingURL=text-generation-BTp593TI.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"text-generation-BTp593TI.js","sources":["../src/runners/text-generation.ts"],"sourcesContent":["/**\n * The built-in text runner — the generic `pipeline('text-generation')` path this\n * provider has always run, extracted from the worker so it is one runner among others.\n *\n * It flattens the conversation to `{ role, content: string }` turns (the tokenizer applies\n * the chat template). Two things it cannot carry, it SAYS: tool turns (their wire syntax is\n * model-specific) and image parts (a text model has no eyes). The second used to vanish\n * silently — a photo attached to a text model produced an answer that pretended — and\n * that silence, not the limitation, was the defect.\n */\n\n// Types only. A runner runs INSIDE the worker, and the worker bundle has no way to\n// resolve `@aparte/core` at runtime (an import map does not reach a worker) — so a value\n// import here is not externalised, it is inlined: the first build that imported\n// `contentToText` shipped all of core, components included, in a 426 kB runner chunk.\nimport type { AparteChatMessage, AparteContentPart } from '@aparte/core';\nimport type { CreateRunner, RunnerContext, RunnerGenerateInput } from './types.js';\nimport { TOOL_TURNS_DROPPED, generationOptions, interruptOn, loadOptions, textStreamer } from './shared.js';\n\ntype SimpleMessage = { role: 'user' | 'assistant' | 'system'; content: string };\n\nexport const IMAGES_DROPPED =\n 'This model has no vision runner: image parts were dropped from the prompt, so the model '\n + 'answers as if there were none. Register the model with task: \"image-text-to-text\", or '\n + 'point `runner` at a module of your own.';\n\n/** The text parts of a message, joined — what a text-only chat template can take. */\nfunction textOf(content: string | AparteContentPart[]): string {\n if (typeof content === 'string') return content;\n return content\n .filter((p): p is Extract<AparteContentPart, { type: 'text' }> => p.type === 'text')\n .map((p) => p.text)\n .join('');\n}\n\n/** Flatten to what the chat template takes; say what was left out. */\nexport function flattenForChatTemplate(messages: AparteChatMessage[], warn: RunnerContext['warn']): SimpleMessage[] {\n const result: SimpleMessage[] = [];\n let droppedImages = 0;\n let droppedToolTurns = 0;\n for (const m of messages) {\n if (m.role === 'user' || m.role === 'assistant' || m.role === 'system') {\n if (Array.isArray(m.content)) droppedImages += m.content.filter((p) => p.type === 'image').length;\n const text = textOf(m.content);\n if (text) result.push({ role: m.role, content: text });\n } else {\n droppedToolTurns++;\n }\n }\n if (droppedImages > 0) warn(IMAGES_DROPPED);\n if (droppedToolTurns > 0) warn(TOOL_TURNS_DROPPED);\n return result;\n}\n\n/** The pipeline, as this runner calls it. Transformers.js types it per task; this is the one task. */\ninterface TextPipeline {\n (messages: SimpleMessage[], options: Record<string, unknown>): Promise<unknown>;\n tokenizer: unknown;\n dispose?: () => Promise<void>;\n}\n\nexport const createRunner: CreateRunner = async (ctx) => {\n const pipeline = ctx.transformers.pipeline as unknown as (task: string, model: string, options: unknown) => Promise<TextPipeline>;\n const pipe = await pipeline('text-generation', ctx.modelId, loadOptions(ctx));\n\n return {\n async generate({ messages, options, emit, signal }: RunnerGenerateInput): Promise<void> {\n const { stopping, release } = interruptOn(signal, ctx.transformers);\n try {\n await pipe(flattenForChatTemplate(messages, ctx.warn), {\n ...generationOptions(options),\n streamer: textStreamer(ctx.transformers, pipe.tokenizer, emit),\n stopping_criteria: stopping,\n });\n } finally {\n release();\n }\n },\n dispose() {\n return pipe.dispose?.();\n },\n };\n};\n"],"names":[],"mappings":";AAqBO,MAAM,iBACT;AAKJ,SAAS,OAAO,SAA+C;AAC3D,MAAI,OAAO,YAAY,SAAU,QAAO;AACxC,SAAO,QACF,OAAO,CAAC,MAAyD,EAAE,SAAS,MAAM,EAClF,IAAI,CAAC,MAAM,EAAE,IAAI,EACjB,KAAK,EAAE;AAChB;AAGO,SAAS,uBAAuB,UAA+B,MAA8C;AAChH,QAAM,SAA0B,CAAA;AAChC,MAAI,gBAAgB;AACpB,MAAI,mBAAmB;AACvB,aAAW,KAAK,UAAU;AACtB,QAAI,EAAE,SAAS,UAAU,EAAE,SAAS,eAAe,EAAE,SAAS,UAAU;AACpE,UAAI,MAAM,QAAQ,EAAE,OAAO,EAAG,kBAAiB,EAAE,QAAQ,OAAO,CAAC,MAAM,EAAE,SAAS,OAAO,EAAE;AAC3F,YAAM,OAAO,OAAO,EAAE,OAAO;AAC7B,UAAI,aAAa,KAAK,EAAE,MAAM,EAAE,MAAM,SAAS,MAAM;AAAA,IACzD,OAAO;AACH;AAAA,IACJ;AAAA,EACJ;AACA,MAAI,gBAAgB,EAAG,MAAK,cAAc;AAC1C,MAAI,mBAAmB,EAAG,MAAK,kBAAkB;AACjD,SAAO;AACX;AASO,MAAM,eAA6B,OAAO,QAAQ;AACrD,QAAM,WAAW,IAAI,aAAa;AAClC,QAAM,OAAO,MAAM,SAAS,mBAAmB,IAAI,SAAS,YAAY,GAAG,CAAC;AAE5E,SAAO;AAAA,IACH,MAAM,SAAS,EAAE,UAAU,SAAS,MAAM,UAA8C;AACpF,YAAM,EAAE,UAAU,QAAA,IAAY,YAAY,QAAQ,IAAI,YAAY;AAClE,UAAI;AACA,cAAM,KAAK,uBAAuB,UAAU,IAAI,IAAI,GAAG;AAAA,UACnD,GAAG,kBAAkB,OAAO;AAAA,UAC5B,UAAU,aAAa,IAAI,cAAc,KAAK,WAAW,IAAI;AAAA,UAC7D,mBAAmB;AAAA,QAAA,CACtB;AAAA,MACL,UAAA;AACI,gBAAA;AAAA,MACJ;AAAA,IACJ;AAAA,IACA,UAAU;AACN,aAAO,KAAK,UAAA;AAAA,IAChB;AAAA,EAAA;AAER;"}
@@ -0,0 +1,153 @@
1
+ const errorText = (err, fallback) => err instanceof Error && err.message || fallback;
2
+ function createWorkerHost(deps) {
3
+ let moduleUrl;
4
+ let current = null;
5
+ const generates = /* @__PURE__ */ new Map();
6
+ const warned = /* @__PURE__ */ new Set();
7
+ const warn = (message) => {
8
+ if (warned.has(message)) return;
9
+ warned.add(message);
10
+ deps.post({ type: "warning", message });
11
+ };
12
+ async function ensureRunner(sel, progressId) {
13
+ const task = sel.task ?? "text-generation";
14
+ const key = `${sel.modelId}::${sel.runner ?? task}`;
15
+ if (current?.key === key) return current.runner;
16
+ if (current) {
17
+ const previous = current;
18
+ current = null;
19
+ await previous.runner.dispose?.();
20
+ }
21
+ const transformers = await deps.loadTransformers(moduleUrl);
22
+ const { createRunner } = await deps.importRunner({ task, runner: sel.runner });
23
+ const runner = await createRunner({
24
+ transformers,
25
+ modelId: sel.modelId,
26
+ dtype: sel.dtype,
27
+ device: sel.device,
28
+ progress: (p) => {
29
+ if (progressId) deps.post({ type: "progress", id: progressId, ...p });
30
+ },
31
+ warn
32
+ });
33
+ current = { key, runner };
34
+ deps.post({ type: "pipeline-ready", modelId: sel.modelId });
35
+ return runner;
36
+ }
37
+ async function handlePrepare(msg) {
38
+ try {
39
+ await ensureRunner(msg, msg.id);
40
+ deps.post({ type: "progress", id: msg.id, status: "ready" });
41
+ } catch (err) {
42
+ deps.post({ type: "prepare-error", id: msg.id, message: errorText(err, "Failed to load model") });
43
+ }
44
+ }
45
+ async function handleGenerate(msg) {
46
+ const controller = new AbortController();
47
+ generates.set(msg.id, controller);
48
+ let usage;
49
+ let closed = false;
50
+ try {
51
+ const runner = await ensureRunner(msg, msg.id);
52
+ await runner.generate({
53
+ messages: msg.messages,
54
+ options: msg.options,
55
+ signal: controller.signal,
56
+ emit: (event) => {
57
+ if (closed) return;
58
+ if (event.type === "done") {
59
+ usage = event.usage;
60
+ return;
61
+ }
62
+ if (event.type === "error") {
63
+ closed = true;
64
+ deps.post({ type: "gen-error", id: msg.id, message: event.message });
65
+ return;
66
+ }
67
+ deps.post({ type: "gen-event", id: msg.id, event });
68
+ }
69
+ });
70
+ if (!closed) deps.post({ type: "gen-done", id: msg.id, ...usage ? { usage } : {} });
71
+ } catch (err) {
72
+ if (!closed) deps.post({ type: "gen-error", id: msg.id, message: errorText(err, "Generation failed") });
73
+ } finally {
74
+ generates.delete(msg.id);
75
+ }
76
+ }
77
+ async function handleCommand(msg) {
78
+ try {
79
+ const runner = await ensureRunner(msg);
80
+ if (!runner.command) {
81
+ deps.post({ type: "command-result", id: msg.id, error: `This runner has no command handler (asked for "${msg.name}")` });
82
+ return;
83
+ }
84
+ const result = await runner.command(msg.name, msg.payload);
85
+ deps.post({ type: "command-result", id: msg.id, result });
86
+ } catch (err) {
87
+ deps.post({ type: "command-result", id: msg.id, error: errorText(err, "Command failed") });
88
+ }
89
+ }
90
+ return {
91
+ onMessage(msg) {
92
+ switch (msg.type) {
93
+ case "init":
94
+ moduleUrl = msg.transformersUrl;
95
+ break;
96
+ case "prepare":
97
+ void handlePrepare(msg);
98
+ break;
99
+ case "generate":
100
+ void handleGenerate(msg);
101
+ break;
102
+ case "cancel":
103
+ generates.get(msg.id)?.abort();
104
+ break;
105
+ case "command":
106
+ void handleCommand(msg);
107
+ break;
108
+ }
109
+ }
110
+ };
111
+ }
112
+ let _tf = null;
113
+ function loadTransformers(moduleUrl) {
114
+ _tf ??= (async () => {
115
+ let mod;
116
+ try {
117
+ mod = await import("@huggingface/transformers");
118
+ } catch (bundlerPathFailed) {
119
+ if (!moduleUrl) throw bundlerPathFailed;
120
+ mod = await import(
121
+ /* @vite-ignore */
122
+ moduleUrl
123
+ );
124
+ }
125
+ mod.env.allowLocalModels = false;
126
+ mod.env.useBrowserCache = true;
127
+ return mod;
128
+ })();
129
+ return _tf;
130
+ }
131
+ const BUILT_IN = {
132
+ "text-generation": () => import("./text-generation-BTp593TI.js"),
133
+ "image-text-to-text": () => import("./image-text-to-text-BR39-mrM.js")
134
+ };
135
+ function importRunner({ task, runner }) {
136
+ if (runner) return import(
137
+ /* @vite-ignore */
138
+ runner
139
+ );
140
+ return BUILT_IN[task]();
141
+ }
142
+ const ctx = self;
143
+ const host = createWorkerHost({
144
+ post: (message) => {
145
+ ctx.postMessage(message);
146
+ },
147
+ loadTransformers,
148
+ importRunner
149
+ });
150
+ ctx.addEventListener("message", (event) => {
151
+ host.onMessage(event.data);
152
+ });
153
+ //# sourceMappingURL=worker-Bm0eBd4l.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"worker-Bm0eBd4l.js","sources":["../src/worker-host.ts","../src/worker.ts"],"sourcesContent":["/**\n * The worker's logic, as a function of its three seams.\n *\n * `worker.ts` is the two-line shell that binds this to `self`; everything it decides is\n * here, with `post`, `loadTransformers` and `importRunner` injected — so the protocol is\n * tested with fakes where no Worker and no model can load (`worker-host.test.ts`).\n *\n * The protocol (main ⇄ worker):\n *\n * → init { transformersUrl? } once, first\n * → prepare { id, modelId, task?, runner?, dtype?, device? }\n * → generate { id, modelId, messages, options, task?, runner?, dtype?, device? }\n * → cancel { id }\n * → command { id, modelId, name, payload, task?, runner?, dtype?, device? }\n * ← progress { id, status, file?, progress?, detail? }\n * ← pipeline-ready { modelId } a runner is up\n * ← prepare-error { id, message }\n * ← gen-event { id, event } text | thinking | tool_use\n * ← gen-done { id, usage? } the host closes; `done` is never forwarded raw\n * ← gen-error { id, message } emitted or thrown\n * ← warning { message } once per distinct text\n * ← command-result { id, result } | { id, error }\n *\n * One runner is resident at a time, keyed by model AND runner: a switch disposes the\n * previous one first, which is the \"one pipeline per tab\" rule this package has always\n * kept — a local model is gigabytes, and two of them resident is the failure to avoid.\n */\n\nimport type { AparteChatMessage, AparteStreamEvent, AparteUsage } from '@aparte/core';\nimport type { BuiltInRunner, Device, Dtype, GenerationOptions, RunnerModule, RunnerProgress, TransformersModule, TransformersRunner } from './runners/types.js';\n\ninterface RunnerSelection {\n modelId: string;\n task?: BuiltInRunner;\n runner?: string;\n dtype?: Dtype;\n device?: Device;\n}\n\nexport type InMessage =\n | { type: 'init'; transformersUrl?: string }\n | ({ type: 'prepare'; id: string } & RunnerSelection)\n | ({ type: 'generate'; id: string; messages: AparteChatMessage[]; options: GenerationOptions } & RunnerSelection)\n | { type: 'cancel'; id: string }\n | ({ type: 'command'; id: string; name: string; payload: unknown } & RunnerSelection);\n\nexport type OutMessage =\n | ({ type: 'progress'; id: string } & RunnerProgress)\n | { type: 'pipeline-ready'; modelId: string }\n | { type: 'prepare-error'; id: string; message: string }\n | { type: 'gen-event'; id: string; event: AparteStreamEvent }\n | { type: 'gen-done'; id: string; usage?: AparteUsage }\n | { type: 'gen-error'; id: string; message: string }\n | { type: 'warning'; message: string }\n | { type: 'command-result'; id: string; result?: unknown; error?: string };\n\nexport interface WorkerHostDeps {\n post(message: OutMessage): void;\n /** Resolve Transformers.js — the bundled specifier first, else the URL `init` carried. */\n loadTransformers(moduleUrl?: string): Promise<TransformersModule>;\n /** Import the runner module a selection names: a built-in by `task`, a custom one by `runner` URL. */\n importRunner(spec: { task: BuiltInRunner; runner?: string }): Promise<RunnerModule>;\n}\n\nconst errorText = (err: unknown, fallback: string): string => (err instanceof Error && err.message) || fallback;\n\nexport function createWorkerHost(deps: WorkerHostDeps): { onMessage(msg: InMessage): void } {\n let moduleUrl: string | undefined;\n let current: { key: string; runner: TransformersRunner } | null = null;\n const generates = new Map<string, AbortController>();\n const warned = new Set<string>();\n\n const warn = (message: string): void => {\n if (warned.has(message)) return;\n warned.add(message);\n deps.post({ type: 'warning', message });\n };\n\n async function ensureRunner(sel: RunnerSelection, progressId?: string): Promise<TransformersRunner> {\n const task = sel.task ?? 'text-generation';\n const key = `${sel.modelId}::${sel.runner ?? task}`;\n if (current?.key === key) return current.runner;\n if (current) {\n const previous = current;\n current = null;\n await previous.runner.dispose?.();\n }\n const transformers = await deps.loadTransformers(moduleUrl);\n const { createRunner } = await deps.importRunner({ task, runner: sel.runner });\n const runner = await createRunner({\n transformers,\n modelId: sel.modelId,\n dtype: sel.dtype,\n device: sel.device,\n progress: (p) => { if (progressId) deps.post({ type: 'progress', id: progressId, ...p }); },\n warn,\n });\n current = { key, runner };\n deps.post({ type: 'pipeline-ready', modelId: sel.modelId });\n return runner;\n }\n\n async function handlePrepare(msg: Extract<InMessage, { type: 'prepare' }>): Promise<void> {\n try {\n await ensureRunner(msg, msg.id);\n deps.post({ type: 'progress', id: msg.id, status: 'ready' });\n } catch (err) {\n deps.post({ type: 'prepare-error', id: msg.id, message: errorText(err, 'Failed to load model') });\n }\n }\n\n async function handleGenerate(msg: Extract<InMessage, { type: 'generate' }>): Promise<void> {\n const controller = new AbortController();\n generates.set(msg.id, controller);\n let usage: AparteUsage | undefined;\n let closed = false;\n try {\n const runner = await ensureRunner(msg, msg.id);\n await runner.generate({\n messages: msg.messages,\n options: msg.options,\n signal: controller.signal,\n emit: (event) => {\n if (closed) return;\n // `done` and `error` close the stream, and closing is the host's: it\n // releases the queue slot on the main thread through gen-done/gen-error.\n if (event.type === 'done') { usage = event.usage; return; }\n if (event.type === 'error') { closed = true; deps.post({ type: 'gen-error', id: msg.id, message: event.message }); return; }\n deps.post({ type: 'gen-event', id: msg.id, event });\n },\n });\n if (!closed) deps.post({ type: 'gen-done', id: msg.id, ...(usage ? { usage } : {}) });\n } catch (err) {\n if (!closed) deps.post({ type: 'gen-error', id: msg.id, message: errorText(err, 'Generation failed') });\n } finally {\n generates.delete(msg.id);\n }\n }\n\n async function handleCommand(msg: Extract<InMessage, { type: 'command' }>): Promise<void> {\n try {\n const runner = await ensureRunner(msg);\n if (!runner.command) {\n deps.post({ type: 'command-result', id: msg.id, error: `This runner has no command handler (asked for \"${msg.name}\")` });\n return;\n }\n const result = await runner.command(msg.name, msg.payload);\n deps.post({ type: 'command-result', id: msg.id, result });\n } catch (err) {\n deps.post({ type: 'command-result', id: msg.id, error: errorText(err, 'Command failed') });\n }\n }\n\n return {\n onMessage(msg: InMessage): void {\n switch (msg.type) {\n case 'init': moduleUrl = msg.transformersUrl; break;\n case 'prepare': void handlePrepare(msg); break;\n case 'generate': void handleGenerate(msg); break;\n case 'cancel': generates.get(msg.id)?.abort(); break;\n case 'command': void handleCommand(msg); break;\n }\n },\n };\n}\n","/**\n * The Transformers.js inference worker — the shell.\n *\n * Runs entirely off the main thread. What it decides lives in `worker-host.ts` (the\n * protocol, the one-runner-at-a-time rule, cancel, warnings); what a model IS lives in a\n * runner (`runners/`). This file binds the two to `self` and owns the two things only a\n * real worker can do: resolve Transformers.js, and import a runner module.\n */\n\nimport type { BuiltInRunner, RunnerModule, TransformersModule } from './runners/types.js';\nimport { createWorkerHost } from './worker-host.js';\n\n/**\n * Where Transformers.js comes from, resolved once, from whichever path has it.\n *\n * A static `import … from '@huggingface/transformers'` is unresolvable in a worker\n * served without a bundler: an import map lives on the DOCUMENT and, by spec, does not\n * reach a worker — so the page can map the specifier for itself and the worker still\n * cannot. The two paths, in order:\n *\n * 1. `import('@huggingface/transformers')` — a bare specifier, statically visible, so a\n * consumer's bundler resolves and bundles the peer exactly as it did before.\n * 2. the absolute URL the main thread read from the page's own import map and sent in\n * the first message — the CDN path, where that map is the consumer's manifest.\n *\n * The order matters: a bundled app must never reach for the network copy. Whichever\n * path won is the module every runner receives as `ctx.transformers` — one copy per\n * worker, the version the CONSUMER installed or pinned.\n */\nlet _tf: Promise<TransformersModule> | null = null;\n\nfunction loadTransformers(moduleUrl?: string): Promise<TransformersModule> {\n _tf ??= (async () => {\n let mod: TransformersModule;\n try {\n mod = await import('@huggingface/transformers');\n } catch (bundlerPathFailed) {\n if (!moduleUrl) throw bundlerPathFailed;\n mod = await import(/* @vite-ignore */ moduleUrl) as TransformersModule;\n }\n // Fetch weights from the Hugging Face hub (not local paths) and cache them in the\n // browser Cache API — this is what `listCachedModels()` scans on the main thread.\n mod.env.allowLocalModels = false;\n mod.env.useBrowserCache = true;\n return mod;\n })();\n return _tf;\n}\n\n/**\n * The runners this package ships, each behind a dynamic import so the bundler splits\n * it into its own chunk and a page loads only the one its model asks for. The imports\n * are RELATIVE on purpose: a module script resolves them against its own URL, so they\n * follow the worker wherever it is served from — the same origin, a CDN, or through the\n * blob shim `_spawnWorker` builds for a cross-origin copy.\n */\nconst BUILT_IN: Record<BuiltInRunner, () => Promise<RunnerModule>> = {\n 'text-generation': () => import('./runners/text-generation.js'),\n 'image-text-to-text': () => import('./runners/image-text-to-text.js'),\n};\n\nfunction importRunner({ task, runner }: { task: BuiltInRunner; runner?: string }): Promise<RunnerModule> {\n // A custom runner is an absolute URL by the time it gets here (the main thread\n // resolved it against the page), and it wins over `task`.\n if (runner) return import(/* @vite-ignore */ runner) as Promise<RunnerModule>;\n return BUILT_IN[task]();\n}\n\n// DOM's `Worker` interface types `postMessage` + typed `addEventListener('message')`,\n// which is enough for the worker scope — avoids pulling the WebWorker lib (it clashes\n// with DOM's global `postMessage`).\nconst ctx = self as unknown as Worker;\n\nconst host = createWorkerHost({\n post: (message) => { ctx.postMessage(message); },\n loadTransformers,\n importRunner,\n});\n\nctx.addEventListener('message', (event: MessageEvent) => { host.onMessage(event.data); });\n"],"names":[],"mappings":"AAgEA,MAAM,YAAY,CAAC,KAAc,aAA8B,eAAe,SAAS,IAAI,WAAY;AAEhG,SAAS,iBAAiB,MAA2D;AACxF,MAAI;AACJ,MAAI,UAA8D;AAClE,QAAM,gCAAgB,IAAA;AACtB,QAAM,6BAAa,IAAA;AAEnB,QAAM,OAAO,CAAC,YAA0B;AACpC,QAAI,OAAO,IAAI,OAAO,EAAG;AACzB,WAAO,IAAI,OAAO;AAClB,SAAK,KAAK,EAAE,MAAM,WAAW,SAAS;AAAA,EAC1C;AAEA,iBAAe,aAAa,KAAsB,YAAkD;AAChG,UAAM,OAAO,IAAI,QAAQ;AACzB,UAAM,MAAM,GAAG,IAAI,OAAO,KAAK,IAAI,UAAU,IAAI;AACjD,QAAI,SAAS,QAAQ,IAAK,QAAO,QAAQ;AACzC,QAAI,SAAS;AACT,YAAM,WAAW;AACjB,gBAAU;AACV,YAAM,SAAS,OAAO,UAAA;AAAA,IAC1B;AACA,UAAM,eAAe,MAAM,KAAK,iBAAiB,SAAS;AAC1D,UAAM,EAAE,iBAAiB,MAAM,KAAK,aAAa,EAAE,MAAM,QAAQ,IAAI,QAAQ;AAC7E,UAAM,SAAS,MAAM,aAAa;AAAA,MAC9B;AAAA,MACA,SAAS,IAAI;AAAA,MACb,OAAO,IAAI;AAAA,MACX,QAAQ,IAAI;AAAA,MACZ,UAAU,CAAC,MAAM;AAAE,YAAI,WAAY,MAAK,KAAK,EAAE,MAAM,YAAY,IAAI,YAAY,GAAG,GAAG;AAAA,MAAG;AAAA,MAC1F;AAAA,IAAA,CACH;AACD,cAAU,EAAE,KAAK,OAAA;AACjB,SAAK,KAAK,EAAE,MAAM,kBAAkB,SAAS,IAAI,SAAS;AAC1D,WAAO;AAAA,EACX;AAEA,iBAAe,cAAc,KAA6D;AACtF,QAAI;AACA,YAAM,aAAa,KAAK,IAAI,EAAE;AAC9B,WAAK,KAAK,EAAE,MAAM,YAAY,IAAI,IAAI,IAAI,QAAQ,SAAS;AAAA,IAC/D,SAAS,KAAK;AACV,WAAK,KAAK,EAAE,MAAM,iBAAiB,IAAI,IAAI,IAAI,SAAS,UAAU,KAAK,sBAAsB,EAAA,CAAG;AAAA,IACpG;AAAA,EACJ;AAEA,iBAAe,eAAe,KAA8D;AACxF,UAAM,aAAa,IAAI,gBAAA;AACvB,cAAU,IAAI,IAAI,IAAI,UAAU;AAChC,QAAI;AACJ,QAAI,SAAS;AACb,QAAI;AACA,YAAM,SAAS,MAAM,aAAa,KAAK,IAAI,EAAE;AAC7C,YAAM,OAAO,SAAS;AAAA,QAClB,UAAU,IAAI;AAAA,QACd,SAAS,IAAI;AAAA,QACb,QAAQ,WAAW;AAAA,QACnB,MAAM,CAAC,UAAU;AACb,cAAI,OAAQ;AAGZ,cAAI,MAAM,SAAS,QAAQ;AAAE,oBAAQ,MAAM;AAAO;AAAA,UAAQ;AAC1D,cAAI,MAAM,SAAS,SAAS;AAAE,qBAAS;AAAM,iBAAK,KAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,SAAS,MAAM,QAAA,CAAS;AAAG;AAAA,UAAQ;AAC3H,eAAK,KAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,OAAO;AAAA,QACtD;AAAA,MAAA,CACH;AACD,UAAI,CAAC,OAAQ,MAAK,KAAK,EAAE,MAAM,YAAY,IAAI,IAAI,IAAI,GAAI,QAAQ,EAAE,UAAU,CAAA,GAAK;AAAA,IACxF,SAAS,KAAK;AACV,UAAI,CAAC,OAAQ,MAAK,KAAK,EAAE,MAAM,aAAa,IAAI,IAAI,IAAI,SAAS,UAAU,KAAK,mBAAmB,GAAG;AAAA,IAC1G,UAAA;AACI,gBAAU,OAAO,IAAI,EAAE;AAAA,IAC3B;AAAA,EACJ;AAEA,iBAAe,cAAc,KAA6D;AACtF,QAAI;AACA,YAAM,SAAS,MAAM,aAAa,GAAG;AACrC,UAAI,CAAC,OAAO,SAAS;AACjB,aAAK,KAAK,EAAE,MAAM,kBAAkB,IAAI,IAAI,IAAI,OAAO,kDAAkD,IAAI,IAAI,KAAA,CAAM;AACvH;AAAA,MACJ;AACA,YAAM,SAAS,MAAM,OAAO,QAAQ,IAAI,MAAM,IAAI,OAAO;AACzD,WAAK,KAAK,EAAE,MAAM,kBAAkB,IAAI,IAAI,IAAI,QAAQ;AAAA,IAC5D,SAAS,KAAK;AACV,WAAK,KAAK,EAAE,MAAM,kBAAkB,IAAI,IAAI,IAAI,OAAO,UAAU,KAAK,gBAAgB,EAAA,CAAG;AAAA,IAC7F;AAAA,EACJ;AAEA,SAAO;AAAA,IACH,UAAU,KAAsB;AAC5B,cAAQ,IAAI,MAAA;AAAA,QACR,KAAK;AAAQ,sBAAY,IAAI;AAAiB;AAAA,QAC9C,KAAK;AAAW,eAAK,cAAc,GAAG;AAAG;AAAA,QACzC,KAAK;AAAY,eAAK,eAAe,GAAG;AAAG;AAAA,QAC3C,KAAK;AAAU,oBAAU,IAAI,IAAI,EAAE,GAAG,MAAA;AAAS;AAAA,QAC/C,KAAK;AAAW,eAAK,cAAc,GAAG;AAAG;AAAA,MAAA;AAAA,IAEjD;AAAA,EAAA;AAER;ACvIA,IAAI,MAA0C;AAE9C,SAAS,iBAAiB,WAAiD;AACvE,WAAS,YAAY;AACjB,QAAI;AACJ,QAAI;AACA,YAAM,MAAM,OAAO,2BAA2B;AAAA,IAClD,SAAS,mBAAmB;AACxB,UAAI,CAAC,UAAW,OAAM;AACtB,YAAM,MAAM;AAAA;AAAA,QAA0B;AAAA;AAAA,IAC1C;AAGA,QAAI,IAAI,mBAAmB;AAC3B,QAAI,IAAI,kBAAkB;AAC1B,WAAO;AAAA,EACX,GAAA;AACA,SAAO;AACX;AASA,MAAM,WAA+D;AAAA,EACjE,mBAAmB,MAAM,OAAO,+BAA8B;AAAA,EAC9D,sBAAsB,MAAM,OAAO,kCAAiC;AACxE;AAEA,SAAS,aAAa,EAAE,MAAM,UAA2E;AAGrG,MAAI,OAAQ,QAAO;AAAA;AAAA,IAA0B;AAAA;AAC7C,SAAO,SAAS,IAAI,EAAA;AACxB;AAKA,MAAM,MAAM;AAEZ,MAAM,OAAO,iBAAiB;AAAA,EAC1B,MAAM,CAAC,YAAY;AAAE,QAAI,YAAY,OAAO;AAAA,EAAG;AAAA,EAC/C;AAAA,EACA;AACJ,CAAC;AAED,IAAI,iBAAiB,WAAW,CAAC,UAAwB;AAAE,OAAK,UAAU,MAAM,IAAI;AAAG,CAAC;"}
package/dist/index.d.ts CHANGED
@@ -5,9 +5,11 @@
5
5
  * thread in a Web Worker) so `AparteDirectTransport` delegates to its `chat()`. Model
6
6
  * weights download once and persist in the Cache API.
7
7
  *
8
- * Scope (v1): generic **text-generation** streaming. Tool-calling for local models is
9
- * model-specific (every family has its own wire format) and is out of scope here — the
10
- * app registers models and streams plain replies. Vision / embeddings can follow on demand.
8
+ * Scope: the worker runs a **runner** — the built-in `text-generation` (any chat model
9
+ * behind Transformers.js' `pipeline()`), or a module of the app's own named by
10
+ * `TransformersModelConfig.runner` (see `runners/types.ts` for the contract). Tool-calling
11
+ * for local models is model-specific (every family has its own wire format), so the
12
+ * built-in drops tool turns and says so; a custom runner may render them.
11
13
  *
12
14
  * ## This provider's state is TAB-scoped, on purpose
13
15
  *
@@ -30,6 +32,7 @@
30
32
  * Same model in both chats is free and correct: they share the load.
31
33
  */
32
34
  import type { AparteAIProvider, AparteAIModel } from '@aparte/core';
35
+ import type { BuiltInRunner, Device, Dtype } from './runners/types.js';
33
36
  export interface HardwareProfile {
34
37
  hasGpu: boolean;
35
38
  ramGb: number;
@@ -52,12 +55,21 @@ export interface TransformersModelConfig {
52
55
  name: string;
53
56
  description?: string;
54
57
  capabilities: AparteAIModel['capabilities'];
55
- /** Transformers.js pipeline task — determines the model architecture / load path. */
56
- task: 'text-generation';
58
+ /**
59
+ * Which built-in runner loads and drives the model. `'text-generation'` (the default)
60
+ * is any chat model behind Transformers.js' `pipeline()`. Ignored when `runner` is set.
61
+ */
62
+ task?: BuiltInRunner;
63
+ /**
64
+ * A runner of your own: the URL of an ES module exporting `createRunner` (see
65
+ * `TransformersRunner`). Resolved against the page, imported by the worker, and handed
66
+ * the same Transformers.js instance the built-ins use. Wins over `task`.
67
+ */
68
+ runner?: string;
57
69
  /** ONNX dtype or per-part dtype map (e.g. `'q4'` or `{ decoder_model_merged: 'q4' }`). */
58
- dtype?: string | Record<string, string>;
70
+ dtype?: Dtype;
59
71
  /** Preferred device. Defaults to WebGPU when available, else WASM. */
60
- device?: 'webgpu' | 'wasm' | 'auto';
72
+ device?: Device;
61
73
  metadata?: Record<string, unknown>;
62
74
  }
63
75
  /**
@@ -97,8 +109,16 @@ export declare function getComputeDevice(): ComputeDevice;
97
109
  export declare const TransformersProvider: AparteAIProvider & Required<Pick<AparteAIProvider, 'prepareModel' | 'getModelStatus' | 'chat'>>;
98
110
  export default TransformersProvider;
99
111
  export type { AparteAIProvider, AparteAIModel, ModelStatus, ModelLoadProgress } from '@aparte/core';
112
+ export type { TransformersRunner, RunnerContext, RunnerGenerateInput, RunnerProgress, RunnerModule, CreateRunner, BuiltInRunner, TransformersModule, } from './runners/types.js';
100
113
  /** Returns the modelId currently loaded in the worker's pipeline, or null. */
101
114
  export declare function getLoadedModelId(): string | null;
115
+ /**
116
+ * Send a runner something that is not a generation — swap an adapter, warm a cache, ask
117
+ * a capability — and get its answer. The name and payload are the runner's vocabulary
118
+ * (the built-in runners answer none). Queued behind the generates in flight: the worker
119
+ * holds one runner, and a command on it mid-stream would race the stream.
120
+ */
121
+ export declare function runnerCommand(modelId: string, name: string, payload: unknown): Promise<unknown>;
102
122
  /** Terminate the shared worker and reset in-memory state. Safe to call any time. */
103
123
  export declare function terminateWorker(): void;
104
124
  export interface CachedModelEntry {
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8BG;AAEH,OAAO,KAAK,EACR,gBAAgB,EAChB,aAAa,EAMhB,MAAM,cAAc,CAAC;AAkBtB,MAAM,WAAW,eAAe;IAC5B,MAAM,EAAE,OAAO,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;IACd,IAAI,EAAE,KAAK,GAAG,KAAK,GAAG,MAAM,CAAC;IAC7B,kBAAkB,EAAE,MAAM,CAAC;CAC9B;AAKD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,KAAK,EAAE;IAAE,GAAG,EAAE,MAAM,CAAC;IAAC,GAAG,CAAC,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAAG,IAAI,CAE9F;AAED,wBAAsB,cAAc,IAAI,OAAO,CAAC,eAAe,CAAC,CA8B/D;AAMD,8DAA8D;AAC9D,MAAM,WAAW,uBAAuB;IACpC,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,aAAa,CAAC,cAAc,CAAC,CAAC;IAC5C,qFAAqF;IACrF,IAAI,EAAE,iBAAiB,CAAC;IACxB,0FAA0F;IAC1F,KAAK,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACxC,sEAAsE;IACtE,MAAM,CAAC,EAAE,QAAQ,GAAG,MAAM,GAAG,MAAM,CAAC;IACpC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACtC;AAQD;;GAEG;AACH,wBAAgB,aAAa,CAAC,MAAM,EAAE,uBAAuB,GAAG,IAAI,CAUnE;AAaD;;;GAGG;AACH,wBAAgB,kBAAkB,CAAC,GAAG,EAAE,MAAM,GAAG,IAAI,CAEpD;AAED,qDAAqD;AACrD,wBAAgB,kBAAkB,IAAI,MAAM,CAE3C;AAED;;;;;GAKG;AACH,MAAM,MAAM,aAAa,GAAG,MAAM,GAAG,QAAQ,GAAG,MAAM,CAAC;AAGvD,wBAAgB,gBAAgB,CAAC,CAAC,EAAE,aAAa,GAAG,IAAI,CAEvD;AAED,wBAAgB,gBAAgB,IAAI,aAAa,CAEhD;AA+UD;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,oBAAoB,EAAE,gBAAgB,GAC7C,QAAQ,CAAC,IAAI,CAAC,gBAAgB,EAAE,cAAc,GAAG,gBAAgB,GAAG,MAAM,CAAC,CAmIhF,CAAC;AAEF,eAAe,oBAAoB,CAAC;AACpC,YAAY,EAAE,gBAAgB,EAAE,aAAa,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,cAAc,CAAC;AAMpG,8EAA8E;AAC9E,wBAAgB,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAEhD;AAED,oFAAoF;AACpF,wBAAgB,eAAe,IAAI,IAAI,CA6BtC;AAED,MAAM,WAAW,gBAAgB;IAC7B,OAAO,EAAE,MAAM,CAAC;IAChB,IAAI,EAAE,MAAM,CAAC;IACb,6EAA6E;IAC7E,SAAS,EAAE,MAAM,CAAC;IAClB,2DAA2D;IAC3D,MAAM,EAAE,OAAO,CAAC;CACnB;AAED;;;GAGG;AACH,wBAAsB,gBAAgB,IAAI,OAAO,CAAC,gBAAgB,EAAE,CAAC,CAsDpE;AAED;;;GAGG;AACH,wBAAsB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAoBtE"}
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAgCG;AAEH,OAAO,KAAK,EACR,gBAAgB,EAChB,aAAa,EAKhB,MAAM,cAAc,CAAC;AAEtB,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,oBAAoB,CAAC;AAcvE,MAAM,WAAW,eAAe;IAC5B,MAAM,EAAE,OAAO,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;IACd,IAAI,EAAE,KAAK,GAAG,KAAK,GAAG,MAAM,CAAC;IAC7B,kBAAkB,EAAE,MAAM,CAAC;CAC9B;AAKD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,KAAK,EAAE;IAAE,GAAG,EAAE,MAAM,CAAC;IAAC,GAAG,CAAC,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAAG,IAAI,CAE9F;AAED,wBAAsB,cAAc,IAAI,OAAO,CAAC,eAAe,CAAC,CA8B/D;AAMD,8DAA8D;AAC9D,MAAM,WAAW,uBAAuB;IACpC,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,aAAa,CAAC,cAAc,CAAC,CAAC;IAC5C;;;OAGG;IACH,IAAI,CAAC,EAAE,aAAa,CAAC;IACrB;;;;OAIG;IACH,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,0FAA0F;IAC1F,KAAK,CAAC,EAAE,KAAK,CAAC;IACd,sEAAsE;IACtE,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACtC;AAQD;;GAEG;AACH,wBAAgB,aAAa,CAAC,MAAM,EAAE,uBAAuB,GAAG,IAAI,CAUnE;AAaD;;;GAGG;AACH,wBAAgB,kBAAkB,CAAC,GAAG,EAAE,MAAM,GAAG,IAAI,CAEpD;AAED,qDAAqD;AACrD,wBAAgB,kBAAkB,IAAI,MAAM,CAE3C;AAED;;;;;GAKG;AACH,MAAM,MAAM,aAAa,GAAG,MAAM,GAAG,QAAQ,GAAG,MAAM,CAAC;AAGvD,wBAAgB,gBAAgB,CAAC,CAAC,EAAE,aAAa,GAAG,IAAI,CAEvD;AAED,wBAAgB,gBAAgB,IAAI,aAAa,CAEhD;AAsVD;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,oBAAoB,EAAE,gBAAgB,GAC7C,QAAQ,CAAC,IAAI,CAAC,gBAAgB,EAAE,cAAc,GAAG,gBAAgB,GAAG,MAAM,CAAC,CAmKhF,CAAC;AAEF,eAAe,oBAAoB,CAAC;AACpC,YAAY,EAAE,gBAAgB,EAAE,aAAa,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,cAAc,CAAC;AACpG,YAAY,EACR,kBAAkB,EAClB,aAAa,EACb,mBAAmB,EACnB,cAAc,EACd,YAAY,EACZ,YAAY,EACZ,aAAa,EACb,kBAAkB,GACrB,MAAM,oBAAoB,CAAC;AAM5B,8EAA8E;AAC9E,wBAAgB,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAEhD;AAED;;;;;GAKG;AACH,wBAAgB,aAAa,CAAC,OAAO,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO,GAAG,OAAO,CAAC,OAAO,CAAC,CAa/F;AAED,oFAAoF;AACpF,wBAAgB,eAAe,IAAI,IAAI,CA+BtC;AAED,MAAM,WAAW,gBAAgB;IAC7B,OAAO,EAAE,MAAM,CAAC;IAChB,IAAI,EAAE,MAAM,CAAC;IACb,6EAA6E;IAC7E,SAAS,EAAE,MAAM,CAAC;IAClB,2DAA2D;IAC3D,MAAM,EAAE,OAAO,CAAC;CACnB;AAED;;;GAGG;AACH,wBAAsB,gBAAgB,IAAI,OAAO,CAAC,gBAAgB,EAAE,CAAC,CAsDpE;AAED;;;GAGG;AACH,wBAAsB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAoBtE"}