foxmind 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +308 -2
  3. package/dist/bin.d.ts +2 -0
  4. package/dist/bin.js +6 -0
  5. package/dist/browser/gliner2-model.d.ts +57 -0
  6. package/dist/browser/gliner2-model.js +233 -0
  7. package/dist/browser/gliner2.d.ts +13 -0
  8. package/dist/browser/gliner2.js +50 -0
  9. package/dist/browser/index.d.ts +4 -0
  10. package/dist/browser/index.js +6 -0
  11. package/dist/browser/runtime.d.ts +61 -0
  12. package/dist/browser/runtime.js +205 -0
  13. package/dist/browser/toolcalls.d.ts +6 -0
  14. package/dist/browser/toolcalls.js +20 -0
  15. package/dist/browser/transformers.d.ts +18 -0
  16. package/dist/browser/transformers.js +81 -0
  17. package/dist/browser/trialml.d.ts +32 -0
  18. package/dist/browser/trialml.js +196 -0
  19. package/dist/cli.d.ts +6 -0
  20. package/dist/cli.js +46 -0
  21. package/dist/doctor.d.ts +23 -0
  22. package/dist/doctor.js +58 -0
  23. package/dist/errors.d.ts +48 -0
  24. package/dist/errors.js +36 -0
  25. package/dist/http.d.ts +46 -0
  26. package/dist/http.js +164 -0
  27. package/dist/index.d.ts +7 -0
  28. package/dist/index.js +7 -0
  29. package/dist/mind.d.ts +60 -0
  30. package/dist/mind.js +122 -0
  31. package/dist/providers/anthropic.d.ts +16 -0
  32. package/dist/providers/anthropic.js +200 -0
  33. package/dist/providers/openai.d.ts +39 -0
  34. package/dist/providers/openai.js +177 -0
  35. package/dist/providers/presets.d.ts +33 -0
  36. package/dist/providers/presets.js +89 -0
  37. package/dist/reply.d.ts +10 -0
  38. package/dist/reply.js +60 -0
  39. package/dist/sse.d.ts +6 -0
  40. package/dist/sse.js +38 -0
  41. package/dist/types.d.ts +106 -0
  42. package/dist/types.js +3 -0
  43. package/package.json +59 -4
@@ -0,0 +1,50 @@
1
+ import { Gliner2 } from "./gliner2-model.js";
2
+ import { loadOnce, onProgress, pickDevice, toBrowserError } from "./runtime.js";
3
+ export function gliner2(options = {}) {
4
+ const model = options.model ?? "pooria/gliner2-multi-v1-agent-batch-ONNX";
5
+ const name = options.name ?? "gliner2";
6
+ const origin = { provider: name, tier: "browser", secrets: [] };
7
+ const capabilities = ["extract", "classify"];
8
+ const loader = loadOnce(origin, model, options.device ?? "auto", async (device, progress) => {
9
+ const loaded = await Gliner2.load(model, { device, dtype: options.dtype ?? "fp16", progress_callback: onProgress(progress) });
10
+ // Compile the WebGPU shaders now, not on the first real call.
11
+ await loaded.classify("warm up", "warmup", { a: undefined, b: undefined });
12
+ return loaded;
13
+ });
14
+ async function run(work) {
15
+ const loaded = await loader.get();
16
+ try {
17
+ return await work(loaded);
18
+ }
19
+ catch (error) {
20
+ throw toBrowserError(origin, error, model);
21
+ }
22
+ }
23
+ return {
24
+ name,
25
+ tier: "browser",
26
+ model,
27
+ capabilities,
28
+ async probe() {
29
+ const picked = await pickDevice(options.device ?? "auto");
30
+ return "code" in picked ? { ok: false, code: picked.code, reason: picked.reason } : { ok: true, where: picked.device, ...(picked.note ? { reason: picked.note } : {}) };
31
+ },
32
+ status() {
33
+ const { state, where, progress, reason } = loader.state;
34
+ return { name, tier: "browser", model, capabilities, state, ...(where ? { where } : {}), ...(progress === undefined ? {} : { progress }), ...(reason ? { reason } : {}) };
35
+ },
36
+ async load() {
37
+ await loader.get();
38
+ },
39
+ async extract(text, labels, extractOptions) {
40
+ if (!Object.keys(labels).length)
41
+ return {};
42
+ return run((loaded) => loaded.extractEntities(text, labels, extractOptions.threshold ?? 0.5));
43
+ },
44
+ async classify(texts, prompt, labels) {
45
+ if (!Object.keys(labels).length)
46
+ return texts.map(() => ({}));
47
+ return run((loaded) => loaded.classifyMany(texts, prompt, labels));
48
+ },
49
+ };
50
+ }
@@ -0,0 +1,4 @@
1
+ export { configureRuntime, hasWebGPU, purgeModel, type Device, type RuntimeOptions } from "./runtime.js";
2
+ export { transformers, type TransformersOptions } from "./transformers.js";
3
+ export { requestTrialML, trialML, wllama, type TrialMLOptions } from "./trialml.js";
4
+ export { gliner2, type Gliner2Options } from "./gliner2.js";
@@ -0,0 +1,6 @@
1
+ // foxmind/browser: the providers that run models inside the browser.
2
+ // They need the optional peer dependency @huggingface/transformers.
3
+ export { configureRuntime, hasWebGPU, purgeModel } from "./runtime.js";
4
+ export { transformers } from "./transformers.js";
5
+ export { requestTrialML, trialML, wllama } from "./trialml.js";
6
+ export { gliner2 } from "./gliner2.js";
@@ -0,0 +1,61 @@
1
+ import { type Origin } from "../http.js";
2
+ import { FoxmindError } from "../errors.js";
3
+ import type { CallOptions } from "../types.js";
4
+ type Transformers = typeof import("@huggingface/transformers");
5
+ export type Device = "auto" | "webgpu" | "wasm";
6
+ export interface RuntimeOptions {
7
+ /** Where ONNX Runtime's asyncify .mjs and .wasm files are. Default: ort/ in the extension. */
8
+ wasmPaths?: {
9
+ mjs: string;
10
+ wasm: string;
11
+ };
12
+ /** The model hub, for a mirror. It applies to every model on the page. Default "https://huggingface.co/". */
13
+ remoteHost?: string;
14
+ }
15
+ /** Change the runtime settings. Call it before the first model loads. */
16
+ export declare function configureRuntime(next: RuntimeOptions): void;
17
+ /** transformers.js, loaded once, set for extension pages: no remote code, one WASM thread when there is no SharedArrayBuffer. */
18
+ export declare function transformersJs(): Promise<Transformers>;
19
+ /** True when the browser gives a WebGPU adapter. */
20
+ export declare function hasWebGPU(): Promise<boolean>;
21
+ /** The device to load on, or why the asked device cannot run. */
22
+ export declare function pickDevice(device: Device): Promise<{
23
+ device: "webgpu" | "wasm";
24
+ note?: string;
25
+ } | {
26
+ code: "webgpu_missing";
27
+ reason: string;
28
+ }>;
29
+ /** Delete one model's files from Cache Storage. Returns how many it deleted. */
30
+ export declare function purgeModel(model: string): Promise<number>;
31
+ export declare function isCached(model: string): Promise<boolean>;
32
+ /**
33
+ * Wait for work, but stop waiting at the caller's abort or after timeoutMs
34
+ * (when either is given). onStop runs then, to stop the work where it can.
35
+ */
36
+ export declare function bounded<T>(origin: Origin, work: Promise<T>, limits: CallOptions, onStop?: () => void): Promise<T>;
37
+ /** Map an error from a model load or run to a FoxmindError code. */
38
+ export declare function toBrowserError(origin: Origin, error: unknown, model: string): FoxmindError;
39
+ export interface LoadState {
40
+ state: "idle" | "loading" | "ready" | "error";
41
+ where?: "webgpu" | "wasm";
42
+ progress?: number;
43
+ reason?: string;
44
+ }
45
+ /**
46
+ * Loads a model once. Calls that come during a load wait for it. A failed
47
+ * load is not kept, so the next call tries again. With device "auto", a model
48
+ * that fails on WebGPU loads on WASM before anything is deleted. Cached files
49
+ * are deleted and downloaded once more only when every device failed, or when
50
+ * the error points at the files.
51
+ */
52
+ export declare function loadOnce<T>(origin: Origin, model: string, device: Device, open: (device: "webgpu" | "wasm", progress: (p: number) => void) => Promise<T>): {
53
+ state: LoadState;
54
+ get(): Promise<T>;
55
+ };
56
+ /** The progress callback transformers.js takes, as one number from 0 to 1. */
57
+ export declare function onProgress(progress: (p: number) => void): (info: {
58
+ status?: string;
59
+ progress?: number;
60
+ }) => void;
61
+ export {};
@@ -0,0 +1,205 @@
1
+ // What every in-browser model needs: transformers.js set up for an extension
2
+ // page, WebGPU detection, the model cache, and load errors mapped to codes.
3
+ import { failure } from "../http.js";
4
+ import { FoxmindError } from "../errors.js";
5
+ let options = {};
6
+ let loaded;
7
+ let adapter;
8
+ /** Change the runtime settings. Call it before the first model loads. */
9
+ export function configureRuntime(next) {
10
+ options = { ...options, ...next };
11
+ }
12
+ function extensionURL(path) {
13
+ const runtime = globalThis.browser?.runtime;
14
+ return runtime?.getURL?.(path);
15
+ }
16
+ /** transformers.js, loaded once, set for extension pages: no remote code, one WASM thread when there is no SharedArrayBuffer. */
17
+ export async function transformersJs() {
18
+ const module = await (loaded ??= import("@huggingface/transformers").then((fresh) => {
19
+ const { env } = fresh;
20
+ env.allowLocalModels = false;
21
+ // MV3 forbids blob: and remote scripts, so ONNX Runtime loads its own files.
22
+ env.useWasmCache = false;
23
+ const wasm = env.backends.onnx.wasm;
24
+ const mjs = extensionURL("ort/ort-wasm-simd-threaded.asyncify.mjs");
25
+ const paths = options.wasmPaths ?? (mjs ? { mjs, wasm: extensionURL("ort/ort-wasm-simd-threaded.asyncify.wasm") } : undefined);
26
+ if (paths)
27
+ wasm.wasmPaths = paths;
28
+ if (!globalThis.crossOriginIsolated)
29
+ wasm.numThreads = 1;
30
+ return fresh;
31
+ }));
32
+ module.env.remoteHost = options.remoteHost ?? "https://huggingface.co/";
33
+ return module;
34
+ }
35
+ /** True when the browser gives a WebGPU adapter. */
36
+ export function hasWebGPU() {
37
+ adapter ??= (async () => {
38
+ const gpu = globalThis.navigator?.gpu;
39
+ if (!gpu)
40
+ return false;
41
+ try {
42
+ return Boolean(await gpu.requestAdapter());
43
+ }
44
+ catch {
45
+ return false;
46
+ }
47
+ })();
48
+ return adapter;
49
+ }
50
+ /** The device to load on, or why the asked device cannot run. */
51
+ export async function pickDevice(device) {
52
+ if (device === "wasm")
53
+ return { device: "wasm" };
54
+ if (await hasWebGPU())
55
+ return { device: "webgpu" };
56
+ if (device === "webgpu")
57
+ return { code: "webgpu_missing", reason: "This browser gives no WebGPU adapter (Firefox has none on Linux, Intel Macs and Android). Use device \"auto\" or \"wasm\"." };
58
+ return { device: "wasm", note: "WebGPU is missing, so it runs on WASM." };
59
+ }
60
+ const CACHE = "transformers-cache";
61
+ /** The cached requests that belong to one model. */
62
+ async function cachedFiles(model) {
63
+ if (typeof caches === "undefined")
64
+ return undefined;
65
+ const cache = await caches.open(CACHE);
66
+ const keys = (await cache.keys()).filter((request) => request.url.includes(`/${model}/`));
67
+ return { cache, keys };
68
+ }
69
+ /** Delete one model's files from Cache Storage. Returns how many it deleted. */
70
+ export async function purgeModel(model) {
71
+ const files = await cachedFiles(model);
72
+ if (!files)
73
+ return 0;
74
+ await Promise.all(files.keys.map((key) => files.cache.delete(key)));
75
+ return files.keys.length;
76
+ }
77
+ export async function isCached(model) {
78
+ return Boolean((await cachedFiles(model))?.keys.length);
79
+ }
80
+ /**
81
+ * Wait for work, but stop waiting at the caller's abort or after timeoutMs
82
+ * (when either is given). onStop runs then, to stop the work where it can.
83
+ */
84
+ export async function bounded(origin, work, limits, onStop) {
85
+ const { signal, timeoutMs } = limits;
86
+ if (!signal && timeoutMs === undefined)
87
+ return work;
88
+ let timer;
89
+ let onAbort;
90
+ const stopped = new Promise((_, reject) => {
91
+ const stop = (error) => {
92
+ onStop?.();
93
+ reject(error);
94
+ };
95
+ onAbort = () => stop(failure(origin, "aborted", "The caller stopped the call."));
96
+ if (signal?.aborted)
97
+ onAbort();
98
+ signal?.addEventListener("abort", onAbort, { once: true });
99
+ if (timeoutMs !== undefined)
100
+ timer = setTimeout(() => stop(failure(origin, "timeout", `No answer within ${timeoutMs} ms.`)), timeoutMs);
101
+ });
102
+ try {
103
+ return await Promise.race([work, stopped]);
104
+ }
105
+ finally {
106
+ clearTimeout(timer);
107
+ if (onAbort)
108
+ signal?.removeEventListener("abort", onAbort);
109
+ }
110
+ }
111
+ /** Map an error from a model load or run to a FoxmindError code. */
112
+ export function toBrowserError(origin, error, model) {
113
+ if (error instanceof FoxmindError)
114
+ return error;
115
+ const message = error instanceof Error ? error.message : String(error);
116
+ if (/out of memory|allocation failed|could not allocate|memory access out of bounds|device (was )?lost|maximum buffer size|exceeds the max/i.test(message)) {
117
+ return failure(origin, "out_of_memory", `${model} needs more memory than this device has (${message}). Try a smaller model or a smaller dtype such as "q4".`, { cause: error });
118
+ }
119
+ if (/Could not locate file|Unauthorized access to file|Forbidden access to file/i.test(message)) {
120
+ return failure(origin, "model_not_found", `The hub has no ${model}, or it is private: ${message}`, { cause: error });
121
+ }
122
+ if (/NetworkError|Failed to fetch|network|body stream|input stream|terminated|load failed|Gateway|Service unavailable/i.test(message)) {
123
+ return failure(origin, "download_failed", `The download of ${model} stopped: ${message}. Call again to download it again.`, { cause: error });
124
+ }
125
+ return failure(origin, "bad_response", `${model} did not load: ${message}`, { cause: error });
126
+ }
127
+ /** Errors that point at the model files, not at the device. */
128
+ const FILE_ERROR = /protobuf|pars(e|ing)|Unexpected (token|end)|JSON|invalid model|not a valid|corrupt|magic/i;
129
+ /**
130
+ * Loads a model once. Calls that come during a load wait for it. A failed
131
+ * load is not kept, so the next call tries again. With device "auto", a model
132
+ * that fails on WebGPU loads on WASM before anything is deleted. Cached files
133
+ * are deleted and downloaded once more only when every device failed, or when
134
+ * the error points at the files.
135
+ */
136
+ export function loadOnce(origin, model, device, open) {
137
+ const state = { state: "idle" };
138
+ let pending;
139
+ const progress = (p) => { state.progress = p; };
140
+ async function attempt() {
141
+ const picked = await pickDevice(device);
142
+ if ("code" in picked)
143
+ throw failure(origin, picked.code, picked.reason);
144
+ Object.assign(state, { state: "loading", where: picked.device, progress: 0, reason: picked.note });
145
+ const hadCache = await isCached(model);
146
+ const devices = device === "auto" && picked.device === "webgpu" ? ["webgpu", "wasm"] : [picked.device];
147
+ let first;
148
+ let last;
149
+ for (const where of devices) {
150
+ state.where = where;
151
+ try {
152
+ const value = await open(where, progress);
153
+ if (first)
154
+ state.reason = `WebGPU failed (${first.message ?? first}), so it runs on WASM.`;
155
+ return value;
156
+ }
157
+ catch (error) {
158
+ const mapped = toBrowserError(origin, error, model);
159
+ // A missing model or a broken download is the same on every device.
160
+ if (mapped.code === "model_not_found" || mapped.code === "download_failed")
161
+ throw mapped;
162
+ first ??= error;
163
+ last = error;
164
+ }
165
+ }
166
+ const mapped = toBrowserError(origin, last, model);
167
+ const message = last instanceof Error ? last.message : String(last);
168
+ if (mapped.code !== "bad_response" || !hadCache || (devices.length === 1 && !FILE_ERROR.test(message)))
169
+ throw mapped;
170
+ const where = devices.at(-1);
171
+ const removed = await purgeModel(model);
172
+ try {
173
+ const value = await open(where, progress);
174
+ state.reason = `Repaired: deleted ${removed} cached files of ${model} that did not load, and downloaded them again.`;
175
+ return value;
176
+ }
177
+ catch (second) {
178
+ const again = toBrowserError(origin, second, model);
179
+ if (again.code !== "bad_response")
180
+ throw again;
181
+ throw failure(origin, "cache_corrupt", `${model} did not load from a fresh download either: ${again.message}`, { cause: second });
182
+ }
183
+ }
184
+ return {
185
+ state,
186
+ get() {
187
+ pending ??= attempt().then((value) => {
188
+ Object.assign(state, { state: "ready", progress: 1 });
189
+ return value;
190
+ }, (error) => {
191
+ pending = undefined;
192
+ Object.assign(state, { state: "error", reason: error.message });
193
+ throw error;
194
+ });
195
+ return pending;
196
+ },
197
+ };
198
+ }
199
+ /** The progress callback transformers.js takes, as one number from 0 to 1. */
200
+ export function onProgress(progress) {
201
+ return (info) => {
202
+ if (info.status === "progress_total" && typeof info.progress === "number")
203
+ progress(info.progress / 100);
204
+ };
205
+ }
@@ -0,0 +1,6 @@
1
+ import type { ToolCall } from "../types.js";
2
+ /** Qwen3 and Hermes-style models write tool calls as <tool_call>{"name": …, "arguments": …}</tool_call>. */
3
+ export declare function parseToolCalls(text: string): {
4
+ content: string;
5
+ toolCalls: ToolCall[];
6
+ };
@@ -0,0 +1,20 @@
1
+ /** Qwen3 and Hermes-style models write tool calls as <tool_call>{"name": …, "arguments": …}</tool_call>. */
2
+ export function parseToolCalls(text) {
3
+ const toolCalls = [];
4
+ const content = text.replace(/<tool_call>([\s\S]*?)(<\/tool_call>|$)/g, (_, body) => {
5
+ const raw = body.trim();
6
+ let name = raw.match(/"name"\s*:\s*"([^"]+)"/)?.[1] ?? "unknown";
7
+ let args = raw;
8
+ try {
9
+ const parsed = JSON.parse(raw);
10
+ name = parsed.name ?? name;
11
+ args = typeof parsed.arguments === "string" ? parsed.arguments : JSON.stringify(parsed.arguments ?? {});
12
+ }
13
+ catch {
14
+ // checkReply() reports the raw text as a bad tool call.
15
+ }
16
+ toolCalls.push({ id: `call_${toolCalls.length}`, type: "function", function: { name, arguments: args } });
17
+ return "";
18
+ });
19
+ return { content: content.trim(), toolCalls };
20
+ }
@@ -0,0 +1,18 @@
1
+ import type { Provider } from "../types.js";
2
+ import { type Device } from "./runtime.js";
3
+ export interface TransformersOptions {
4
+ task: "embed" | "chat";
5
+ /** Default "Xenova/all-MiniLM-L6-v2" for embed and "onnx-community/Qwen3-0.6B-ONNX" for chat. */
6
+ model?: string;
7
+ /** Default "auto": WebGPU when the browser has it, else WASM. */
8
+ device?: Device;
9
+ /** Default on WebGPU: "fp16" for embed, "q4f16" for chat. On WASM: "q8" for embed, "q4" for chat. */
10
+ dtype?: string;
11
+ /** Default "transformers-embed" or "transformers-chat". */
12
+ name?: string;
13
+ /** Chat: the most tokens to write. Default 256. */
14
+ maxNewTokens?: number;
15
+ /** Chat: let Qwen3 think before it answers. Default false. */
16
+ thinking?: boolean;
17
+ }
18
+ export declare function transformers(options: TransformersOptions): Provider;
@@ -0,0 +1,81 @@
1
+ import { checkReply } from "../reply.js";
2
+ import { bounded, loadOnce, onProgress, pickDevice, toBrowserError, transformersJs } from "./runtime.js";
3
+ import { parseToolCalls } from "./toolcalls.js";
4
+ export function transformers(options) {
5
+ const chat = options.task === "chat";
6
+ const model = options.model ?? (chat ? "onnx-community/Qwen3-0.6B-ONNX" : "Xenova/all-MiniLM-L6-v2");
7
+ const name = options.name ?? `transformers-${options.task}`;
8
+ const origin = { provider: name, tier: "browser", secrets: [] };
9
+ const capabilities = chat ? ["chat"] : ["embed"];
10
+ const loader = loadOnce(origin, model, options.device ?? "auto", async (device, progress) => {
11
+ const { pipeline } = await transformersJs();
12
+ const dtype = options.dtype ?? (device === "webgpu" ? (chat ? "q4f16" : "fp16") : chat ? "q4" : "q8");
13
+ const task = chat ? "text-generation" : "feature-extraction";
14
+ return (await pipeline(task, model, { device, dtype, progress_callback: onProgress(progress) }));
15
+ });
16
+ async function run(work) {
17
+ const pipe = await loader.get();
18
+ try {
19
+ return await work(pipe);
20
+ }
21
+ catch (error) {
22
+ throw toBrowserError(origin, error, model);
23
+ }
24
+ }
25
+ return {
26
+ name,
27
+ tier: "browser",
28
+ model,
29
+ capabilities,
30
+ async probe() {
31
+ const picked = await pickDevice(options.device ?? "auto");
32
+ return "code" in picked ? { ok: false, code: picked.code, reason: picked.reason } : { ok: true, where: picked.device, ...(picked.note ? { reason: picked.note } : {}) };
33
+ },
34
+ status() {
35
+ const { state, where, progress, reason } = loader.state;
36
+ return { name, tier: "browser", model, capabilities, state, ...(where ? { where } : {}), ...(progress === undefined ? {} : { progress }), ...(reason ? { reason } : {}) };
37
+ },
38
+ async load() {
39
+ await loader.get();
40
+ },
41
+ ...(chat
42
+ ? {
43
+ async chat(messages, chatOptions) {
44
+ const rule = chatOptions.json ? [{ role: "system", content: "Reply with one JSON object and nothing else." }] : [];
45
+ return run(async (pipe) => {
46
+ const { TextStreamer, InterruptableStoppingCriteria } = await transformersJs();
47
+ // Stops the generation itself when the caller aborts or the time runs out.
48
+ const stopper = new InterruptableStoppingCriteria();
49
+ const prompt = pipe.tokenizer.apply_chat_template([...rule, ...messages.map(({ reasoning: _r, provider_data: _p, ...m }) => m)], {
50
+ tokenize: false,
51
+ add_generation_prompt: true,
52
+ enable_thinking: options.thinking ?? false,
53
+ ...(chatOptions.tools?.length ? { tools: chatOptions.tools } : {}),
54
+ });
55
+ const onDelta = chatOptions.onDelta;
56
+ const streamer = onDelta ? new TextStreamer(pipe.tokenizer, { skip_prompt: true, skip_special_tokens: true, callback_function: onDelta }) : undefined;
57
+ const generation = pipe(prompt, {
58
+ max_new_tokens: chatOptions.maxTokens ?? options.maxNewTokens ?? 256,
59
+ do_sample: (chatOptions.temperature ?? 0) > 0,
60
+ ...(chatOptions.temperature ? { temperature: chatOptions.temperature } : {}),
61
+ return_full_text: false,
62
+ stopping_criteria: stopper,
63
+ ...(streamer ? { streamer } : {}),
64
+ });
65
+ const [output] = (await bounded(origin, generation, chatOptions, () => stopper.interrupt()));
66
+ const { content, toolCalls } = parseToolCalls(output?.generated_text ?? "");
67
+ const message = { role: "assistant", content, ...(toolCalls.length ? { tool_calls: toolCalls } : {}) };
68
+ return checkReply(origin, { message, finishReason: toolCalls.length ? "tool_calls" : "stop" }, chatOptions.json);
69
+ });
70
+ },
71
+ }
72
+ : {
73
+ async embed(texts) {
74
+ return run(async (pipe) => {
75
+ const tensor = (await pipe(texts, { pooling: "mean", normalize: true }));
76
+ return tensor.tolist();
77
+ });
78
+ },
79
+ }),
80
+ };
81
+ }
@@ -0,0 +1,32 @@
1
+ import type { Provider } from "../types.js";
2
+ export interface TrialMLOptions {
3
+ task: "embed" | "chat";
4
+ /** Default "Xenova/all-MiniLM-L6-v2" for embed and "Xenova/Qwen1.5-0.5B-Chat" for chat. */
5
+ model?: string;
6
+ /** Default "wasm". "gpu" uses WebGPU where Firefox has it. */
7
+ device?: "wasm" | "gpu";
8
+ /** Default "trialml-embed" or "trialml-chat". */
9
+ name?: string;
10
+ /** "llama.cpp" runs a GGUF file (set `modelFile`). Default: Firefox's ONNX backend. */
11
+ backend?: "llama.cpp";
12
+ /** The GGUF file in the model repo, for the llama.cpp backend. */
13
+ modelFile?: string;
14
+ /** Refuse model files larger than this many bytes before the download. Default 4 GB. */
15
+ maxBytes?: number;
16
+ }
17
+ /** Ask the user for the optional trialML permission. Call it from a click handler. */
18
+ export declare function requestTrialML(): Promise<boolean>;
19
+ /** One vector per text from whatever shape the engine returns. */
20
+ export declare function toVectors(result: unknown, count: number): number[][] | undefined;
21
+ export declare function trialML(options: TrialMLOptions): Provider;
22
+ /**
23
+ * Experimental: a GGUF model on Firefox's llama.cpp backend, through trial ML.
24
+ * Small models only: trial ML takes models from the Mozilla and Xenova orgs,
25
+ * and foxmind refuses files over `maxBytes` (default 4 GB) before the download.
26
+ */
27
+ export declare function wllama(options: {
28
+ model: string;
29
+ modelFile: string;
30
+ name?: string;
31
+ maxBytes?: number;
32
+ }): Provider;