foxmind 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +308 -2
  3. package/dist/bin.d.ts +2 -0
  4. package/dist/bin.js +6 -0
  5. package/dist/browser/gliner2-model.d.ts +57 -0
  6. package/dist/browser/gliner2-model.js +233 -0
  7. package/dist/browser/gliner2.d.ts +13 -0
  8. package/dist/browser/gliner2.js +50 -0
  9. package/dist/browser/index.d.ts +4 -0
  10. package/dist/browser/index.js +6 -0
  11. package/dist/browser/runtime.d.ts +61 -0
  12. package/dist/browser/runtime.js +205 -0
  13. package/dist/browser/toolcalls.d.ts +6 -0
  14. package/dist/browser/toolcalls.js +20 -0
  15. package/dist/browser/transformers.d.ts +18 -0
  16. package/dist/browser/transformers.js +81 -0
  17. package/dist/browser/trialml.d.ts +32 -0
  18. package/dist/browser/trialml.js +196 -0
  19. package/dist/cli.d.ts +6 -0
  20. package/dist/cli.js +46 -0
  21. package/dist/doctor.d.ts +23 -0
  22. package/dist/doctor.js +58 -0
  23. package/dist/errors.d.ts +48 -0
  24. package/dist/errors.js +36 -0
  25. package/dist/http.d.ts +46 -0
  26. package/dist/http.js +164 -0
  27. package/dist/index.d.ts +7 -0
  28. package/dist/index.js +7 -0
  29. package/dist/mind.d.ts +60 -0
  30. package/dist/mind.js +122 -0
  31. package/dist/providers/anthropic.d.ts +16 -0
  32. package/dist/providers/anthropic.js +200 -0
  33. package/dist/providers/openai.d.ts +39 -0
  34. package/dist/providers/openai.js +177 -0
  35. package/dist/providers/presets.d.ts +33 -0
  36. package/dist/providers/presets.js +89 -0
  37. package/dist/reply.d.ts +10 -0
  38. package/dist/reply.js +60 -0
  39. package/dist/sse.d.ts +6 -0
  40. package/dist/sse.js +38 -0
  41. package/dist/types.d.ts +106 -0
  42. package/dist/types.js +3 -0
  43. package/package.json +59 -4
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pooria Arab
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -1,3 +1,309 @@
1
- # Temporary Holding Version
1
+ # foxmind
2
2
 
3
- This version is a temporary placeholder for this package. An operational version to replace this has been submitted for review and is awaiting a staged release.
3
+ One API for on-device AI models in the browser and for your own model key.
4
+
5
+ foxmind gives you one `chat`, `embed`, `extract` and `classify` call over three
6
+ tiers of models: models that run in the browser, a model server on your own
7
+ machine, and a cloud model that you pay for with your own key. Every result
8
+ says which tier answered. When a tier cannot run, foxmind says why. It does not
9
+ switch to another tier in secret.
10
+
11
+ ## Install
12
+
13
+ ```bash
14
+ npm i foxmind
15
+ ```
16
+
17
+ For the in-browser models, also install the optional peer dependency:
18
+ `npm i @huggingface/transformers`.
19
+
20
+ ## Example
21
+
22
+ This runs in Node 24 or later. Start a local server first, for example
23
+ `ollama pull qwen3:0.6b`.
24
+
25
+ ```js
26
+ import { createMind, ollama, saluki } from "foxmind";
27
+
28
+ const mind = createMind({
29
+ providers: [saluki(), ollama({ model: "qwen3:0.6b" })],
30
+ only: ["browser", "local"], // private mode: no cloud provider is ever called
31
+ });
32
+
33
+ const result = await mind.chat([{ role: "user", content: "Say hello in five words." }], { maxTokens: 1024 });
34
+ console.log(result.message.content);
35
+ console.log(`answered by ${result.provider} (${result.tier})`);
36
+ console.log(result.skipped.map((skip) => `${skip.provider}: ${skip.code}`));
37
+ ```
38
+
39
+ Output on our test machine. Saluki was not running, so the router skipped it
40
+ and said why:
41
+
42
+ ```text
43
+ Hi there!
44
+ answered by ollama (local)
45
+ [ 'saluki: unreachable' ]
46
+ ```
47
+
48
+ Messages, tools and tool calls use the OpenAI chat completions shape. Code that
49
+ you wrote for that API works here.
50
+
51
+ ## Use cases
52
+
53
+ | Who | What they build | How foxmind helps |
54
+ |---|---|---|
55
+ | A Firefox extension developer | Local AI features in an extension: search, summaries, form help | `foxmind/browser` runs embeddings and small chat models in the extension's background page, on WebGPU or on single-thread WASM. |
56
+ | A web app team | A "bring your own key" AI feature | `anthropic()` and `openaiCompatible()` take the user's key at call time and map both APIs to one shape. foxmind does not store the key. |
57
+ | An agent builder who wants privacy | A private-mode agent planner (for example in foxloop or foxmate) | `saluki()` runs Underdog Saluki 27B on llama-server on the same machine. `only: ["browser", "local"]` drops the cloud tier, so page text cannot leave the machine through foxmind. |
58
+ | A form-filling tool | Pull names, dates and places out of a request | `gliner2()` runs GLiNER2 entity extraction and label scoring in the browser, as foxpilot does. |
59
+ | A notes or bookmarks app | Local semantic search | `mind.embed()` with `transformers()` or Firefox's own `trialML()` gives vectors without a server. |
60
+ | A developer or a CI job | A check of which model servers run on a machine | `foxmind doctor` probes llama-server, Ollama and LM Studio, and prints the command that starts each one that is down. |
61
+
62
+ ## How it works
63
+
64
+ ```mermaid
65
+ flowchart TB
66
+ caller["Your code: mind.chat / embed / extract / classify"] --> router["createMind router"]
67
+ router -->|"1. drop tiers outside only, order by prefer"| order["providers in order"]
68
+ order -->|"2. probe (cached 30 s)"| probe{"can it run now?"}
69
+ probe -->|no| skip["add to result.skipped with code and reason"]
70
+ skip --> order
71
+ probe -->|yes| call["call the provider"]
72
+ call -->|ok| result["result + provider + tier + skipped"]
73
+ call -->|error| error["throw FoxmindError with code, provider, tier"]
74
+ subgraph browser["Tier 1: in the browser"]
75
+ T["transformers() embed or chat"]
76
+ G["gliner2() extract and classify"]
77
+ M["trialML() / wllama() on browser.trial.ml"]
78
+ end
79
+ subgraph local["Tier 2: a server on this machine"]
80
+ L["llamaServer() / saluki() :8080"]
81
+ O["ollama() :11434"]
82
+ S["lmStudio() :1234"]
83
+ end
84
+ subgraph cloud["Tier 3: your own key"]
85
+ A["anthropic()"]
86
+ C["openaiCompatible({ baseURL, apiKey })"]
87
+ end
88
+ call --- browser
89
+ call --- local
90
+ call --- cloud
91
+ ```
92
+
93
+ The router drops every provider whose tier is not in `only`, then tries the
94
+ rest in the `prefer` order. `prefer` only changes the order; use `only` when
95
+ text must not reach a tier. A provider that cannot run
96
+ now (no WebGPU, server down, model missing, permission not granted) is skipped,
97
+ and the skip goes into `result.skipped`. When a provider fails during a call,
98
+ the router throws that error. It tries the next provider only when you set
99
+ `fallbackOnError: true`, and never after text has streamed or the caller has
100
+ aborted.
101
+
102
+ ```mermaid
103
+ sequenceDiagram
104
+ participant Panel as Demo panel (popup or sidebar)
105
+ participant BG as Background page
106
+ participant Mind as createMind
107
+ participant Server as llama-server
108
+ Panel->>BG: runtime.sendMessage({ op: "local-chat" })
109
+ BG->>Mind: chat(messages)
110
+ Mind->>Server: GET /v1/models (probe)
111
+ Mind->>Server: POST /v1/chat/completions
112
+ Server-->>Mind: reply
113
+ Mind-->>BG: reply, provider "llama-server", tier "local"
114
+ BG-->>Panel: answer and "answered by llama-server (local)"
115
+ ```
116
+
117
+ In an extension, the models live in the background page (an event page with a
118
+ DOM in Firefox). Each view asks it with `runtime.sendMessage`, so the views
119
+ share one loaded model.
120
+
121
+ ## API
122
+
123
+ ### `foxmind` (Node and the browser)
124
+
125
+ | Export | What it does |
126
+ |---|---|
127
+ | `createMind({ providers, only?, prefer?, fallbackOnError?, probeTtlMs? })` | Makes the router. `only` is the list of tiers it may use (private mode: `["browser", "local"]`); it never probes or calls the others. `prefer` takes provider names or tiers and only changes the order. |
128
+ | `mind.chat(messages, { tools?, json?, onDelta?, temperature?, maxTokens?, timeoutMs?, signal? })` | Chat in the OpenAI shape. `onDelta` streams text. `json: true` fails with `bad_json` when the reply is not JSON. Tool call arguments are checked: they must be a JSON object. `timeoutMs` is the longest wait for the headers and then for each next piece of the body, so a slow stream that keeps sending is not cut. |
129
+ | `mind.embed(texts)` | One vector per text: `{ vectors, provider, tier, model, ms, skipped }`. |
130
+ | `mind.extract(text, labels, { threshold? })` | GLiNER2 entities per label: `{ entities: { label: [{ text, confidence, start, end }] } }`. |
131
+ | `mind.classify(texts, prompt, labels)` | GLiNER2 label scores per text: `{ scores: [{ label: probability }] }`. |
132
+ | `mind.status()` | Each provider's state (`idle`, `loading`, `ready`, `unavailable`, `error`), where it runs, download progress, and the last provider that answered. |
133
+ | `mind.probe()` / `mind.load(name)` | Probe every provider now. Download and start one model now. |
134
+ | `openaiCompatible({ baseURL, model, apiKey?, embedModel?, body?, checkModel?, timeoutMs? })` | Any OpenAI-compatible server: llama-server, Ollama, LM Studio, OpenAI, OpenRouter. The tier is `local` for localhost, else `cloud`. |
135
+ | `ollama({ model })`, `llamaServer()`, `lmStudio({ model })` | Presets with the usual local address and the start command in each failed probe. |
136
+ | `saluki({ thinking? })`, `SALUKI` | Underdog Saluki 27B on llama-server, the default planner for private mode. Thinking is off and temperature is 0 by default, the model card's settings for tool calls. |
137
+ | `anthropic({ apiKey, model?, maxTokens? })` | The Anthropic Messages API, mapped to the OpenAI shape. The default model is `claude-opus-5-5`. |
138
+ | `doctor(options)`, `formatDoctor(report)` | The `foxmind doctor` check as a function. |
139
+ | `FoxmindError` | Every failure. `code` is one of `no_provider`, `unreachable`, `timeout`, `aborted`, `rate_limited` (with `retryAfterMs`), `auth`, `model_not_found`, `http`, `stream_interrupted` (with `partial`), `bad_tool_call` (with `raw`), `bad_json`, `bad_response`, `unsupported`, `download_failed`, `cache_corrupt`, `out_of_memory`, `webgpu_missing`, `permission`. |
140
+
141
+ To run Saluki, follow its [model card](https://huggingface.co/ConwayResearch/Underdog-Saluki-27B-1.0):
142
+
143
+ ```bash
144
+ huggingface-cli download ConwayResearch/Underdog-Saluki-27B-1.0 Underdog-Saluki-27B-1.0-IQ2-mix.gguf --local-dir .
145
+ llama-server -m Underdog-Saluki-27B-1.0-IQ2-mix.gguf --jinja -ngl 99 -fa on -c 32768
146
+ ```
147
+
148
+ ### `foxmind/browser` (extension pages and web pages)
149
+
150
+ | Export | What it does |
151
+ |---|---|
152
+ | `transformers({ task: "embed" \| "chat", model?, device?, dtype? })` | transformers.js. Defaults: `Xenova/all-MiniLM-L6-v2` for embed, `onnx-community/Qwen3-0.6B-ONNX` for chat. `device: "auto"` uses WebGPU when the browser has an adapter, else WASM, and `status().where` says which. |
153
+ | `gliner2({ model?, device?, dtype? })` | GLiNER2 extract and classify. Default model: `pooria/gliner2-multi-v1-agent-batch-ONNX` (614 MB at fp16). |
154
+ | `trialML({ task, model?, device? })`, `requestTrialML()` | Firefox's own engine, `browser.trial.ml`. Call `requestTrialML()` from a click to get the optional `trialML` permission. |
155
+ | `wllama({ model, modelFile, maxBytes? })` | Experimental. A GGUF model on Firefox's llama.cpp backend through trial ML. It refuses files over `maxBytes` (default 4 GB) before the download. |
156
+ | `configureRuntime({ wasmPaths?, remoteHost? })` | Where ONNX Runtime's WASM files are (default: `ort/` in the extension) and which model hub to use. |
157
+ | `hasWebGPU()`, `purgeModel(model)` | Detect a WebGPU adapter. Delete one model's files from Cache Storage. |
158
+
159
+ Each cached model file that does not load is deleted and downloaded once more,
160
+ and `status().reason` says so. A download that stops is not kept, so the next
161
+ call downloads again.
162
+
163
+ ### CLI: `foxmind doctor`
164
+
165
+ ```bash
166
+ npx foxmind doctor
167
+ ```
168
+
169
+ ```text
170
+ foxmind doctor · Node v25.9.0 · darwin arm64
171
+
172
+ local ollama yes http://127.0.0.1:11434/v1 models: qwen3:0.6b
173
+ local llama-server yes http://127.0.0.1:8080/v1 models: qwen3-0.6b.gguf
174
+ local saluki no http://127.0.0.1:8080/v1 llama-server serves qwen3-0.6b.gguf, not Saluki. Download it with "huggingface-cli download ConwayResearch/Underdog-Saluki-27B-1.0 Underdog-Saluki-27B-1.0-IQ2-mix.gguf --local-dir .", then start it with "llama-server -m Underdog-Saluki-27B-1.0-IQ2-mix.gguf --jinja -ngl 99 -fa on -c 32768".
175
+ local lm-studio no http://127.0.0.1:1234/v1 Cannot reach http://127.0.0.1:1234/v1/models: connect ECONNREFUSED 127.0.0.1:1234. Start it with "lms server start".
176
+ browser WebGPU, trial.ml and in-browser models: doctor runs in Node and cannot check them. Load the demo extension in Firefox.
177
+ cloud anthropic and cloud openaiCompatible: these need your own key. doctor does not read keys.
178
+ ```
179
+
180
+ Flags: `--json`, `--ollama URL`, `--llama-server URL`, `--lm-studio URL`,
181
+ `--timeout MS`. The exit code is 0 when at least one server works, 1 when none
182
+ does, and 2 for bad input. There is no MCP server.
183
+
184
+ ### Demo extension
185
+
186
+ `extension/` is a small Firefox extension. Its panel opens from the toolbar and
187
+ in the sidebar. It shows which tiers work, sends a test prompt to a local
188
+ server, and compares two sentences with an in-browser embedding model. Build it
189
+ with `pnpm build:ext` and load `dist-ext/manifest.json` from `about:debugging`.
190
+ `pnpm build:ext` makes the build AMO signs. `node scripts/build-ext.mjs --e2e`
191
+ makes the test build that `pnpm e2e` loads; only it has the test-only ops and
192
+ the content script. To let it reach Ollama, start Ollama with
193
+ `OLLAMA_ORIGINS="moz-extension://*"`.
194
+ Ollama refuses extension origins without it.
195
+
196
+ ## Tests
197
+
198
+ `pnpm ci:local` runs lint, typecheck, 71 tests and the build. The tests start
199
+ fake servers and call them over real HTTP. `docs/failure-modes.md` lists each
200
+ failure mode (F1 to F72) and its test. The tests went in before the code.
201
+
202
+ `pnpm e2e` runs the real thing and writes `artifacts/e2e-<date>.json`. First it
203
+ sends chat, tool calls, streams and JSON requests to each local server that
204
+ runs. Then it starts Firefox with the demo extension. `FOXMIND_E2E_HEAVY=1`
205
+ adds Qwen3-0.6B chat and GLiNER2 in the browser. Results from our run on
206
+ 2026-10-09, in `artifacts/e2e-2026-10-09.json` (Firefox 157.0.1, Apple M3 Pro,
207
+ headless, so WASM; all 32 checks passed):
208
+
209
+ | Check | Result |
210
+ |---|---|
211
+ | llama-server, Qwen3-0.6B GGUF: chat / tool call / stream | 82 ms / 632 ms / 210 ms |
212
+ | Ollama, qwen3:0.6b: chat / tool call / stream (it thinks first) | 1.7 s / 0.9 s / 1.3 s |
213
+ | MiniLM embedding in Firefox: first load with download / cached load / one call | 1,124 ms / 180 ms / 9 ms |
214
+ | Firefox trial ML, MiniLM embedding: first call / next call | 2.0 s / 17 ms |
215
+ | Qwen3-0.6B chat in Firefox (WASM, q4): first answer with download / tool call | 34.2 s / 28.3 s |
216
+ | GLiNER2 in Firefox (WASM, fp16): load with 614 MB download / extract / classify | 19.6 s / 1.6 s / 1.5 s |
217
+ | GLiNER2 against the Python gliner2 library | 14 of 14 reference cases match |
218
+ | Saluki 27B in the browser | refused before download: 7.90 GB is over the 4 GB limit |
219
+
220
+ After the review fixes (redirects, aborts, `only`, idle timeouts, the release
221
+ build), we ran `pnpm e2e` again without the heavy models:
222
+ `artifacts/e2e-review-fixes-2026-10-09.json`, 27 of 27 checks passed. The heavy
223
+ run was not repeated after the fixes: the test machine was under heavy load from
224
+ other jobs, and the in-browser Qwen3 call passed the test's 180 s limit.
225
+
226
+ ## Firefox APIs used
227
+
228
+ | API | MDN | Why |
229
+ |---|---|---|
230
+ | `browser.trial.ml` (`createEngine`, `runEngine`, `onProgress`) | No MDN page: [Firefox source docs](https://firefox-source-docs.mozilla.org/toolkit/components/ml/extensions.html) | `trialML()` and `wllama()` run Firefox's own inference engine. |
231
+ | `permissions.request` / `permissions.contains` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/API/permissions/request) | Ask for the optional `trialML` permission, and check it before each call. |
232
+ | `optional_permissions` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/manifest.json/optional_permissions) | `trialML` is an optional-only permission. |
233
+ | `host_permissions` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/manifest.json/host_permissions) | The demo calls local servers on `127.0.0.1` and `localhost`. |
234
+ | `runtime.sendMessage` / `runtime.onMessage` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/API/runtime/sendMessage) | The panel asks the background page, which hosts the models. |
235
+ | `runtime.getURL` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/API/runtime/getURL) | ONNX Runtime loads its WASM files from inside the extension, because MV3 allows no remote code. |
236
+ | Background scripts (event page) | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/Background_scripts) | The page that hosts the models. It has a DOM, Cache Storage and WebGPU. |
237
+ | `action` popup and `sidebar_action` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/manifest.json/sidebar_action) | The demo panel opens in both places. |
238
+ | `content_security_policy` with `'wasm-unsafe-eval'` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/manifest.json/content_security_policy) | Lets the extension compile WebAssembly. |
239
+ | `unlimitedStorage` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/manifest.json/permissions#unlimitedstorage) | Model files can be hundreds of MB. |
240
+ | `browser_specific_settings.gecko.data_collection_permissions` | [MDN](https://developer.mozilla.org/en-US/docs/Mozilla/Add-ons/WebExtensions/manifest.json/browser_specific_settings) | The demo collects no data. If your extension sends page text to a cloud key, declare `websiteContent` under `optional`. |
241
+ | WebGPU (`navigator.gpu.requestAdapter`) | [MDN](https://developer.mozilla.org/en-US/docs/Web/API/GPU/requestAdapter) | `device: "auto"` picks WebGPU only when there is an adapter. |
242
+ | WebAssembly | [MDN](https://developer.mozilla.org/en-US/docs/WebAssembly) | ONNX Runtime runs models on the CPU when there is no WebGPU. |
243
+ | `crossOriginIsolated` / `SharedArrayBuffer` | [MDN](https://developer.mozilla.org/en-US/docs/Web/API/Window/crossOriginIsolated) | Extension pages have no `SharedArrayBuffer`, so the runtime uses one WASM thread. |
244
+ | Cache Storage (`caches`) | [MDN](https://developer.mozilla.org/en-US/docs/Web/API/CacheStorage) | transformers.js caches model files there. foxmind deletes a model's files when they do not load. |
245
+ | `fetch`, `AbortSignal.timeout`, `AbortSignal.any` | [MDN](https://developer.mozilla.org/en-US/docs/Web/API/AbortSignal/any_static) | Every server call has a timeout and honors the caller's abort. |
246
+
247
+ ## Limits
248
+
249
+ - We did not run Saluki 27B. We tested the `saluki()` preset against a fake
250
+ server and checked that it refuses to run in the browser. The Saluki
251
+ benchmark numbers are from its vendor (Underdog Bench, 120 tasks), not from
252
+ us.
253
+ - A 27B model in the browser is not proven. `wllama()` refuses GGUF files over
254
+ 4 GB, and trial ML takes models only from the Mozilla and Xenova orgs.
255
+ - `wllama()` is experimental. In our Firefox 157 runs, the llama.cpp engine
256
+ started but never answered, so the call ends with code `timeout`.
257
+ - We tested `anthropic()` and cloud `openaiCompatible()` against fake servers
258
+ only. No real API key was used.
259
+ - Small in-browser chat models are slow on WASM (about 28 s per answer for
260
+ Qwen3-0.6B), and they call a tool only when the prompt asks for it.
261
+ - We ran GLiNER2 on WASM only. Its WebGPU read-back code comes from foxpilot
262
+ and has no foxmind test yet.
263
+ - For a model as small as MiniLM, WebGPU was not faster than WASM in our
264
+ headed run.
265
+ - `json: true` on Anthropic is a system-prompt rule plus a JSON check, not
266
+ constrained decoding.
267
+ - foxmind does not retry. On `rate_limited`, use `retryAfterMs` to decide.
268
+ - Firefox unloads an idle background page, and the loaded model goes with it.
269
+ An open popup or sidebar keeps the page loaded.
270
+ - Firefox allows one trial ML engine per extension.
271
+ - The in-browser providers do not run in Node. The server and cloud tiers do.
272
+ - Ollama can run a model on ollama.com. foxmind treats a model named
273
+ `*-cloud` or `*:cloud` as tier `cloud` (for `ollama()` and for
274
+ `openaiCompatible()` on localhost), so `only: ["browser", "local"]` drops it.
275
+ For other names, `ollama().probe()` reads `/api/tags` and refuses a model
276
+ with a `remote_host` unless you pass `tier: "cloud"`. Other local servers
277
+ that forward to a remote model are not detected.
278
+ - ONNX Runtime adds about 27 MB of WASM to an extension.
279
+ - foxmind does not check model files against a hash after the download, and
280
+ transformers.js does not either. Hugging Face lists a SHA-256 for each LFS
281
+ file in its API, so a check is possible later.
282
+
283
+ ## Part of the fox primitives
284
+
285
+ ```mermaid
286
+ flowchart LR
287
+ foxkit[foxkit] -- template --> foxmind[foxmind]
288
+ foxmind --> foxpaw[foxpaw]
289
+ foxmind --> foxloop[foxloop]
290
+ foxmind -. optional .-> foxshield[foxshield]
291
+ foxmind --> foxmemory[foxmemory]
292
+ foxmind --> foxlens[foxlens]
293
+ foxvault[foxvault] -. will hold the keys .-> foxmind
294
+ click foxkit "https://github.com/pooriaarab/foxkit"
295
+ click foxmind "https://github.com/pooriaarab/foxmind"
296
+ click foxpaw "https://github.com/pooriaarab/foxpaw"
297
+ click foxloop "https://github.com/pooriaarab/foxloop"
298
+ click foxshield "https://github.com/pooriaarab/foxshield"
299
+ click foxmemory "https://github.com/pooriaarab/foxmemory"
300
+ click foxlens "https://github.com/pooriaarab/foxlens"
301
+ click foxvault "https://github.com/pooriaarab/foxvault"
302
+ ```
303
+
304
+ foxmind depends on no other fox primitive. The GLiNER2 code comes from
305
+ [foxpilot](https://github.com/pooriaarab/foxpilot) (MIT, same author).
306
+
307
+ ## License
308
+
309
+ [MIT](LICENSE)
package/dist/bin.d.ts ADDED
@@ -0,0 +1,2 @@
1
+ #!/usr/bin/env node
2
+ export {};
package/dist/bin.js ADDED
@@ -0,0 +1,6 @@
1
+ #!/usr/bin/env node
2
+ import { main } from "./cli.js";
3
+ process.exitCode = await main(process.argv.slice(2), {
4
+ out: (text) => process.stdout.write(text),
5
+ err: (text) => process.stderr.write(text),
6
+ });
@@ -0,0 +1,57 @@
1
+ import type { Entity, Labels } from "../types.js";
2
+ export type Encoded = {
3
+ inputIds: number[];
4
+ wordPositions: number[];
5
+ schemaPositions: number[];
6
+ /** Character offsets of each text word in the (period-terminated) text. */
7
+ starts: number[];
8
+ ends: number[];
9
+ text: string;
10
+ };
11
+ export type EncodeTask = {
12
+ name: string;
13
+ marker: "[E]" | "[L]";
14
+ labels: Labels;
15
+ };
16
+ export declare class Gliner2 {
17
+ private readonly js;
18
+ private readonly model;
19
+ private readonly tokenizer;
20
+ private readonly ids;
21
+ private constructor();
22
+ static load(modelId: string, options?: {
23
+ device?: "webgpu" | "wasm";
24
+ dtype?: "fp32" | "fp16";
25
+ progress_callback?: (info: {
26
+ status?: string;
27
+ progress?: number;
28
+ }) => void;
29
+ session_options?: Record<string, unknown>;
30
+ }): Promise<Gliner2>;
31
+ private pieces;
32
+ /**
33
+ * `( [P] name [DESCRIPTION] label: desc … ( [E] label … ) ) [SEP_TEXT] words`.
34
+ * The prompt and labels keep their case and are tokenized whole; the text
35
+ * gets a terminal "." if it has none, is split with WORD and lowercased,
36
+ * and each word is tokenized on its own (its first piece is its position).
37
+ */
38
+ encode(input: string, task: EncodeTask): Encoded;
39
+ /**
40
+ * Runs the graph once on every row and reads back only `names`. Rows are
41
+ * padded to the longest with 0 (attention mask 0); a row's outputs past its
42
+ * own label and word counts are padding. Every output on the GPU is released.
43
+ */
44
+ private run;
45
+ /** Softmax over the labels, like ClassificationSchema().single(..., activation="softmax"). */
46
+ classify(text: string, name: string, labels: Labels): Promise<Record<string, number>>;
47
+ /**
48
+ * classify() for several texts against one prompt and label set, in one run
49
+ * of the graph and one read back. Each call costs a fixed 300-500 ms in
50
+ * Firefox (foxpilot #70); a row adds much less (foxpilot #90).
51
+ */
52
+ classifyMany(texts: string[], name: string, labels: Labels): Promise<Record<string, number>[]>;
53
+ /** extract_entities(text, types) with include_confidence and include_spans. */
54
+ extractEntities(text: string, types: Labels, threshold?: number): Promise<Record<string, Entity[]>>;
55
+ }
56
+ /** gliner2's default span decoder: confidence-first greedy, no character overlaps. */
57
+ export declare function finalizeSpans(raw: Entity[]): Entity[];
@@ -0,0 +1,233 @@
1
+ import { transformersJs } from "./runtime.js";
2
+ /** Same pattern as gliner2's WhitespaceTokenSplitter. */
3
+ const WORD = /(?:https?:\/\/[^\s]+|www\.[^\s]+)|[a-z0-9._%+-]+@[a-z0-9.-]+\.[a-z]{2,}|@[a-z0-9_]+|\w+(?:[-_]\w+)*|\S/giu;
4
+ const SPECIAL = ["[P]", "[E]", "[L]", "[SEP_TEXT]", "[DESCRIPTION]"];
5
+ export class Gliner2 {
6
+ js;
7
+ model;
8
+ tokenizer;
9
+ ids;
10
+ constructor(js, model, tokenizer, ids) {
11
+ this.js = js;
12
+ this.model = model;
13
+ this.tokenizer = tokenizer;
14
+ this.ids = ids;
15
+ }
16
+ static async load(modelId, options = {}) {
17
+ const js = await transformersJs();
18
+ const device = options.device ?? "webgpu";
19
+ const tokenizer = await js.AutoTokenizer.from_pretrained(modelId);
20
+ const model = await js.AutoModel.from_pretrained(modelId, {
21
+ device,
22
+ dtype: options.dtype ?? "fp16",
23
+ progress_callback: options.progress_callback,
24
+ // On WebGPU, outputs stay on the GPU. Reading every output back costs
25
+ // about 200 ms per call in Firefox (foxpilot #70), so run() reads only what it needs.
26
+ session_options: {
27
+ ...(device === "webgpu" ? { preferredOutputLocation: "gpu-buffer" } : {}),
28
+ ...options.session_options,
29
+ },
30
+ });
31
+ // Added tokens encode to exactly one id.
32
+ const ids = Object.fromEntries(SPECIAL.map((token) => {
33
+ const { input_ids } = tokenizer(token, { add_special_tokens: false });
34
+ const encoded = Array.from(input_ids.data, Number);
35
+ if (encoded.length !== 1)
36
+ throw new Error(`tokenizer has no single id for ${token}`);
37
+ return [token, encoded[0]];
38
+ }));
39
+ return new Gliner2(js, model, tokenizer, ids);
40
+ }
41
+ pieces(text) {
42
+ const { input_ids } = this.tokenizer(text, { add_special_tokens: false });
43
+ return Array.from(input_ids.data, Number);
44
+ }
45
+ /**
46
+ * `( [P] name [DESCRIPTION] label: desc … ( [E] label … ) ) [SEP_TEXT] words`.
47
+ * The prompt and labels keep their case and are tokenized whole; the text
48
+ * gets a terminal "." if it has none, is split with WORD and lowercased,
49
+ * and each word is tokenized on its own (its first piece is its position).
50
+ */
51
+ encode(input, task) {
52
+ const text = !input ? "." : /[.!?]$/.test(input) ? input : `${input}.`;
53
+ const labelNames = Object.keys(task.labels);
54
+ const prompt = task.name +
55
+ labelNames
56
+ .map((label) => (task.labels[label] ? ` [DESCRIPTION] ${label}: ${task.labels[label]}` : ""))
57
+ .join("");
58
+ const inputIds = [];
59
+ const schemaPositions = [];
60
+ const push = (ids) => inputIds.push(...ids);
61
+ const special = (token) => {
62
+ if (token === "[P]" || token === task.marker)
63
+ schemaPositions.push(inputIds.length);
64
+ inputIds.push(this.ids[token]);
65
+ };
66
+ push(this.pieces("("));
67
+ special("[P]");
68
+ // Whole string, as the processor does; the tokenizer maps [DESCRIPTION] itself.
69
+ push(this.pieces(prompt));
70
+ push(this.pieces("("));
71
+ for (const label of labelNames) {
72
+ special(task.marker);
73
+ push(this.pieces(label));
74
+ }
75
+ push(this.pieces(")"));
76
+ push(this.pieces(")"));
77
+ special("[SEP_TEXT]");
78
+ const wordPositions = [];
79
+ const starts = [];
80
+ const ends = [];
81
+ for (const match of text.matchAll(WORD)) {
82
+ wordPositions.push(inputIds.length);
83
+ starts.push(match.index);
84
+ ends.push(match.index + match[0].length);
85
+ push(this.pieces(match[0].toLowerCase()));
86
+ }
87
+ return { inputIds, wordPositions, schemaPositions, starts, ends, text };
88
+ }
89
+ /**
90
+ * Runs the graph once on every row and reads back only `names`. Rows are
91
+ * padded to the longest with 0 (attention mask 0); a row's outputs past its
92
+ * own label and word counts are padding. Every output on the GPU is released.
93
+ */
94
+ async run(rows, names) {
95
+ const long = (values) => {
96
+ const n = Math.max(...values.map((row) => row.length));
97
+ const data = BigInt64Array.from(values.flatMap((row) => [...row, ...Array.from({ length: n - row.length }, () => 0)]), BigInt);
98
+ return new this.js.Tensor("int64", data, [values.length, n]);
99
+ };
100
+ const outputs = (await this.model({
101
+ input_ids: long(rows.map((row) => row.inputIds)),
102
+ attention_mask: long(rows.map((row) => Array.from({ length: row.inputIds.length }, () => 1))),
103
+ word_positions: long(rows.map((row) => row.wordPositions)),
104
+ schema_positions: long(rows.map((row) => row.schemaPositions)),
105
+ }));
106
+ try {
107
+ const read = await readBack(this.js, names.map((name) => outputs[name]));
108
+ return Object.fromEntries(names.map((name, i) => [name, { data: Array.from(read[i].to("float32").data), dims: read[i].dims }]));
109
+ }
110
+ finally {
111
+ for (const tensor of Object.values(outputs))
112
+ if (tensor.location === "gpu-buffer")
113
+ tensor.dispose();
114
+ }
115
+ }
116
+ /** Softmax over the labels, like ClassificationSchema().single(..., activation="softmax"). */
117
+ async classify(text, name, labels) {
118
+ return (await this.classifyMany([text], name, labels))[0];
119
+ }
120
+ /**
121
+ * classify() for several texts against one prompt and label set, in one run
122
+ * of the graph and one read back. Each call costs a fixed 300-500 ms in
123
+ * Firefox (foxpilot #70); a row adds much less (foxpilot #90).
124
+ */
125
+ async classifyMany(texts, name, labels) {
126
+ if (!texts.length)
127
+ return [];
128
+ const rows = texts.map((text) => this.encode(text, { name, marker: "[L]", labels }));
129
+ const read = (await this.run(rows, ["cls_logits"])).cls_logits;
130
+ const width = read.dims[1];
131
+ const names = Object.keys(labels);
132
+ const results = texts.map((_, row) => {
133
+ const cls = read.data.slice(row * width, row * width + names.length);
134
+ const max = Math.max(...cls);
135
+ const exp = cls.map((x) => Math.exp(x - max));
136
+ const sum = exp.reduce((a, b) => a + b, 0);
137
+ return Object.fromEntries(names.map((label, i) => [label, exp[i] / sum]));
138
+ });
139
+ return results;
140
+ }
141
+ /** extract_entities(text, types) with include_confidence and include_spans. */
142
+ async extractEntities(text, types, threshold = 0.5) {
143
+ const encoded = this.encode(text, { name: "entities", marker: "[E]", labels: types });
144
+ const read = await this.run([encoded], ["count_logits", "span_logits"]);
145
+ const count = read.count_logits.data;
146
+ const span = read.span_logits.data;
147
+ const names = Object.keys(types);
148
+ const result = Object.fromEntries(names.map((name) => [name, []]));
149
+ if (argmax(count) <= 0)
150
+ return result;
151
+ const [, , words, width] = read.span_logits.dims;
152
+ names.forEach((name, li) => {
153
+ const raw = [];
154
+ for (let start = 0; start < words; start++) {
155
+ for (let w = 0; w < width; w++) {
156
+ const end = start + w + 1;
157
+ if (end > words)
158
+ continue;
159
+ const confidence = sigmoid(span[(li * words + start) * width + w]);
160
+ if (confidence < threshold)
161
+ continue;
162
+ const charStart = encoded.starts[start];
163
+ const charEnd = encoded.ends[end - 1];
164
+ const surface = encoded.text.slice(charStart, charEnd).trim();
165
+ if (surface)
166
+ raw.push({ text: surface, confidence, start: charStart, end: charEnd });
167
+ }
168
+ }
169
+ result[name] = finalizeSpans(raw);
170
+ });
171
+ return result;
172
+ }
173
+ }
174
+ /**
175
+ * Copies every GPU tensor in `tensors` to the CPU with one mapAsync: each
176
+ * GPU to CPU round trip costs about 90 ms in Firefox (foxpilot #70). CPU tensors pass
177
+ * through unchanged.
178
+ */
179
+ async function readBack(js, tensors) {
180
+ const onGpu = tensors.filter((tensor) => tensor.location === "gpu-buffer");
181
+ if (!onGpu.length)
182
+ return tensors;
183
+ const device = js.env.backends.onnx.webgpu?.device;
184
+ if (!device)
185
+ throw new Error("ONNX Runtime has no WebGPU device");
186
+ // ONNX Runtime pads GPU buffers to 16 bytes, so each padded copy stays inside its buffer.
187
+ const bytes = onGpu.map((tensor) => {
188
+ if (tensor.type !== "float32" && tensor.type !== "float16")
189
+ throw new Error(`cannot read back ${tensor.type}`);
190
+ return tensor.size * (tensor.type === "float16" ? 2 : 4);
191
+ });
192
+ const padded = bytes.map((size) => Math.ceil(size / 16) * 16);
193
+ const offsets = padded.map((_, i) => padded.slice(0, i).reduce((a, b) => a + b, 0));
194
+ // MAP_READ | COPY_DST
195
+ const staging = device.createBuffer({ size: padded.reduce((a, b) => a + b, 0), usage: 0x01 | 0x08 });
196
+ try {
197
+ const encoder = device.createCommandEncoder();
198
+ onGpu.forEach((tensor, i) => encoder.copyBufferToBuffer(tensor.ort_tensor.gpuBuffer, 0, staging, offsets[i], padded[i]));
199
+ device.queue.submit([encoder.finish()]);
200
+ await staging.mapAsync(0x01); // GPUMapMode.READ
201
+ const data = staging.getMappedRange().slice(0);
202
+ return tensors.map((tensor) => {
203
+ const i = onGpu.indexOf(tensor);
204
+ if (i < 0)
205
+ return tensor;
206
+ const values = tensor.type === "float16" ? new Uint16Array(data, offsets[i], tensor.size) : new Float32Array(data, offsets[i], tensor.size);
207
+ return new js.Tensor(tensor.type, values, tensor.dims);
208
+ });
209
+ }
210
+ finally {
211
+ staging.destroy();
212
+ }
213
+ }
214
+ /** gliner2's default span decoder: confidence-first greedy, no character overlaps. */
215
+ export function finalizeSpans(raw) {
216
+ const kept = [];
217
+ for (const candidate of raw.toSorted((a, b) => b.confidence - a.confidence)) {
218
+ if (kept.some((existing) => candidate.start < existing.end && existing.start < candidate.end))
219
+ continue;
220
+ kept.push(candidate);
221
+ }
222
+ return kept;
223
+ }
224
+ function sigmoid(x) {
225
+ return x >= 0 ? 1 / (1 + Math.exp(-x)) : Math.exp(x) / (1 + Math.exp(x));
226
+ }
227
+ function argmax(values) {
228
+ let best = 0;
229
+ for (let i = 1; i < values.length; i++)
230
+ if (values[i] > values[best])
231
+ best = i;
232
+ return best;
233
+ }
@@ -0,0 +1,13 @@
1
+ import type { Provider } from "../types.js";
2
+ import { type Device } from "./runtime.js";
3
+ export interface Gliner2Options {
4
+ /** Default "pooria/gliner2-multi-v1-agent-batch-ONNX". */
5
+ model?: string;
6
+ /** Default "auto": WebGPU when the browser has it, else WASM. */
7
+ device?: Device;
8
+ /** Default "fp16". */
9
+ dtype?: "fp32" | "fp16";
10
+ /** Default "gliner2". */
11
+ name?: string;
12
+ }
13
+ export declare function gliner2(options?: Gliner2Options): Provider;