foxmind 0.0.0-stage → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +308 -2
- package/dist/bin.d.ts +2 -0
- package/dist/bin.js +6 -0
- package/dist/browser/gliner2-model.d.ts +57 -0
- package/dist/browser/gliner2-model.js +233 -0
- package/dist/browser/gliner2.d.ts +13 -0
- package/dist/browser/gliner2.js +50 -0
- package/dist/browser/index.d.ts +4 -0
- package/dist/browser/index.js +6 -0
- package/dist/browser/runtime.d.ts +61 -0
- package/dist/browser/runtime.js +205 -0
- package/dist/browser/toolcalls.d.ts +6 -0
- package/dist/browser/toolcalls.js +20 -0
- package/dist/browser/transformers.d.ts +18 -0
- package/dist/browser/transformers.js +81 -0
- package/dist/browser/trialml.d.ts +32 -0
- package/dist/browser/trialml.js +196 -0
- package/dist/cli.d.ts +6 -0
- package/dist/cli.js +46 -0
- package/dist/doctor.d.ts +23 -0
- package/dist/doctor.js +58 -0
- package/dist/errors.d.ts +48 -0
- package/dist/errors.js +36 -0
- package/dist/http.d.ts +46 -0
- package/dist/http.js +164 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.js +7 -0
- package/dist/mind.d.ts +60 -0
- package/dist/mind.js +122 -0
- package/dist/providers/anthropic.d.ts +16 -0
- package/dist/providers/anthropic.js +200 -0
- package/dist/providers/openai.d.ts +39 -0
- package/dist/providers/openai.js +177 -0
- package/dist/providers/presets.d.ts +33 -0
- package/dist/providers/presets.js +89 -0
- package/dist/reply.d.ts +10 -0
- package/dist/reply.js +60 -0
- package/dist/sse.d.ts +6 -0
- package/dist/sse.js +38 -0
- package/dist/types.d.ts +106 -0
- package/dist/types.js +3 -0
- package/package.json +59 -4
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
// Firefox's own on-device inference engine, browser.trial.ml. It needs the
|
|
2
|
+
// optional "trialML" permission, takes models only from the Mozilla hub and the
|
|
3
|
+
// Mozilla and Xenova orgs on Hugging Face, and allows one engine per extension.
|
|
4
|
+
import { FoxmindError } from "../errors.js";
|
|
5
|
+
import { failure } from "../http.js";
|
|
6
|
+
import { checkReply } from "../reply.js";
|
|
7
|
+
import { bounded } from "./runtime.js";
|
|
8
|
+
const GB = 1e9;
|
|
9
|
+
/** The size of one file in a Hugging Face repo, from the hub API. */
|
|
10
|
+
async function fileSize(model, file) {
|
|
11
|
+
const response = await fetch(`https://huggingface.co/api/models/${model}?blobs=true`).catch(() => undefined);
|
|
12
|
+
if (!response?.ok)
|
|
13
|
+
return undefined;
|
|
14
|
+
const info = (await response.json());
|
|
15
|
+
return info.siblings?.find((sibling) => sibling.rfilename === file)?.size;
|
|
16
|
+
}
|
|
17
|
+
const api = () => globalThis.browser;
|
|
18
|
+
/** Ask the user for the optional trialML permission. Call it from a click handler. */
|
|
19
|
+
export function requestTrialML() {
|
|
20
|
+
const permissions = api()?.permissions;
|
|
21
|
+
if (!permissions)
|
|
22
|
+
return Promise.resolve(false);
|
|
23
|
+
return permissions.request({ permissions: ["trialML"] });
|
|
24
|
+
}
|
|
25
|
+
/** Firefox allows one engine per extension, so the first provider to start one keeps it. */
|
|
26
|
+
let engine;
|
|
27
|
+
/** The status that engine progress goes to. One listener per page, however often the engine restarts. */
|
|
28
|
+
let progressTo;
|
|
29
|
+
let listening = false;
|
|
30
|
+
/** One vector per text from whatever shape the engine returns. */
|
|
31
|
+
export function toVectors(result, count) {
|
|
32
|
+
let rows = result;
|
|
33
|
+
const tensor = result;
|
|
34
|
+
if (tensor && !Array.isArray(result) && tensor.data && tensor.dims) {
|
|
35
|
+
const width = tensor.dims.at(-1);
|
|
36
|
+
const flat = Array.from(tensor.data);
|
|
37
|
+
rows = Array.from({ length: flat.length / width }, (_, i) => flat.slice(i * width, (i + 1) * width));
|
|
38
|
+
}
|
|
39
|
+
while (Array.isArray(rows) && rows.length === 1 && Array.isArray(rows[0]) && Array.isArray(rows[0][0]) && count === rows[0].length)
|
|
40
|
+
rows = rows[0];
|
|
41
|
+
const ok = Array.isArray(rows) && rows.length === count && rows.every((row) => Array.isArray(row) && row.every((x) => typeof x === "number"));
|
|
42
|
+
return ok ? rows : undefined;
|
|
43
|
+
}
|
|
44
|
+
/** Why trial.ml cannot run, or undefined when it can. The namespace exists only after the grant. */
|
|
45
|
+
async function unavailable() {
|
|
46
|
+
const permissions = api()?.permissions;
|
|
47
|
+
if (permissions && !(await permissions.contains({ permissions: ["trialML"] }).catch(() => false))) {
|
|
48
|
+
return { code: "permission", reason: 'The optional "trialML" permission is not granted. Call requestTrialML() from a click.' };
|
|
49
|
+
}
|
|
50
|
+
if (!api()?.trial?.ml)
|
|
51
|
+
return { code: "unsupported", reason: "browser.trial.ml is missing: this is not Firefox, or trial ML is turned off." };
|
|
52
|
+
return undefined;
|
|
53
|
+
}
|
|
54
|
+
/** The text in a text-generation result, in any of the shapes the backends return. */
|
|
55
|
+
function generatedText(output) {
|
|
56
|
+
if (typeof output === "string")
|
|
57
|
+
return output;
|
|
58
|
+
const record = output;
|
|
59
|
+
if (typeof record?.finalOutput === "string")
|
|
60
|
+
return record.finalOutput;
|
|
61
|
+
if (typeof record?.output === "string")
|
|
62
|
+
return record.output;
|
|
63
|
+
const text = output?.[0]?.generated_text;
|
|
64
|
+
return Array.isArray(text) ? text.at(-1)?.content : text;
|
|
65
|
+
}
|
|
66
|
+
export function trialML(options) {
|
|
67
|
+
const chat = options.task === "chat";
|
|
68
|
+
const model = options.model ?? (chat ? "Xenova/Qwen1.5-0.5B-Chat" : "Xenova/all-MiniLM-L6-v2");
|
|
69
|
+
const name = options.name ?? `trialml-${options.task}`;
|
|
70
|
+
const device = options.device ?? "wasm";
|
|
71
|
+
const origin = { provider: name, tier: "browser", secrets: [] };
|
|
72
|
+
const llama = options.backend === "llama.cpp";
|
|
73
|
+
const key = `${options.task}:${model}:${options.modelFile ?? ""}:${device}`;
|
|
74
|
+
/** Checks that need no engine: the file size, and the orgs trial ML takes models from. */
|
|
75
|
+
async function refused() {
|
|
76
|
+
if (llama && options.modelFile) {
|
|
77
|
+
const size = await fileSize(model, options.modelFile);
|
|
78
|
+
const max = options.maxBytes ?? 4 * GB;
|
|
79
|
+
if (size !== undefined && size > max) {
|
|
80
|
+
return { code: "out_of_memory", reason: `${options.modelFile} is ${(size / GB).toFixed(2)} GB, more than the ${(max / GB).toFixed(2)} GB limit (maxBytes) for a model in the browser. Run it on a local server instead.` };
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
if (!/^(Mozilla|Xenova)\//.test(model))
|
|
84
|
+
return { code: "unsupported", reason: `trial.ml takes models only from the Mozilla and Xenova orgs on Hugging Face (or the Mozilla hub), not ${model}.` };
|
|
85
|
+
return undefined;
|
|
86
|
+
}
|
|
87
|
+
const state = { state: "idle" };
|
|
88
|
+
async function ready() {
|
|
89
|
+
const blocked = (await unavailable()) ?? (await refused());
|
|
90
|
+
if (blocked)
|
|
91
|
+
throw failure(origin, blocked.code, blocked.reason);
|
|
92
|
+
const ml = api().trial.ml;
|
|
93
|
+
if (engine && engine.key !== key)
|
|
94
|
+
throw failure(origin, "unsupported", `Firefox allows one trial.ml engine per extension, and ${engine.owner} holds it (${engine.key}).`);
|
|
95
|
+
if (!engine) {
|
|
96
|
+
state.state = "loading";
|
|
97
|
+
progressTo = state;
|
|
98
|
+
if (!listening) {
|
|
99
|
+
listening = true;
|
|
100
|
+
ml.onProgress.addListener((data) => {
|
|
101
|
+
const progress = Number(data.progress ?? data.totalProgress);
|
|
102
|
+
if (progressTo && Number.isFinite(progress))
|
|
103
|
+
progressTo.progress = progress > 1 ? progress / 100 : progress;
|
|
104
|
+
});
|
|
105
|
+
}
|
|
106
|
+
const request = { taskName: chat ? "text-generation" : "feature-extraction", modelHub: "huggingface", modelId: model, ...(llama ? { backend: "llama.cpp", modelFile: options.modelFile } : { device }) };
|
|
107
|
+
const started = { owner: name, key, ready: ml.createEngine(request) };
|
|
108
|
+
engine = started;
|
|
109
|
+
started.ready.catch(() => { if (engine === started)
|
|
110
|
+
engine = undefined; });
|
|
111
|
+
}
|
|
112
|
+
try {
|
|
113
|
+
await engine.ready;
|
|
114
|
+
}
|
|
115
|
+
catch (error) {
|
|
116
|
+
throw mapped(error);
|
|
117
|
+
}
|
|
118
|
+
Object.assign(state, { state: "ready", progress: 1 });
|
|
119
|
+
return ml;
|
|
120
|
+
}
|
|
121
|
+
function mapped(error) {
|
|
122
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
123
|
+
state.state = "error";
|
|
124
|
+
state.reason = message;
|
|
125
|
+
if (/disabled/i.test(message))
|
|
126
|
+
return failure(origin, "unsupported", `Trial ML is turned off in this Firefox: ${message}`);
|
|
127
|
+
if (/memory/i.test(message))
|
|
128
|
+
return failure(origin, "out_of_memory", `Firefox stopped the engine for lack of memory: ${message}`);
|
|
129
|
+
return failure(origin, "bad_response", `trial.ml failed: ${message}`);
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* One engine call. It stops waiting at the caller's abort or after
|
|
133
|
+
* timeoutMs (default 120 s): an engine that never answers must not hang the
|
|
134
|
+
* caller. Firefox has no call to stop the engine itself.
|
|
135
|
+
*/
|
|
136
|
+
async function run(args, runOptions, callOptions = {}) {
|
|
137
|
+
const ml = await bounded(origin, ready(), { signal: callOptions.signal });
|
|
138
|
+
try {
|
|
139
|
+
return await bounded(origin, ml.runEngine({ args, options: runOptions, ...(llama ? runOptions : {}) }), { signal: callOptions.signal, timeoutMs: callOptions.timeoutMs ?? 120_000 });
|
|
140
|
+
}
|
|
141
|
+
catch (error) {
|
|
142
|
+
if (error instanceof FoxmindError)
|
|
143
|
+
throw error;
|
|
144
|
+
throw mapped(error);
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
return {
|
|
148
|
+
name,
|
|
149
|
+
tier: "browser",
|
|
150
|
+
model,
|
|
151
|
+
capabilities: chat ? ["chat"] : ["embed"],
|
|
152
|
+
async probe() {
|
|
153
|
+
const blocked = (await unavailable()) ?? (await refused());
|
|
154
|
+
if (blocked)
|
|
155
|
+
return { ok: false, ...blocked };
|
|
156
|
+
if (engine && engine.key !== key)
|
|
157
|
+
return { ok: false, code: "unsupported", reason: `${engine.owner} holds the one trial.ml engine.` };
|
|
158
|
+
return { ok: true, where: `trial.ml (${device})` };
|
|
159
|
+
},
|
|
160
|
+
status: () => ({ name, tier: "browser", model, capabilities: chat ? ["chat"] : ["embed"], where: `trial.ml (${device})`, ...state }),
|
|
161
|
+
async load() {
|
|
162
|
+
await ready();
|
|
163
|
+
},
|
|
164
|
+
...(chat
|
|
165
|
+
? {
|
|
166
|
+
async chat(messages, chatOptions) {
|
|
167
|
+
if (chatOptions.tools?.length)
|
|
168
|
+
throw failure(origin, "unsupported", "trial.ml has no tool calls. Use transformers() or a server for tools.");
|
|
169
|
+
const plain = messages.map(({ role, content }) => ({ role, content }));
|
|
170
|
+
const maxTokens = chatOptions.maxTokens ?? 256;
|
|
171
|
+
// The ONNX backend takes { args, options }; the llama.cpp backend takes { prompt, nPredict }.
|
|
172
|
+
const output = await run([plain], llama ? { prompt: plain, nPredict: maxTokens } : { max_new_tokens: maxTokens }, chatOptions);
|
|
173
|
+
const content = generatedText(output);
|
|
174
|
+
if (typeof content !== "string")
|
|
175
|
+
throw failure(origin, "bad_response", "trial.ml returned no generated text.");
|
|
176
|
+
return checkReply(origin, { message: { role: "assistant", content }, finishReason: "stop" }, chatOptions.json);
|
|
177
|
+
},
|
|
178
|
+
}
|
|
179
|
+
: {
|
|
180
|
+
async embed(texts, embedOptions) {
|
|
181
|
+
const vectors = toVectors(await run([texts], { pooling: "mean", normalize: true }, embedOptions), texts.length);
|
|
182
|
+
if (!vectors)
|
|
183
|
+
throw failure(origin, "bad_response", "trial.ml returned a shape that is not one vector per text.");
|
|
184
|
+
return vectors;
|
|
185
|
+
},
|
|
186
|
+
}),
|
|
187
|
+
};
|
|
188
|
+
}
|
|
189
|
+
/**
|
|
190
|
+
* Experimental: a GGUF model on Firefox's llama.cpp backend, through trial ML.
|
|
191
|
+
* Small models only: trial ML takes models from the Mozilla and Xenova orgs,
|
|
192
|
+
* and foxmind refuses files over `maxBytes` (default 4 GB) before the download.
|
|
193
|
+
*/
|
|
194
|
+
export function wllama(options) {
|
|
195
|
+
return trialML({ task: "chat", backend: "llama.cpp", name: options.name ?? "wllama", ...options });
|
|
196
|
+
}
|
package/dist/cli.d.ts
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
export declare const USAGE = "Usage: foxmind doctor [--json] [--ollama URL] [--llama-server URL] [--lm-studio URL] [--timeout MS]\n\nChecks which local model servers run on this machine and lists their models.\nExit code 0 when at least one server works, 1 when none does, 2 for bad input.\n";
|
|
2
|
+
export interface Output {
|
|
3
|
+
out(text: string): void;
|
|
4
|
+
err(text: string): void;
|
|
5
|
+
}
|
|
6
|
+
export declare function main(argv: string[], output: Output): Promise<number>;
|
package/dist/cli.js
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
// The foxmind command. Today it has one subcommand: doctor.
|
|
2
|
+
import { doctor, format } from "./doctor.js";
|
|
3
|
+
export const USAGE = `Usage: foxmind doctor [--json] [--ollama URL] [--llama-server URL] [--lm-studio URL] [--timeout MS]
|
|
4
|
+
|
|
5
|
+
Checks which local model servers run on this machine and lists their models.
|
|
6
|
+
Exit code 0 when at least one server works, 1 when none does, 2 for bad input.
|
|
7
|
+
`;
|
|
8
|
+
const URL_FLAGS = { "--ollama": "ollama", "--llama-server": "llamaServer", "--lm-studio": "lmStudio" };
|
|
9
|
+
export async function main(argv, output) {
|
|
10
|
+
const [command, ...rest] = argv;
|
|
11
|
+
if (command === "--help" || command === "-h") {
|
|
12
|
+
output.out(USAGE);
|
|
13
|
+
return 0;
|
|
14
|
+
}
|
|
15
|
+
const bad = (why) => {
|
|
16
|
+
output.err(`${why}\n\n${USAGE}`);
|
|
17
|
+
return 2;
|
|
18
|
+
};
|
|
19
|
+
if (command !== "doctor")
|
|
20
|
+
return bad(command ? `Unknown command "${command}".` : "Give a command.");
|
|
21
|
+
const options = {};
|
|
22
|
+
let asJson = false;
|
|
23
|
+
for (let i = 0; i < rest.length; i++) {
|
|
24
|
+
const flag = rest[i];
|
|
25
|
+
const value = rest[i + 1];
|
|
26
|
+
if (flag === "--json")
|
|
27
|
+
asJson = true;
|
|
28
|
+
else if (flag in URL_FLAGS) {
|
|
29
|
+
if (!value || !/^https?:\/\//.test(value))
|
|
30
|
+
return bad(`${flag} needs an http:// or https:// URL.`);
|
|
31
|
+
options[URL_FLAGS[flag]] = value;
|
|
32
|
+
i++;
|
|
33
|
+
}
|
|
34
|
+
else if (flag === "--timeout") {
|
|
35
|
+
if (!value || !/^\d+$/.test(value))
|
|
36
|
+
return bad("--timeout needs a number of milliseconds.");
|
|
37
|
+
options.timeoutMs = Number(value);
|
|
38
|
+
i++;
|
|
39
|
+
}
|
|
40
|
+
else
|
|
41
|
+
return bad(`Unknown flag "${flag}".`);
|
|
42
|
+
}
|
|
43
|
+
const report = await doctor(options);
|
|
44
|
+
output.out(asJson ? `${JSON.stringify(report, null, 2)}\n` : format(report));
|
|
45
|
+
return report.checks.some((check) => check.ok) ? 0 : 1;
|
|
46
|
+
}
|
package/dist/doctor.d.ts
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
export interface DoctorOptions {
|
|
2
|
+
ollama?: string;
|
|
3
|
+
llamaServer?: string;
|
|
4
|
+
lmStudio?: string;
|
|
5
|
+
timeoutMs?: number;
|
|
6
|
+
}
|
|
7
|
+
export interface Check {
|
|
8
|
+
tier: "local";
|
|
9
|
+
name: string;
|
|
10
|
+
ok: boolean;
|
|
11
|
+
where: string;
|
|
12
|
+
models?: string[];
|
|
13
|
+
reason?: string;
|
|
14
|
+
}
|
|
15
|
+
export interface Report {
|
|
16
|
+
node: string;
|
|
17
|
+
platform: string;
|
|
18
|
+
checks: Check[];
|
|
19
|
+
browser: string;
|
|
20
|
+
cloud: string;
|
|
21
|
+
}
|
|
22
|
+
export declare function doctor(options?: DoctorOptions): Promise<Report>;
|
|
23
|
+
export declare function format(report: Report): string;
|
package/dist/doctor.js
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
// What works on this machine: probes each local server and lists its models.
|
|
2
|
+
// It never reads API keys, so it cannot print one.
|
|
3
|
+
import { call } from "./http.js";
|
|
4
|
+
import { SALUKI } from "./providers/presets.js";
|
|
5
|
+
const START = {
|
|
6
|
+
ollama: 'Start it with "ollama serve".',
|
|
7
|
+
"llama-server": 'Start it with "llama-server -m <model.gguf> --jinja".',
|
|
8
|
+
"lm-studio": 'Start it with "lms server start".',
|
|
9
|
+
};
|
|
10
|
+
function api(url) {
|
|
11
|
+
const trimmed = url.replace(/\/+$/, "");
|
|
12
|
+
return trimmed.endsWith("/v1") ? trimmed : `${trimmed}/v1`;
|
|
13
|
+
}
|
|
14
|
+
async function models(name, where, timeoutMs) {
|
|
15
|
+
const origin = { provider: name, tier: "local", secrets: [] };
|
|
16
|
+
try {
|
|
17
|
+
const fetched = await call(origin, `${where}/models`, { timeoutMs }, timeoutMs);
|
|
18
|
+
const ids = ((await fetched.json()).data ?? []).map((model) => model.id);
|
|
19
|
+
return { tier: "local", name, ok: true, where, models: ids };
|
|
20
|
+
}
|
|
21
|
+
catch (error) {
|
|
22
|
+
const message = error instanceof Error ? error.message.replace(/^[^:]+\(local\): /, "") : String(error);
|
|
23
|
+
return { tier: "local", name, ok: false, where, reason: `${message.replace(/\.?$/, ".")} ${START[name]}` };
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
export async function doctor(options = {}) {
|
|
27
|
+
const timeoutMs = options.timeoutMs ?? 2000;
|
|
28
|
+
const llama = api(options.llamaServer ?? "http://127.0.0.1:8080");
|
|
29
|
+
const [ollama, llamaServer, lmStudio] = await Promise.all([
|
|
30
|
+
models("ollama", api(options.ollama ?? "http://127.0.0.1:11434"), timeoutMs),
|
|
31
|
+
models("llama-server", llama, timeoutMs),
|
|
32
|
+
models("lm-studio", api(options.lmStudio ?? "http://127.0.0.1:1234"), timeoutMs),
|
|
33
|
+
]);
|
|
34
|
+
const serving = llamaServer.models?.find((id) => /saluki/i.test(id));
|
|
35
|
+
const saluki = serving
|
|
36
|
+
? { tier: "local", name: "saluki", ok: true, where: llama, models: [serving] }
|
|
37
|
+
: {
|
|
38
|
+
tier: "local",
|
|
39
|
+
name: "saluki",
|
|
40
|
+
ok: false,
|
|
41
|
+
where: llama,
|
|
42
|
+
reason: `${llamaServer.ok ? `llama-server serves ${llamaServer.models?.join(", ") || "no model"}, not Saluki.` : "llama-server is not running."} Download it with "${SALUKI.download}", then start it with "${SALUKI.serve}".`,
|
|
43
|
+
};
|
|
44
|
+
return {
|
|
45
|
+
node: process.version,
|
|
46
|
+
platform: `${process.platform} ${process.arch}`,
|
|
47
|
+
checks: [ollama, llamaServer, saluki, lmStudio],
|
|
48
|
+
browser: "WebGPU, trial.ml and in-browser models: doctor runs in Node and cannot check them. Load the demo extension in Firefox.",
|
|
49
|
+
cloud: "anthropic and cloud openaiCompatible: these need your own key. doctor does not read keys.",
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
export function format(report) {
|
|
53
|
+
const rows = report.checks.map((check) => {
|
|
54
|
+
const detail = check.ok ? `models: ${check.models?.join(", ") || "none"}` : check.reason;
|
|
55
|
+
return `${check.tier.padEnd(8)} ${check.name.padEnd(13)} ${(check.ok ? "yes" : "no").padEnd(4)} ${check.where.padEnd(28)} ${detail}`;
|
|
56
|
+
});
|
|
57
|
+
return [`foxmind doctor · Node ${report.node} · ${report.platform}`, "", ...rows, `browser ${report.browser}`, `cloud ${report.cloud}`, ""].join("\n");
|
|
58
|
+
}
|
package/dist/errors.d.ts
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import type { Tier } from "./types.js";
|
|
2
|
+
export type ErrorCode = "no_provider" | "unreachable" | "timeout" | "aborted" | "rate_limited" | "auth" | "model_not_found" | "http" | "stream_interrupted" | "bad_tool_call" | "bad_json" | "bad_response" | "unsupported" | "download_failed" | "cache_corrupt" | "out_of_memory" | "webgpu_missing" | "permission";
|
|
3
|
+
/** A provider the router did not use, and why. */
|
|
4
|
+
export interface Skip {
|
|
5
|
+
provider: string;
|
|
6
|
+
tier: Tier;
|
|
7
|
+
code: string;
|
|
8
|
+
reason: string;
|
|
9
|
+
}
|
|
10
|
+
export interface ErrorDetails {
|
|
11
|
+
provider?: string;
|
|
12
|
+
tier?: Tier;
|
|
13
|
+
status?: number;
|
|
14
|
+
retryAfterMs?: number;
|
|
15
|
+
/** The text that failed to parse (tool arguments or JSON), cut to 200 characters. */
|
|
16
|
+
raw?: string;
|
|
17
|
+
/** Text streamed before the failure. */
|
|
18
|
+
partial?: string;
|
|
19
|
+
skipped?: Skip[];
|
|
20
|
+
cause?: unknown;
|
|
21
|
+
}
|
|
22
|
+
export declare class FoxmindError extends Error {
|
|
23
|
+
readonly code: ErrorCode;
|
|
24
|
+
provider?: string;
|
|
25
|
+
tier?: Tier;
|
|
26
|
+
status?: number;
|
|
27
|
+
retryAfterMs?: number;
|
|
28
|
+
raw?: string;
|
|
29
|
+
partial?: string;
|
|
30
|
+
skipped?: Skip[];
|
|
31
|
+
constructor(code: ErrorCode, message: string, details?: ErrorDetails);
|
|
32
|
+
toJSON(): {
|
|
33
|
+
name: string;
|
|
34
|
+
message: string;
|
|
35
|
+
code: ErrorCode;
|
|
36
|
+
provider: string | undefined;
|
|
37
|
+
tier: Tier | undefined;
|
|
38
|
+
status: number | undefined;
|
|
39
|
+
retryAfterMs: number | undefined;
|
|
40
|
+
raw: string | undefined;
|
|
41
|
+
partial: string | undefined;
|
|
42
|
+
skipped: Skip[] | undefined;
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
/** Replace every secret and every key-shaped string in text with "[redacted]". */
|
|
46
|
+
export declare function redact(text: string, secrets?: (string | undefined)[]): string;
|
|
47
|
+
/** Cut text for an error message. */
|
|
48
|
+
export declare function clip(text: string, max?: number): string;
|
package/dist/errors.js
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
export class FoxmindError extends Error {
|
|
2
|
+
code;
|
|
3
|
+
provider;
|
|
4
|
+
tier;
|
|
5
|
+
status;
|
|
6
|
+
retryAfterMs;
|
|
7
|
+
raw;
|
|
8
|
+
partial;
|
|
9
|
+
skipped;
|
|
10
|
+
constructor(code, message, details = {}) {
|
|
11
|
+
const where = details.provider ? `${details.provider}${details.tier ? ` (${details.tier})` : ""}: ` : "";
|
|
12
|
+
super(`${where}${message}`, details.cause === undefined ? undefined : { cause: details.cause });
|
|
13
|
+
this.name = "FoxmindError";
|
|
14
|
+
this.code = code;
|
|
15
|
+
const { cause: _cause, ...rest } = details;
|
|
16
|
+
Object.assign(this, rest);
|
|
17
|
+
}
|
|
18
|
+
toJSON() {
|
|
19
|
+
const { code, provider, tier, status, retryAfterMs, raw, partial, skipped } = this;
|
|
20
|
+
return { name: this.name, message: this.message, code, provider, tier, status, retryAfterMs, raw, partial, skipped };
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
/** Matches common API key shapes, so a key the caller did not give us is still hidden. */
|
|
24
|
+
const KEY_SHAPES = /\b(sk-[A-Za-z0-9_-]{8,}|sk-ant-[A-Za-z0-9_-]{8,}|hf_[A-Za-z0-9]{8,}|AIza[A-Za-z0-9_-]{20,})/g;
|
|
25
|
+
/** Replace every secret and every key-shaped string in text with "[redacted]". */
|
|
26
|
+
export function redact(text, secrets = []) {
|
|
27
|
+
let out = text;
|
|
28
|
+
for (const secret of secrets)
|
|
29
|
+
if (secret && secret.length >= 4)
|
|
30
|
+
out = out.split(secret).join("[redacted]");
|
|
31
|
+
return out.replace(KEY_SHAPES, "[redacted]");
|
|
32
|
+
}
|
|
33
|
+
/** Cut text for an error message. */
|
|
34
|
+
export function clip(text, max = 200) {
|
|
35
|
+
return text.length > max ? `${text.slice(0, max)}…` : text;
|
|
36
|
+
}
|
package/dist/http.d.ts
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { FoxmindError, type ErrorDetails } from "./errors.js";
|
|
2
|
+
import type { CallOptions, Tier } from "./types.js";
|
|
3
|
+
export interface Origin {
|
|
4
|
+
provider: string;
|
|
5
|
+
tier: Tier;
|
|
6
|
+
/** Secrets to hide from every error message. */
|
|
7
|
+
secrets: (string | undefined)[];
|
|
8
|
+
}
|
|
9
|
+
export interface Request extends CallOptions {
|
|
10
|
+
method?: "GET" | "POST";
|
|
11
|
+
headers?: Record<string, string>;
|
|
12
|
+
body?: unknown;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* One signal for the caller's abort and a timeout that can restart. The
|
|
16
|
+
* timeout covers the wait for the response headers and then each gap between
|
|
17
|
+
* two body chunks, never the whole request, so a slow healthy stream lives.
|
|
18
|
+
*/
|
|
19
|
+
export declare function watchdog(options: CallOptions, fallbackMs: number): {
|
|
20
|
+
signal: AbortSignal;
|
|
21
|
+
arm: () => void;
|
|
22
|
+
stop: () => void;
|
|
23
|
+
ms: number;
|
|
24
|
+
timedOut: () => boolean;
|
|
25
|
+
};
|
|
26
|
+
export declare function failure(origin: Origin, code: ConstructorParameters<typeof FoxmindError>[0], message: string, details?: ErrorDetails): FoxmindError;
|
|
27
|
+
/**
|
|
28
|
+
* Map an exception from fetch or a body read to a FoxmindError. When a stream
|
|
29
|
+
* was running, a dropped connection is `stream_interrupted` and the error
|
|
30
|
+
* keeps the text streamed so far.
|
|
31
|
+
*/
|
|
32
|
+
export declare function fetchFailure(origin: Origin, error: unknown, timedOut: boolean, url: string, timeoutMs: number, partial?: string): FoxmindError;
|
|
33
|
+
/** "Retry-After" in seconds or as a date, or "retry-after-ms". */
|
|
34
|
+
export declare function retryAfter(headers: Headers): number | undefined;
|
|
35
|
+
/** Throw the FoxmindError for a non-2xx response. */
|
|
36
|
+
export declare function httpFailure(origin: Origin, response: Response): Promise<FoxmindError>;
|
|
37
|
+
export interface Fetched {
|
|
38
|
+
response: Response;
|
|
39
|
+
/** The body. Each chunk restarts the timeout; read this, not response.body. */
|
|
40
|
+
body: ReadableStream<Uint8Array> | null;
|
|
41
|
+
/** Map an error from reading the body (a timeout, a dropped connection) to a FoxmindError. */
|
|
42
|
+
fail(error: unknown, partial?: string): FoxmindError;
|
|
43
|
+
json<T>(): Promise<T>;
|
|
44
|
+
}
|
|
45
|
+
/** fetch() that returns an ok response or throws a FoxmindError. */
|
|
46
|
+
export declare function call(origin: Origin, url: string, request: Request, fallbackMs: number): Promise<Fetched>;
|
package/dist/http.js
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
// fetch with a timeout, a caller abort, and HTTP failures mapped to FoxmindError
|
|
2
|
+
// codes. Every message goes through redact() before it leaves this file.
|
|
3
|
+
import { clip, FoxmindError, redact } from "./errors.js";
|
|
4
|
+
/**
|
|
5
|
+
* One signal for the caller's abort and a timeout that can restart. The
|
|
6
|
+
* timeout covers the wait for the response headers and then each gap between
|
|
7
|
+
* two body chunks, never the whole request, so a slow healthy stream lives.
|
|
8
|
+
*/
|
|
9
|
+
export function watchdog(options, fallbackMs) {
|
|
10
|
+
const ms = options.timeoutMs ?? fallbackMs;
|
|
11
|
+
const controller = new AbortController();
|
|
12
|
+
let fired = false;
|
|
13
|
+
let timer;
|
|
14
|
+
const stop = () => clearTimeout(timer);
|
|
15
|
+
const arm = () => {
|
|
16
|
+
stop();
|
|
17
|
+
timer = setTimeout(() => {
|
|
18
|
+
fired = true;
|
|
19
|
+
controller.abort();
|
|
20
|
+
}, ms);
|
|
21
|
+
};
|
|
22
|
+
arm();
|
|
23
|
+
const signal = options.signal ? AbortSignal.any([options.signal, controller.signal]) : controller.signal;
|
|
24
|
+
return { signal, arm, stop, ms, timedOut: () => fired && !options.signal?.aborted };
|
|
25
|
+
}
|
|
26
|
+
export function failure(origin, code, message, details = {}) {
|
|
27
|
+
return new FoxmindError(code, redact(message, origin.secrets), {
|
|
28
|
+
...details,
|
|
29
|
+
provider: origin.provider,
|
|
30
|
+
tier: origin.tier,
|
|
31
|
+
raw: details.raw === undefined ? undefined : clip(redact(details.raw, origin.secrets)),
|
|
32
|
+
partial: details.partial === undefined ? undefined : redact(details.partial, origin.secrets),
|
|
33
|
+
});
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Map an exception from fetch or a body read to a FoxmindError. When a stream
|
|
37
|
+
* was running, a dropped connection is `stream_interrupted` and the error
|
|
38
|
+
* keeps the text streamed so far.
|
|
39
|
+
*/
|
|
40
|
+
export function fetchFailure(origin, error, timedOut, url, timeoutMs, partial) {
|
|
41
|
+
if (error instanceof FoxmindError)
|
|
42
|
+
return error;
|
|
43
|
+
const details = partial === undefined ? {} : { partial };
|
|
44
|
+
if (timedOut)
|
|
45
|
+
return failure(origin, "timeout", `${url} sent nothing for ${timeoutMs} ms.`, details);
|
|
46
|
+
if (error instanceof Error && error.name === "AbortError")
|
|
47
|
+
return failure(origin, "aborted", "The caller stopped the call.", details);
|
|
48
|
+
const cause = error instanceof Error && error.cause instanceof Error ? error.cause.message : String(error);
|
|
49
|
+
if (partial !== undefined)
|
|
50
|
+
return failure(origin, "stream_interrupted", `The stream from ${url} stopped after ${partial.length} characters: ${cause}`, details);
|
|
51
|
+
return failure(origin, "unreachable", `Cannot reach ${url}: ${cause}`);
|
|
52
|
+
}
|
|
53
|
+
/** "Retry-After" in seconds or as a date, or "retry-after-ms". */
|
|
54
|
+
export function retryAfter(headers) {
|
|
55
|
+
const ms = Number(headers.get("retry-after-ms"));
|
|
56
|
+
if (ms > 0)
|
|
57
|
+
return ms;
|
|
58
|
+
const value = headers.get("retry-after");
|
|
59
|
+
if (!value)
|
|
60
|
+
return undefined;
|
|
61
|
+
const seconds = Number(value);
|
|
62
|
+
if (Number.isFinite(seconds))
|
|
63
|
+
return Math.max(0, seconds * 1000);
|
|
64
|
+
const date = Date.parse(value);
|
|
65
|
+
return Number.isNaN(date) ? undefined : Math.max(0, date - Date.now());
|
|
66
|
+
}
|
|
67
|
+
/** The server's own words from an error body, in the OpenAI or Anthropic shape. */
|
|
68
|
+
function serverMessage(text) {
|
|
69
|
+
try {
|
|
70
|
+
const body = JSON.parse(text);
|
|
71
|
+
if (typeof body.error === "string")
|
|
72
|
+
return body.error;
|
|
73
|
+
return body.error?.message ?? body.message ?? text;
|
|
74
|
+
}
|
|
75
|
+
catch {
|
|
76
|
+
return text;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
/** Throw the FoxmindError for a non-2xx response. */
|
|
80
|
+
export async function httpFailure(origin, response) {
|
|
81
|
+
const said = clip(serverMessage(await response.text().catch(() => "")).trim(), 300);
|
|
82
|
+
const status = response.status;
|
|
83
|
+
const text = `HTTP ${status}${said ? `: ${said}` : ""}`;
|
|
84
|
+
if (status === 429)
|
|
85
|
+
return failure(origin, "rate_limited", text, { status, retryAfterMs: retryAfter(response.headers) });
|
|
86
|
+
if (status === 401 || status === 403)
|
|
87
|
+
return failure(origin, "auth", text, { status });
|
|
88
|
+
if (status === 501)
|
|
89
|
+
return failure(origin, "unsupported", text, { status });
|
|
90
|
+
// OpenAI: "model ... does not exist"; Ollama: 'model "x" not found'; Anthropic: "model: x".
|
|
91
|
+
if ((status === 404 || status === 400) && /model/i.test(said) && /not.?found|does not exist|unknown|^model:/i.test(said)) {
|
|
92
|
+
return failure(origin, "model_not_found", text, { status });
|
|
93
|
+
}
|
|
94
|
+
return failure(origin, "http", text, { status });
|
|
95
|
+
}
|
|
96
|
+
/** fetch() that returns an ok response or throws a FoxmindError. */
|
|
97
|
+
export async function call(origin, url, request, fallbackMs) {
|
|
98
|
+
const dog = watchdog(request, fallbackMs);
|
|
99
|
+
const fail = (error, partial) => fetchFailure(origin, error, dog.timedOut(), url, dog.ms, partial);
|
|
100
|
+
let response;
|
|
101
|
+
try {
|
|
102
|
+
response = await fetch(url, {
|
|
103
|
+
method: request.method ?? (request.body === undefined ? "GET" : "POST"),
|
|
104
|
+
headers: { ...(request.body === undefined ? {} : { "content-type": "application/json" }), ...request.headers },
|
|
105
|
+
body: request.body === undefined ? undefined : JSON.stringify(request.body),
|
|
106
|
+
signal: dog.signal,
|
|
107
|
+
// Never follow a redirect: fetch would send the key headers to the new host.
|
|
108
|
+
redirect: "manual",
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
catch (error) {
|
|
112
|
+
dog.stop();
|
|
113
|
+
throw fail(error);
|
|
114
|
+
}
|
|
115
|
+
if (response.type === "opaqueredirect" || (response.status >= 300 && response.status < 400)) {
|
|
116
|
+
dog.stop();
|
|
117
|
+
const target = response.headers.get("location") ?? "a URL the browser does not show";
|
|
118
|
+
await response.body?.cancel().catch(() => { });
|
|
119
|
+
throw failure(origin, "http", `${url} answered ${response.status || "a redirect"} to ${target}. foxmind does not follow redirects, so no key goes to another URL. Use the final URL as baseURL.`, { status: response.status || undefined });
|
|
120
|
+
}
|
|
121
|
+
if (!response.ok) {
|
|
122
|
+
const failed = await httpFailure(origin, response);
|
|
123
|
+
dog.stop();
|
|
124
|
+
throw failed;
|
|
125
|
+
}
|
|
126
|
+
// The headers came: from now on the timeout is the longest gap between chunks.
|
|
127
|
+
dog.arm();
|
|
128
|
+
const reader = response.body?.getReader();
|
|
129
|
+
const body = reader
|
|
130
|
+
? new ReadableStream({
|
|
131
|
+
async pull(controller) {
|
|
132
|
+
try {
|
|
133
|
+
const { value, done } = await reader.read();
|
|
134
|
+
if (done) {
|
|
135
|
+
dog.stop();
|
|
136
|
+
controller.close();
|
|
137
|
+
}
|
|
138
|
+
else {
|
|
139
|
+
dog.arm();
|
|
140
|
+
controller.enqueue(value);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
catch (error) {
|
|
144
|
+
dog.stop();
|
|
145
|
+
controller.error(error);
|
|
146
|
+
}
|
|
147
|
+
},
|
|
148
|
+
cancel(reason) {
|
|
149
|
+
dog.stop();
|
|
150
|
+
return reader.cancel(reason);
|
|
151
|
+
},
|
|
152
|
+
})
|
|
153
|
+
: (dog.stop(), null);
|
|
154
|
+
const json = async () => {
|
|
155
|
+
const text = await new Response(body).text().catch((error) => { throw fail(error); });
|
|
156
|
+
try {
|
|
157
|
+
return JSON.parse(text);
|
|
158
|
+
}
|
|
159
|
+
catch {
|
|
160
|
+
throw failure(origin, "bad_response", `${url} sent a body that is not JSON: ${clip(text, 120)}`);
|
|
161
|
+
}
|
|
162
|
+
};
|
|
163
|
+
return { response, body, fail, json };
|
|
164
|
+
}
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export { FoxmindError, type ErrorCode, type Skip } from "./errors.js";
|
|
2
|
+
export { openaiCompatible, type OpenAICompatibleOptions } from "./providers/openai.js";
|
|
3
|
+
export type * from "./types.js";
|
|
4
|
+
export { createMind, type Answered, type ChatResult, type ClassifyResult, type EmbedResult, type ExtractResult, type Mind, type MindOptions, type MindStatus } from "./mind.js";
|
|
5
|
+
export { llamaServer, lmStudio, ollama, saluki, SALUKI } from "./providers/presets.js";
|
|
6
|
+
export { anthropic, type AnthropicOptions } from "./providers/anthropic.js";
|
|
7
|
+
export { doctor, format as formatDoctor, type Check, type DoctorOptions, type Report } from "./doctor.js";
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
// The public API of foxmind for Node and the browser.
|
|
2
|
+
export { FoxmindError } from "./errors.js";
|
|
3
|
+
export { openaiCompatible } from "./providers/openai.js";
|
|
4
|
+
export { createMind } from "./mind.js";
|
|
5
|
+
export { llamaServer, lmStudio, ollama, saluki, SALUKI } from "./providers/presets.js";
|
|
6
|
+
export { anthropic } from "./providers/anthropic.js";
|
|
7
|
+
export { doctor, format as formatDoctor } from "./doctor.js";
|