foxmind 0.0.0-stage → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +308 -2
- package/dist/bin.d.ts +2 -0
- package/dist/bin.js +6 -0
- package/dist/browser/gliner2-model.d.ts +57 -0
- package/dist/browser/gliner2-model.js +233 -0
- package/dist/browser/gliner2.d.ts +13 -0
- package/dist/browser/gliner2.js +50 -0
- package/dist/browser/index.d.ts +4 -0
- package/dist/browser/index.js +6 -0
- package/dist/browser/runtime.d.ts +61 -0
- package/dist/browser/runtime.js +205 -0
- package/dist/browser/toolcalls.d.ts +6 -0
- package/dist/browser/toolcalls.js +20 -0
- package/dist/browser/transformers.d.ts +18 -0
- package/dist/browser/transformers.js +81 -0
- package/dist/browser/trialml.d.ts +32 -0
- package/dist/browser/trialml.js +196 -0
- package/dist/cli.d.ts +6 -0
- package/dist/cli.js +46 -0
- package/dist/doctor.d.ts +23 -0
- package/dist/doctor.js +58 -0
- package/dist/errors.d.ts +48 -0
- package/dist/errors.js +36 -0
- package/dist/http.d.ts +46 -0
- package/dist/http.js +164 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.js +7 -0
- package/dist/mind.d.ts +60 -0
- package/dist/mind.js +122 -0
- package/dist/providers/anthropic.d.ts +16 -0
- package/dist/providers/anthropic.js +200 -0
- package/dist/providers/openai.d.ts +39 -0
- package/dist/providers/openai.js +177 -0
- package/dist/providers/presets.d.ts +33 -0
- package/dist/providers/presets.js +89 -0
- package/dist/reply.d.ts +10 -0
- package/dist/reply.js +60 -0
- package/dist/sse.d.ts +6 -0
- package/dist/sse.js +38 -0
- package/dist/types.d.ts +106 -0
- package/dist/types.js +3 -0
- package/package.json +59 -4
package/dist/mind.d.ts
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { type Skip } from "./errors.js";
|
|
2
|
+
import type { CallOptions, Capability, ChatOptions, ChatReply, Entity, ExtractOptions, Labels, Message, Probe, Provider, ProviderStatus, Tier } from "./types.js";
|
|
3
|
+
export interface MindOptions {
|
|
4
|
+
providers: Provider[];
|
|
5
|
+
/** Provider names or tiers ("browser", "local", "cloud"), best first. Providers not named come after, in `providers` order. */
|
|
6
|
+
prefer?: string[];
|
|
7
|
+
/**
|
|
8
|
+
* The only tiers the router may use. Providers of other tiers are dropped:
|
|
9
|
+
* never probed, never called. Private mode is `only: ["browser", "local"]`.
|
|
10
|
+
*/
|
|
11
|
+
only?: Tier[];
|
|
12
|
+
/** When a call fails, try the next provider and list the failure in `skipped`. Default false. */
|
|
13
|
+
fallbackOnError?: boolean;
|
|
14
|
+
/** How long a probe result stays good. Default 30000. */
|
|
15
|
+
probeTtlMs?: number;
|
|
16
|
+
}
|
|
17
|
+
export interface Answered {
|
|
18
|
+
provider: string;
|
|
19
|
+
tier: Tier;
|
|
20
|
+
model: string;
|
|
21
|
+
ms: number;
|
|
22
|
+
/** Providers tried before this one, and why each did not answer. */
|
|
23
|
+
skipped: Skip[];
|
|
24
|
+
}
|
|
25
|
+
export type ChatResult = ChatReply & Answered;
|
|
26
|
+
export interface EmbedResult extends Answered {
|
|
27
|
+
vectors: number[][];
|
|
28
|
+
}
|
|
29
|
+
export interface ExtractResult extends Answered {
|
|
30
|
+
entities: Record<string, Entity[]>;
|
|
31
|
+
}
|
|
32
|
+
export interface ClassifyResult extends Answered {
|
|
33
|
+
scores: Record<string, number>[];
|
|
34
|
+
}
|
|
35
|
+
export interface MindStatus {
|
|
36
|
+
providers: ProviderStatus[];
|
|
37
|
+
last?: {
|
|
38
|
+
capability: Capability;
|
|
39
|
+
provider: string;
|
|
40
|
+
tier: Tier;
|
|
41
|
+
model: string;
|
|
42
|
+
at: string;
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
export interface Mind {
|
|
46
|
+
readonly providers: readonly Provider[];
|
|
47
|
+
chat(messages: Message[], options?: ChatOptions): Promise<ChatResult>;
|
|
48
|
+
embed(texts: string[], options?: CallOptions): Promise<EmbedResult>;
|
|
49
|
+
extract(text: string, labels: Labels, options?: ExtractOptions): Promise<ExtractResult>;
|
|
50
|
+
classify(texts: string[], prompt: string, labels: Labels, options?: CallOptions): Promise<ClassifyResult>;
|
|
51
|
+
/** Probe every provider now, past the cache. */
|
|
52
|
+
probe(options?: CallOptions): Promise<(Probe & {
|
|
53
|
+
provider: string;
|
|
54
|
+
tier: Tier;
|
|
55
|
+
})[]>;
|
|
56
|
+
/** Download and start one provider's model now. */
|
|
57
|
+
load(name: string, options?: CallOptions): Promise<void>;
|
|
58
|
+
status(): MindStatus;
|
|
59
|
+
}
|
|
60
|
+
export declare function createMind(options: MindOptions): Mind;
|
package/dist/mind.js
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
// The router. It picks a provider in `prefer` order, skips the ones that
|
|
2
|
+
// cannot run now, and says in every result and error which provider and tier
|
|
3
|
+
// answered and which ones it skipped and why.
|
|
4
|
+
import { FoxmindError } from "./errors.js";
|
|
5
|
+
const TIERS = new Set(["browser", "local", "cloud"]);
|
|
6
|
+
function order(providers, prefer = []) {
|
|
7
|
+
const names = new Set();
|
|
8
|
+
for (const provider of providers) {
|
|
9
|
+
if (names.has(provider.name))
|
|
10
|
+
throw new TypeError(`createMind got two providers named "${provider.name}". Give each one its own name.`);
|
|
11
|
+
names.add(provider.name);
|
|
12
|
+
}
|
|
13
|
+
for (const entry of prefer) {
|
|
14
|
+
if (!names.has(entry) && !TIERS.has(entry))
|
|
15
|
+
throw new TypeError(`prefer names "${entry}", which is not a provider name or a tier. Providers: ${[...names].join(", ")}.`);
|
|
16
|
+
}
|
|
17
|
+
const ranked = prefer.flatMap((entry) => providers.filter((provider) => provider.name === entry || provider.tier === entry));
|
|
18
|
+
return [...new Set([...ranked, ...providers])];
|
|
19
|
+
}
|
|
20
|
+
const aborted = () => new FoxmindError("aborted", "The caller stopped the call.");
|
|
21
|
+
export function createMind(options) {
|
|
22
|
+
const only = options.only;
|
|
23
|
+
for (const tier of only ?? [])
|
|
24
|
+
if (!TIERS.has(tier))
|
|
25
|
+
throw new TypeError(`only names "${tier}", which is not a tier. Tiers: browser, local, cloud.`);
|
|
26
|
+
const providers = order(options.providers, options.prefer).filter((provider) => !only || only.includes(provider.tier));
|
|
27
|
+
if (only && !providers.length)
|
|
28
|
+
throw new TypeError(`only: [${only.join(", ")}] leaves no provider. Add a provider of one of those tiers.`);
|
|
29
|
+
const excluded = only ? ` only: [${only.join(", ")}] excluded the other tiers.` : "";
|
|
30
|
+
const ttl = options.probeTtlMs ?? 30_000;
|
|
31
|
+
const probes = new Map();
|
|
32
|
+
let last;
|
|
33
|
+
/**
|
|
34
|
+
* A cached or fresh probe. The probe runs on its own timeout, never on the
|
|
35
|
+
* caller's signal, so an abort cannot turn into a cached "unreachable". An
|
|
36
|
+
* abort while the probe runs throws aborted and caches nothing.
|
|
37
|
+
*/
|
|
38
|
+
async function probe(provider, callOptions) {
|
|
39
|
+
const cached = probes.get(provider.name);
|
|
40
|
+
if (cached && Date.now() - cached.at < ttl)
|
|
41
|
+
return cached.result;
|
|
42
|
+
const running = provider.probe().catch((error) => ({ ok: false, code: "unreachable", reason: String(error) }));
|
|
43
|
+
const signal = callOptions.signal;
|
|
44
|
+
const result = signal
|
|
45
|
+
? await Promise.race([running, new Promise((_, reject) => signal.addEventListener("abort", () => reject(aborted()), { once: true }))])
|
|
46
|
+
: await running;
|
|
47
|
+
if (signal?.aborted)
|
|
48
|
+
throw aborted();
|
|
49
|
+
probes.set(provider.name, { at: Date.now(), result });
|
|
50
|
+
return result;
|
|
51
|
+
}
|
|
52
|
+
async function call(capability, callOptions, run) {
|
|
53
|
+
const able = providers.filter((provider) => provider.capabilities.includes(capability));
|
|
54
|
+
if (!able.length) {
|
|
55
|
+
throw new FoxmindError("no_provider", `No provider can ${capability}. Providers: ${providers.map((p) => `${p.name} (${p.capabilities.join(", ")})`).join("; ") || "none"}.${excluded}`, { skipped: [] });
|
|
56
|
+
}
|
|
57
|
+
const skipped = [];
|
|
58
|
+
for (const provider of able) {
|
|
59
|
+
// An abort ends the call here. It never moves the text on to the next provider.
|
|
60
|
+
if (callOptions.signal?.aborted)
|
|
61
|
+
throw aborted();
|
|
62
|
+
const probed = await probe(provider, callOptions);
|
|
63
|
+
if (!probed.ok) {
|
|
64
|
+
skipped.push({ provider: provider.name, tier: provider.tier, code: probed.code ?? "unavailable", reason: probed.reason ?? "The probe failed." });
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
67
|
+
// Once text reached the caller, a second provider would repeat it, so a stream never falls back.
|
|
68
|
+
let streamed = false;
|
|
69
|
+
const onDelta = callOptions.onDelta;
|
|
70
|
+
const attempt = onDelta ? { ...callOptions, onDelta: (text) => { streamed = true; onDelta(text); } } : callOptions;
|
|
71
|
+
const started = Date.now();
|
|
72
|
+
try {
|
|
73
|
+
const value = await run(provider, attempt);
|
|
74
|
+
last = { capability, provider: provider.name, tier: provider.tier, model: provider.model, at: new Date().toISOString() };
|
|
75
|
+
return { value, provider: provider.name, tier: provider.tier, model: provider.model, ms: Date.now() - started, skipped };
|
|
76
|
+
}
|
|
77
|
+
catch (error) {
|
|
78
|
+
const failed = error instanceof FoxmindError ? error : new FoxmindError("http", String(error), { provider: provider.name, tier: provider.tier, cause: error });
|
|
79
|
+
// Whatever went wrong, the cached probe no longer tells the truth.
|
|
80
|
+
probes.delete(provider.name);
|
|
81
|
+
if (options.fallbackOnError && !streamed && failed.code !== "aborted" && !callOptions.signal?.aborted) {
|
|
82
|
+
skipped.push({ provider: provider.name, tier: provider.tier, code: failed.code, reason: failed.message });
|
|
83
|
+
continue;
|
|
84
|
+
}
|
|
85
|
+
failed.skipped = skipped;
|
|
86
|
+
throw failed;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
const list = skipped.map((skip) => `${skip.provider} (${skip.tier}): ${skip.reason}`).join("; ");
|
|
90
|
+
throw new FoxmindError("no_provider", `No provider could ${capability}. ${list}${excluded}`, { skipped });
|
|
91
|
+
}
|
|
92
|
+
return {
|
|
93
|
+
providers,
|
|
94
|
+
async chat(messages, chatOptions = {}) {
|
|
95
|
+
const { value, ...answered } = await call("chat", chatOptions, (provider, o) => provider.chat(messages, o));
|
|
96
|
+
return { ...value, ...answered };
|
|
97
|
+
},
|
|
98
|
+
async embed(texts, embedOptions = {}) {
|
|
99
|
+
const { value, ...answered } = await call("embed", embedOptions, (provider, o) => provider.embed(texts, o));
|
|
100
|
+
return { vectors: value, ...answered };
|
|
101
|
+
},
|
|
102
|
+
async extract(text, labels, extractOptions = {}) {
|
|
103
|
+
const { value, ...answered } = await call("extract", extractOptions, (provider, o) => provider.extract(text, labels, o));
|
|
104
|
+
return { entities: value, ...answered };
|
|
105
|
+
},
|
|
106
|
+
async classify(texts, prompt, labels, classifyOptions = {}) {
|
|
107
|
+
const { value, ...answered } = await call("classify", classifyOptions, (provider, o) => provider.classify(texts, prompt, labels, o));
|
|
108
|
+
return { scores: value, ...answered };
|
|
109
|
+
},
|
|
110
|
+
async probe(probeOptions = {}) {
|
|
111
|
+
probes.clear();
|
|
112
|
+
return Promise.all(providers.map(async (provider) => ({ ...(await probe(provider, probeOptions)), provider: provider.name, tier: provider.tier })));
|
|
113
|
+
},
|
|
114
|
+
async load(name, loadOptions = {}) {
|
|
115
|
+
const provider = providers.find((candidate) => candidate.name === name);
|
|
116
|
+
if (!provider)
|
|
117
|
+
throw new TypeError(`No provider named "${name}".`);
|
|
118
|
+
await provider.load?.(loadOptions);
|
|
119
|
+
},
|
|
120
|
+
status: () => ({ providers: providers.map((provider) => provider.status()), ...(last ? { last } : {}) }),
|
|
121
|
+
};
|
|
122
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import type { Provider } from "../types.js";
|
|
2
|
+
export interface AnthropicOptions {
|
|
3
|
+
apiKey: string;
|
|
4
|
+
/** Default "claude-opus-5-5". */
|
|
5
|
+
model?: string;
|
|
6
|
+
/** Default "https://api.anthropic.com". */
|
|
7
|
+
baseURL?: string;
|
|
8
|
+
/** Default 16000. The Messages API needs a value. */
|
|
9
|
+
maxTokens?: number;
|
|
10
|
+
/** Default "anthropic". */
|
|
11
|
+
name?: string;
|
|
12
|
+
/** Default 300000. */
|
|
13
|
+
timeoutMs?: number;
|
|
14
|
+
headers?: Record<string, string>;
|
|
15
|
+
}
|
|
16
|
+
export declare function anthropic(options: AnthropicOptions): Provider;
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
// The Anthropic Messages API, mapped to the OpenAI shape both ways. The caller
|
|
2
|
+
// passes the key; foxmind keeps it in memory only and never stores it.
|
|
3
|
+
import { FoxmindError } from "../errors.js";
|
|
4
|
+
import { call, failure } from "../http.js";
|
|
5
|
+
import { checkReply } from "../reply.js";
|
|
6
|
+
import { events } from "../sse.js";
|
|
7
|
+
const JSON_RULE = "Reply with one JSON object and nothing else: no prose and no code fence.";
|
|
8
|
+
function finishReason(stop) {
|
|
9
|
+
if (stop === "tool_use")
|
|
10
|
+
return "tool_calls";
|
|
11
|
+
if (stop === "max_tokens")
|
|
12
|
+
return "length";
|
|
13
|
+
if (stop === "refusal")
|
|
14
|
+
return "content_filter";
|
|
15
|
+
return "stop";
|
|
16
|
+
}
|
|
17
|
+
export function anthropic(options) {
|
|
18
|
+
const baseURL = (options.baseURL ?? "https://api.anthropic.com").replace(/\/+$/, "");
|
|
19
|
+
const model = options.model ?? "claude-opus-5-5";
|
|
20
|
+
const name = options.name ?? "anthropic";
|
|
21
|
+
const origin = { provider: name, tier: "cloud", secrets: [options.apiKey] };
|
|
22
|
+
const timeout = options.timeoutMs ?? 300_000;
|
|
23
|
+
let last;
|
|
24
|
+
const headers = () => ({
|
|
25
|
+
...options.headers,
|
|
26
|
+
"x-api-key": options.apiKey,
|
|
27
|
+
"anthropic-version": "2023-06-01",
|
|
28
|
+
// Lets a browser extension page call the API with the user's own key.
|
|
29
|
+
"anthropic-dangerous-direct-browser-access": "true",
|
|
30
|
+
});
|
|
31
|
+
/** OpenAI messages to Anthropic: system messages to `system`, grouped tool results, tool_use blocks. */
|
|
32
|
+
function toWire(messages) {
|
|
33
|
+
const system = [];
|
|
34
|
+
const out = [];
|
|
35
|
+
for (const message of messages) {
|
|
36
|
+
if (message.role === "system") {
|
|
37
|
+
// Many models reject role "system" inside messages, so every system message joins the top-level system, in order.
|
|
38
|
+
system.push(message.content ?? "");
|
|
39
|
+
}
|
|
40
|
+
else if (message.role === "user") {
|
|
41
|
+
out.push({ role: "user", content: message.content ?? "" });
|
|
42
|
+
}
|
|
43
|
+
else if (message.role === "tool") {
|
|
44
|
+
const result = { type: "tool_result", tool_use_id: message.tool_call_id, content: message.content ?? "" };
|
|
45
|
+
const previous = out.at(-1);
|
|
46
|
+
if (previous?.role === "user" && Array.isArray(previous.content) && previous.content.every((block) => block.type === "tool_result"))
|
|
47
|
+
previous.content.push(result);
|
|
48
|
+
else
|
|
49
|
+
out.push({ role: "user", content: [result] });
|
|
50
|
+
}
|
|
51
|
+
else {
|
|
52
|
+
const saved = message.provider_data?.anthropic?.content;
|
|
53
|
+
if (saved) {
|
|
54
|
+
out.push({ role: "assistant", content: saved });
|
|
55
|
+
continue;
|
|
56
|
+
}
|
|
57
|
+
if (!message.tool_calls?.length) {
|
|
58
|
+
out.push({ role: "assistant", content: message.content ?? "" });
|
|
59
|
+
continue;
|
|
60
|
+
}
|
|
61
|
+
const blocks = message.content ? [{ type: "text", text: message.content }] : [];
|
|
62
|
+
for (const toolCall of message.tool_calls) {
|
|
63
|
+
let input;
|
|
64
|
+
try {
|
|
65
|
+
input = JSON.parse(toolCall.function.arguments || "{}");
|
|
66
|
+
}
|
|
67
|
+
catch {
|
|
68
|
+
throw failure(origin, "bad_tool_call", `Past tool call "${toolCall.function.name}" in the messages has arguments that are not JSON.`, { raw: toolCall.function.arguments });
|
|
69
|
+
}
|
|
70
|
+
blocks.push({ type: "tool_use", id: toolCall.id, name: toolCall.function.name, input });
|
|
71
|
+
}
|
|
72
|
+
out.push({ role: "assistant", content: blocks });
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return { system, messages: out };
|
|
76
|
+
}
|
|
77
|
+
function toReply(data) {
|
|
78
|
+
if (!Array.isArray(data.content))
|
|
79
|
+
throw failure(origin, "bad_response", "The reply has no content array.");
|
|
80
|
+
const text = data.content.filter((block) => block.type === "text").map((block) => block.text ?? "");
|
|
81
|
+
const toolCalls = data.content
|
|
82
|
+
.filter((block) => block.type === "tool_use")
|
|
83
|
+
.map((block) => ({ id: block.id ?? "", type: "function", function: { name: block.name ?? "", arguments: JSON.stringify(block.input ?? {}) } }));
|
|
84
|
+
// Thinking blocks must go back unchanged with the next turn.
|
|
85
|
+
const keep = data.content.some((block) => block.type === "thinking" || block.type === "redacted_thinking");
|
|
86
|
+
return {
|
|
87
|
+
message: {
|
|
88
|
+
role: "assistant",
|
|
89
|
+
content: text.length ? text.join("") : null,
|
|
90
|
+
...(toolCalls.length ? { tool_calls: toolCalls } : {}),
|
|
91
|
+
...(keep ? { provider_data: { anthropic: { content: data.content } } } : {}),
|
|
92
|
+
},
|
|
93
|
+
finishReason: finishReason(data.stop_reason),
|
|
94
|
+
usage: { inputTokens: data.usage?.input_tokens ?? 0, outputTokens: data.usage?.output_tokens ?? 0 },
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
/** Join a stream of Messages API events into one reply. */
|
|
98
|
+
async function collect(fetched, onDelta) {
|
|
99
|
+
const blocks = [];
|
|
100
|
+
const json = [];
|
|
101
|
+
const reply = { usage: {} };
|
|
102
|
+
let text = "";
|
|
103
|
+
let done = false;
|
|
104
|
+
try {
|
|
105
|
+
for await (const event of events(fetched.body)) {
|
|
106
|
+
const data = JSON.parse(event.data);
|
|
107
|
+
const index = data.index ?? 0;
|
|
108
|
+
if (data.type === "message_start")
|
|
109
|
+
reply.usage = { ...data.message?.usage };
|
|
110
|
+
else if (data.type === "content_block_start")
|
|
111
|
+
blocks[index] = { ...data.content_block };
|
|
112
|
+
else if (data.type === "content_block_delta" && data.delta) {
|
|
113
|
+
const block = blocks[index];
|
|
114
|
+
const delta = data.delta;
|
|
115
|
+
if (delta.type === "text_delta") {
|
|
116
|
+
block.text = (block.text ?? "") + delta.text;
|
|
117
|
+
text += delta.text;
|
|
118
|
+
onDelta(delta.text);
|
|
119
|
+
}
|
|
120
|
+
else if (delta.type === "input_json_delta")
|
|
121
|
+
json[index] = (json[index] ?? "") + delta.partial_json;
|
|
122
|
+
else if (delta.type === "thinking_delta")
|
|
123
|
+
block.thinking = (block.thinking ?? "") + delta.thinking;
|
|
124
|
+
else if (delta.type === "signature_delta")
|
|
125
|
+
block.signature = delta.signature;
|
|
126
|
+
}
|
|
127
|
+
else if (data.type === "content_block_stop" && blocks[index]?.type === "tool_use") {
|
|
128
|
+
const raw = json[index] ?? "";
|
|
129
|
+
try {
|
|
130
|
+
blocks[index].input = raw.trim() ? JSON.parse(raw) : {};
|
|
131
|
+
}
|
|
132
|
+
catch {
|
|
133
|
+
throw failure(origin, "bad_tool_call", `The arguments of tool call "${blocks[index].name}" are not valid JSON.`, { raw, partial: text });
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
else if (data.type === "message_delta") {
|
|
137
|
+
reply.stop_reason = data.delta?.stop_reason ?? reply.stop_reason;
|
|
138
|
+
reply.usage = { ...reply.usage, ...data.usage };
|
|
139
|
+
}
|
|
140
|
+
else if (data.type === "message_stop")
|
|
141
|
+
done = true;
|
|
142
|
+
else if (data.type === "error") {
|
|
143
|
+
const kind = data.error?.type;
|
|
144
|
+
const said = `The API sent an error in the stream: ${data.error?.message ?? kind}`;
|
|
145
|
+
if (kind === "rate_limit_error")
|
|
146
|
+
throw failure(origin, "rate_limited", said, { partial: text });
|
|
147
|
+
if (kind === "authentication_error")
|
|
148
|
+
throw failure(origin, "auth", said, { partial: text });
|
|
149
|
+
throw failure(origin, "http", said, { partial: text, ...(kind === "overloaded_error" ? { status: 529 } : {}) });
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
catch (error) {
|
|
154
|
+
if (error instanceof FoxmindError)
|
|
155
|
+
throw error;
|
|
156
|
+
if (error instanceof SyntaxError)
|
|
157
|
+
throw failure(origin, "bad_response", "The stream sent data that is not JSON.", { partial: text });
|
|
158
|
+
throw fetched.fail(error, text);
|
|
159
|
+
}
|
|
160
|
+
if (!done)
|
|
161
|
+
throw failure(origin, "stream_interrupted", "The stream ended before message_stop.", { partial: text });
|
|
162
|
+
return { ...reply, content: blocks.filter(Boolean) };
|
|
163
|
+
}
|
|
164
|
+
return {
|
|
165
|
+
name,
|
|
166
|
+
tier: "cloud",
|
|
167
|
+
model,
|
|
168
|
+
capabilities: ["chat"],
|
|
169
|
+
async probe(probeOptions = {}) {
|
|
170
|
+
try {
|
|
171
|
+
await call(origin, `${baseURL}/v1/models/${encodeURIComponent(model)}`, { headers: headers(), ...probeOptions, timeoutMs: probeOptions.timeoutMs ?? 5000 }, 5000);
|
|
172
|
+
last = { ok: true, where: baseURL };
|
|
173
|
+
}
|
|
174
|
+
catch (error) {
|
|
175
|
+
const failed = error;
|
|
176
|
+
last = { ok: false, code: failed.code ?? "unreachable", reason: failed.message, where: baseURL };
|
|
177
|
+
}
|
|
178
|
+
return last;
|
|
179
|
+
},
|
|
180
|
+
status() {
|
|
181
|
+
return { name, tier: "cloud", model, capabilities: ["chat"], state: !last ? "idle" : last.ok ? "ready" : "unavailable", where: baseURL, ...(last?.reason ? { reason: last.reason } : {}) };
|
|
182
|
+
},
|
|
183
|
+
async chat(messages, chat) {
|
|
184
|
+
const wire = toWire(messages);
|
|
185
|
+
const system = [...wire.system, ...(chat.json ? [JSON_RULE] : [])].join("\n\n");
|
|
186
|
+
const body = {
|
|
187
|
+
model,
|
|
188
|
+
max_tokens: chat.maxTokens ?? options.maxTokens ?? 16_000,
|
|
189
|
+
...(system ? { system } : {}),
|
|
190
|
+
...(chat.tools?.length ? { tools: chat.tools.map((tool) => ({ name: tool.function.name, ...(tool.function.description ? { description: tool.function.description } : {}), input_schema: tool.function.parameters ?? { type: "object", properties: {} } })) } : {}),
|
|
191
|
+
...(chat.temperature === undefined ? {} : { temperature: chat.temperature }),
|
|
192
|
+
messages: wire.messages,
|
|
193
|
+
...(chat.onDelta ? { stream: true } : {}),
|
|
194
|
+
};
|
|
195
|
+
const fetched = await call(origin, `${baseURL}/v1/messages`, { ...chat, headers: headers(), body }, timeout);
|
|
196
|
+
const data = chat.onDelta ? await collect(fetched, chat.onDelta) : await fetched.json();
|
|
197
|
+
return checkReply(origin, toReply(data), chat.json);
|
|
198
|
+
},
|
|
199
|
+
};
|
|
200
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { FinishReason, Message, Provider, Tier, ToolCall } from "../types.js";
|
|
2
|
+
export interface OpenAICompatibleOptions {
|
|
3
|
+
/** The API root, with the version: "http://127.0.0.1:8080/v1", "https://api.openai.com/v1". */
|
|
4
|
+
baseURL: string;
|
|
5
|
+
model: string;
|
|
6
|
+
/** Sent as "Authorization: Bearer". foxmind keeps it in memory only and hides it from every error. */
|
|
7
|
+
apiKey?: string;
|
|
8
|
+
/** The model for embed(). Default: `model`. */
|
|
9
|
+
embedModel?: string;
|
|
10
|
+
/** Default "openai-compatible". */
|
|
11
|
+
name?: string;
|
|
12
|
+
/** Default "local" for localhost and 127.0.0.1, else "cloud". */
|
|
13
|
+
tier?: Tier;
|
|
14
|
+
headers?: Record<string, string>;
|
|
15
|
+
/** Extra fields for every chat request, for example llama.cpp's `chat_template_kwargs`. */
|
|
16
|
+
body?: Record<string, unknown>;
|
|
17
|
+
/**
|
|
18
|
+
* How probe() checks the server's model list: true = it must list `model`,
|
|
19
|
+
* a RegExp = one id must match, false = do not check. Default true.
|
|
20
|
+
*/
|
|
21
|
+
checkModel?: boolean | RegExp;
|
|
22
|
+
/** Added to the reason of a failed probe, for example the command that starts the server. */
|
|
23
|
+
hint?: string;
|
|
24
|
+
/** Default 120000 for chat and embed. probe() always uses 3000. */
|
|
25
|
+
timeoutMs?: number;
|
|
26
|
+
}
|
|
27
|
+
export declare function finishReason(value: string | null | undefined): FinishReason;
|
|
28
|
+
/** The wire message without the fields only foxmind uses. */
|
|
29
|
+
export declare function toWire(message: Message): {
|
|
30
|
+
role: import("../types.js").Role;
|
|
31
|
+
content: string | null;
|
|
32
|
+
name?: string;
|
|
33
|
+
tool_calls?: ToolCall[];
|
|
34
|
+
tool_call_id?: string;
|
|
35
|
+
};
|
|
36
|
+
/** Ollama names models that run on ollama.com with a "-cloud" or ":cloud" ending. */
|
|
37
|
+
export declare function isCloudModel(model: string): boolean;
|
|
38
|
+
export declare function isLocalURL(url: string): boolean;
|
|
39
|
+
export declare function openaiCompatible(options: OpenAICompatibleOptions): Provider;
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
// Any server that speaks the OpenAI chat completions API: llama.cpp
|
|
2
|
+
// llama-server, Ollama, LM Studio, OpenAI, OpenRouter and many more.
|
|
3
|
+
import { FoxmindError } from "../errors.js";
|
|
4
|
+
import { call, failure } from "../http.js";
|
|
5
|
+
import { checkReply } from "../reply.js";
|
|
6
|
+
import { events } from "../sse.js";
|
|
7
|
+
export function finishReason(value) {
|
|
8
|
+
if (value === "tool_calls" || value === "function_call")
|
|
9
|
+
return "tool_calls";
|
|
10
|
+
if (value === "length" || value === "content_filter")
|
|
11
|
+
return value;
|
|
12
|
+
return "stop";
|
|
13
|
+
}
|
|
14
|
+
/** The wire message without the fields only foxmind uses. */
|
|
15
|
+
export function toWire(message) {
|
|
16
|
+
const { reasoning: _reasoning, provider_data: _data, ...wire } = message;
|
|
17
|
+
return wire;
|
|
18
|
+
}
|
|
19
|
+
/** Ollama names models that run on ollama.com with a "-cloud" or ":cloud" ending. */
|
|
20
|
+
export function isCloudModel(model) {
|
|
21
|
+
return /[-:]cloud$/i.test(model);
|
|
22
|
+
}
|
|
23
|
+
export function isLocalURL(url) {
|
|
24
|
+
return /^https?:\/\/(localhost|127\.\d+\.\d+\.\d+|\[::1\])(:\d+)?(\/|$)/i.test(url);
|
|
25
|
+
}
|
|
26
|
+
export function openaiCompatible(options) {
|
|
27
|
+
const baseURL = options.baseURL.replace(/\/+$/, "");
|
|
28
|
+
const name = options.name ?? "openai-compatible";
|
|
29
|
+
// A model named *-cloud (Ollama) runs on a remote host even when the server is on localhost.
|
|
30
|
+
const tier = options.tier ?? (isLocalURL(baseURL) && !isCloudModel(options.model) ? "local" : "cloud");
|
|
31
|
+
const origin = { provider: name, tier, secrets: [options.apiKey] };
|
|
32
|
+
const headers = { ...options.headers, ...(options.apiKey ? { authorization: `Bearer ${options.apiKey}` } : {}) };
|
|
33
|
+
const timeout = options.timeoutMs ?? 120_000;
|
|
34
|
+
let last;
|
|
35
|
+
function request(messages, chat, stream) {
|
|
36
|
+
return {
|
|
37
|
+
...options.body,
|
|
38
|
+
model: options.model,
|
|
39
|
+
messages: messages.map(toWire),
|
|
40
|
+
...(chat.tools?.length ? { tools: chat.tools } : {}),
|
|
41
|
+
...(chat.json ? { response_format: { type: "json_object" } } : {}),
|
|
42
|
+
...(chat.temperature === undefined ? {} : { temperature: chat.temperature }),
|
|
43
|
+
...(chat.maxTokens === undefined ? {} : { max_tokens: chat.maxTokens }),
|
|
44
|
+
...(stream ? { stream: true, stream_options: { include_usage: true } } : {}),
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
function reply(data) {
|
|
48
|
+
const choice = data.choices?.[0];
|
|
49
|
+
if (!choice?.message)
|
|
50
|
+
throw failure(origin, "bad_response", "The reply has no choices[0].message.");
|
|
51
|
+
const wire = choice.message;
|
|
52
|
+
const reasoning = wire.reasoning_content ?? wire.reasoning;
|
|
53
|
+
const message = {
|
|
54
|
+
role: "assistant",
|
|
55
|
+
content: wire.content ?? null,
|
|
56
|
+
...(wire.tool_calls?.length ? { tool_calls: wire.tool_calls } : {}),
|
|
57
|
+
...(reasoning ? { reasoning } : {}),
|
|
58
|
+
};
|
|
59
|
+
const usage = data.usage ? { inputTokens: data.usage.prompt_tokens ?? 0, outputTokens: data.usage.completion_tokens ?? 0 } : undefined;
|
|
60
|
+
return { message, finishReason: finishReason(choice.finish_reason), ...(usage ? { usage } : {}) };
|
|
61
|
+
}
|
|
62
|
+
/** Join streamed pieces into one completion, calling onDelta with each piece of text. */
|
|
63
|
+
async function collect(fetched, onDelta) {
|
|
64
|
+
const message = { content: "" };
|
|
65
|
+
const calls = [];
|
|
66
|
+
let finish;
|
|
67
|
+
let done = false;
|
|
68
|
+
let usage;
|
|
69
|
+
try {
|
|
70
|
+
for await (const event of events(fetched.body)) {
|
|
71
|
+
if (event.data === "[DONE]") {
|
|
72
|
+
done = true;
|
|
73
|
+
break;
|
|
74
|
+
}
|
|
75
|
+
let chunk;
|
|
76
|
+
try {
|
|
77
|
+
chunk = JSON.parse(event.data);
|
|
78
|
+
}
|
|
79
|
+
catch {
|
|
80
|
+
throw failure(origin, "bad_response", `The stream sent data that is not JSON: ${event.data.slice(0, 120)}`, { partial: message.content });
|
|
81
|
+
}
|
|
82
|
+
if (chunk.error) {
|
|
83
|
+
const said = typeof chunk.error === "string" ? chunk.error : chunk.error.message;
|
|
84
|
+
throw failure(origin, "http", `The server sent an error in the stream: ${said ?? event.data}`, { partial: message.content });
|
|
85
|
+
}
|
|
86
|
+
usage = chunk.usage ?? usage;
|
|
87
|
+
const choice = chunk.choices?.[0];
|
|
88
|
+
const delta = choice?.delta ?? {};
|
|
89
|
+
if (delta.content) {
|
|
90
|
+
message.content += delta.content;
|
|
91
|
+
onDelta(delta.content);
|
|
92
|
+
}
|
|
93
|
+
const thought = delta.reasoning_content ?? delta.reasoning;
|
|
94
|
+
if (thought)
|
|
95
|
+
message.reasoning_content = (message.reasoning_content ?? "") + thought;
|
|
96
|
+
for (const piece of delta.tool_calls ?? []) {
|
|
97
|
+
// With no index, a new id starts a new call; otherwise the piece belongs to the last call.
|
|
98
|
+
const top = calls.length - 1;
|
|
99
|
+
const index = piece.index ?? (top < 0 || (piece.id && piece.id !== calls[top].id) ? top + 1 : top);
|
|
100
|
+
const slot = (calls[index] ??= { id: "", type: "function", function: { name: "", arguments: "" } });
|
|
101
|
+
if (piece.id)
|
|
102
|
+
slot.id = piece.id;
|
|
103
|
+
// Some servers repeat the name in every piece: keep the first one.
|
|
104
|
+
if (piece.function?.name && !slot.function.name)
|
|
105
|
+
slot.function.name = piece.function.name;
|
|
106
|
+
slot.function.arguments += piece.function?.arguments ?? "";
|
|
107
|
+
}
|
|
108
|
+
finish = choice?.finish_reason ?? finish;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
catch (error) {
|
|
112
|
+
if (error instanceof FoxmindError)
|
|
113
|
+
throw error;
|
|
114
|
+
// A connection that drops after the server said it was done lost nothing.
|
|
115
|
+
if (!finish)
|
|
116
|
+
throw fetched.fail(error, message.content);
|
|
117
|
+
}
|
|
118
|
+
if (!done && !finish)
|
|
119
|
+
throw failure(origin, "stream_interrupted", "The stream ended before the server said it was done.", { partial: message.content });
|
|
120
|
+
const toolCalls = calls.filter(Boolean);
|
|
121
|
+
return {
|
|
122
|
+
choices: [{ message: { ...message, content: message.content || (toolCalls.length ? null : ""), ...(toolCalls.length ? { tool_calls: toolCalls } : {}) }, finish_reason: finish }],
|
|
123
|
+
...(usage ? { usage } : {}),
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
return {
|
|
127
|
+
name,
|
|
128
|
+
tier,
|
|
129
|
+
model: options.model,
|
|
130
|
+
capabilities: ["chat", "embed"],
|
|
131
|
+
async probe(probeOptions = {}) {
|
|
132
|
+
try {
|
|
133
|
+
const fetched = await call(origin, `${baseURL}/models`, { headers, ...probeOptions, timeoutMs: probeOptions.timeoutMs ?? 3000 }, 3000);
|
|
134
|
+
const ids = ((await fetched.json()).data ?? []).map((model) => model.id);
|
|
135
|
+
const check = options.checkModel ?? true;
|
|
136
|
+
const found = check === false || (check instanceof RegExp ? ids.some((id) => check.test(id)) : ids.includes(options.model));
|
|
137
|
+
last = found
|
|
138
|
+
? { ok: true, where: baseURL }
|
|
139
|
+
: { ok: false, code: "model_not_found", where: baseURL, reason: `The server does not list ${check instanceof RegExp ? String(check) : `"${options.model}"`}. It has: ${ids.join(", ") || "no models"}.` };
|
|
140
|
+
}
|
|
141
|
+
catch (error) {
|
|
142
|
+
const failed = error;
|
|
143
|
+
last = { ok: false, code: failed.code ?? "unreachable", where: baseURL, reason: failed.message ?? String(error) };
|
|
144
|
+
}
|
|
145
|
+
if (!last.ok && options.hint)
|
|
146
|
+
last.reason = `${last.reason} ${options.hint}`;
|
|
147
|
+
return last;
|
|
148
|
+
},
|
|
149
|
+
status() {
|
|
150
|
+
return {
|
|
151
|
+
name,
|
|
152
|
+
tier,
|
|
153
|
+
model: options.model,
|
|
154
|
+
capabilities: ["chat", "embed"],
|
|
155
|
+
state: !last ? "idle" : last.ok ? "ready" : "unavailable",
|
|
156
|
+
where: baseURL,
|
|
157
|
+
...(last?.reason ? { reason: last.reason } : {}),
|
|
158
|
+
};
|
|
159
|
+
},
|
|
160
|
+
async chat(messages, chat) {
|
|
161
|
+
const url = `${baseURL}/chat/completions`;
|
|
162
|
+
const stream = chat.onDelta !== undefined;
|
|
163
|
+
const fetched = await call(origin, url, { ...chat, headers, body: request(messages, chat, stream) }, timeout);
|
|
164
|
+
const data = chat.onDelta ? await collect(fetched, chat.onDelta) : await fetched.json();
|
|
165
|
+
return checkReply(origin, reply(data), chat.json);
|
|
166
|
+
},
|
|
167
|
+
async embed(texts, embed) {
|
|
168
|
+
const url = `${baseURL}/embeddings`;
|
|
169
|
+
const fetched = await call(origin, url, { ...embed, headers, body: { model: options.embedModel ?? options.model, input: texts } }, timeout);
|
|
170
|
+
const data = (await fetched.json()).data;
|
|
171
|
+
if (!Array.isArray(data) || data.length !== texts.length) {
|
|
172
|
+
throw failure(origin, "bad_response", `Asked for ${texts.length} embeddings and got ${Array.isArray(data) ? data.length : "none"}.`);
|
|
173
|
+
}
|
|
174
|
+
return data.toSorted((a, b) => a.index - b.index).map((item) => item.embedding);
|
|
175
|
+
},
|
|
176
|
+
};
|
|
177
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { type OpenAICompatibleOptions } from "./openai.js";
|
|
2
|
+
import type { Provider } from "../types.js";
|
|
3
|
+
type PresetOptions = Partial<Omit<OpenAICompatibleOptions, "model">>;
|
|
4
|
+
/** Underdog Saluki 27B 1.0, from its Hugging Face model card (checked 2026-10-08). */
|
|
5
|
+
export declare const SALUKI: {
|
|
6
|
+
readonly repo: "ConwayResearch/Underdog-Saluki-27B-1.0";
|
|
7
|
+
readonly file: "Underdog-Saluki-27B-1.0-IQ2-mix.gguf";
|
|
8
|
+
readonly sizeGB: 7.89;
|
|
9
|
+
readonly license: "Apache-2.0";
|
|
10
|
+
readonly download: "huggingface-cli download ConwayResearch/Underdog-Saluki-27B-1.0 Underdog-Saluki-27B-1.0-IQ2-mix.gguf --local-dir .";
|
|
11
|
+
readonly serve: "llama-server -m Underdog-Saluki-27B-1.0-IQ2-mix.gguf --jinja -ngl 99 -fa on -c 32768";
|
|
12
|
+
};
|
|
13
|
+
/** Ollama's OpenAI-compatible API. "llama3.2" matches "llama3.2:latest". */
|
|
14
|
+
export declare function ollama(options: PresetOptions & {
|
|
15
|
+
model: string;
|
|
16
|
+
}): Provider;
|
|
17
|
+
/** llama.cpp llama-server. It serves one model, so the name is not checked. */
|
|
18
|
+
export declare function llamaServer(options?: PresetOptions & {
|
|
19
|
+
model?: string;
|
|
20
|
+
}): Provider;
|
|
21
|
+
/** LM Studio's local server. */
|
|
22
|
+
export declare function lmStudio(options: PresetOptions & {
|
|
23
|
+
model: string;
|
|
24
|
+
}): Provider;
|
|
25
|
+
/**
|
|
26
|
+
* Underdog Saluki 27B on llama-server: the default planner for private mode.
|
|
27
|
+
* Thinking is off by default with temperature 0, the model card's settings
|
|
28
|
+
* for direct tool calls. `thinking: true` uses its settings for reasoning.
|
|
29
|
+
*/
|
|
30
|
+
export declare function saluki(options?: PresetOptions & {
|
|
31
|
+
thinking?: boolean;
|
|
32
|
+
}): Provider;
|
|
33
|
+
export {};
|