agentfootprint 9.72.0 → 9.74.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +160 -0
- package/dist/adapters/identity/azure.js +306 -0
- package/dist/adapters/identity/azure.js.map +1 -0
- package/dist/adapters/llm/FoundryLocalProvider.js +992 -0
- package/dist/adapters/llm/FoundryLocalProvider.js.map +1 -0
- package/dist/adapters/llm/FoundryProvider.js +273 -0
- package/dist/adapters/llm/FoundryProvider.js.map +1 -0
- package/dist/adapters/llm/OllamaProvider.js +171 -12
- package/dist/adapters/llm/OllamaProvider.js.map +1 -1
- package/dist/adapters/llm/OpenAIProvider.js +117 -5
- package/dist/adapters/llm/OpenAIProvider.js.map +1 -1
- package/dist/adapters/llm/createProvider.js +143 -10
- package/dist/adapters/llm/createProvider.js.map +1 -1
- package/dist/adapters/types.js.map +1 -1
- package/dist/esm/adapters/identity/azure.d.ts +188 -0
- package/dist/esm/adapters/identity/azure.js +302 -0
- package/dist/esm/adapters/identity/azure.js.map +1 -0
- package/dist/esm/adapters/llm/FoundryLocalProvider.d.ts +215 -0
- package/dist/esm/adapters/llm/FoundryLocalProvider.js +986 -0
- package/dist/esm/adapters/llm/FoundryLocalProvider.js.map +1 -0
- package/dist/esm/adapters/llm/FoundryProvider.d.ts +178 -0
- package/dist/esm/adapters/llm/FoundryProvider.js +268 -0
- package/dist/esm/adapters/llm/FoundryProvider.js.map +1 -0
- package/dist/esm/adapters/llm/OllamaProvider.js +171 -12
- package/dist/esm/adapters/llm/OllamaProvider.js.map +1 -1
- package/dist/esm/adapters/llm/OpenAIProvider.d.ts +111 -0
- package/dist/esm/adapters/llm/OpenAIProvider.js +115 -4
- package/dist/esm/adapters/llm/OpenAIProvider.js.map +1 -1
- package/dist/esm/adapters/llm/createProvider.d.ts +48 -12
- package/dist/esm/adapters/llm/createProvider.js +143 -10
- package/dist/esm/adapters/llm/createProvider.js.map +1 -1
- package/dist/esm/adapters/types.d.ts +4 -2
- package/dist/esm/adapters/types.js.map +1 -1
- package/dist/esm/identity.d.ts +1 -0
- package/dist/esm/identity.js +9 -0
- package/dist/esm/identity.js.map +1 -1
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/providers.d.ts +5 -0
- package/dist/esm/providers.js +14 -0
- package/dist/esm/providers.js.map +1 -1
- package/dist/identity.js +14 -1
- package/dist/identity.js.map +1 -1
- package/dist/index.js.map +1 -1
- package/dist/providers.js +20 -1
- package/dist/providers.js.map +1 -1
- package/dist/types/adapters/identity/azure.d.ts +189 -0
- package/dist/types/adapters/identity/azure.d.ts.map +1 -0
- package/dist/types/adapters/llm/FoundryLocalProvider.d.ts +216 -0
- package/dist/types/adapters/llm/FoundryLocalProvider.d.ts.map +1 -0
- package/dist/types/adapters/llm/FoundryProvider.d.ts +179 -0
- package/dist/types/adapters/llm/FoundryProvider.d.ts.map +1 -0
- package/dist/types/adapters/llm/OllamaProvider.d.ts.map +1 -1
- package/dist/types/adapters/llm/OpenAIProvider.d.ts +111 -0
- package/dist/types/adapters/llm/OpenAIProvider.d.ts.map +1 -1
- package/dist/types/adapters/llm/createProvider.d.ts +48 -12
- package/dist/types/adapters/llm/createProvider.d.ts.map +1 -1
- package/dist/types/adapters/types.d.ts +4 -2
- package/dist/types/adapters/types.d.ts.map +1 -1
- package/dist/types/identity.d.ts +1 -0
- package/dist/types/identity.d.ts.map +1 -1
- package/dist/types/index.d.ts +1 -1
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/providers.d.ts +5 -0
- package/dist/types/providers.d.ts.map +1 -1
- package/package.json +5 -1
|
@@ -0,0 +1,992 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* FoundryLocalProvider — on-device models over Foundry Local's
|
|
4
|
+
* OpenAI-compatible `/v1/chat/completions`.
|
|
5
|
+
*
|
|
6
|
+
* Pattern: Adapter (GoF) + Ports-and-Adapters (Cockburn 2005).
|
|
7
|
+
* Role: Outer ring — translates `LLMRequest`/`LLMResponse` to/from the
|
|
8
|
+
* wire Foundry Local serves on localhost. Knows nothing about
|
|
9
|
+
* agents, recorders, or compositions.
|
|
10
|
+
* Emits: N/A.
|
|
11
|
+
*
|
|
12
|
+
* ─── Why this exists ─────────────────────────────────────────────────
|
|
13
|
+
*
|
|
14
|
+
* The adapter ladder is `mock()` → a local model → a paid API, and
|
|
15
|
+
* `ollama()` is the proof that the middle rung is worth owning: zero
|
|
16
|
+
* dependencies, honest refusals, real token counts. Foundry Local is
|
|
17
|
+
* Microsoft's runtime for the same rung — ONNX under the hood, models
|
|
18
|
+
* pulled with `foundry model run <alias>`, no key, no account — and a
|
|
19
|
+
* Windows or macOS machine that has it installed deserves the same
|
|
20
|
+
* one-import experience. So this file owns that wire the way
|
|
21
|
+
* `OllamaProvider.ts` owns Ollama's:
|
|
22
|
+
*
|
|
23
|
+
* • ZERO dependencies — one `fetch` POST and SSE. The official
|
|
24
|
+
* `foundry-local-sdk` is NOT imported; its manager is accepted
|
|
25
|
+
* duck-typed (see {@link FoundryLocalProviderOptions.manager}) so a
|
|
26
|
+
* consumer who already uses it can hand over the discovered URL
|
|
27
|
+
* without this package gaining a dependency.
|
|
28
|
+
* • HONEST REFUSALS — a typed {@link FoundryLocalUnavailableError}
|
|
29
|
+
* that names the endpoint it tried and the command to run. The
|
|
30
|
+
* service's port is DYNAMIC per start, which makes "nothing is
|
|
31
|
+
* answering" the most likely first failure — so that message
|
|
32
|
+
* carries the discovery command, not just the start command. A
|
|
33
|
+
* failure the service reports IN BAND — an `error` frame on an
|
|
34
|
+
* already-200 stream, the out-of-memory a laptop runtime really does
|
|
35
|
+
* hit — is RAISED the same way, never handed over as a shorter
|
|
36
|
+
* answer that reads like a clean stop.
|
|
37
|
+
* • REAL TOKEN COUNTS while streaming — `stream_options:
|
|
38
|
+
* { include_usage: true }` is always sent. This is OUR wire, a
|
|
39
|
+
* documented Foundry Local surface, not an arbitrary
|
|
40
|
+
* OpenAI-compatible server — so the caution that made
|
|
41
|
+
* `openai({ baseURL })` withhold the field (and silently zero every
|
|
42
|
+
* local token count until 9.73.0) does not apply here.
|
|
43
|
+
*
|
|
44
|
+
* ─── Wire realities this file owns ───────────────────────────────────
|
|
45
|
+
*
|
|
46
|
+
* • THE PORT IS DYNAMIC. Every `foundry server start` may pick a new
|
|
47
|
+
* port; the docs' own REST example shows `http://localhost:5272` and
|
|
48
|
+
* that is the default here, but the truthful discovery is
|
|
49
|
+
* `foundry server status` (or the SDK manager's `.urls`). Note the
|
|
50
|
+
* CLI group was RENAMED from `foundry service` to `foundry server` —
|
|
51
|
+
* every message in this file uses the NEW spelling.
|
|
52
|
+
* • ALIASES vs VARIANT IDS. The catalog speaks in aliases
|
|
53
|
+
* (`qwen2.5-0.5b`) that fan out to hardware variants
|
|
54
|
+
* (`qwen2.5-0.5b-instruct-generic-cpu:1`), but REST chat calls take
|
|
55
|
+
* the FULL variant id. This adapter resolves an alias through
|
|
56
|
+
* `GET /foundry/list` — first matching variant wins, because the
|
|
57
|
+
* list's order IS the service's priority order — and caches the
|
|
58
|
+
* answer per provider instance, HIT OR MISS: exactly one catalog
|
|
59
|
+
* attempt per name, so an alias the catalog never answers for cannot
|
|
60
|
+
* re-ask before every call. A fresh provider is the retry. A name that
|
|
61
|
+
* already carries a variant's execution-provider suffix
|
|
62
|
+
* (`-cpu`/`-gpu`/`-npu`, optional `:version`) is used as-is with no
|
|
63
|
+
* catalog round-trip.
|
|
64
|
+
* • NO API KEY EXISTS. The docs' own samples pass placeholders. This
|
|
65
|
+
* adapter sends no `Authorization` header at all — there is nothing
|
|
66
|
+
* to put in one, and an invented value would only end up in somebody's
|
|
67
|
+
* proxy log.
|
|
68
|
+
*
|
|
69
|
+
* ─── Ceilings (stated, not worked around) ────────────────────────────
|
|
70
|
+
*
|
|
71
|
+
* • NO FORCED TOOL CHOICE. `tool_choice` support is UNDOCUMENTED on
|
|
72
|
+
* this wire, so `carriesForcedToolChoice` is `false` and an agent
|
|
73
|
+
* using `.outputSchema(parser, { strategy: 'tool-forced' })` refuses
|
|
74
|
+
* at run start, naming this provider. Claiming an undocumented field
|
|
75
|
+
* works would turn a guarantee into a suggestion.
|
|
76
|
+
* • TOOL CALLING IS MODEL-DEPENDENT. `/foundry/list` reports
|
|
77
|
+
* `supportsToolCalling` per variant, but this adapter does not
|
|
78
|
+
* preflight-refuse on it — a wrong refusal is worse than a weak
|
|
79
|
+
* answer, the same stance `ollama()` takes on `/api/show`. Pick a
|
|
80
|
+
* tool-capable variant.
|
|
81
|
+
* • NO MULTI-MODAL. `LLMMessage.content` is a string. Same ceiling as
|
|
82
|
+
* every other adapter here.
|
|
83
|
+
* • NO PROMPT CACHING — resolves to the NoOp cache strategy.
|
|
84
|
+
* • NO STRUCTURED THINKING. The wire has no thinking field; a reasoning
|
|
85
|
+
* model's `<think>` tags ride the answer text untouched.
|
|
86
|
+
*/
|
|
87
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
88
|
+
exports.FoundryLocalProvider = exports.foundryLocal = exports.FoundryLocalUnavailableError = void 0;
|
|
89
|
+
/** Cap for any text the wire supplies before it becomes one of our messages. */
|
|
90
|
+
const ERROR_TEXT_CAP = 200;
|
|
91
|
+
// ─── Errors ─────────────────────────────────────────────────────────
|
|
92
|
+
/**
|
|
93
|
+
* The two failures an on-device runtime actually has, told in words that
|
|
94
|
+
* contain the fix.
|
|
95
|
+
*
|
|
96
|
+
* Both are things the person at the keyboard can resolve in one command,
|
|
97
|
+
* which is exactly why they get a type instead of a wrapped
|
|
98
|
+
* `ECONNREFUSED` or a bare `404`. `reason` is the discriminator; the
|
|
99
|
+
* message already reads as instructions — and because Foundry Local's
|
|
100
|
+
* port changes per start, the unreachable message teaches the discovery
|
|
101
|
+
* command (`foundry server status`) alongside the start command.
|
|
102
|
+
*/
|
|
103
|
+
class FoundryLocalUnavailableError extends Error {
|
|
104
|
+
name = 'FoundryLocalUnavailableError';
|
|
105
|
+
/** Which of the two situations this is. */
|
|
106
|
+
reason;
|
|
107
|
+
/** The endpoint that was tried — the thing to check or change. */
|
|
108
|
+
endpoint;
|
|
109
|
+
/** The model asked for. Absent when the service never answered at all. */
|
|
110
|
+
model;
|
|
111
|
+
/** Models this machine DOES have cached, when the service could tell us. */
|
|
112
|
+
availableModels;
|
|
113
|
+
constructor(init) {
|
|
114
|
+
super(buildUnavailableMessage(init));
|
|
115
|
+
this.reason = init.reason;
|
|
116
|
+
this.endpoint = init.endpoint;
|
|
117
|
+
if (init.model !== undefined)
|
|
118
|
+
this.model = init.model;
|
|
119
|
+
if (init.availableModels !== undefined)
|
|
120
|
+
this.availableModels = init.availableModels;
|
|
121
|
+
if (init.cause !== undefined)
|
|
122
|
+
this.cause = init.cause;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
exports.FoundryLocalUnavailableError = FoundryLocalUnavailableError;
|
|
126
|
+
function buildUnavailableMessage(init) {
|
|
127
|
+
if (init.reason === 'service-unreachable') {
|
|
128
|
+
return (`foundryLocal: nothing is answering at ${init.endpoint}. ` +
|
|
129
|
+
'Start it with `foundry server start` (Foundry Local, install: ' +
|
|
130
|
+
'https://learn.microsoft.com/azure/foundry-local). ' +
|
|
131
|
+
'The port is dynamic — `foundry server status` prints the live URL; ' +
|
|
132
|
+
'pin one with `foundry server start --port <p>`. ' +
|
|
133
|
+
"Running elsewhere? Pass foundryLocal('<model>', { endpoint }) or set FOUNDRY_LOCAL_ENDPOINT.");
|
|
134
|
+
}
|
|
135
|
+
const model = init.model ?? '(unnamed)';
|
|
136
|
+
const have = init.availableModels && init.availableModels.length > 0
|
|
137
|
+
? ` Models on this machine: ${init.availableModels.join(', ')}.`
|
|
138
|
+
: '';
|
|
139
|
+
// A 404 that did not speak the dialect is as likely a wrong route as a
|
|
140
|
+
// missing model. Say both rather than one confidently.
|
|
141
|
+
const route = init.routeUnconfirmed
|
|
142
|
+
? ` That 404 named no model error, so ${init.endpoint} may not be Foundry Local's chat route at all — \`foundry server status\` prints the live URL.`
|
|
143
|
+
: '';
|
|
144
|
+
return (`foundryLocal: model '${model}' is not available on the service at ${init.endpoint}. ` +
|
|
145
|
+
`Run: foundry model run ${model} — it downloads the model if needed (aliases work too).${have}${route}`);
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* The docs' own REST example port. Only a default, never a promise —
|
|
149
|
+
* see {@link FoundryLocalProviderOptions.endpoint} for the truth about
|
|
150
|
+
* dynamic ports.
|
|
151
|
+
*/
|
|
152
|
+
const DEFAULT_ENDPOINT = 'http://localhost:5272';
|
|
153
|
+
/** The alias the Foundry Local docs use in their own REST walkthrough. */
|
|
154
|
+
const DEFAULT_MODEL = 'qwen2.5-0.5b';
|
|
155
|
+
const DEFAULT_TIMEOUT_MS = 10_000;
|
|
156
|
+
/** Request `model` values that mean "whatever this provider was configured with". */
|
|
157
|
+
const MODEL_SHORTHANDS = new Set(['foundry-local']);
|
|
158
|
+
/**
|
|
159
|
+
* Which roles this wire carries inside `messages`.
|
|
160
|
+
*
|
|
161
|
+
* The OpenAI dialect takes the system prompt as a message like any other
|
|
162
|
+
* (no separate top-level `system` field), so all three roles survive the
|
|
163
|
+
* trip.
|
|
164
|
+
*/
|
|
165
|
+
const CARRIES_IN_MESSAGES = Object.freeze(['system', 'user', 'assistant']);
|
|
166
|
+
function foundryLocal(modelOrOptions, maybeOptions) {
|
|
167
|
+
const options = typeof modelOrOptions === 'string' ? maybeOptions ?? {} : modelOrOptions ?? {};
|
|
168
|
+
const positionalModel = typeof modelOrOptions === 'string' ? modelOrOptions : undefined;
|
|
169
|
+
const endpoint = resolveEndpoint(options);
|
|
170
|
+
const defaultModel = positionalModel ?? options.defaultModel ?? DEFAULT_MODEL;
|
|
171
|
+
const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
172
|
+
const fetchImpl = options._fetch ?? ((...args) => fetch(...args));
|
|
173
|
+
const chatUrl = `${endpoint}/v1/chat/completions`;
|
|
174
|
+
const cfg = {
|
|
175
|
+
defaultModel,
|
|
176
|
+
...(options.defaultMaxTokens !== undefined && { defaultMaxTokens: options.defaultMaxTokens }),
|
|
177
|
+
};
|
|
178
|
+
// Tool-call ids are synthesized per provider instance — the dialect says
|
|
179
|
+
// every call carries one, small local models sometimes disagree, and the
|
|
180
|
+
// whole tool round-trip in this library is keyed by id.
|
|
181
|
+
let toolCallSeq = 0;
|
|
182
|
+
const nextToolCallId = () => `foundry-call-${++toolCallSeq}`;
|
|
183
|
+
// Alias → variant-id resolutions, cached per provider instance — HIT OR
|
|
184
|
+
// MISS. The catalog does not change under a running service often enough
|
|
185
|
+
// to be worth re-asking on every call, and the cache is what keeps an
|
|
186
|
+
// alias-configured agent at one catalog fetch per process. Caching the
|
|
187
|
+
// FALLBACK too is what keeps that promise true when the catalog has no
|
|
188
|
+
// answer for the alias.
|
|
189
|
+
const resolutionCache = new Map();
|
|
190
|
+
/**
|
|
191
|
+
* REST chat calls take the FULL variant id, so an alias must be
|
|
192
|
+
* resolved first:
|
|
193
|
+
*
|
|
194
|
+
* • A name already shaped like a variant id (it ends in the
|
|
195
|
+
* execution-provider suffix `-cpu`/`-gpu`/`-npu`, optionally
|
|
196
|
+
* `:version`) is used AS-IS — no catalog round-trip. That suffix is
|
|
197
|
+
* how Foundry Local itself distinguishes a variant from an alias.
|
|
198
|
+
* • Anything else is treated as an alias: `GET /foundry/list`, first
|
|
199
|
+
* variant whose alias matches wins (the list's order is the
|
|
200
|
+
* service's priority order), and the answer is cached.
|
|
201
|
+
* • A silent or alias-less catalog resolves to the name UNCHANGED —
|
|
202
|
+
* the chat call's own 404 then reports honestly, with the model
|
|
203
|
+
* name the caller actually wrote. A failed lookup must never
|
|
204
|
+
* replace the error that was coming anyway. That fallback is CACHED
|
|
205
|
+
* exactly like a success: ONE catalog attempt per name per provider,
|
|
206
|
+
* so an alias the service has no answer for cannot put a
|
|
207
|
+
* `/foundry/list` round-trip — and, against a host that accepts but
|
|
208
|
+
* never answers, a whole `timeoutMs` — in front of every single
|
|
209
|
+
* call. A model pulled with `foundry model run` after the fact is
|
|
210
|
+
* picked up by a FRESH provider; that is the retry.
|
|
211
|
+
*/
|
|
212
|
+
const resolveModel = async (requested) => {
|
|
213
|
+
const named = MODEL_SHORTHANDS.has(requested) ? cfg.defaultModel : requested;
|
|
214
|
+
if (looksLikeVariantId(named))
|
|
215
|
+
return named;
|
|
216
|
+
const cached = resolutionCache.get(named);
|
|
217
|
+
if (cached !== undefined)
|
|
218
|
+
return cached;
|
|
219
|
+
const catalog = await fetchJsonBounded(fetchImpl, `${endpoint}/foundry/list`, timeoutMs);
|
|
220
|
+
// The fallback is cached like an answer, so the catalog is asked once
|
|
221
|
+
// per name whatever it says. A fresh provider re-asks.
|
|
222
|
+
const answer = firstVariantForAlias(catalog, named) ?? named;
|
|
223
|
+
resolutionCache.set(named, answer);
|
|
224
|
+
return answer;
|
|
225
|
+
};
|
|
226
|
+
const post = async (body, req) => {
|
|
227
|
+
// No Authorization header: no key exists on this wire, and nothing is
|
|
228
|
+
// sent where a secret could be invented, logged, or leaked.
|
|
229
|
+
const response = await fetchUntilHeaders(fetchImpl, chatUrl, {
|
|
230
|
+
method: 'POST',
|
|
231
|
+
headers: { 'content-type': 'application/json' },
|
|
232
|
+
body: JSON.stringify(body),
|
|
233
|
+
}, { timeoutMs, endpoint, ...(req.signal && { signal: req.signal }) });
|
|
234
|
+
if (!response.ok) {
|
|
235
|
+
throw await describeFailure(response, body.model, endpoint, fetchImpl, timeoutMs);
|
|
236
|
+
}
|
|
237
|
+
return response;
|
|
238
|
+
};
|
|
239
|
+
const provider = {
|
|
240
|
+
name: 'foundry-local',
|
|
241
|
+
carriesInMessages: CARRIES_IN_MESSAGES,
|
|
242
|
+
// `tool_choice` support is undocumented on this wire. Absence would mean
|
|
243
|
+
// the same thing; saying it out loud documents that this was checked
|
|
244
|
+
// rather than forgotten. See the header's ceilings.
|
|
245
|
+
carriesForcedToolChoice: false,
|
|
246
|
+
async complete(req) {
|
|
247
|
+
// A signal that fired BEFORE the call stops it here, before any socket
|
|
248
|
+
// opens — the catalog lookup included. An already-aborted signal never
|
|
249
|
+
// dispatches another 'abort' event, so a listener alone cannot see it.
|
|
250
|
+
throwIfAborted(req.signal);
|
|
251
|
+
const model = await resolveModel(req.model);
|
|
252
|
+
const body = buildBody(req, cfg, model, false);
|
|
253
|
+
const response = await post(body, req);
|
|
254
|
+
// `fetch` resolves on HEADERS, so reading the body is the stretch where
|
|
255
|
+
// a caller abort would otherwise be inert — honor it for that too.
|
|
256
|
+
const json = (await untilAborted(response.json(), req.signal));
|
|
257
|
+
return fromFoundryResponse(json, nextToolCallId);
|
|
258
|
+
},
|
|
259
|
+
async *stream(req) {
|
|
260
|
+
throwIfAborted(req.signal);
|
|
261
|
+
const model = await resolveModel(req.model);
|
|
262
|
+
const body = buildBody(req, cfg, model, true);
|
|
263
|
+
const response = await post(body, req);
|
|
264
|
+
if (!response.body)
|
|
265
|
+
throw new Error('[foundry-local] response has no body');
|
|
266
|
+
const textParts = [];
|
|
267
|
+
// The dialect streams tool calls as DELTAS — id and name on the first
|
|
268
|
+
// fragment, argument JSON split across the rest — assembled by index.
|
|
269
|
+
const toolCallsByIndex = new Map();
|
|
270
|
+
let finishReason = null;
|
|
271
|
+
let usage;
|
|
272
|
+
let lastId = '';
|
|
273
|
+
let tokenIndex = 0;
|
|
274
|
+
// The signal rides INTO the parser: it is what stops a generation the
|
|
275
|
+
// caller no longer wants, and what cancels the body when it does.
|
|
276
|
+
for await (const parsed of parseSse(response.body, req.signal)) {
|
|
277
|
+
const frame = parsed.data;
|
|
278
|
+
// An already-200 stream can still FAIL mid-generation, and this
|
|
279
|
+
// dialect says so IN BAND. Such a frame carries no `choices`, so the
|
|
280
|
+
// guard below would drop it and the terminal chunk would report a
|
|
281
|
+
// truncated answer as `stopReason: 'stop'` with nobody told. Raise
|
|
282
|
+
// instead: a failed generation is a failure, never a shorter answer.
|
|
283
|
+
const failure = frameFailureText(parsed);
|
|
284
|
+
if (failure) {
|
|
285
|
+
throw providerError(`[foundry-local] the stream failed mid-generation — ${failure}`);
|
|
286
|
+
}
|
|
287
|
+
// Usage FIRST, and OUTSIDE the choice guard. With
|
|
288
|
+
// `stream_options.include_usage` (set in buildBody) the token counts
|
|
289
|
+
// ride a FINAL chunk whose `choices` array is EMPTY — the exact bug
|
|
290
|
+
// class 9.73.0 fixed in the OpenAI adapter: a `continue` on a missing
|
|
291
|
+
// choice threw away the only usage the stream ever reports, and every
|
|
292
|
+
// streamed local call read zero tokens everywhere usage is consumed.
|
|
293
|
+
if (frame.id)
|
|
294
|
+
lastId = frame.id;
|
|
295
|
+
if (frame.usage)
|
|
296
|
+
usage = frame.usage;
|
|
297
|
+
const choice = frame.choices?.[0];
|
|
298
|
+
if (!choice)
|
|
299
|
+
continue;
|
|
300
|
+
if (choice.finish_reason)
|
|
301
|
+
finishReason = choice.finish_reason;
|
|
302
|
+
const delta = choice.delta;
|
|
303
|
+
if (!delta)
|
|
304
|
+
continue;
|
|
305
|
+
if (delta.content) {
|
|
306
|
+
textParts.push(delta.content);
|
|
307
|
+
yield { tokenIndex, content: delta.content, done: false };
|
|
308
|
+
tokenIndex++;
|
|
309
|
+
}
|
|
310
|
+
if (delta.tool_calls) {
|
|
311
|
+
for (const tcDelta of delta.tool_calls) {
|
|
312
|
+
const idx = tcDelta.index ?? 0;
|
|
313
|
+
const existing = toolCallsByIndex.get(idx) ?? { id: '', name: '', argsJson: '' };
|
|
314
|
+
if (tcDelta.id)
|
|
315
|
+
existing.id = tcDelta.id;
|
|
316
|
+
if (tcDelta.function?.name)
|
|
317
|
+
existing.name = tcDelta.function.name;
|
|
318
|
+
if (tcDelta.function?.arguments)
|
|
319
|
+
existing.argsJson += tcDelta.function.arguments;
|
|
320
|
+
toolCallsByIndex.set(idx, existing);
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
// Sorted by INDEX, not by first-seen order: a wire is free to open
|
|
325
|
+
// index 1 before index 0, and a consumer reading `toolCalls[0]` as
|
|
326
|
+
// "the first tool the model asked for" would then name the wrong one.
|
|
327
|
+
const toolCalls = Array.from(toolCallsByIndex.entries())
|
|
328
|
+
.sort(([a], [b]) => a - b)
|
|
329
|
+
.map(([, tc]) => ({
|
|
330
|
+
id: tc.id.length > 0 ? tc.id : nextToolCallId(),
|
|
331
|
+
name: tc.name,
|
|
332
|
+
args: coerceArgs(tc.argsJson),
|
|
333
|
+
}));
|
|
334
|
+
const authoritative = {
|
|
335
|
+
content: textParts.join(''),
|
|
336
|
+
toolCalls,
|
|
337
|
+
usage: {
|
|
338
|
+
input: usage?.prompt_tokens ?? 0,
|
|
339
|
+
output: usage?.completion_tokens ?? 0,
|
|
340
|
+
},
|
|
341
|
+
stopReason: normalizeStopReason(finishReason ?? 'stop', toolCalls.length > 0),
|
|
342
|
+
...(lastId && { providerRef: lastId }),
|
|
343
|
+
};
|
|
344
|
+
yield { tokenIndex, content: '', done: true, response: authoritative };
|
|
345
|
+
},
|
|
346
|
+
};
|
|
347
|
+
return provider;
|
|
348
|
+
}
|
|
349
|
+
exports.foundryLocal = foundryLocal;
|
|
350
|
+
/**
|
|
351
|
+
* Class form for consumers who prefer `new FoundryLocalProvider(...)`.
|
|
352
|
+
*/
|
|
353
|
+
class FoundryLocalProvider {
|
|
354
|
+
name = 'foundry-local';
|
|
355
|
+
carriesInMessages = CARRIES_IN_MESSAGES;
|
|
356
|
+
carriesForcedToolChoice = false;
|
|
357
|
+
inner;
|
|
358
|
+
constructor(model, options) {
|
|
359
|
+
this.inner =
|
|
360
|
+
typeof model === 'string'
|
|
361
|
+
? foundryLocal(model, options)
|
|
362
|
+
: foundryLocal(model);
|
|
363
|
+
}
|
|
364
|
+
// `hooks` is FORWARDED, not dropped — see LLMCallHooks in adapters/types.ts.
|
|
365
|
+
complete(req, hooks) {
|
|
366
|
+
return this.inner.complete(req, hooks);
|
|
367
|
+
}
|
|
368
|
+
stream(req, hooks) {
|
|
369
|
+
if (!this.inner.stream)
|
|
370
|
+
throw new Error('stream() unavailable');
|
|
371
|
+
return this.inner.stream(req, hooks);
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
exports.FoundryLocalProvider = FoundryLocalProvider;
|
|
375
|
+
// ─── Internals ──────────────────────────────────────────────────────
|
|
376
|
+
/**
|
|
377
|
+
* Resolve where the service lives. Most specific wins: an explicit
|
|
378
|
+
* `endpoint`, then a duck-typed SDK manager's discovered URL, then the
|
|
379
|
+
* two env spellings, then the docs' example port.
|
|
380
|
+
*
|
|
381
|
+
* A `/v1` suffix is trimmed rather than rejected — someone copying the
|
|
382
|
+
* chat URL out of `foundry server status` output means the same
|
|
383
|
+
* machine — and a bare `host:port` gets `http://`.
|
|
384
|
+
*/
|
|
385
|
+
function resolveEndpoint(options) {
|
|
386
|
+
const env = typeof process !== 'undefined' ? process.env : undefined;
|
|
387
|
+
const raw = firstConfigured(options.endpoint, options.manager?.urls?.[0], env?.FOUNDRY_LOCAL_ENDPOINT, env?.FOUNDRY_LOCAL_BASE_URL) ?? DEFAULT_ENDPOINT;
|
|
388
|
+
const withScheme = /^https?:\/\//i.test(raw) ? raw : `http://${raw}`;
|
|
389
|
+
return withScheme.replace(/\/+$/, '').replace(/\/v1$/i, '');
|
|
390
|
+
}
|
|
391
|
+
/**
|
|
392
|
+
* First candidate that actually says something.
|
|
393
|
+
*
|
|
394
|
+
* A blank value is treated as ABSENT rather than as an endpoint. `??` would
|
|
395
|
+
* accept `''`, and `'' → 'http://' → strip trailing slashes` leaves the URL
|
|
396
|
+
* `http:` — a refusal naming nothing the reader can check or fix. An env key
|
|
397
|
+
* present with an empty value is a config that forgot to fill it in, so the
|
|
398
|
+
* next candidate (and ultimately the default) gets its turn.
|
|
399
|
+
*/
|
|
400
|
+
function firstConfigured(...candidates) {
|
|
401
|
+
for (const candidate of candidates) {
|
|
402
|
+
if (typeof candidate === 'string' && candidate.trim().length > 0)
|
|
403
|
+
return candidate.trim();
|
|
404
|
+
}
|
|
405
|
+
return undefined;
|
|
406
|
+
}
|
|
407
|
+
/**
|
|
408
|
+
* Does this name already carry a variant's shape?
|
|
409
|
+
*
|
|
410
|
+
* Full Foundry Local variant ids name their execution provider as the
|
|
411
|
+
* terminal segment — `-cpu`, `-gpu` or `-npu`, optionally followed by a
|
|
412
|
+
* `:version` — e.g. `qwen2.5-0.5b-instruct-generic-cpu:1`. Aliases never
|
|
413
|
+
* do. This structural fact is what lets a fully-qualified id skip the
|
|
414
|
+
* catalog round-trip entirely.
|
|
415
|
+
*/
|
|
416
|
+
function looksLikeVariantId(model) {
|
|
417
|
+
const bare = model.replace(/:\d+$/, '');
|
|
418
|
+
return /-(cpu|gpu|npu)$/i.test(bare);
|
|
419
|
+
}
|
|
420
|
+
/**
|
|
421
|
+
* First catalog variant whose alias matches — `/foundry/list` order is
|
|
422
|
+
* the service's priority order, so first wins. Tolerates both a bare
|
|
423
|
+
* array and a `{ models: [...] }` wrapper, and both id spellings.
|
|
424
|
+
*/
|
|
425
|
+
function firstVariantForAlias(catalog, alias) {
|
|
426
|
+
const entries = Array.isArray(catalog)
|
|
427
|
+
? catalog
|
|
428
|
+
: catalog?.models;
|
|
429
|
+
if (!Array.isArray(entries))
|
|
430
|
+
return undefined;
|
|
431
|
+
for (const raw of entries) {
|
|
432
|
+
if (typeof raw !== 'object' || raw === null)
|
|
433
|
+
continue;
|
|
434
|
+
const entry = raw;
|
|
435
|
+
if (entry.alias !== alias)
|
|
436
|
+
continue;
|
|
437
|
+
const id = entry.name ?? entry.id;
|
|
438
|
+
if (typeof id === 'string' && id.length > 0)
|
|
439
|
+
return id;
|
|
440
|
+
}
|
|
441
|
+
return undefined;
|
|
442
|
+
}
|
|
443
|
+
function buildBody(req, cfg, model, stream) {
|
|
444
|
+
const body = {
|
|
445
|
+
model,
|
|
446
|
+
messages: toFoundryMessages(req.messages, req.systemPrompt),
|
|
447
|
+
stream,
|
|
448
|
+
};
|
|
449
|
+
// Always ask for usage on a stream. This wire is a documented Foundry
|
|
450
|
+
// Local surface, not an arbitrary OpenAI-compatible server, so the
|
|
451
|
+
// reject-unknown-field caution that cost `openai({ baseURL })` its token
|
|
452
|
+
// counts (fixed in 9.73.0) has no purchase here — and a streamed call
|
|
453
|
+
// that reports zero tokens silently disarms `.compaction()` and budgets.
|
|
454
|
+
if (stream)
|
|
455
|
+
body.stream_options = { include_usage: true };
|
|
456
|
+
if (req.tools && req.tools.length > 0)
|
|
457
|
+
body.tools = req.tools.map(toFoundryTool);
|
|
458
|
+
const maxTokens = req.maxTokens ?? cfg.defaultMaxTokens;
|
|
459
|
+
if (maxTokens !== undefined)
|
|
460
|
+
body.max_tokens = maxTokens;
|
|
461
|
+
if (req.temperature !== undefined)
|
|
462
|
+
body.temperature = req.temperature;
|
|
463
|
+
if (req.stop && req.stop.length > 0)
|
|
464
|
+
body.stop = [...req.stop];
|
|
465
|
+
// `req.toolChoice` is intentionally NOT translated: support for it is
|
|
466
|
+
// undocumented on this wire, `carriesForcedToolChoice` says so, and the
|
|
467
|
+
// agent refuses before it ever reaches here.
|
|
468
|
+
return body;
|
|
469
|
+
}
|
|
470
|
+
/**
|
|
471
|
+
* messages → wire messages.
|
|
472
|
+
*
|
|
473
|
+
* Roles map 1:1. The system prompt is prepended as an ordinary `system`
|
|
474
|
+
* message (this dialect has no separate system field). Assistant turns
|
|
475
|
+
* carry their `tool_calls` back — arguments re-serialized to the JSON
|
|
476
|
+
* STRING the dialect expects — and tool results carry `tool_call_id`,
|
|
477
|
+
* which is how this wire correlates a result with the call that asked
|
|
478
|
+
* for it. A role the port does not define is dropped, not forwarded
|
|
479
|
+
* blindly.
|
|
480
|
+
*/
|
|
481
|
+
function toFoundryMessages(messages, systemPrompt) {
|
|
482
|
+
const result = [];
|
|
483
|
+
if (systemPrompt)
|
|
484
|
+
result.push({ role: 'system', content: systemPrompt });
|
|
485
|
+
for (const m of messages) {
|
|
486
|
+
if (m.role === 'system' || m.role === 'user') {
|
|
487
|
+
result.push({ role: m.role, content: m.content });
|
|
488
|
+
continue;
|
|
489
|
+
}
|
|
490
|
+
if (m.role === 'assistant') {
|
|
491
|
+
const msg = { role: 'assistant', content: m.content };
|
|
492
|
+
if (m.toolCalls && m.toolCalls.length > 0) {
|
|
493
|
+
msg.tool_calls = m.toolCalls.map((tc) => ({
|
|
494
|
+
id: tc.id,
|
|
495
|
+
type: 'function',
|
|
496
|
+
function: { name: tc.name, arguments: JSON.stringify(tc.args) },
|
|
497
|
+
}));
|
|
498
|
+
}
|
|
499
|
+
result.push(msg);
|
|
500
|
+
continue;
|
|
501
|
+
}
|
|
502
|
+
if (m.role === 'tool') {
|
|
503
|
+
result.push({
|
|
504
|
+
role: 'tool',
|
|
505
|
+
content: m.content,
|
|
506
|
+
...(m.toolCallId && { tool_call_id: m.toolCallId }),
|
|
507
|
+
});
|
|
508
|
+
continue;
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
return result;
|
|
512
|
+
}
|
|
513
|
+
function toFoundryTool(schema) {
|
|
514
|
+
return {
|
|
515
|
+
type: 'function',
|
|
516
|
+
function: {
|
|
517
|
+
name: schema.name,
|
|
518
|
+
description: schema.description,
|
|
519
|
+
parameters: { ...schema.inputSchema },
|
|
520
|
+
},
|
|
521
|
+
};
|
|
522
|
+
}
|
|
523
|
+
/**
|
|
524
|
+
* Wire response → the port's shape.
|
|
525
|
+
*
|
|
526
|
+
* @throws a `FoundryLocalProviderError` when the 200 body is really a
|
|
527
|
+
* failure (`{"error": ...}`). Reading it as a response would yield empty
|
|
528
|
+
* content with `stopReason: 'stop'` and zero tokens — a failed call
|
|
529
|
+
* dressed as a successful one.
|
|
530
|
+
*/
|
|
531
|
+
function fromFoundryResponse(response, nextToolCallId) {
|
|
532
|
+
if (response.error !== undefined && response.error !== null) {
|
|
533
|
+
const detail = extractErrorPayload(response.error) || 'the service named no reason';
|
|
534
|
+
throw providerError(`[foundry-local] the service answered 200 with an error — ${detail}`);
|
|
535
|
+
}
|
|
536
|
+
const choice = response.choices?.[0];
|
|
537
|
+
const message = choice?.message;
|
|
538
|
+
const wireToolCalls = message?.tool_calls ?? [];
|
|
539
|
+
const toolCalls = wireToolCalls.map((tc) => toLLMToolCall(tc, nextToolCallId));
|
|
540
|
+
return {
|
|
541
|
+
content: message?.content ?? '',
|
|
542
|
+
toolCalls,
|
|
543
|
+
usage: {
|
|
544
|
+
input: response.usage?.prompt_tokens ?? 0,
|
|
545
|
+
output: response.usage?.completion_tokens ?? 0,
|
|
546
|
+
},
|
|
547
|
+
stopReason: normalizeStopReason(choice?.finish_reason ?? 'stop', toolCalls.length > 0),
|
|
548
|
+
...(response.id && { providerRef: response.id }),
|
|
549
|
+
};
|
|
550
|
+
}
|
|
551
|
+
/**
|
|
552
|
+
* One wire tool call → the port's shape.
|
|
553
|
+
*
|
|
554
|
+
* Two wire facts handled here:
|
|
555
|
+
* • `id` should always be present on this dialect, but small local
|
|
556
|
+
* models have been seen to omit it — and the whole tool round-trip
|
|
557
|
+
* in this library is keyed by id. So an id is SYNTHESIZED when
|
|
558
|
+
* missing, unique per provider instance.
|
|
559
|
+
* • `arguments` is a JSON STRING per the dialect (Ollama sends an
|
|
560
|
+
* object); the object form is still tolerated in case a proxy in the
|
|
561
|
+
* middle reshaped it.
|
|
562
|
+
*/
|
|
563
|
+
function toLLMToolCall(tc, nextToolCallId) {
|
|
564
|
+
return {
|
|
565
|
+
id: tc.id && tc.id.length > 0 ? tc.id : nextToolCallId(),
|
|
566
|
+
name: tc.function?.name ?? '',
|
|
567
|
+
args: coerceArgs(tc.function?.arguments),
|
|
568
|
+
};
|
|
569
|
+
}
|
|
570
|
+
function coerceArgs(args) {
|
|
571
|
+
if (args === undefined || args === null)
|
|
572
|
+
return {};
|
|
573
|
+
if (typeof args === 'string') {
|
|
574
|
+
if (args.length === 0)
|
|
575
|
+
return {};
|
|
576
|
+
try {
|
|
577
|
+
const parsed = JSON.parse(args);
|
|
578
|
+
return typeof parsed === 'object' && parsed !== null
|
|
579
|
+
? parsed
|
|
580
|
+
: {};
|
|
581
|
+
}
|
|
582
|
+
catch {
|
|
583
|
+
// Malformed args are rare but observed on small local models. Surface
|
|
584
|
+
// empty rather than crash — the tool-call event still fires, so the
|
|
585
|
+
// problem is visible in the trace.
|
|
586
|
+
return {};
|
|
587
|
+
}
|
|
588
|
+
}
|
|
589
|
+
return { ...args };
|
|
590
|
+
}
|
|
591
|
+
/**
|
|
592
|
+
* `finish_reason` → the port's stop vocabulary.
|
|
593
|
+
*
|
|
594
|
+
* `tool_calls` is the dialect's own word for a tool-ending turn, but a
|
|
595
|
+
* local model has been seen to report plain `stop` with tool calls
|
|
596
|
+
* attached — so the presence of tool calls also decides, the same
|
|
597
|
+
* correction the Ollama adapter makes.
|
|
598
|
+
*/
|
|
599
|
+
function normalizeStopReason(raw, hasToolCalls) {
|
|
600
|
+
if (hasToolCalls && (raw === 'stop' || raw === ''))
|
|
601
|
+
return 'tool_use';
|
|
602
|
+
switch (raw) {
|
|
603
|
+
case 'stop':
|
|
604
|
+
return 'stop';
|
|
605
|
+
case 'tool_calls':
|
|
606
|
+
return 'tool_use';
|
|
607
|
+
case 'length':
|
|
608
|
+
return 'max_tokens';
|
|
609
|
+
case 'content_filter':
|
|
610
|
+
return 'content_filter';
|
|
611
|
+
default:
|
|
612
|
+
return raw;
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
/**
|
|
616
|
+
* Parse an SSE body — `data: {...}` lines, one JSON payload each, each
|
|
617
|
+
* carrying its frame's `event:` name when it had one.
|
|
618
|
+
*
|
|
619
|
+
* The same discipline as the Ollama adapter's NDJSON parser, adapted to
|
|
620
|
+
* SSE: a cross-read buffer so a frame split anywhere (even mid-byte
|
|
621
|
+
* sequence — TextDecoder streams) reassembles; a malformed line is
|
|
622
|
+
* SKIPPED, never fatal; comments and every other field are ignored; and
|
|
623
|
+
* the `[DONE]` sentinel ends the stream, so anything a broken server
|
|
624
|
+
* writes after it never reaches a consumer.
|
|
625
|
+
*
|
|
626
|
+
* Two things here that a plain SSE reader would not do, both because the
|
|
627
|
+
* other end is a model on THIS machine: the `event:` name survives (it is
|
|
628
|
+
* one of the two ways this dialect spells a mid-generation failure), and
|
|
629
|
+
* the caller's `signal` both interrupts the read and CANCELS the body —
|
|
630
|
+
* a generation nobody is reading still occupies the GPU.
|
|
631
|
+
*/
|
|
632
|
+
async function* parseSse(body, signal) {
|
|
633
|
+
const reader = body.getReader();
|
|
634
|
+
const decoder = new TextDecoder();
|
|
635
|
+
let buf = '';
|
|
636
|
+
let eventName;
|
|
637
|
+
try {
|
|
638
|
+
for (;;) {
|
|
639
|
+
// The read is RACED against the caller's signal. `fetch` resolved on
|
|
640
|
+
// headers, so this loop is the whole rest of the call — a signal that
|
|
641
|
+
// only reached the headers would be a cancellation that cancels nothing.
|
|
642
|
+
const { value, done } = await untilAborted(reader.read(), signal);
|
|
643
|
+
if (done)
|
|
644
|
+
break;
|
|
645
|
+
buf += decoder.decode(value, { stream: true });
|
|
646
|
+
let idx;
|
|
647
|
+
while ((idx = buf.indexOf('\n')) >= 0) {
|
|
648
|
+
const line = buf.slice(0, idx).trim(); // trim eats the \r of a \r\n wire
|
|
649
|
+
buf = buf.slice(idx + 1);
|
|
650
|
+
if (line.length === 0) {
|
|
651
|
+
eventName = undefined; // the blank line ends a frame
|
|
652
|
+
continue;
|
|
653
|
+
}
|
|
654
|
+
if (line.startsWith('event:')) {
|
|
655
|
+
eventName = line.slice('event:'.length).trim();
|
|
656
|
+
continue;
|
|
657
|
+
}
|
|
658
|
+
if (!line.startsWith('data:'))
|
|
659
|
+
continue;
|
|
660
|
+
const payload = line.slice('data:'.length).trim();
|
|
661
|
+
if (payload === '[DONE]')
|
|
662
|
+
return;
|
|
663
|
+
let data;
|
|
664
|
+
try {
|
|
665
|
+
data = JSON.parse(payload);
|
|
666
|
+
}
|
|
667
|
+
catch {
|
|
668
|
+
/* skip malformed line */
|
|
669
|
+
eventName = undefined;
|
|
670
|
+
continue;
|
|
671
|
+
}
|
|
672
|
+
yield { ...(eventName !== undefined && { event: eventName }), data };
|
|
673
|
+
eventName = undefined;
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
// A final data line with no trailing newline still counts.
|
|
677
|
+
const tail = buf.trim();
|
|
678
|
+
if (tail.startsWith('data:')) {
|
|
679
|
+
const payload = tail.slice('data:'.length).trim();
|
|
680
|
+
if (payload !== '[DONE]') {
|
|
681
|
+
let data;
|
|
682
|
+
try {
|
|
683
|
+
data = JSON.parse(payload);
|
|
684
|
+
}
|
|
685
|
+
catch {
|
|
686
|
+
return; // skip a malformed tail — there is nothing after it anyway
|
|
687
|
+
}
|
|
688
|
+
yield { ...(eventName !== undefined && { event: eventName }), data };
|
|
689
|
+
}
|
|
690
|
+
}
|
|
691
|
+
}
|
|
692
|
+
finally {
|
|
693
|
+
// CANCEL, not merely release the lock. `[DONE]`, a consumer that breaks
|
|
694
|
+
// out of the loop, an error frame, a caller abort — every one of them
|
|
695
|
+
// leaves an open body, and an open body means the on-device model keeps
|
|
696
|
+
// generating for nobody and the socket stays up.
|
|
697
|
+
try {
|
|
698
|
+
await reader.cancel();
|
|
699
|
+
}
|
|
700
|
+
catch {
|
|
701
|
+
/* already closed or errored — nothing left to close */
|
|
702
|
+
}
|
|
703
|
+
try {
|
|
704
|
+
reader.releaseLock();
|
|
705
|
+
}
|
|
706
|
+
catch {
|
|
707
|
+
/* a lock the runtime already dropped */
|
|
708
|
+
}
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
/**
|
|
712
|
+
* POST, but never hang waiting for a service that is not there.
|
|
713
|
+
*
|
|
714
|
+
* The timer bounds the wait for RESPONSE HEADERS and is cleared the moment
|
|
715
|
+
* they arrive, so a model that takes four minutes to write a long answer is
|
|
716
|
+
* unaffected — `fetch` resolves on headers, and the body streams afterwards.
|
|
717
|
+
*
|
|
718
|
+
* The deadline is a RACE, not just an `AbortSignal`. Aborting is the polite
|
|
719
|
+
* request — it releases the socket and is what a real `fetch` acts on — but a
|
|
720
|
+
* promise that never settles is exactly the failure this guards against, and
|
|
721
|
+
* "never hangs" cannot be a promise the caller's `fetch` implementation gets
|
|
722
|
+
* to break on our behalf. So the timeout wins on its own.
|
|
723
|
+
*
|
|
724
|
+
* A caller's own `AbortSignal` is forwarded and, if IT is what fired, the
|
|
725
|
+
* abort is re-thrown as an abort rather than blamed on the service —
|
|
726
|
+
* including a signal that was ALREADY aborted when the call was made, which
|
|
727
|
+
* never dispatches an event for a listener to hear.
|
|
728
|
+
*/
|
|
729
|
+
async function fetchUntilHeaders(fetchImpl, url, init, opts) {
|
|
730
|
+
// An already-aborted signal never dispatches another 'abort' event, so the
|
|
731
|
+
// bridge below could not fire and the request would go out on a turn the
|
|
732
|
+
// caller has cancelled. Refuse before the socket opens, and refuse with the
|
|
733
|
+
// caller's own reason: an abort is their call, never a service fault.
|
|
734
|
+
throwIfAborted(opts.signal);
|
|
735
|
+
const controller = new AbortController();
|
|
736
|
+
let timedOut = false;
|
|
737
|
+
let timer;
|
|
738
|
+
const onCallerAbort = () => controller.abort();
|
|
739
|
+
opts.signal?.addEventListener('abort', onCallerAbort, { once: true });
|
|
740
|
+
const deadline = new Promise((_resolve, reject) => {
|
|
741
|
+
timer = setTimeout(() => {
|
|
742
|
+
timedOut = true;
|
|
743
|
+
controller.abort();
|
|
744
|
+
reject(new FoundryLocalUnavailableError({
|
|
745
|
+
reason: 'service-unreachable',
|
|
746
|
+
endpoint: opts.endpoint,
|
|
747
|
+
}));
|
|
748
|
+
}, opts.timeoutMs);
|
|
749
|
+
});
|
|
750
|
+
try {
|
|
751
|
+
return await Promise.race([fetchImpl(url, { ...init, signal: controller.signal }), deadline]);
|
|
752
|
+
}
|
|
753
|
+
catch (err) {
|
|
754
|
+
if (err instanceof FoundryLocalUnavailableError)
|
|
755
|
+
throw err; // the deadline fired
|
|
756
|
+
if (opts.signal?.aborted && !timedOut)
|
|
757
|
+
throw err; // the caller's call, not ours
|
|
758
|
+
// Timed out, or connection refused / DNS failure / TLS failure — either
|
|
759
|
+
// way, nothing answered at that endpoint. The original is preserved as
|
|
760
|
+
// `cause` for anyone who wants it — just never in the message.
|
|
761
|
+
throw new FoundryLocalUnavailableError({
|
|
762
|
+
reason: 'service-unreachable',
|
|
763
|
+
endpoint: opts.endpoint,
|
|
764
|
+
cause: err,
|
|
765
|
+
});
|
|
766
|
+
}
|
|
767
|
+
finally {
|
|
768
|
+
if (timer !== undefined)
|
|
769
|
+
clearTimeout(timer);
|
|
770
|
+
opts.signal?.removeEventListener('abort', onCallerAbort);
|
|
771
|
+
// When the fetch wins the race, the deadline promise may still reject with
|
|
772
|
+
// nobody listening. Swallow it deliberately: the answer already arrived, so
|
|
773
|
+
// an unhandled-rejection warning here would be noise about a non-event.
|
|
774
|
+
deadline.catch(() => undefined);
|
|
775
|
+
}
|
|
776
|
+
}
|
|
777
|
+
/**
|
|
778
|
+
* Best-effort GET-and-parse, bounded the same way the main POST is.
|
|
779
|
+
*
|
|
780
|
+
* Used for the two side lookups (`/foundry/list`, `/openai/models`),
|
|
781
|
+
* where the answer improves an outcome but its absence must never worsen
|
|
782
|
+
* one. Any failure — non-2xx, bad JSON, a fetch that hangs past the
|
|
783
|
+
* deadline — resolves to `undefined`; the race (not just the abort)
|
|
784
|
+
* keeps the "never hangs" promise even against a fetch implementation
|
|
785
|
+
* that ignores its signal.
|
|
786
|
+
*/
|
|
787
|
+
async function fetchJsonBounded(fetchImpl, url, timeoutMs) {
|
|
788
|
+
const controller = new AbortController();
|
|
789
|
+
let timer;
|
|
790
|
+
const deadline = new Promise((resolve) => {
|
|
791
|
+
timer = setTimeout(() => {
|
|
792
|
+
controller.abort();
|
|
793
|
+
resolve(undefined);
|
|
794
|
+
}, timeoutMs);
|
|
795
|
+
});
|
|
796
|
+
const attempt = (async () => {
|
|
797
|
+
const response = await fetchImpl(url, { signal: controller.signal });
|
|
798
|
+
if (!response.ok)
|
|
799
|
+
return undefined;
|
|
800
|
+
return (await response.json());
|
|
801
|
+
})();
|
|
802
|
+
try {
|
|
803
|
+
return await Promise.race([attempt, deadline]);
|
|
804
|
+
}
|
|
805
|
+
catch {
|
|
806
|
+
// Best-effort lookup. Its failure must never replace the error the
|
|
807
|
+
// caller is on the way to reporting.
|
|
808
|
+
return undefined;
|
|
809
|
+
}
|
|
810
|
+
finally {
|
|
811
|
+
if (timer !== undefined)
|
|
812
|
+
clearTimeout(timer);
|
|
813
|
+
// If the deadline won, the losing attempt may still reject later with
|
|
814
|
+
// nobody listening — swallow that deliberately.
|
|
815
|
+
attempt.catch(() => undefined);
|
|
816
|
+
}
|
|
817
|
+
}
|
|
818
|
+
/**
|
|
819
|
+
* Turn a non-2xx into the most actionable error available.
|
|
820
|
+
*
|
|
821
|
+
* A 404 from chat means the model is not on this service. Before saying
|
|
822
|
+
* so we ask `/openai/models` — a cheap local call — so the message can
|
|
823
|
+
* also name what IS cached here, which is usually enough to spot a typo.
|
|
824
|
+
* If that lookup fails too, the `foundry model run` instruction stands
|
|
825
|
+
* on its own.
|
|
826
|
+
*
|
|
827
|
+
* A 404 whose body does not speak this dialect's `{"error": ...}` may not
|
|
828
|
+
* be a model 404 at all — the endpoint may point at something that is not
|
|
829
|
+
* Foundry Local's chat route. The refusal still names the model command
|
|
830
|
+
* (guessing the other way would be just as confident and just as wrong),
|
|
831
|
+
* but it also names the endpoint as a suspect instead of pretending to know.
|
|
832
|
+
*/
|
|
833
|
+
async function describeFailure(response, model, endpoint, fetchImpl, timeoutMs) {
|
|
834
|
+
const bodyText = await safeText(response);
|
|
835
|
+
if (response.status === 404) {
|
|
836
|
+
const availableModels = await listCachedModels(fetchImpl, endpoint, timeoutMs);
|
|
837
|
+
return new FoundryLocalUnavailableError({
|
|
838
|
+
reason: 'model-not-available',
|
|
839
|
+
endpoint,
|
|
840
|
+
model,
|
|
841
|
+
...(availableModels && { availableModels }),
|
|
842
|
+
...(speaksDialectError(bodyText) ? {} : { routeUnconfirmed: true }),
|
|
843
|
+
});
|
|
844
|
+
}
|
|
845
|
+
const detail = extractErrorText(bodyText);
|
|
846
|
+
const said = detail ? ` — ${detail}` : '';
|
|
847
|
+
return providerError(`[foundry-local] ${response.status} ${response.statusText}${said}`.trim(), response.status);
|
|
848
|
+
}
|
|
849
|
+
/** `GET /openai/models` — a bare JSON array of cached model names. */
|
|
850
|
+
async function listCachedModels(fetchImpl, endpoint, timeoutMs) {
|
|
851
|
+
const json = await fetchJsonBounded(fetchImpl, `${endpoint}/openai/models`, timeoutMs);
|
|
852
|
+
if (!Array.isArray(json))
|
|
853
|
+
return undefined;
|
|
854
|
+
const names = json.filter((n) => typeof n === 'string' && n.length > 0);
|
|
855
|
+
return names.length > 0 ? names : undefined;
|
|
856
|
+
}
|
|
857
|
+
async function safeText(response) {
|
|
858
|
+
try {
|
|
859
|
+
return await response.text();
|
|
860
|
+
}
|
|
861
|
+
catch {
|
|
862
|
+
return '';
|
|
863
|
+
}
|
|
864
|
+
}
|
|
865
|
+
/**
|
|
866
|
+
* Errors are `{"error": {"message": "..."}}` per the OpenAI dialect; a
|
|
867
|
+
* bare `{"error": "..."}` string and raw text are tolerated. Every path
|
|
868
|
+
* is capped at 200 chars — the wire's words help diagnose, but a server
|
|
869
|
+
* echoing something enormous (or poisoned) must not become the message.
|
|
870
|
+
*/
|
|
871
|
+
function extractErrorText(bodyText) {
|
|
872
|
+
if (!bodyText)
|
|
873
|
+
return '';
|
|
874
|
+
try {
|
|
875
|
+
const parsed = JSON.parse(bodyText);
|
|
876
|
+
const detail = extractErrorPayload(parsed.error);
|
|
877
|
+
if (detail)
|
|
878
|
+
return detail;
|
|
879
|
+
}
|
|
880
|
+
catch {
|
|
881
|
+
/* not JSON — fall through */
|
|
882
|
+
}
|
|
883
|
+
return bodyText.slice(0, ERROR_TEXT_CAP);
|
|
884
|
+
}
|
|
885
|
+
/**
|
|
886
|
+
* Did this body speak the dialect's `{"error": ...}`?
|
|
887
|
+
*
|
|
888
|
+
* The one fact {@link describeFailure} needs to tell a model 404 from a 404
|
|
889
|
+
* that is really "this route is not ours".
|
|
890
|
+
*/
|
|
891
|
+
function speaksDialectError(bodyText) {
|
|
892
|
+
if (!bodyText)
|
|
893
|
+
return false;
|
|
894
|
+
try {
|
|
895
|
+
const parsed = JSON.parse(bodyText);
|
|
896
|
+
return parsed.error !== undefined && parsed.error !== null;
|
|
897
|
+
}
|
|
898
|
+
catch {
|
|
899
|
+
return false;
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
/**
|
|
903
|
+
* The words out of an `error` payload — `{"message": "..."}` per the
|
|
904
|
+
* dialect, a bare string tolerated — capped like every other piece of wire
|
|
905
|
+
* text this file repeats. Empty when there is nothing usable to say, so a
|
|
906
|
+
* caller can tell "no error" from "an error that named no reason".
|
|
907
|
+
*/
|
|
908
|
+
function extractErrorPayload(err) {
|
|
909
|
+
if (typeof err === 'string')
|
|
910
|
+
return err.slice(0, ERROR_TEXT_CAP);
|
|
911
|
+
if (typeof err === 'object' && err !== null) {
|
|
912
|
+
const message = err.message;
|
|
913
|
+
if (typeof message === 'string' && message.length > 0) {
|
|
914
|
+
return message.slice(0, ERROR_TEXT_CAP);
|
|
915
|
+
}
|
|
916
|
+
}
|
|
917
|
+
return '';
|
|
918
|
+
}
|
|
919
|
+
/**
|
|
920
|
+
* What a mid-stream frame says went wrong — '' when nothing did.
|
|
921
|
+
*
|
|
922
|
+
* Two spellings on this dialect: a `data:` frame carrying `error`, and an
|
|
923
|
+
* `event: error` frame whose payload names the failure at the top level.
|
|
924
|
+
* Both mean the generation failed; neither carries `choices`, which is
|
|
925
|
+
* exactly why an unchecked one looks like an ordinary skippable frame.
|
|
926
|
+
*/
|
|
927
|
+
function frameFailureText(frame) {
|
|
928
|
+
const data = typeof frame.data === 'object' && frame.data !== null
|
|
929
|
+
? frame.data
|
|
930
|
+
: undefined;
|
|
931
|
+
const carriesError = data?.error !== undefined && data?.error !== null;
|
|
932
|
+
const isErrorEvent = frame.event === 'error';
|
|
933
|
+
if (!carriesError && !isErrorEvent)
|
|
934
|
+
return '';
|
|
935
|
+
const detail = (carriesError ? extractErrorPayload(data?.error) : '') ||
|
|
936
|
+
(typeof data?.message === 'string' ? data.message.slice(0, ERROR_TEXT_CAP) : '');
|
|
937
|
+
return detail || 'the service reported an error but named no reason';
|
|
938
|
+
}
|
|
939
|
+
/**
|
|
940
|
+
* The provider's own labelled error — one shape for every failure that is
|
|
941
|
+
* not one of the two typed, actionable ones.
|
|
942
|
+
*/
|
|
943
|
+
function providerError(message, status) {
|
|
944
|
+
return Object.assign(new Error(message), {
|
|
945
|
+
name: 'FoundryLocalProviderError',
|
|
946
|
+
...(status !== undefined && { status }),
|
|
947
|
+
});
|
|
948
|
+
}
|
|
949
|
+
/**
|
|
950
|
+
* An abort is the CALLER's word, so it must reach them AS an abort — never
|
|
951
|
+
* dressed up as a service failure. The signal's own `reason` is used when it
|
|
952
|
+
* is an Error (what every runtime supplies: a DOMException named
|
|
953
|
+
* 'AbortError'); anything else becomes one, keeping the reason as `cause`.
|
|
954
|
+
*/
|
|
955
|
+
function asAbortError(reason) {
|
|
956
|
+
if (reason instanceof Error)
|
|
957
|
+
return reason;
|
|
958
|
+
return Object.assign(new Error('This operation was aborted'), {
|
|
959
|
+
name: 'AbortError',
|
|
960
|
+
...(reason !== undefined && { cause: reason }),
|
|
961
|
+
});
|
|
962
|
+
}
|
|
963
|
+
/** Stop right here when the caller's signal has already fired. */
|
|
964
|
+
function throwIfAborted(signal) {
|
|
965
|
+
if (signal?.aborted)
|
|
966
|
+
throw asAbortError(signal.reason);
|
|
967
|
+
}
|
|
968
|
+
/**
|
|
969
|
+
* `promise`, except that a caller abort ends the wait immediately.
|
|
970
|
+
*
|
|
971
|
+
* `fetch` resolves on HEADERS. Everything after that — a streamed body, a
|
|
972
|
+
* large non-streaming JSON read — used to be deaf to the caller's signal, so
|
|
973
|
+
* a cancelled turn kept the local model generating to completion. This is
|
|
974
|
+
* what makes `LLMRequest.signal` honest for the whole call rather than only
|
|
975
|
+
* until the headers land.
|
|
976
|
+
*/
|
|
977
|
+
function untilAborted(promise, signal) {
|
|
978
|
+
if (!signal)
|
|
979
|
+
return promise;
|
|
980
|
+
if (signal.aborted) {
|
|
981
|
+
void promise.catch(() => undefined); // the abandoned work must not warn
|
|
982
|
+
return Promise.reject(asAbortError(signal.reason));
|
|
983
|
+
}
|
|
984
|
+
return new Promise((resolve, reject) => {
|
|
985
|
+
const onAbort = () => reject(asAbortError(signal.reason));
|
|
986
|
+
signal.addEventListener('abort', onAbort, { once: true });
|
|
987
|
+
promise.then(resolve, reject).finally(() => {
|
|
988
|
+
signal.removeEventListener('abort', onAbort);
|
|
989
|
+
});
|
|
990
|
+
});
|
|
991
|
+
}
|
|
992
|
+
//# sourceMappingURL=FoundryLocalProvider.js.map
|