agentfootprint 9.72.0 → 9.74.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/CHANGELOG.md +160 -0
  2. package/dist/adapters/identity/azure.js +306 -0
  3. package/dist/adapters/identity/azure.js.map +1 -0
  4. package/dist/adapters/llm/FoundryLocalProvider.js +992 -0
  5. package/dist/adapters/llm/FoundryLocalProvider.js.map +1 -0
  6. package/dist/adapters/llm/FoundryProvider.js +273 -0
  7. package/dist/adapters/llm/FoundryProvider.js.map +1 -0
  8. package/dist/adapters/llm/OllamaProvider.js +171 -12
  9. package/dist/adapters/llm/OllamaProvider.js.map +1 -1
  10. package/dist/adapters/llm/OpenAIProvider.js +117 -5
  11. package/dist/adapters/llm/OpenAIProvider.js.map +1 -1
  12. package/dist/adapters/llm/createProvider.js +143 -10
  13. package/dist/adapters/llm/createProvider.js.map +1 -1
  14. package/dist/adapters/types.js.map +1 -1
  15. package/dist/esm/adapters/identity/azure.d.ts +188 -0
  16. package/dist/esm/adapters/identity/azure.js +302 -0
  17. package/dist/esm/adapters/identity/azure.js.map +1 -0
  18. package/dist/esm/adapters/llm/FoundryLocalProvider.d.ts +215 -0
  19. package/dist/esm/adapters/llm/FoundryLocalProvider.js +986 -0
  20. package/dist/esm/adapters/llm/FoundryLocalProvider.js.map +1 -0
  21. package/dist/esm/adapters/llm/FoundryProvider.d.ts +178 -0
  22. package/dist/esm/adapters/llm/FoundryProvider.js +268 -0
  23. package/dist/esm/adapters/llm/FoundryProvider.js.map +1 -0
  24. package/dist/esm/adapters/llm/OllamaProvider.js +171 -12
  25. package/dist/esm/adapters/llm/OllamaProvider.js.map +1 -1
  26. package/dist/esm/adapters/llm/OpenAIProvider.d.ts +111 -0
  27. package/dist/esm/adapters/llm/OpenAIProvider.js +115 -4
  28. package/dist/esm/adapters/llm/OpenAIProvider.js.map +1 -1
  29. package/dist/esm/adapters/llm/createProvider.d.ts +48 -12
  30. package/dist/esm/adapters/llm/createProvider.js +143 -10
  31. package/dist/esm/adapters/llm/createProvider.js.map +1 -1
  32. package/dist/esm/adapters/types.d.ts +4 -2
  33. package/dist/esm/adapters/types.js.map +1 -1
  34. package/dist/esm/identity.d.ts +1 -0
  35. package/dist/esm/identity.js +9 -0
  36. package/dist/esm/identity.js.map +1 -1
  37. package/dist/esm/index.d.ts +1 -1
  38. package/dist/esm/index.js.map +1 -1
  39. package/dist/esm/providers.d.ts +5 -0
  40. package/dist/esm/providers.js +14 -0
  41. package/dist/esm/providers.js.map +1 -1
  42. package/dist/identity.js +14 -1
  43. package/dist/identity.js.map +1 -1
  44. package/dist/index.js.map +1 -1
  45. package/dist/providers.js +20 -1
  46. package/dist/providers.js.map +1 -1
  47. package/dist/types/adapters/identity/azure.d.ts +189 -0
  48. package/dist/types/adapters/identity/azure.d.ts.map +1 -0
  49. package/dist/types/adapters/llm/FoundryLocalProvider.d.ts +216 -0
  50. package/dist/types/adapters/llm/FoundryLocalProvider.d.ts.map +1 -0
  51. package/dist/types/adapters/llm/FoundryProvider.d.ts +179 -0
  52. package/dist/types/adapters/llm/FoundryProvider.d.ts.map +1 -0
  53. package/dist/types/adapters/llm/OllamaProvider.d.ts.map +1 -1
  54. package/dist/types/adapters/llm/OpenAIProvider.d.ts +111 -0
  55. package/dist/types/adapters/llm/OpenAIProvider.d.ts.map +1 -1
  56. package/dist/types/adapters/llm/createProvider.d.ts +48 -12
  57. package/dist/types/adapters/llm/createProvider.d.ts.map +1 -1
  58. package/dist/types/adapters/types.d.ts +4 -2
  59. package/dist/types/adapters/types.d.ts.map +1 -1
  60. package/dist/types/identity.d.ts +1 -0
  61. package/dist/types/identity.d.ts.map +1 -1
  62. package/dist/types/index.d.ts +1 -1
  63. package/dist/types/index.d.ts.map +1 -1
  64. package/dist/types/providers.d.ts +5 -0
  65. package/dist/types/providers.d.ts.map +1 -1
  66. package/package.json +5 -1
@@ -0,0 +1,986 @@
1
+ /**
2
+ * FoundryLocalProvider — on-device models over Foundry Local's
3
+ * OpenAI-compatible `/v1/chat/completions`.
4
+ *
5
+ * Pattern: Adapter (GoF) + Ports-and-Adapters (Cockburn 2005).
6
+ * Role: Outer ring — translates `LLMRequest`/`LLMResponse` to/from the
7
+ * wire Foundry Local serves on localhost. Knows nothing about
8
+ * agents, recorders, or compositions.
9
+ * Emits: N/A.
10
+ *
11
+ * ─── Why this exists ─────────────────────────────────────────────────
12
+ *
13
+ * The adapter ladder is `mock()` → a local model → a paid API, and
14
+ * `ollama()` is the proof that the middle rung is worth owning: zero
15
+ * dependencies, honest refusals, real token counts. Foundry Local is
16
+ * Microsoft's runtime for the same rung — ONNX under the hood, models
17
+ * pulled with `foundry model run <alias>`, no key, no account — and a
18
+ * Windows or macOS machine that has it installed deserves the same
19
+ * one-import experience. So this file owns that wire the way
20
+ * `OllamaProvider.ts` owns Ollama's:
21
+ *
22
+ * • ZERO dependencies — one `fetch` POST and SSE. The official
23
+ * `foundry-local-sdk` is NOT imported; its manager is accepted
24
+ * duck-typed (see {@link FoundryLocalProviderOptions.manager}) so a
25
+ * consumer who already uses it can hand over the discovered URL
26
+ * without this package gaining a dependency.
27
+ * • HONEST REFUSALS — a typed {@link FoundryLocalUnavailableError}
28
+ * that names the endpoint it tried and the command to run. The
29
+ * service's port is DYNAMIC per start, which makes "nothing is
30
+ * answering" the most likely first failure — so that message
31
+ * carries the discovery command, not just the start command. A
32
+ * failure the service reports IN BAND — an `error` frame on an
33
+ * already-200 stream, the out-of-memory a laptop runtime really does
34
+ * hit — is RAISED the same way, never handed over as a shorter
35
+ * answer that reads like a clean stop.
36
+ * • REAL TOKEN COUNTS while streaming — `stream_options:
37
+ * { include_usage: true }` is always sent. This is OUR wire, a
38
+ * documented Foundry Local surface, not an arbitrary
39
+ * OpenAI-compatible server — so the caution that made
40
+ * `openai({ baseURL })` withhold the field (and silently zero every
41
+ * local token count until 9.73.0) does not apply here.
42
+ *
43
+ * ─── Wire realities this file owns ───────────────────────────────────
44
+ *
45
+ * • THE PORT IS DYNAMIC. Every `foundry server start` may pick a new
46
+ * port; the docs' own REST example shows `http://localhost:5272` and
47
+ * that is the default here, but the truthful discovery is
48
+ * `foundry server status` (or the SDK manager's `.urls`). Note the
49
+ * CLI group was RENAMED from `foundry service` to `foundry server` —
50
+ * every message in this file uses the NEW spelling.
51
+ * • ALIASES vs VARIANT IDS. The catalog speaks in aliases
52
+ * (`qwen2.5-0.5b`) that fan out to hardware variants
53
+ * (`qwen2.5-0.5b-instruct-generic-cpu:1`), but REST chat calls take
54
+ * the FULL variant id. This adapter resolves an alias through
55
+ * `GET /foundry/list` — first matching variant wins, because the
56
+ * list's order IS the service's priority order — and caches the
57
+ * answer per provider instance, HIT OR MISS: exactly one catalog
58
+ * attempt per name, so an alias the catalog never answers for cannot
59
+ * re-ask before every call. A fresh provider is the retry. A name that
60
+ * already carries a variant's execution-provider suffix
61
+ * (`-cpu`/`-gpu`/`-npu`, optional `:version`) is used as-is with no
62
+ * catalog round-trip.
63
+ * • NO API KEY EXISTS. The docs' own samples pass placeholders. This
64
+ * adapter sends no `Authorization` header at all — there is nothing
65
+ * to put in one, and an invented value would only end up in somebody's
66
+ * proxy log.
67
+ *
68
+ * ─── Ceilings (stated, not worked around) ────────────────────────────
69
+ *
70
+ * • NO FORCED TOOL CHOICE. `tool_choice` support is UNDOCUMENTED on
71
+ * this wire, so `carriesForcedToolChoice` is `false` and an agent
72
+ * using `.outputSchema(parser, { strategy: 'tool-forced' })` refuses
73
+ * at run start, naming this provider. Claiming an undocumented field
74
+ * works would turn a guarantee into a suggestion.
75
+ * • TOOL CALLING IS MODEL-DEPENDENT. `/foundry/list` reports
76
+ * `supportsToolCalling` per variant, but this adapter does not
77
+ * preflight-refuse on it — a wrong refusal is worse than a weak
78
+ * answer, the same stance `ollama()` takes on `/api/show`. Pick a
79
+ * tool-capable variant.
80
+ * • NO MULTI-MODAL. `LLMMessage.content` is a string. Same ceiling as
81
+ * every other adapter here.
82
+ * • NO PROMPT CACHING — resolves to the NoOp cache strategy.
83
+ * • NO STRUCTURED THINKING. The wire has no thinking field; a reasoning
84
+ * model's `<think>` tags ride the answer text untouched.
85
+ */
86
+ /** Cap for any text the wire supplies before it becomes one of our messages. */
87
+ const ERROR_TEXT_CAP = 200;
88
+ // ─── Errors ─────────────────────────────────────────────────────────
89
+ /**
90
+ * The two failures an on-device runtime actually has, told in words that
91
+ * contain the fix.
92
+ *
93
+ * Both are things the person at the keyboard can resolve in one command,
94
+ * which is exactly why they get a type instead of a wrapped
95
+ * `ECONNREFUSED` or a bare `404`. `reason` is the discriminator; the
96
+ * message already reads as instructions — and because Foundry Local's
97
+ * port changes per start, the unreachable message teaches the discovery
98
+ * command (`foundry server status`) alongside the start command.
99
+ */
100
+ export class FoundryLocalUnavailableError extends Error {
101
+ name = 'FoundryLocalUnavailableError';
102
+ /** Which of the two situations this is. */
103
+ reason;
104
+ /** The endpoint that was tried — the thing to check or change. */
105
+ endpoint;
106
+ /** The model asked for. Absent when the service never answered at all. */
107
+ model;
108
+ /** Models this machine DOES have cached, when the service could tell us. */
109
+ availableModels;
110
+ constructor(init) {
111
+ super(buildUnavailableMessage(init));
112
+ this.reason = init.reason;
113
+ this.endpoint = init.endpoint;
114
+ if (init.model !== undefined)
115
+ this.model = init.model;
116
+ if (init.availableModels !== undefined)
117
+ this.availableModels = init.availableModels;
118
+ if (init.cause !== undefined)
119
+ this.cause = init.cause;
120
+ }
121
+ }
122
+ function buildUnavailableMessage(init) {
123
+ if (init.reason === 'service-unreachable') {
124
+ return (`foundryLocal: nothing is answering at ${init.endpoint}. ` +
125
+ 'Start it with `foundry server start` (Foundry Local, install: ' +
126
+ 'https://learn.microsoft.com/azure/foundry-local). ' +
127
+ 'The port is dynamic — `foundry server status` prints the live URL; ' +
128
+ 'pin one with `foundry server start --port <p>`. ' +
129
+ "Running elsewhere? Pass foundryLocal('<model>', { endpoint }) or set FOUNDRY_LOCAL_ENDPOINT.");
130
+ }
131
+ const model = init.model ?? '(unnamed)';
132
+ const have = init.availableModels && init.availableModels.length > 0
133
+ ? ` Models on this machine: ${init.availableModels.join(', ')}.`
134
+ : '';
135
+ // A 404 that did not speak the dialect is as likely a wrong route as a
136
+ // missing model. Say both rather than one confidently.
137
+ const route = init.routeUnconfirmed
138
+ ? ` That 404 named no model error, so ${init.endpoint} may not be Foundry Local's chat route at all — \`foundry server status\` prints the live URL.`
139
+ : '';
140
+ return (`foundryLocal: model '${model}' is not available on the service at ${init.endpoint}. ` +
141
+ `Run: foundry model run ${model} — it downloads the model if needed (aliases work too).${have}${route}`);
142
+ }
143
+ /**
144
+ * The docs' own REST example port. Only a default, never a promise —
145
+ * see {@link FoundryLocalProviderOptions.endpoint} for the truth about
146
+ * dynamic ports.
147
+ */
148
+ const DEFAULT_ENDPOINT = 'http://localhost:5272';
149
+ /** The alias the Foundry Local docs use in their own REST walkthrough. */
150
+ const DEFAULT_MODEL = 'qwen2.5-0.5b';
151
+ const DEFAULT_TIMEOUT_MS = 10_000;
152
+ /** Request `model` values that mean "whatever this provider was configured with". */
153
+ const MODEL_SHORTHANDS = new Set(['foundry-local']);
154
+ /**
155
+ * Which roles this wire carries inside `messages`.
156
+ *
157
+ * The OpenAI dialect takes the system prompt as a message like any other
158
+ * (no separate top-level `system` field), so all three roles survive the
159
+ * trip.
160
+ */
161
+ const CARRIES_IN_MESSAGES = Object.freeze(['system', 'user', 'assistant']);
162
+ export function foundryLocal(modelOrOptions, maybeOptions) {
163
+ const options = typeof modelOrOptions === 'string' ? maybeOptions ?? {} : modelOrOptions ?? {};
164
+ const positionalModel = typeof modelOrOptions === 'string' ? modelOrOptions : undefined;
165
+ const endpoint = resolveEndpoint(options);
166
+ const defaultModel = positionalModel ?? options.defaultModel ?? DEFAULT_MODEL;
167
+ const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
168
+ const fetchImpl = options._fetch ?? ((...args) => fetch(...args));
169
+ const chatUrl = `${endpoint}/v1/chat/completions`;
170
+ const cfg = {
171
+ defaultModel,
172
+ ...(options.defaultMaxTokens !== undefined && { defaultMaxTokens: options.defaultMaxTokens }),
173
+ };
174
+ // Tool-call ids are synthesized per provider instance — the dialect says
175
+ // every call carries one, small local models sometimes disagree, and the
176
+ // whole tool round-trip in this library is keyed by id.
177
+ let toolCallSeq = 0;
178
+ const nextToolCallId = () => `foundry-call-${++toolCallSeq}`;
179
+ // Alias → variant-id resolutions, cached per provider instance — HIT OR
180
+ // MISS. The catalog does not change under a running service often enough
181
+ // to be worth re-asking on every call, and the cache is what keeps an
182
+ // alias-configured agent at one catalog fetch per process. Caching the
183
+ // FALLBACK too is what keeps that promise true when the catalog has no
184
+ // answer for the alias.
185
+ const resolutionCache = new Map();
186
+ /**
187
+ * REST chat calls take the FULL variant id, so an alias must be
188
+ * resolved first:
189
+ *
190
+ * • A name already shaped like a variant id (it ends in the
191
+ * execution-provider suffix `-cpu`/`-gpu`/`-npu`, optionally
192
+ * `:version`) is used AS-IS — no catalog round-trip. That suffix is
193
+ * how Foundry Local itself distinguishes a variant from an alias.
194
+ * • Anything else is treated as an alias: `GET /foundry/list`, first
195
+ * variant whose alias matches wins (the list's order is the
196
+ * service's priority order), and the answer is cached.
197
+ * • A silent or alias-less catalog resolves to the name UNCHANGED —
198
+ * the chat call's own 404 then reports honestly, with the model
199
+ * name the caller actually wrote. A failed lookup must never
200
+ * replace the error that was coming anyway. That fallback is CACHED
201
+ * exactly like a success: ONE catalog attempt per name per provider,
202
+ * so an alias the service has no answer for cannot put a
203
+ * `/foundry/list` round-trip — and, against a host that accepts but
204
+ * never answers, a whole `timeoutMs` — in front of every single
205
+ * call. A model pulled with `foundry model run` after the fact is
206
+ * picked up by a FRESH provider; that is the retry.
207
+ */
208
+ const resolveModel = async (requested) => {
209
+ const named = MODEL_SHORTHANDS.has(requested) ? cfg.defaultModel : requested;
210
+ if (looksLikeVariantId(named))
211
+ return named;
212
+ const cached = resolutionCache.get(named);
213
+ if (cached !== undefined)
214
+ return cached;
215
+ const catalog = await fetchJsonBounded(fetchImpl, `${endpoint}/foundry/list`, timeoutMs);
216
+ // The fallback is cached like an answer, so the catalog is asked once
217
+ // per name whatever it says. A fresh provider re-asks.
218
+ const answer = firstVariantForAlias(catalog, named) ?? named;
219
+ resolutionCache.set(named, answer);
220
+ return answer;
221
+ };
222
+ const post = async (body, req) => {
223
+ // No Authorization header: no key exists on this wire, and nothing is
224
+ // sent where a secret could be invented, logged, or leaked.
225
+ const response = await fetchUntilHeaders(fetchImpl, chatUrl, {
226
+ method: 'POST',
227
+ headers: { 'content-type': 'application/json' },
228
+ body: JSON.stringify(body),
229
+ }, { timeoutMs, endpoint, ...(req.signal && { signal: req.signal }) });
230
+ if (!response.ok) {
231
+ throw await describeFailure(response, body.model, endpoint, fetchImpl, timeoutMs);
232
+ }
233
+ return response;
234
+ };
235
+ const provider = {
236
+ name: 'foundry-local',
237
+ carriesInMessages: CARRIES_IN_MESSAGES,
238
+ // `tool_choice` support is undocumented on this wire. Absence would mean
239
+ // the same thing; saying it out loud documents that this was checked
240
+ // rather than forgotten. See the header's ceilings.
241
+ carriesForcedToolChoice: false,
242
+ async complete(req) {
243
+ // A signal that fired BEFORE the call stops it here, before any socket
244
+ // opens — the catalog lookup included. An already-aborted signal never
245
+ // dispatches another 'abort' event, so a listener alone cannot see it.
246
+ throwIfAborted(req.signal);
247
+ const model = await resolveModel(req.model);
248
+ const body = buildBody(req, cfg, model, false);
249
+ const response = await post(body, req);
250
+ // `fetch` resolves on HEADERS, so reading the body is the stretch where
251
+ // a caller abort would otherwise be inert — honor it for that too.
252
+ const json = (await untilAborted(response.json(), req.signal));
253
+ return fromFoundryResponse(json, nextToolCallId);
254
+ },
255
+ async *stream(req) {
256
+ throwIfAborted(req.signal);
257
+ const model = await resolveModel(req.model);
258
+ const body = buildBody(req, cfg, model, true);
259
+ const response = await post(body, req);
260
+ if (!response.body)
261
+ throw new Error('[foundry-local] response has no body');
262
+ const textParts = [];
263
+ // The dialect streams tool calls as DELTAS — id and name on the first
264
+ // fragment, argument JSON split across the rest — assembled by index.
265
+ const toolCallsByIndex = new Map();
266
+ let finishReason = null;
267
+ let usage;
268
+ let lastId = '';
269
+ let tokenIndex = 0;
270
+ // The signal rides INTO the parser: it is what stops a generation the
271
+ // caller no longer wants, and what cancels the body when it does.
272
+ for await (const parsed of parseSse(response.body, req.signal)) {
273
+ const frame = parsed.data;
274
+ // An already-200 stream can still FAIL mid-generation, and this
275
+ // dialect says so IN BAND. Such a frame carries no `choices`, so the
276
+ // guard below would drop it and the terminal chunk would report a
277
+ // truncated answer as `stopReason: 'stop'` with nobody told. Raise
278
+ // instead: a failed generation is a failure, never a shorter answer.
279
+ const failure = frameFailureText(parsed);
280
+ if (failure) {
281
+ throw providerError(`[foundry-local] the stream failed mid-generation — ${failure}`);
282
+ }
283
+ // Usage FIRST, and OUTSIDE the choice guard. With
284
+ // `stream_options.include_usage` (set in buildBody) the token counts
285
+ // ride a FINAL chunk whose `choices` array is EMPTY — the exact bug
286
+ // class 9.73.0 fixed in the OpenAI adapter: a `continue` on a missing
287
+ // choice threw away the only usage the stream ever reports, and every
288
+ // streamed local call read zero tokens everywhere usage is consumed.
289
+ if (frame.id)
290
+ lastId = frame.id;
291
+ if (frame.usage)
292
+ usage = frame.usage;
293
+ const choice = frame.choices?.[0];
294
+ if (!choice)
295
+ continue;
296
+ if (choice.finish_reason)
297
+ finishReason = choice.finish_reason;
298
+ const delta = choice.delta;
299
+ if (!delta)
300
+ continue;
301
+ if (delta.content) {
302
+ textParts.push(delta.content);
303
+ yield { tokenIndex, content: delta.content, done: false };
304
+ tokenIndex++;
305
+ }
306
+ if (delta.tool_calls) {
307
+ for (const tcDelta of delta.tool_calls) {
308
+ const idx = tcDelta.index ?? 0;
309
+ const existing = toolCallsByIndex.get(idx) ?? { id: '', name: '', argsJson: '' };
310
+ if (tcDelta.id)
311
+ existing.id = tcDelta.id;
312
+ if (tcDelta.function?.name)
313
+ existing.name = tcDelta.function.name;
314
+ if (tcDelta.function?.arguments)
315
+ existing.argsJson += tcDelta.function.arguments;
316
+ toolCallsByIndex.set(idx, existing);
317
+ }
318
+ }
319
+ }
320
+ // Sorted by INDEX, not by first-seen order: a wire is free to open
321
+ // index 1 before index 0, and a consumer reading `toolCalls[0]` as
322
+ // "the first tool the model asked for" would then name the wrong one.
323
+ const toolCalls = Array.from(toolCallsByIndex.entries())
324
+ .sort(([a], [b]) => a - b)
325
+ .map(([, tc]) => ({
326
+ id: tc.id.length > 0 ? tc.id : nextToolCallId(),
327
+ name: tc.name,
328
+ args: coerceArgs(tc.argsJson),
329
+ }));
330
+ const authoritative = {
331
+ content: textParts.join(''),
332
+ toolCalls,
333
+ usage: {
334
+ input: usage?.prompt_tokens ?? 0,
335
+ output: usage?.completion_tokens ?? 0,
336
+ },
337
+ stopReason: normalizeStopReason(finishReason ?? 'stop', toolCalls.length > 0),
338
+ ...(lastId && { providerRef: lastId }),
339
+ };
340
+ yield { tokenIndex, content: '', done: true, response: authoritative };
341
+ },
342
+ };
343
+ return provider;
344
+ }
345
+ /**
346
+ * Class form for consumers who prefer `new FoundryLocalProvider(...)`.
347
+ */
348
+ export class FoundryLocalProvider {
349
+ name = 'foundry-local';
350
+ carriesInMessages = CARRIES_IN_MESSAGES;
351
+ carriesForcedToolChoice = false;
352
+ inner;
353
+ constructor(model, options) {
354
+ this.inner =
355
+ typeof model === 'string'
356
+ ? foundryLocal(model, options)
357
+ : foundryLocal(model);
358
+ }
359
+ // `hooks` is FORWARDED, not dropped — see LLMCallHooks in adapters/types.ts.
360
+ complete(req, hooks) {
361
+ return this.inner.complete(req, hooks);
362
+ }
363
+ stream(req, hooks) {
364
+ if (!this.inner.stream)
365
+ throw new Error('stream() unavailable');
366
+ return this.inner.stream(req, hooks);
367
+ }
368
+ }
369
+ // ─── Internals ──────────────────────────────────────────────────────
370
+ /**
371
+ * Resolve where the service lives. Most specific wins: an explicit
372
+ * `endpoint`, then a duck-typed SDK manager's discovered URL, then the
373
+ * two env spellings, then the docs' example port.
374
+ *
375
+ * A `/v1` suffix is trimmed rather than rejected — someone copying the
376
+ * chat URL out of `foundry server status` output means the same
377
+ * machine — and a bare `host:port` gets `http://`.
378
+ */
379
+ function resolveEndpoint(options) {
380
+ const env = typeof process !== 'undefined' ? process.env : undefined;
381
+ const raw = firstConfigured(options.endpoint, options.manager?.urls?.[0], env?.FOUNDRY_LOCAL_ENDPOINT, env?.FOUNDRY_LOCAL_BASE_URL) ?? DEFAULT_ENDPOINT;
382
+ const withScheme = /^https?:\/\//i.test(raw) ? raw : `http://${raw}`;
383
+ return withScheme.replace(/\/+$/, '').replace(/\/v1$/i, '');
384
+ }
385
+ /**
386
+ * First candidate that actually says something.
387
+ *
388
+ * A blank value is treated as ABSENT rather than as an endpoint. `??` would
389
+ * accept `''`, and `'' → 'http://' → strip trailing slashes` leaves the URL
390
+ * `http:` — a refusal naming nothing the reader can check or fix. An env key
391
+ * present with an empty value is a config that forgot to fill it in, so the
392
+ * next candidate (and ultimately the default) gets its turn.
393
+ */
394
+ function firstConfigured(...candidates) {
395
+ for (const candidate of candidates) {
396
+ if (typeof candidate === 'string' && candidate.trim().length > 0)
397
+ return candidate.trim();
398
+ }
399
+ return undefined;
400
+ }
401
+ /**
402
+ * Does this name already carry a variant's shape?
403
+ *
404
+ * Full Foundry Local variant ids name their execution provider as the
405
+ * terminal segment — `-cpu`, `-gpu` or `-npu`, optionally followed by a
406
+ * `:version` — e.g. `qwen2.5-0.5b-instruct-generic-cpu:1`. Aliases never
407
+ * do. This structural fact is what lets a fully-qualified id skip the
408
+ * catalog round-trip entirely.
409
+ */
410
+ function looksLikeVariantId(model) {
411
+ const bare = model.replace(/:\d+$/, '');
412
+ return /-(cpu|gpu|npu)$/i.test(bare);
413
+ }
414
+ /**
415
+ * First catalog variant whose alias matches — `/foundry/list` order is
416
+ * the service's priority order, so first wins. Tolerates both a bare
417
+ * array and a `{ models: [...] }` wrapper, and both id spellings.
418
+ */
419
+ function firstVariantForAlias(catalog, alias) {
420
+ const entries = Array.isArray(catalog)
421
+ ? catalog
422
+ : catalog?.models;
423
+ if (!Array.isArray(entries))
424
+ return undefined;
425
+ for (const raw of entries) {
426
+ if (typeof raw !== 'object' || raw === null)
427
+ continue;
428
+ const entry = raw;
429
+ if (entry.alias !== alias)
430
+ continue;
431
+ const id = entry.name ?? entry.id;
432
+ if (typeof id === 'string' && id.length > 0)
433
+ return id;
434
+ }
435
+ return undefined;
436
+ }
437
+ function buildBody(req, cfg, model, stream) {
438
+ const body = {
439
+ model,
440
+ messages: toFoundryMessages(req.messages, req.systemPrompt),
441
+ stream,
442
+ };
443
+ // Always ask for usage on a stream. This wire is a documented Foundry
444
+ // Local surface, not an arbitrary OpenAI-compatible server, so the
445
+ // reject-unknown-field caution that cost `openai({ baseURL })` its token
446
+ // counts (fixed in 9.73.0) has no purchase here — and a streamed call
447
+ // that reports zero tokens silently disarms `.compaction()` and budgets.
448
+ if (stream)
449
+ body.stream_options = { include_usage: true };
450
+ if (req.tools && req.tools.length > 0)
451
+ body.tools = req.tools.map(toFoundryTool);
452
+ const maxTokens = req.maxTokens ?? cfg.defaultMaxTokens;
453
+ if (maxTokens !== undefined)
454
+ body.max_tokens = maxTokens;
455
+ if (req.temperature !== undefined)
456
+ body.temperature = req.temperature;
457
+ if (req.stop && req.stop.length > 0)
458
+ body.stop = [...req.stop];
459
+ // `req.toolChoice` is intentionally NOT translated: support for it is
460
+ // undocumented on this wire, `carriesForcedToolChoice` says so, and the
461
+ // agent refuses before it ever reaches here.
462
+ return body;
463
+ }
464
+ /**
465
+ * messages → wire messages.
466
+ *
467
+ * Roles map 1:1. The system prompt is prepended as an ordinary `system`
468
+ * message (this dialect has no separate system field). Assistant turns
469
+ * carry their `tool_calls` back — arguments re-serialized to the JSON
470
+ * STRING the dialect expects — and tool results carry `tool_call_id`,
471
+ * which is how this wire correlates a result with the call that asked
472
+ * for it. A role the port does not define is dropped, not forwarded
473
+ * blindly.
474
+ */
475
+ function toFoundryMessages(messages, systemPrompt) {
476
+ const result = [];
477
+ if (systemPrompt)
478
+ result.push({ role: 'system', content: systemPrompt });
479
+ for (const m of messages) {
480
+ if (m.role === 'system' || m.role === 'user') {
481
+ result.push({ role: m.role, content: m.content });
482
+ continue;
483
+ }
484
+ if (m.role === 'assistant') {
485
+ const msg = { role: 'assistant', content: m.content };
486
+ if (m.toolCalls && m.toolCalls.length > 0) {
487
+ msg.tool_calls = m.toolCalls.map((tc) => ({
488
+ id: tc.id,
489
+ type: 'function',
490
+ function: { name: tc.name, arguments: JSON.stringify(tc.args) },
491
+ }));
492
+ }
493
+ result.push(msg);
494
+ continue;
495
+ }
496
+ if (m.role === 'tool') {
497
+ result.push({
498
+ role: 'tool',
499
+ content: m.content,
500
+ ...(m.toolCallId && { tool_call_id: m.toolCallId }),
501
+ });
502
+ continue;
503
+ }
504
+ }
505
+ return result;
506
+ }
507
+ function toFoundryTool(schema) {
508
+ return {
509
+ type: 'function',
510
+ function: {
511
+ name: schema.name,
512
+ description: schema.description,
513
+ parameters: { ...schema.inputSchema },
514
+ },
515
+ };
516
+ }
517
+ /**
518
+ * Wire response → the port's shape.
519
+ *
520
+ * @throws a `FoundryLocalProviderError` when the 200 body is really a
521
+ * failure (`{"error": ...}`). Reading it as a response would yield empty
522
+ * content with `stopReason: 'stop'` and zero tokens — a failed call
523
+ * dressed as a successful one.
524
+ */
525
+ function fromFoundryResponse(response, nextToolCallId) {
526
+ if (response.error !== undefined && response.error !== null) {
527
+ const detail = extractErrorPayload(response.error) || 'the service named no reason';
528
+ throw providerError(`[foundry-local] the service answered 200 with an error — ${detail}`);
529
+ }
530
+ const choice = response.choices?.[0];
531
+ const message = choice?.message;
532
+ const wireToolCalls = message?.tool_calls ?? [];
533
+ const toolCalls = wireToolCalls.map((tc) => toLLMToolCall(tc, nextToolCallId));
534
+ return {
535
+ content: message?.content ?? '',
536
+ toolCalls,
537
+ usage: {
538
+ input: response.usage?.prompt_tokens ?? 0,
539
+ output: response.usage?.completion_tokens ?? 0,
540
+ },
541
+ stopReason: normalizeStopReason(choice?.finish_reason ?? 'stop', toolCalls.length > 0),
542
+ ...(response.id && { providerRef: response.id }),
543
+ };
544
+ }
545
+ /**
546
+ * One wire tool call → the port's shape.
547
+ *
548
+ * Two wire facts handled here:
549
+ * • `id` should always be present on this dialect, but small local
550
+ * models have been seen to omit it — and the whole tool round-trip
551
+ * in this library is keyed by id. So an id is SYNTHESIZED when
552
+ * missing, unique per provider instance.
553
+ * • `arguments` is a JSON STRING per the dialect (Ollama sends an
554
+ * object); the object form is still tolerated in case a proxy in the
555
+ * middle reshaped it.
556
+ */
557
+ function toLLMToolCall(tc, nextToolCallId) {
558
+ return {
559
+ id: tc.id && tc.id.length > 0 ? tc.id : nextToolCallId(),
560
+ name: tc.function?.name ?? '',
561
+ args: coerceArgs(tc.function?.arguments),
562
+ };
563
+ }
564
+ function coerceArgs(args) {
565
+ if (args === undefined || args === null)
566
+ return {};
567
+ if (typeof args === 'string') {
568
+ if (args.length === 0)
569
+ return {};
570
+ try {
571
+ const parsed = JSON.parse(args);
572
+ return typeof parsed === 'object' && parsed !== null
573
+ ? parsed
574
+ : {};
575
+ }
576
+ catch {
577
+ // Malformed args are rare but observed on small local models. Surface
578
+ // empty rather than crash — the tool-call event still fires, so the
579
+ // problem is visible in the trace.
580
+ return {};
581
+ }
582
+ }
583
+ return { ...args };
584
+ }
585
+ /**
586
+ * `finish_reason` → the port's stop vocabulary.
587
+ *
588
+ * `tool_calls` is the dialect's own word for a tool-ending turn, but a
589
+ * local model has been seen to report plain `stop` with tool calls
590
+ * attached — so the presence of tool calls also decides, the same
591
+ * correction the Ollama adapter makes.
592
+ */
593
+ function normalizeStopReason(raw, hasToolCalls) {
594
+ if (hasToolCalls && (raw === 'stop' || raw === ''))
595
+ return 'tool_use';
596
+ switch (raw) {
597
+ case 'stop':
598
+ return 'stop';
599
+ case 'tool_calls':
600
+ return 'tool_use';
601
+ case 'length':
602
+ return 'max_tokens';
603
+ case 'content_filter':
604
+ return 'content_filter';
605
+ default:
606
+ return raw;
607
+ }
608
+ }
609
+ /**
610
+ * Parse an SSE body — `data: {...}` lines, one JSON payload each, each
611
+ * carrying its frame's `event:` name when it had one.
612
+ *
613
+ * The same discipline as the Ollama adapter's NDJSON parser, adapted to
614
+ * SSE: a cross-read buffer so a frame split anywhere (even mid-byte
615
+ * sequence — TextDecoder streams) reassembles; a malformed line is
616
+ * SKIPPED, never fatal; comments and every other field are ignored; and
617
+ * the `[DONE]` sentinel ends the stream, so anything a broken server
618
+ * writes after it never reaches a consumer.
619
+ *
620
+ * Two things here that a plain SSE reader would not do, both because the
621
+ * other end is a model on THIS machine: the `event:` name survives (it is
622
+ * one of the two ways this dialect spells a mid-generation failure), and
623
+ * the caller's `signal` both interrupts the read and CANCELS the body —
624
+ * a generation nobody is reading still occupies the GPU.
625
+ */
626
+ async function* parseSse(body, signal) {
627
+ const reader = body.getReader();
628
+ const decoder = new TextDecoder();
629
+ let buf = '';
630
+ let eventName;
631
+ try {
632
+ for (;;) {
633
+ // The read is RACED against the caller's signal. `fetch` resolved on
634
+ // headers, so this loop is the whole rest of the call — a signal that
635
+ // only reached the headers would be a cancellation that cancels nothing.
636
+ const { value, done } = await untilAborted(reader.read(), signal);
637
+ if (done)
638
+ break;
639
+ buf += decoder.decode(value, { stream: true });
640
+ let idx;
641
+ while ((idx = buf.indexOf('\n')) >= 0) {
642
+ const line = buf.slice(0, idx).trim(); // trim eats the \r of a \r\n wire
643
+ buf = buf.slice(idx + 1);
644
+ if (line.length === 0) {
645
+ eventName = undefined; // the blank line ends a frame
646
+ continue;
647
+ }
648
+ if (line.startsWith('event:')) {
649
+ eventName = line.slice('event:'.length).trim();
650
+ continue;
651
+ }
652
+ if (!line.startsWith('data:'))
653
+ continue;
654
+ const payload = line.slice('data:'.length).trim();
655
+ if (payload === '[DONE]')
656
+ return;
657
+ let data;
658
+ try {
659
+ data = JSON.parse(payload);
660
+ }
661
+ catch {
662
+ /* skip malformed line */
663
+ eventName = undefined;
664
+ continue;
665
+ }
666
+ yield { ...(eventName !== undefined && { event: eventName }), data };
667
+ eventName = undefined;
668
+ }
669
+ }
670
+ // A final data line with no trailing newline still counts.
671
+ const tail = buf.trim();
672
+ if (tail.startsWith('data:')) {
673
+ const payload = tail.slice('data:'.length).trim();
674
+ if (payload !== '[DONE]') {
675
+ let data;
676
+ try {
677
+ data = JSON.parse(payload);
678
+ }
679
+ catch {
680
+ return; // skip a malformed tail — there is nothing after it anyway
681
+ }
682
+ yield { ...(eventName !== undefined && { event: eventName }), data };
683
+ }
684
+ }
685
+ }
686
+ finally {
687
+ // CANCEL, not merely release the lock. `[DONE]`, a consumer that breaks
688
+ // out of the loop, an error frame, a caller abort — every one of them
689
+ // leaves an open body, and an open body means the on-device model keeps
690
+ // generating for nobody and the socket stays up.
691
+ try {
692
+ await reader.cancel();
693
+ }
694
+ catch {
695
+ /* already closed or errored — nothing left to close */
696
+ }
697
+ try {
698
+ reader.releaseLock();
699
+ }
700
+ catch {
701
+ /* a lock the runtime already dropped */
702
+ }
703
+ }
704
+ }
705
+ /**
706
+ * POST, but never hang waiting for a service that is not there.
707
+ *
708
+ * The timer bounds the wait for RESPONSE HEADERS and is cleared the moment
709
+ * they arrive, so a model that takes four minutes to write a long answer is
710
+ * unaffected — `fetch` resolves on headers, and the body streams afterwards.
711
+ *
712
+ * The deadline is a RACE, not just an `AbortSignal`. Aborting is the polite
713
+ * request — it releases the socket and is what a real `fetch` acts on — but a
714
+ * promise that never settles is exactly the failure this guards against, and
715
+ * "never hangs" cannot be a promise the caller's `fetch` implementation gets
716
+ * to break on our behalf. So the timeout wins on its own.
717
+ *
718
+ * A caller's own `AbortSignal` is forwarded and, if IT is what fired, the
719
+ * abort is re-thrown as an abort rather than blamed on the service —
720
+ * including a signal that was ALREADY aborted when the call was made, which
721
+ * never dispatches an event for a listener to hear.
722
+ */
723
+ async function fetchUntilHeaders(fetchImpl, url, init, opts) {
724
+ // An already-aborted signal never dispatches another 'abort' event, so the
725
+ // bridge below could not fire and the request would go out on a turn the
726
+ // caller has cancelled. Refuse before the socket opens, and refuse with the
727
+ // caller's own reason: an abort is their call, never a service fault.
728
+ throwIfAborted(opts.signal);
729
+ const controller = new AbortController();
730
+ let timedOut = false;
731
+ let timer;
732
+ const onCallerAbort = () => controller.abort();
733
+ opts.signal?.addEventListener('abort', onCallerAbort, { once: true });
734
+ const deadline = new Promise((_resolve, reject) => {
735
+ timer = setTimeout(() => {
736
+ timedOut = true;
737
+ controller.abort();
738
+ reject(new FoundryLocalUnavailableError({
739
+ reason: 'service-unreachable',
740
+ endpoint: opts.endpoint,
741
+ }));
742
+ }, opts.timeoutMs);
743
+ });
744
+ try {
745
+ return await Promise.race([fetchImpl(url, { ...init, signal: controller.signal }), deadline]);
746
+ }
747
+ catch (err) {
748
+ if (err instanceof FoundryLocalUnavailableError)
749
+ throw err; // the deadline fired
750
+ if (opts.signal?.aborted && !timedOut)
751
+ throw err; // the caller's call, not ours
752
+ // Timed out, or connection refused / DNS failure / TLS failure — either
753
+ // way, nothing answered at that endpoint. The original is preserved as
754
+ // `cause` for anyone who wants it — just never in the message.
755
+ throw new FoundryLocalUnavailableError({
756
+ reason: 'service-unreachable',
757
+ endpoint: opts.endpoint,
758
+ cause: err,
759
+ });
760
+ }
761
+ finally {
762
+ if (timer !== undefined)
763
+ clearTimeout(timer);
764
+ opts.signal?.removeEventListener('abort', onCallerAbort);
765
+ // When the fetch wins the race, the deadline promise may still reject with
766
+ // nobody listening. Swallow it deliberately: the answer already arrived, so
767
+ // an unhandled-rejection warning here would be noise about a non-event.
768
+ deadline.catch(() => undefined);
769
+ }
770
+ }
771
+ /**
772
+ * Best-effort GET-and-parse, bounded the same way the main POST is.
773
+ *
774
+ * Used for the two side lookups (`/foundry/list`, `/openai/models`),
775
+ * where the answer improves an outcome but its absence must never worsen
776
+ * one. Any failure — non-2xx, bad JSON, a fetch that hangs past the
777
+ * deadline — resolves to `undefined`; the race (not just the abort)
778
+ * keeps the "never hangs" promise even against a fetch implementation
779
+ * that ignores its signal.
780
+ */
781
+ async function fetchJsonBounded(fetchImpl, url, timeoutMs) {
782
+ const controller = new AbortController();
783
+ let timer;
784
+ const deadline = new Promise((resolve) => {
785
+ timer = setTimeout(() => {
786
+ controller.abort();
787
+ resolve(undefined);
788
+ }, timeoutMs);
789
+ });
790
+ const attempt = (async () => {
791
+ const response = await fetchImpl(url, { signal: controller.signal });
792
+ if (!response.ok)
793
+ return undefined;
794
+ return (await response.json());
795
+ })();
796
+ try {
797
+ return await Promise.race([attempt, deadline]);
798
+ }
799
+ catch {
800
+ // Best-effort lookup. Its failure must never replace the error the
801
+ // caller is on the way to reporting.
802
+ return undefined;
803
+ }
804
+ finally {
805
+ if (timer !== undefined)
806
+ clearTimeout(timer);
807
+ // If the deadline won, the losing attempt may still reject later with
808
+ // nobody listening — swallow that deliberately.
809
+ attempt.catch(() => undefined);
810
+ }
811
+ }
812
+ /**
813
+ * Turn a non-2xx into the most actionable error available.
814
+ *
815
+ * A 404 from chat means the model is not on this service. Before saying
816
+ * so we ask `/openai/models` — a cheap local call — so the message can
817
+ * also name what IS cached here, which is usually enough to spot a typo.
818
+ * If that lookup fails too, the `foundry model run` instruction stands
819
+ * on its own.
820
+ *
821
+ * A 404 whose body does not speak this dialect's `{"error": ...}` may not
822
+ * be a model 404 at all — the endpoint may point at something that is not
823
+ * Foundry Local's chat route. The refusal still names the model command
824
+ * (guessing the other way would be just as confident and just as wrong),
825
+ * but it also names the endpoint as a suspect instead of pretending to know.
826
+ */
827
+ async function describeFailure(response, model, endpoint, fetchImpl, timeoutMs) {
828
+ const bodyText = await safeText(response);
829
+ if (response.status === 404) {
830
+ const availableModels = await listCachedModels(fetchImpl, endpoint, timeoutMs);
831
+ return new FoundryLocalUnavailableError({
832
+ reason: 'model-not-available',
833
+ endpoint,
834
+ model,
835
+ ...(availableModels && { availableModels }),
836
+ ...(speaksDialectError(bodyText) ? {} : { routeUnconfirmed: true }),
837
+ });
838
+ }
839
+ const detail = extractErrorText(bodyText);
840
+ const said = detail ? ` — ${detail}` : '';
841
+ return providerError(`[foundry-local] ${response.status} ${response.statusText}${said}`.trim(), response.status);
842
+ }
843
+ /** `GET /openai/models` — a bare JSON array of cached model names. */
844
+ async function listCachedModels(fetchImpl, endpoint, timeoutMs) {
845
+ const json = await fetchJsonBounded(fetchImpl, `${endpoint}/openai/models`, timeoutMs);
846
+ if (!Array.isArray(json))
847
+ return undefined;
848
+ const names = json.filter((n) => typeof n === 'string' && n.length > 0);
849
+ return names.length > 0 ? names : undefined;
850
+ }
851
+ async function safeText(response) {
852
+ try {
853
+ return await response.text();
854
+ }
855
+ catch {
856
+ return '';
857
+ }
858
+ }
859
+ /**
860
+ * Errors are `{"error": {"message": "..."}}` per the OpenAI dialect; a
861
+ * bare `{"error": "..."}` string and raw text are tolerated. Every path
862
+ * is capped at 200 chars — the wire's words help diagnose, but a server
863
+ * echoing something enormous (or poisoned) must not become the message.
864
+ */
865
+ function extractErrorText(bodyText) {
866
+ if (!bodyText)
867
+ return '';
868
+ try {
869
+ const parsed = JSON.parse(bodyText);
870
+ const detail = extractErrorPayload(parsed.error);
871
+ if (detail)
872
+ return detail;
873
+ }
874
+ catch {
875
+ /* not JSON — fall through */
876
+ }
877
+ return bodyText.slice(0, ERROR_TEXT_CAP);
878
+ }
879
+ /**
880
+ * Did this body speak the dialect's `{"error": ...}`?
881
+ *
882
+ * The one fact {@link describeFailure} needs to tell a model 404 from a 404
883
+ * that is really "this route is not ours".
884
+ */
885
+ function speaksDialectError(bodyText) {
886
+ if (!bodyText)
887
+ return false;
888
+ try {
889
+ const parsed = JSON.parse(bodyText);
890
+ return parsed.error !== undefined && parsed.error !== null;
891
+ }
892
+ catch {
893
+ return false;
894
+ }
895
+ }
896
+ /**
897
+ * The words out of an `error` payload — `{"message": "..."}` per the
898
+ * dialect, a bare string tolerated — capped like every other piece of wire
899
+ * text this file repeats. Empty when there is nothing usable to say, so a
900
+ * caller can tell "no error" from "an error that named no reason".
901
+ */
902
+ function extractErrorPayload(err) {
903
+ if (typeof err === 'string')
904
+ return err.slice(0, ERROR_TEXT_CAP);
905
+ if (typeof err === 'object' && err !== null) {
906
+ const message = err.message;
907
+ if (typeof message === 'string' && message.length > 0) {
908
+ return message.slice(0, ERROR_TEXT_CAP);
909
+ }
910
+ }
911
+ return '';
912
+ }
913
+ /**
914
+ * What a mid-stream frame says went wrong — '' when nothing did.
915
+ *
916
+ * Two spellings on this dialect: a `data:` frame carrying `error`, and an
917
+ * `event: error` frame whose payload names the failure at the top level.
918
+ * Both mean the generation failed; neither carries `choices`, which is
919
+ * exactly why an unchecked one looks like an ordinary skippable frame.
920
+ */
921
+ function frameFailureText(frame) {
922
+ const data = typeof frame.data === 'object' && frame.data !== null
923
+ ? frame.data
924
+ : undefined;
925
+ const carriesError = data?.error !== undefined && data?.error !== null;
926
+ const isErrorEvent = frame.event === 'error';
927
+ if (!carriesError && !isErrorEvent)
928
+ return '';
929
+ const detail = (carriesError ? extractErrorPayload(data?.error) : '') ||
930
+ (typeof data?.message === 'string' ? data.message.slice(0, ERROR_TEXT_CAP) : '');
931
+ return detail || 'the service reported an error but named no reason';
932
+ }
933
+ /**
934
+ * The provider's own labelled error — one shape for every failure that is
935
+ * not one of the two typed, actionable ones.
936
+ */
937
+ function providerError(message, status) {
938
+ return Object.assign(new Error(message), {
939
+ name: 'FoundryLocalProviderError',
940
+ ...(status !== undefined && { status }),
941
+ });
942
+ }
943
+ /**
944
+ * An abort is the CALLER's word, so it must reach them AS an abort — never
945
+ * dressed up as a service failure. The signal's own `reason` is used when it
946
+ * is an Error (what every runtime supplies: a DOMException named
947
+ * 'AbortError'); anything else becomes one, keeping the reason as `cause`.
948
+ */
949
+ function asAbortError(reason) {
950
+ if (reason instanceof Error)
951
+ return reason;
952
+ return Object.assign(new Error('This operation was aborted'), {
953
+ name: 'AbortError',
954
+ ...(reason !== undefined && { cause: reason }),
955
+ });
956
+ }
957
+ /** Stop right here when the caller's signal has already fired. */
958
+ function throwIfAborted(signal) {
959
+ if (signal?.aborted)
960
+ throw asAbortError(signal.reason);
961
+ }
962
+ /**
963
+ * `promise`, except that a caller abort ends the wait immediately.
964
+ *
965
+ * `fetch` resolves on HEADERS. Everything after that — a streamed body, a
966
+ * large non-streaming JSON read — used to be deaf to the caller's signal, so
967
+ * a cancelled turn kept the local model generating to completion. This is
968
+ * what makes `LLMRequest.signal` honest for the whole call rather than only
969
+ * until the headers land.
970
+ */
971
+ function untilAborted(promise, signal) {
972
+ if (!signal)
973
+ return promise;
974
+ if (signal.aborted) {
975
+ void promise.catch(() => undefined); // the abandoned work must not warn
976
+ return Promise.reject(asAbortError(signal.reason));
977
+ }
978
+ return new Promise((resolve, reject) => {
979
+ const onAbort = () => reject(asAbortError(signal.reason));
980
+ signal.addEventListener('abort', onAbort, { once: true });
981
+ promise.then(resolve, reject).finally(() => {
982
+ signal.removeEventListener('abort', onAbort);
983
+ });
984
+ });
985
+ }
986
+ //# sourceMappingURL=FoundryLocalProvider.js.map