@remnic/core 9.36.0 → 9.38.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/access-cli.js +10 -10
  2. package/dist/access-http.js +3 -3
  3. package/dist/access-mcp-output-schemas.js +1 -1
  4. package/dist/access-mcp.js +2 -2
  5. package/dist/{chunk-NN2MCG5N.js → chunk-2YJPCRUM.js} +5 -5
  6. package/dist/chunk-2YJPCRUM.js.map +1 -0
  7. package/dist/{chunk-FT4RF53O.js → chunk-EJUEXZHX.js} +2 -2
  8. package/dist/{chunk-WAJPTY5Y.js → chunk-HQYYJRU7.js} +3 -3
  9. package/dist/{chunk-XNJIND4I.js → chunk-IKKWIPYN.js} +12 -7
  10. package/dist/chunk-IKKWIPYN.js.map +1 -0
  11. package/dist/{chunk-SYBHD2IU.js → chunk-QZT2UWL6.js} +12 -12
  12. package/dist/{chunk-ACKB2THH.js → chunk-VU7732DU.js} +2 -2
  13. package/dist/{chunk-JQSBRLJ6.js → chunk-WFTIGIIS.js} +39 -37
  14. package/dist/chunk-WFTIGIIS.js.map +1 -0
  15. package/dist/{chunk-OA6PIPDE.js → chunk-XN3EW6P2.js} +2 -2
  16. package/dist/cli.js +4 -4
  17. package/dist/extraction.js +2 -2
  18. package/dist/index.js +10 -10
  19. package/dist/local-llm.js +1 -1
  20. package/dist/orchestrator.js +10 -10
  21. package/dist/summarizer.js +2 -2
  22. package/package.json +2 -2
  23. package/src/access-mcp-output-schemas.ts +14 -6
  24. package/src/access-mcp.ts +3 -3
  25. package/src/local-llm-headers-timeout.test.ts +194 -0
  26. package/src/local-llm.ts +38 -38
  27. package/dist/chunk-JQSBRLJ6.js.map +0 -1
  28. package/dist/chunk-NN2MCG5N.js.map +0 -1
  29. package/dist/chunk-XNJIND4I.js.map +0 -1
  30. /package/dist/{chunk-FT4RF53O.js.map → chunk-EJUEXZHX.js.map} +0 -0
  31. /package/dist/{chunk-WAJPTY5Y.js.map → chunk-HQYYJRU7.js.map} +0 -0
  32. /package/dist/{chunk-SYBHD2IU.js.map → chunk-QZT2UWL6.js.map} +0 -0
  33. /package/dist/{chunk-ACKB2THH.js.map → chunk-VU7732DU.js.map} +0 -0
  34. /package/dist/{chunk-OA6PIPDE.js.map → chunk-XN3EW6P2.js.map} +0 -0
package/dist/index.js CHANGED
@@ -109,7 +109,7 @@ import {
109
109
  watchForChanges,
110
110
  writeManifest,
111
111
  writeMeetingEpisodeMemory
112
- } from "./chunk-SYBHD2IU.js";
112
+ } from "./chunk-QZT2UWL6.js";
113
113
  import {
114
114
  WEARABLE_SOURCE_PREFIX,
115
115
  buildExtractionTurns,
@@ -198,10 +198,10 @@ import "./chunk-PBGWYM3H.js";
198
198
  import "./chunk-LTAVSOZW.js";
199
199
  import "./chunk-7N4KAIGN.js";
200
200
  import "./chunk-V5OCT34X.js";
201
- import "./chunk-TECVW3JP.js";
202
- import "./chunk-C5JF6EZ7.js";
203
201
  import "./chunk-IJAQCBGV.js";
204
202
  import "./chunk-AS7YMSL6.js";
203
+ import "./chunk-TECVW3JP.js";
204
+ import "./chunk-C5JF6EZ7.js";
205
205
  import "./chunk-LXOM6IQU.js";
206
206
  import "./chunk-VOUOLGIP.js";
207
207
  import {
@@ -215,7 +215,7 @@ import "./chunk-I6WYQSX4.js";
215
215
  import {
216
216
  CODEX_THREAD_KEY_PREFIX
217
217
  } from "./chunk-4EX5WZ4M.js";
218
- import "./chunk-FT4RF53O.js";
218
+ import "./chunk-EJUEXZHX.js";
219
219
  import "./chunk-6DNMUYEE.js";
220
220
  import {
221
221
  buildTargetedFactRecallSection,
@@ -286,7 +286,7 @@ import {
286
286
  import "./chunk-LTHVM4XN.js";
287
287
  import {
288
288
  ExtractionEngine
289
- } from "./chunk-OA6PIPDE.js";
289
+ } from "./chunk-XN3EW6P2.js";
290
290
  import "./chunk-4RA3C3EV.js";
291
291
  import "./chunk-B6XHIBIB.js";
292
292
  import "./chunk-CMQZNEIF.js";
@@ -295,7 +295,7 @@ import "./chunk-2CGLBUW3.js";
295
295
  import {
296
296
  ModelRegistry
297
297
  } from "./chunk-WLZYGLJ4.js";
298
- import "./chunk-JQSBRLJ6.js";
298
+ import "./chunk-WFTIGIIS.js";
299
299
  import "./chunk-WKDFJXW5.js";
300
300
  import "./chunk-JUXOTY6J.js";
301
301
  import "./chunk-YIAA2X3D.js";
@@ -360,7 +360,7 @@ import {
360
360
  runBulkImportCliCommand,
361
361
  runMeetingsCliCommand,
362
362
  runWearablesCliCommand
363
- } from "./chunk-WAJPTY5Y.js";
363
+ } from "./chunk-HQYYJRU7.js";
364
364
  import "./chunk-MMACM5G2.js";
365
365
  import "./chunk-54BGIBCR.js";
366
366
  import "./chunk-H4CKS5B2.js";
@@ -567,7 +567,7 @@ import "./chunk-LBJBNWS2.js";
567
567
  import {
568
568
  EngramAccessHttpServer,
569
569
  MeetingsInputError
570
- } from "./chunk-ACKB2THH.js";
570
+ } from "./chunk-VU7732DU.js";
571
571
  import {
572
572
  WearablesInputError,
573
573
  describeErrorForOperator
@@ -582,7 +582,7 @@ import "./chunk-7RXCMVFQ.js";
582
582
  import "./chunk-MKQBXQKU.js";
583
583
  import {
584
584
  EngramMcpServer
585
- } from "./chunk-NN2MCG5N.js";
585
+ } from "./chunk-2YJPCRUM.js";
586
586
  import "./chunk-57QKLRBV.js";
587
587
  import {
588
588
  buildCitationGuidance,
@@ -652,7 +652,7 @@ import {
652
652
  validateRequest
653
653
  } from "./chunk-3F2LKUMU.js";
654
654
  import "./chunk-56BSXSHC.js";
655
- import "./chunk-XNJIND4I.js";
655
+ import "./chunk-IKKWIPYN.js";
656
656
  import "./chunk-5IVLTFOV.js";
657
657
  import "./chunk-VCD4PSK2.js";
658
658
  import "./chunk-OADWQ5CR.js";
package/dist/local-llm.js CHANGED
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  LocalLlmClient
3
- } from "./chunk-JQSBRLJ6.js";
3
+ } from "./chunk-WFTIGIIS.js";
4
4
  import "./chunk-WKDFJXW5.js";
5
5
  import "./chunk-JUXOTY6J.js";
6
6
  import "./chunk-O75CRYGF.js";
@@ -4,7 +4,7 @@ import {
4
4
  filterRecallCandidates,
5
5
  graphPathRelativeToStorage,
6
6
  mergeGraphExpandedResults
7
- } from "./chunk-SYBHD2IU.js";
7
+ } from "./chunk-QZT2UWL6.js";
8
8
  import "./chunk-UV72DEZD.js";
9
9
  import "./chunk-I74SUMNI.js";
10
10
  import "./chunk-ILNPU5W5.js";
@@ -26,10 +26,10 @@ import "./chunk-PBGWYM3H.js";
26
26
  import "./chunk-LTAVSOZW.js";
27
27
  import "./chunk-7N4KAIGN.js";
28
28
  import "./chunk-V5OCT34X.js";
29
- import "./chunk-TECVW3JP.js";
30
- import "./chunk-C5JF6EZ7.js";
31
29
  import "./chunk-IJAQCBGV.js";
32
30
  import "./chunk-AS7YMSL6.js";
31
+ import "./chunk-TECVW3JP.js";
32
+ import "./chunk-C5JF6EZ7.js";
33
33
  import "./chunk-LXOM6IQU.js";
34
34
  import "./chunk-VOUOLGIP.js";
35
35
  import "./chunk-S75M5ZRK.js";
@@ -38,7 +38,7 @@ import "./chunk-KWMXR7XS.js";
38
38
  import "./chunk-UHGBNIOS.js";
39
39
  import "./chunk-I6WYQSX4.js";
40
40
  import "./chunk-4EX5WZ4M.js";
41
- import "./chunk-FT4RF53O.js";
41
+ import "./chunk-EJUEXZHX.js";
42
42
  import "./chunk-6DNMUYEE.js";
43
43
  import "./chunk-T67LUYCR.js";
44
44
  import "./chunk-QFN4U7AG.js";
@@ -66,14 +66,14 @@ import "./chunk-PHRLJFWZ.js";
66
66
  import "./chunk-64NJRYU2.js";
67
67
  import "./chunk-K2TDK7GG.js";
68
68
  import "./chunk-LTHVM4XN.js";
69
- import "./chunk-OA6PIPDE.js";
69
+ import "./chunk-XN3EW6P2.js";
70
70
  import "./chunk-4RA3C3EV.js";
71
71
  import "./chunk-B6XHIBIB.js";
72
72
  import "./chunk-CMQZNEIF.js";
73
73
  import "./chunk-54V4BZWP.js";
74
74
  import "./chunk-2CGLBUW3.js";
75
75
  import "./chunk-WLZYGLJ4.js";
76
- import "./chunk-JQSBRLJ6.js";
76
+ import "./chunk-WFTIGIIS.js";
77
77
  import "./chunk-WKDFJXW5.js";
78
78
  import "./chunk-JUXOTY6J.js";
79
79
  import "./chunk-YIAA2X3D.js";
@@ -95,7 +95,7 @@ import "./chunk-VEWZZM3H.js";
95
95
  import "./chunk-Q6IP2ARR.js";
96
96
  import "./chunk-6HEM6HTQ.js";
97
97
  import "./chunk-3I4FUAVJ.js";
98
- import "./chunk-WAJPTY5Y.js";
98
+ import "./chunk-HQYYJRU7.js";
99
99
  import "./chunk-MMACM5G2.js";
100
100
  import "./chunk-54BGIBCR.js";
101
101
  import "./chunk-H4CKS5B2.js";
@@ -183,7 +183,7 @@ import "./chunk-OO42R444.js";
183
183
  import "./chunk-PIG4WJFQ.js";
184
184
  import "./chunk-KFZY2OFH.js";
185
185
  import "./chunk-LBJBNWS2.js";
186
- import "./chunk-ACKB2THH.js";
186
+ import "./chunk-VU7732DU.js";
187
187
  import "./chunk-V7IZKJI3.js";
188
188
  import "./chunk-SEDEKFYQ.js";
189
189
  import "./chunk-RKNJBZ55.js";
@@ -193,7 +193,7 @@ import "./chunk-42NQ7AVG.js";
193
193
  import "./chunk-TMSXWOBZ.js";
194
194
  import "./chunk-7RXCMVFQ.js";
195
195
  import "./chunk-MKQBXQKU.js";
196
- import "./chunk-NN2MCG5N.js";
196
+ import "./chunk-2YJPCRUM.js";
197
197
  import "./chunk-57QKLRBV.js";
198
198
  import "./chunk-D24OXEPB.js";
199
199
  import "./chunk-GORHSN5C.js";
@@ -207,7 +207,7 @@ import "./chunk-CEN6DW6X.js";
207
207
  import "./chunk-T4WDJPEZ.js";
208
208
  import "./chunk-3F2LKUMU.js";
209
209
  import "./chunk-56BSXSHC.js";
210
- import "./chunk-XNJIND4I.js";
210
+ import "./chunk-IKKWIPYN.js";
211
211
  import "./chunk-5IVLTFOV.js";
212
212
  import "./chunk-VCD4PSK2.js";
213
213
  import "./chunk-OADWQ5CR.js";
@@ -1,9 +1,9 @@
1
1
  import {
2
2
  HourlySummarizer
3
- } from "./chunk-FT4RF53O.js";
3
+ } from "./chunk-EJUEXZHX.js";
4
4
  import "./chunk-6DNMUYEE.js";
5
5
  import "./chunk-WLZYGLJ4.js";
6
- import "./chunk-JQSBRLJ6.js";
6
+ import "./chunk-WFTIGIIS.js";
7
7
  import "./chunk-WKDFJXW5.js";
8
8
  import "./chunk-JUXOTY6J.js";
9
9
  import "./chunk-S4DDLTPX.js";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@remnic/core",
3
- "version": "9.36.0",
3
+ "version": "9.38.1",
4
4
  "description": "Framework-agnostic Remnic memory engine — orchestrator, storage, extraction, search, trust zones",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -3086,7 +3086,7 @@
3086
3086
  "core"
3087
3087
  ],
3088
3088
  "peerDependencies": {
3089
- "@remnic/coding-graph": "^9.36.0"
3089
+ "@remnic/coding-graph": "^9.38.1"
3090
3090
  },
3091
3091
  "peerDependenciesMeta": {
3092
3092
  "@remnic/coding-graph": {
@@ -16,8 +16,16 @@ const T_NULLABLE_OBJECT = { type: ["object", "null"] } as const;
16
16
  const T_NULLABLE_STRING = { type: ["string", "null"] } as const;
17
17
 
18
18
  /** Build a JSON Schema object type with the given properties. */
19
- function objectSchema(properties: Record<string, Readonly<Record<string, unknown>>>): Record<string, unknown> {
20
- return { type: "object", properties, additionalProperties: true };
19
+ function objectSchema(
20
+ properties: Record<string, Readonly<Record<string, unknown>>>,
21
+ required?: readonly string[],
22
+ ): Record<string, unknown> {
23
+ return {
24
+ type: "object",
25
+ properties,
26
+ ...(required && required.length > 0 ? { required } : {}),
27
+ additionalProperties: true,
28
+ };
21
29
  }
22
30
 
23
31
  /**
@@ -53,10 +61,10 @@ const TOOL_OUTPUT_SCHEMAS: Readonly<Record<string, Record<string, unknown>>> = {
53
61
  sources: T_ARRAY,
54
62
  connectorsInstalled: T_ARRAY,
55
63
  }),
56
- wearables_sync: objectSchema({ summaries: T_ARRAY }),
57
- transcript_day: objectSchema({ transcripts: T_ARRAY }),
58
- transcript_search: objectSchema({ results: T_ARRAY }),
59
- transcript_memories: objectSchema({ memories: T_ARRAY }),
64
+ wearables_sync: objectSchema({ summaries: T_ARRAY }, ["summaries"]),
65
+ transcript_day: objectSchema({ transcripts: T_ARRAY }, ["transcripts"]),
66
+ transcript_search: objectSchema({ results: T_ARRAY }, ["results"]),
67
+ transcript_memories: objectSchema({ memories: T_ARRAY }, ["memories"]),
60
68
  meetings_list: objectSchema({ enabled: T_BOOLEAN, days: T_ARRAY }),
61
69
  meetings_get: objectSchema({ enabled: T_BOOLEAN, found: T_BOOLEAN, id: T_STRING, record: T_NULLABLE_STRING }),
62
70
  meetings_build: objectSchema({
package/src/access-mcp.ts CHANGED
@@ -2445,13 +2445,13 @@ export class EngramMcpServer {
2445
2445
  if (isReadOnlyToolName(name)) {
2446
2446
  throwMcpAbort(options?.abortSignal, "MCP request aborted before response");
2447
2447
  }
2448
- const structuredContent = result ?? {};
2448
+ const structuredContent = result;
2449
2449
  return {
2450
2450
  jsonrpc: "2.0",
2451
2451
  id,
2452
2452
  result: {
2453
- content: [{ type: "text", text: JSON.stringify(structuredContent, null, 2) }],
2454
- structuredContent,
2453
+ content: [{ type: "text", text: JSON.stringify(result ?? null, null, 2) }],
2454
+ ...(structuredContent == null ? {} : { structuredContent }),
2455
2455
  isError: false,
2456
2456
  },
2457
2457
  };
@@ -68,6 +68,11 @@ function primeClient(client: LocalLlmClient): void {
68
68
  internals.detectedType = "generic";
69
69
  }
70
70
 
71
+ function clientIsAvailable(client: LocalLlmClient): boolean {
72
+ const internals = client as unknown as { isAvailable: boolean };
73
+ return internals.isAvailable;
74
+ }
75
+
71
76
  /** Test seam: the pool is private, but its budget is what we assert on. */
72
77
  function dispatcherOf(client: LocalLlmClient): Agent {
73
78
  const seam = client as unknown as {
@@ -122,6 +127,63 @@ async function startSlowHeaderServer(delayMs: number): Promise<{
122
127
  };
123
128
  }
124
129
 
130
+ async function startTricklingBodyServer(chunkIntervalMs: number, status = 200): Promise<{
131
+ url: string;
132
+ requestBodies: string[];
133
+ close: () => Promise<void>;
134
+ }> {
135
+ const pending = new Set<NodeJS.Timeout>();
136
+ const requestBodies: string[] = [];
137
+ const payload = JSON.stringify({
138
+ choices: [{ message: { content: "late completion" } }],
139
+ usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
140
+ });
141
+ const chunkSize = Math.ceil(payload.length / 16);
142
+ const chunks = Array.from(
143
+ { length: 16 },
144
+ (_, index) => payload.slice(index * chunkSize, (index + 1) * chunkSize),
145
+ ).filter((chunk) => chunk.length > 0);
146
+ const server = http.createServer((req, res) => {
147
+ let requestBody = "";
148
+ req.setEncoding("utf8");
149
+ req.on("data", (chunk: string) => {
150
+ requestBody += chunk;
151
+ });
152
+ req.on("end", () => {
153
+ requestBodies.push(requestBody);
154
+ res.writeHead(status, { "content-type": "application/json" });
155
+ res.flushHeaders();
156
+ let index = 0;
157
+ const interval = setInterval(() => {
158
+ const chunk = chunks[index];
159
+ index += 1;
160
+ if (chunk !== undefined) res.write(chunk);
161
+ if (index >= chunks.length) {
162
+ clearInterval(interval);
163
+ pending.delete(interval);
164
+ res.end();
165
+ }
166
+ }, chunkIntervalMs);
167
+ pending.add(interval);
168
+ res.once("close", () => {
169
+ clearInterval(interval);
170
+ pending.delete(interval);
171
+ });
172
+ });
173
+ });
174
+ await new Promise<void>((resolve) => server.listen(0, "127.0.0.1", resolve));
175
+ const { port } = server.address() as AddressInfo;
176
+ return {
177
+ url: `http://127.0.0.1:${port}/v1`,
178
+ requestBodies,
179
+ close: async () => {
180
+ for (const interval of pending) clearInterval(interval);
181
+ pending.clear();
182
+ await new Promise<void>((resolve) => server.close(() => resolve()));
183
+ },
184
+ };
185
+ }
186
+
125
187
  test("chat completion passes an undici dispatcher to fetch", async () => {
126
188
  const original = globalThis.fetch;
127
189
  let seenDispatcher: unknown;
@@ -244,6 +306,138 @@ test("a completion whose headers lag the request still succeeds", async () => {
244
306
  }
245
307
  });
246
308
 
309
+ test("the request budget covers a trickling response body", async () => {
310
+ const timeoutMs = 250;
311
+ const server = await startTricklingBodyServer(75);
312
+ const client = new LocalLlmClient(
313
+ createConfig({ localLlmUrl: server.url, localLlmTimeoutMs: timeoutMs }),
314
+ );
315
+ try {
316
+ primeClient(client);
317
+ const startedAt = Date.now();
318
+ const result = await client.chatCompletion([
319
+ { role: "user", content: "extract facts" },
320
+ ]);
321
+ const elapsedMs = Date.now() - startedAt;
322
+
323
+ assert.equal(result, null);
324
+ assert.ok(
325
+ elapsedMs < 800,
326
+ `response body must abort near the ${timeoutMs}ms request budget; elapsed ${elapsedMs}ms`,
327
+ );
328
+ const requestBody: unknown = JSON.parse(server.requestBodies[0] ?? "{}");
329
+ assert.ok(requestBody && typeof requestBody === "object" && "stream" in requestBody);
330
+ assert.equal(requestBody.stream, false);
331
+ } finally {
332
+ await dispatcherOf(client).close();
333
+ await server.close();
334
+ }
335
+ });
336
+
337
+ test("caller abort during body consumption remains cancellation for an arbitrary reason", async () => {
338
+ const original = globalThis.fetch;
339
+ const caller = new AbortController();
340
+ let calls = 0;
341
+ globalThis.fetch = (async (_url: string, init?: RequestInit) => {
342
+ calls += 1;
343
+ const signal = init?.signal;
344
+ assert.ok(signal);
345
+ const body = new ReadableStream<Uint8Array>({
346
+ start(controller) {
347
+ signal.addEventListener("abort", () => controller.error(signal.reason), { once: true });
348
+ },
349
+ });
350
+ queueMicrotask(() => caller.abort(new Error("caller cancelled")));
351
+ return new Response(body, {
352
+ status: 200,
353
+ headers: { "content-type": "application/json" },
354
+ });
355
+ }) as typeof fetch;
356
+
357
+ const client = new LocalLlmClient(createConfig({ localLlmRetry5xxCount: 2 }));
358
+ try {
359
+ primeClient(client);
360
+ const result = await client.chatCompletion(
361
+ [{ role: "user", content: "extract facts" }],
362
+ { signal: caller.signal },
363
+ );
364
+ assert.equal(result, null);
365
+ assert.equal(calls, 1);
366
+ assert.equal(clientIsAvailable(client), true);
367
+ } finally {
368
+ globalThis.fetch = original;
369
+ await dispatcherOf(client).close();
370
+ }
371
+ });
372
+
373
+ test("a timed-out 400 body is accounted once and enters cooldown", async () => {
374
+ const server = await startTricklingBodyServer(75, 400);
375
+ const client = new LocalLlmClient(
376
+ createConfig({
377
+ localLlmUrl: server.url,
378
+ localLlmTimeoutMs: 250,
379
+ localLlmRetry5xxCount: 2,
380
+ localLlm400TripThreshold: 1,
381
+ }),
382
+ );
383
+ try {
384
+ primeClient(client);
385
+ assert.equal(
386
+ await client.chatCompletion([{ role: "user", content: "extract facts" }]),
387
+ null,
388
+ );
389
+ assert.equal(server.requestBodies.length, 1);
390
+ assert.equal(
391
+ await client.chatCompletion([{ role: "user", content: "extract facts again" }]),
392
+ null,
393
+ );
394
+ assert.equal(server.requestBodies.length, 1, "cooldown must stop a second request");
395
+ } finally {
396
+ await dispatcherOf(client).close();
397
+ await server.close();
398
+ }
399
+ });
400
+
401
+ test("a truncated 503 body retains the configured retry", async () => {
402
+ const original = globalThis.fetch;
403
+ const encoder = new TextEncoder();
404
+ let calls = 0;
405
+ globalThis.fetch = (async () => {
406
+ calls += 1;
407
+ if (calls === 1) {
408
+ const body = new ReadableStream<Uint8Array>({
409
+ start(controller) {
410
+ controller.enqueue(encoder.encode('{"error":'));
411
+ controller.error(new TypeError("terminated"));
412
+ },
413
+ });
414
+ return new Response(body, {
415
+ status: 503,
416
+ headers: { "content-type": "application/json" },
417
+ });
418
+ }
419
+ return new Response(
420
+ JSON.stringify({ choices: [{ message: { content: "retry succeeded" } }] }),
421
+ { status: 200, headers: { "content-type": "application/json" } },
422
+ );
423
+ }) as typeof fetch;
424
+
425
+ const client = new LocalLlmClient(
426
+ createConfig({ localLlmRetry5xxCount: 1, localLlmRetryBackoffMs: 1 }),
427
+ );
428
+ try {
429
+ primeClient(client);
430
+ const result = await client.chatCompletion([
431
+ { role: "user", content: "extract facts" },
432
+ ]);
433
+ assert.equal(result?.content, "retry succeeded");
434
+ assert.equal(calls, 2);
435
+ } finally {
436
+ globalThis.fetch = original;
437
+ await dispatcherOf(client).close();
438
+ }
439
+ });
440
+
247
441
  test("a process-wide custom dispatcher is left in place", async () => {
248
442
  // A deployment may reach the local endpoint only through a ProxyAgent (or
249
443
  // another custom connect/TLS/DNS transport) installed with
package/src/local-llm.ts CHANGED
@@ -962,6 +962,7 @@ export class LocalLlmClient {
962
962
  temperature: options.temperature ?? 0.7,
963
963
  // Use max_tokens consistent with cloud models
964
964
  max_tokens: options.maxTokens ?? 4096,
965
+ stream: false,
965
966
  };
966
967
 
967
968
  // Skip response_format for local LLMs - they don't support json_object type
@@ -1034,9 +1035,11 @@ export class LocalLlmClient {
1034
1035
  ? Math.min(this.config.localLlmTimeoutMs, options.timeoutMs)
1035
1036
  : this.config.localLlmTimeoutMs;
1036
1037
  const maxAttempts = 1 + Math.max(0, this.config.localLlmRetry5xxCount);
1037
- let response: Response | null = null;
1038
+ let response: Response | null = null, responseBody = "";
1038
1039
  let lastAbortError: Error | null = null;
1039
1040
  for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
1041
+ response = null;
1042
+ responseBody = "";
1040
1043
  const attemptAbort = new AbortController();
1041
1044
  const onCallerAbort = (): void => {
1042
1045
  attemptAbort.abort(options.signal?.reason);
@@ -1059,18 +1062,28 @@ export class LocalLlmClient {
1059
1062
  signal: attemptAbort.signal,
1060
1063
  budgetMs: this.config.localLlmTimeoutMs,
1061
1064
  });
1065
+ responseBody = await response.text();
1062
1066
  } catch (err) {
1063
- if (!isAbortError(err)) throw err;
1064
- lastAbortError = err instanceof Error ? err : new Error(String(err));
1065
- if (options.signal?.aborted || attempt >= maxAttempts) {
1067
+ const error = err instanceof Error ? err : new Error(String(err));
1068
+ if (options.signal?.aborted) {
1069
+ response = null;
1070
+ lastAbortError = error;
1066
1071
  break;
1067
1072
  }
1068
- const backoffMs = this.config.localLlmRetryBackoffMs * attempt;
1069
- log.warn(
1070
- `local LLM request aborted: op=${operation} attempt=${attempt}/${maxAttempts} timeoutMs=${effectiveTimeoutMs} model=${this.config.localLlmModel}; retrying after ${backoffMs}ms`,
1071
- );
1072
- if (!(await waitForRetryBackoff(backoffMs, options.signal))) return null;
1073
- continue;
1073
+ if (response && !response.ok) {
1074
+ log.debug(`local LLM failed to read ${response.status} response body: ${error.message}`);
1075
+ } else {
1076
+ response = null;
1077
+ if (!isAbortError(err)) throw err;
1078
+ lastAbortError = error;
1079
+ if (attempt >= maxAttempts) break;
1080
+ const backoffMs = this.config.localLlmRetryBackoffMs * attempt;
1081
+ log.warn(
1082
+ `local LLM request aborted: op=${operation} attempt=${attempt}/${maxAttempts} timeoutMs=${effectiveTimeoutMs} model=${this.config.localLlmModel}; retrying after ${backoffMs}ms`,
1083
+ );
1084
+ if (!(await waitForRetryBackoff(backoffMs, options.signal))) return null;
1085
+ continue;
1086
+ }
1074
1087
  } finally {
1075
1088
  clearTimeout(attemptTimeout);
1076
1089
  options.signal?.removeEventListener("abort", onCallerAbort);
@@ -1078,20 +1091,15 @@ export class LocalLlmClient {
1078
1091
 
1079
1092
  if (response.ok) break;
1080
1093
  if (response.status >= 500 && attempt < maxAttempts) {
1081
- try {
1082
- const errorText = await response.clone().text();
1083
- const nonRecoverableReason =
1084
- extractNonRecoverableBackendReasonFromErrorText(errorText);
1085
- if (nonRecoverableReason) {
1086
- this.markBackendUnavailable(
1087
- nonRecoverableReason,
1088
- this.config.localLlm400CooldownMs,
1089
- );
1090
- this.consecutive400s = 0;
1091
- return null;
1092
- }
1093
- } catch (e) {
1094
- log.debug(`local LLM failed to inspect retryable error body: ${e}`);
1094
+ const nonRecoverableReason =
1095
+ extractNonRecoverableBackendReasonFromErrorText(responseBody);
1096
+ if (nonRecoverableReason) {
1097
+ this.markBackendUnavailable(
1098
+ nonRecoverableReason,
1099
+ this.config.localLlm400CooldownMs,
1100
+ );
1101
+ this.consecutive400s = 0;
1102
+ return null;
1095
1103
  }
1096
1104
  }
1097
1105
  if (response.status < 500 || attempt >= maxAttempts) break;
@@ -1120,19 +1128,11 @@ export class LocalLlmClient {
1120
1128
 
1121
1129
  if (!response.ok) {
1122
1130
  let reason = "";
1123
- let errorText = "";
1124
1131
  try {
1125
- errorText = await response.text();
1126
- // Try to extract a stable error message without logging content.
1127
- try {
1128
- const parsed = JSON.parse(errorText) as { error?: { message?: string } };
1129
- reason = parsed?.error?.message ? ` — ${parsed.error.message}` : "";
1130
- } catch {
1131
- // Keep a short preview in debug only.
1132
- log.debug(`local LLM error body: ${errorText.slice(0, 500)}`);
1133
- }
1134
- } catch (e) {
1135
- log.debug(`local LLM failed to read error body: ${e}`);
1132
+ const parsed = JSON.parse(responseBody) as { error?: { message?: string } };
1133
+ reason = parsed?.error?.message ? ` ${parsed.error.message}` : "";
1134
+ } catch {
1135
+ log.debug(`local LLM error body: ${responseBody.slice(0, 500)}`);
1136
1136
  }
1137
1137
  log.warn(
1138
1138
  `local LLM request failed: ${response.status} ${response.statusText}${reason} ` +
@@ -1140,7 +1140,7 @@ export class LocalLlmClient {
1140
1140
  );
1141
1141
  const nonRecoverableReason =
1142
1142
  extractNonRecoverableBackendReason(reason) ??
1143
- extractNonRecoverableBackendReasonFromErrorText(errorText);
1143
+ extractNonRecoverableBackendReasonFromErrorText(responseBody);
1144
1144
  if (nonRecoverableReason) {
1145
1145
  this.markBackendUnavailable(
1146
1146
  nonRecoverableReason,
@@ -1166,7 +1166,7 @@ export class LocalLlmClient {
1166
1166
  }
1167
1167
  this.consecutive400s = 0;
1168
1168
 
1169
- const data = (await response.json()) as {
1169
+ const data = JSON.parse(responseBody) as {
1170
1170
  choices?: Array<{
1171
1171
  message?: { content?: string; reasoning_content?: string };
1172
1172
  }>;