newmark-agent 0.5.14 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/dist/conversation-utility-host.bundle.cjs +4565 -2998
  2. package/dist/conversation-utility-host.js +1 -1
  3. package/dist/core/agent.d.ts +108 -16
  4. package/dist/core/agent.js +698 -195
  5. package/dist/core/agentKernel/agent.d.ts +1 -0
  6. package/dist/core/agentKernel/agent.js +9 -0
  7. package/dist/core/agentKernelDiagnostics.d.ts +4 -6
  8. package/dist/core/agentKernelDiagnostics.js +12 -9
  9. package/dist/core/agentKernelRunner.d.ts +5 -0
  10. package/dist/core/agentKernelRunner.js +250 -107
  11. package/dist/core/autoRouter.js +17 -27
  12. package/dist/core/config.d.ts +2 -0
  13. package/dist/core/config.js +7 -1
  14. package/dist/core/conversationCommandState.d.ts +17 -0
  15. package/dist/core/conversationCommandState.js +140 -0
  16. package/dist/core/conversationKernel.d.ts +25 -4
  17. package/dist/core/conversationKernel.js +360 -89
  18. package/dist/core/conversationListEvent.d.ts +6 -0
  19. package/dist/core/conversationListEvent.js +14 -0
  20. package/dist/core/electronUtilityAgentClient.d.ts +4 -1
  21. package/dist/core/electronUtilityAgentClient.js +4 -4
  22. package/dist/core/electronUtilityRuntimePool.d.ts +8 -2
  23. package/dist/core/electronUtilityRuntimePool.js +11 -5
  24. package/dist/core/installUpdate.js +41 -29
  25. package/dist/core/modelResponseHealth.d.ts +17 -0
  26. package/dist/core/modelResponseHealth.js +193 -0
  27. package/dist/core/providerUsageAccounting.d.ts +47 -0
  28. package/dist/core/providerUsageAccounting.js +71 -0
  29. package/dist/core/requestContextEstimate.d.ts +18 -0
  30. package/dist/core/requestContextEstimate.js +54 -0
  31. package/dist/core/subagent.d.ts +89 -5
  32. package/dist/core/subagent.js +264 -41
  33. package/dist/core/subagentCommunication.d.ts +43 -0
  34. package/dist/core/subagentCommunication.js +167 -0
  35. package/dist/core/types.d.ts +34 -1
  36. package/dist/core/utilityAgentProtocol.d.ts +4 -0
  37. package/dist/core/workEventCoalescer.js +4 -1
  38. package/dist/core/wslAgentClient.d.ts +18 -5
  39. package/dist/core/wslAgentClient.js +79 -27
  40. package/dist/core/wslAgentProtocol.d.ts +7 -0
  41. package/dist/core/wslAgentRuntimePool.d.ts +8 -2
  42. package/dist/core/wslAgentRuntimePool.js +19 -6
  43. package/dist/core/wslRuntimeProcessTree.d.ts +10 -0
  44. package/dist/core/wslRuntimeProcessTree.js +104 -0
  45. package/dist/llm/provider.d.ts +5 -12
  46. package/dist/llm/provider.js +150 -269
  47. package/dist/main.js +582 -356
  48. package/dist/preload.js +3 -3
  49. package/dist/providers/chat-completions.adapter.d.ts +2 -0
  50. package/dist/providers/chat-completions.adapter.js +84 -36
  51. package/dist/providers/provider-adapter.d.ts +2 -0
  52. package/dist/providers/provider-events.d.ts +23 -14
  53. package/dist/providers/provider-events.js +148 -43
  54. package/dist/providers/provider-headers.d.ts +7 -3
  55. package/dist/providers/provider-headers.js +109 -39
  56. package/dist/providers/provider-request-compat.d.ts +6 -0
  57. package/dist/providers/provider-request-compat.js +37 -0
  58. package/dist/providers/responses.adapter.d.ts +2 -0
  59. package/dist/providers/responses.adapter.js +192 -126
  60. package/dist/server.d.ts +9 -2
  61. package/dist/server.js +100 -24
  62. package/dist/tools/index.js +17 -1
  63. package/dist/ui/index.html +2515 -693
  64. package/dist/ui/lucide-sprite.svg +7 -0
  65. package/dist/ui/startup.html +6 -6
  66. package/dist/wsl-agent-host.bundle.cjs +4575 -3006
  67. package/dist/wsl-agent-host.js +9 -4
  68. package/package.json +23 -4
@@ -41,6 +41,8 @@ const os = __importStar(require("os"));
41
41
  const path = __importStar(require("path"));
42
42
  const child_process_1 = require("child_process");
43
43
  const agentKernelDiagnostics_1 = require("../core/agentKernelDiagnostics");
44
+ const provider_request_compat_1 = require("../providers/provider-request-compat");
45
+ const provider_headers_1 = require("../providers/provider-headers");
44
46
  const chat_messages_1 = require("../providers/chat-messages");
45
47
  const providers_1 = require("../providers");
46
48
  const undici_1 = require("undici");
@@ -188,14 +190,16 @@ class LLMProvider {
188
190
  }
189
191
  return this.nodeProxyAgent;
190
192
  }
191
- async providerFetch(input, init = {}) {
192
- const dispatcher = this.resolveProxyDispatcher();
193
+ async providerFetch(input, init = {}, streaming = false) {
194
+ const proxyDispatcher = this.resolveProxyDispatcher();
193
195
  const url = typeof input === 'string'
194
196
  ? input
195
197
  : input instanceof URL
196
198
  ? input.toString()
197
199
  : String(input.url || '');
198
- if (!dispatcher || this.isPlainHttpLoopback(url))
200
+ const selectedDispatcher = this.isPlainHttpLoopback(url) ? null : proxyDispatcher;
201
+ const dispatcher = streaming ? (0, providers_1.providerStreamingDispatcher)(selectedDispatcher) : selectedDispatcher;
202
+ if (!dispatcher)
199
203
  return fetch(input, init);
200
204
  return fetch(input, { ...init, dispatcher });
201
205
  }
@@ -305,40 +309,23 @@ class LLMProvider {
305
309
  return 'chat_stream';
306
310
  }
307
311
  cleanBaseUrl() {
308
- return this.baseUrl.replace(/\/+$/, '');
312
+ return this.baseUrl.trim();
309
313
  }
310
314
  githubModelsBaseUrl() {
311
315
  const base = this.cleanBaseUrl();
312
316
  if (!base)
313
317
  return 'https://models.github.ai';
314
- if (/\/inference$/i.test(base))
315
- return base.replace(/\/inference$/i, '');
316
318
  return base;
317
319
  }
318
320
  githubModelsUrl(pathname) {
319
- const base = this.githubModelsBaseUrl();
320
- const path = pathname.startsWith('/') ? pathname : `/${pathname}`;
321
- return `${base}${path}`;
321
+ return (0, provider_request_compat_1.providerEndpoint)(this.githubModelsBaseUrl(), pathname);
322
322
  }
323
- openAIHeaders() {
324
- if (this.protocol() === 'github_models')
325
- return this.githubModelsHeaders();
326
- return {
327
- 'Content-Type': 'application/json',
328
- 'Authorization': `Bearer ${this.apiKey}`,
329
- };
323
+ openAIHeaders(stream = false) {
324
+ return (0, provider_request_compat_1.providerRequestHeaders)(this.protocol() === 'github_models' ? 'github_models' : 'openai', this.apiKey, stream);
330
325
  }
331
326
  llmErrorText(response, body) {
332
- const rawRetryAfter = response.headers?.get('retry-after')?.trim() || '';
333
- let retryAfter = '';
334
- if (/^\d+(?:\.\d+)?$/.test(rawRetryAfter)) {
335
- retryAfter = ` Retry-After: ${rawRetryAfter}s`;
336
- }
337
- else if (rawRetryAfter) {
338
- const retryAt = Date.parse(rawRetryAfter);
339
- if (Number.isFinite(retryAt))
340
- retryAfter = ` Retry-After: ${Math.max(0, Math.ceil((retryAt - Date.now()) / 1_000))}s`;
341
- }
327
+ const seconds = response.headers ? (0, provider_headers_1.normalizeProviderHeaders)(response.headers).retryAfterSeconds : undefined;
328
+ const retryAfter = seconds === undefined ? '' : ` Retry-After: ${seconds}s`;
342
329
  return `[LLM Error: ${response.status}]${retryAfter} ${body}`;
343
330
  }
344
331
  headerReader(headers) {
@@ -369,12 +356,8 @@ class LLMProvider {
369
356
  return;
370
357
  console.error(`[NewmarkProvider] ${stage}${detail ? ` ${detail}` : ''}`);
371
358
  }
372
- githubModelsHeaders() {
373
- return {
374
- 'Content-Type': 'application/json',
375
- 'Authorization': `Bearer ${this.apiKey}`,
376
- 'X-GitHub-Api-Version': '2022-11-28',
377
- };
359
+ githubModelsHeaders(stream = false) {
360
+ return (0, provider_request_compat_1.providerRequestHeaders)('github_models', this.apiKey, stream);
378
361
  }
379
362
  temperatureCapabilityKey(url, body) {
380
363
  return `${url}|${String(body.model || '')}`;
@@ -460,8 +443,17 @@ class LLMProvider {
460
443
  headers,
461
444
  body: JSON.stringify(body),
462
445
  signal: abort.signal,
463
- });
464
- return response;
446
+ }, true);
447
+ // Keep the caller's cancellation and explicit deadline attached through
448
+ // the JSON body, not just until response headers arrive.
449
+ const responseText = await response.text();
450
+ return {
451
+ ok: response.ok,
452
+ status: response.status,
453
+ headers: response.headers,
454
+ text: async () => responseText,
455
+ json: async () => JSON.parse(responseText || '{}'),
456
+ };
465
457
  }
466
458
  catch (e) {
467
459
  if (signal?.aborted)
@@ -713,12 +705,8 @@ class LLMProvider {
713
705
  out = out.split(this.apiKey).join('sk-***REDACTED***');
714
706
  return out.replace(/sk-[A-Za-z0-9_\-.]{8,}/g, 'sk-***REDACTED***');
715
707
  }
716
- anthropicHeaders() {
717
- return {
718
- 'Content-Type': 'application/json',
719
- 'x-api-key': this.apiKey,
720
- 'anthropic-version': '2023-06-01',
721
- };
708
+ anthropicHeaders(stream = false) {
709
+ return (0, provider_request_compat_1.providerRequestHeaders)('anthropic', this.apiKey, stream);
722
710
  }
723
711
  stringifyContent(value) {
724
712
  return (0, chat_messages_1.stringifyContent)(value);
@@ -944,12 +932,17 @@ class LLMProvider {
944
932
  ? current
945
933
  : {};
946
934
  }
947
- async openAIResponsesChat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier) {
948
- const response = await this.postJsonWithFetchFallback(`${this.cleanBaseUrl()}/responses`, this.openAIHeaders(), this.responsesBody(model, messages, systemPrompt, temperature, maxTokens, [], reasoningTier), 120000, signal);
935
+ async openAIResponsesChat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier, onUsage) {
936
+ const response = await this.postJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/responses'), this.openAIHeaders(), this.responsesBody(model, messages, systemPrompt, temperature, maxTokens, [], reasoningTier), 120000, signal);
949
937
  if (!response.ok) {
950
938
  return this.llmErrorText(response, await response.text());
951
939
  }
952
- return this.extractResponsesText(this.normalizeResponsesPayload(await response.json()));
940
+ const json = this.normalizeResponsesPayload(await response.json());
941
+ onUsage?.((0, agentKernelDiagnostics_1.extractProviderUsage)(json));
942
+ const failure = this.nativeJsonCompletionError(json, 'responses');
943
+ if (failure)
944
+ return failure;
945
+ return this.extractResponsesText(json);
953
946
  }
954
947
  anthropicTools(tools) {
955
948
  const converted = [];
@@ -1052,6 +1045,7 @@ class LLMProvider {
1052
1045
  };
1053
1046
  const serialized = await adapter.serializeRequest(request);
1054
1047
  serialized.body.stream = mode === 'chat' ? false : true;
1048
+ serialized.headers = this.openAIHeaders(serialized.body.stream === true);
1055
1049
  const execSignal = signal ?? new AbortController().signal;
1056
1050
  const transport = this.buildProviderAdapterTransport();
1057
1051
  const streaming = mode === 'chat_stream';
@@ -1091,7 +1085,7 @@ class LLMProvider {
1091
1085
  yield* this.adapterResponsesBridge(model, messages, systemPrompt, temperature, maxTokens, tools, signal, downgradeTier);
1092
1086
  return;
1093
1087
  }
1094
- yield { type: 'text', text: event.error };
1088
+ yield { type: 'text', text: event.error, providerError: true };
1095
1089
  return;
1096
1090
  }
1097
1091
  }
@@ -1113,7 +1107,7 @@ class LLMProvider {
1113
1107
  };
1114
1108
  const serialized = await adapter.serializeRequest(request);
1115
1109
  serialized.body.stream = true;
1116
- serialized.headers['Accept'] = 'text/event-stream';
1110
+ serialized.headers = this.openAIHeaders(true);
1117
1111
  const execSignal = signal ?? new AbortController().signal;
1118
1112
  const transport = this.buildProviderAdapterTransport();
1119
1113
  let reasoningSummary = '';
@@ -1147,7 +1141,7 @@ class LLMProvider {
1147
1141
  continue;
1148
1142
  }
1149
1143
  if (event.type === 'response.failed') {
1150
- yield { type: 'text', text: event.error };
1144
+ yield { type: 'text', text: event.error, providerError: true };
1151
1145
  return;
1152
1146
  }
1153
1147
  }
@@ -1158,6 +1152,7 @@ class LLMProvider {
1158
1152
  output: usage.outputTokens,
1159
1153
  cacheRead: usage.cacheReadTokens,
1160
1154
  cacheWrite: usage.cacheWriteTokens,
1155
+ reported: usage.reported,
1161
1156
  };
1162
1157
  }
1163
1158
  shouldDowngradeToResponses(errorText) {
@@ -1195,9 +1190,10 @@ class LLMProvider {
1195
1190
  headers: request.headers,
1196
1191
  body: JSON.stringify(request.body),
1197
1192
  signal: abort.signal,
1198
- });
1193
+ }, true);
1199
1194
  if (request.body.temperature !== undefined && response.status === 400) {
1200
- const errorText = await response.clone().text();
1195
+ const buffered = await this.bufferTransportResponse(response);
1196
+ const errorText = await buffered.text();
1201
1197
  if (this.unsupportedTemperatureError(response.status, errorText)) {
1202
1198
  this.temperatureUnsupported.add(this.temperatureCapabilityKey(request.url, request.body));
1203
1199
  const retryBody = { ...request.body };
@@ -1207,8 +1203,17 @@ class LLMProvider {
1207
1203
  headers: request.headers,
1208
1204
  body: JSON.stringify(retryBody),
1209
1205
  signal: abort.signal,
1210
- });
1206
+ }, true);
1211
1207
  }
1208
+ else {
1209
+ return buffered;
1210
+ }
1211
+ }
1212
+ // Compatible providers can answer a stream request with JSON or
1213
+ // an error body. Read those once while cancellation still owns
1214
+ // the socket; only SSE readers take ownership after this returns.
1215
+ if (!response.ok || !/text\/event-stream/i.test(response.headers.get('content-type') || '')) {
1216
+ return await this.bufferTransportResponse(response);
1212
1217
  }
1213
1218
  return response;
1214
1219
  }
@@ -1220,7 +1225,7 @@ class LLMProvider {
1220
1225
  if (!this.shouldUseNodeHttpFallback(error))
1221
1226
  throw error;
1222
1227
  const fallbackHeaders = { ...request.headers };
1223
- delete fallbackHeaders['Accept'];
1228
+ fallbackHeaders.Accept = (0, provider_request_compat_1.providerRequestHeaders)(this.protocol(), this.apiKey).Accept;
1224
1229
  const fallback = await this.postJsonWithFetchFallback(request.url, fallbackHeaders, { ...request.body, stream: false }, effectiveTimeout, signal);
1225
1230
  return this.toTransportResponse(fallback);
1226
1231
  }
@@ -1234,6 +1239,17 @@ class LLMProvider {
1234
1239
  return this.toTransportResponse(response);
1235
1240
  };
1236
1241
  }
1242
+ async bufferTransportResponse(response) {
1243
+ const body = await response.text();
1244
+ return {
1245
+ ok: response.ok,
1246
+ status: response.status,
1247
+ headers: response.headers,
1248
+ body: null,
1249
+ text: async () => body,
1250
+ json: async () => JSON.parse(body),
1251
+ };
1252
+ }
1237
1253
  toTransportResponse(response) {
1238
1254
  if (response.body !== undefined) {
1239
1255
  return response;
@@ -1310,10 +1326,8 @@ class LLMProvider {
1310
1326
  throw new Error('LLMProvider legacy OpenAI streaming was removed in dev-0.3.0: enable provider_adapters_v2 (useProviderAdaptersV2).');
1311
1327
  }
1312
1328
  /**
1313
- * GitHub Models streaming path. Preserved as a dedicated implementation
1314
- * because the provider adapters (V2) do not serialize the GitHub Models
1315
- * inference URL or its X-GitHub-Api-Version headers. OpenAI protocol
1316
- * providers route exclusively through chatStreamWithToolsV2.
1329
+ * GitHub Models keeps its dedicated URL, headers and request serialization.
1330
+ * Response parsing shares the Chat adapter's completion validation.
1317
1331
  */
1318
1332
  async *githubModelsChatStreamWithTools(model, messages, systemPrompt, temperature, maxTokens, tools, signal) {
1319
1333
  const url = this.githubModelsUrl('/inference/chat/completions');
@@ -1329,191 +1343,30 @@ class LLMProvider {
1329
1343
  tool_choice: 'auto',
1330
1344
  stream: true,
1331
1345
  };
1332
- const abort = new AbortController();
1333
- const forwardAbort = () => abort.abort(signal?.reason);
1334
- if (signal?.aborted)
1335
- forwardAbort();
1336
- else
1337
- signal?.addEventListener('abort', forwardAbort, { once: true });
1338
- // Streaming provider responses are intentionally unbounded. Do not turn
1339
- // silence into an empty-response failure.
1340
- const effectiveTimeout = 0;
1341
- const timeout = effectiveTimeout > 0
1342
- ? setTimeout(() => abort.abort(providerTimeoutError(effectiveTimeout)), effectiveTimeout)
1343
- : undefined;
1344
- let reader = null;
1345
- try {
1346
- let response;
1347
- try {
1348
- response = await this.providerFetch(url, {
1349
- method: 'POST',
1350
- headers: this.openAIHeaders(),
1351
- body: JSON.stringify(body),
1352
- signal: abort.signal,
1353
- });
1354
- }
1355
- catch (e) {
1356
- if (signal?.aborted)
1357
- throw abortFailure(signal);
1358
- if (abort.signal.aborted)
1359
- throw abortFailure(abort.signal);
1360
- if (!this.shouldUseNodeHttpFallback(e))
1361
- throw e;
1362
- clearTimeout(timeout);
1363
- yield* this.githubModelsChatNonStreaming(url, body, signal);
1364
- return;
1346
+ // Keep GitHub's endpoint, headers and payload, while sharing the validated
1347
+ // Chat Completions parser and fetch-to-node transport fallback.
1348
+ const adapter = (0, providers_1.createProviderAdapter)(this.name, 'chat_completions');
1349
+ const request = { url, headers: this.openAIHeaders(true), body };
1350
+ let currentReasoningContent = '';
1351
+ for await (const event of adapter.execute(request, signal || new AbortController().signal, this.buildProviderAdapterTransport())) {
1352
+ if (event.type === 'reasoning.summary.delta') {
1353
+ currentReasoningContent += event.delta;
1354
+ yield { type: 'status', text: '', reasoningContent: currentReasoningContent };
1365
1355
  }
1366
- if (!response.ok) {
1367
- const err = await response.text();
1368
- yield { type: 'text', text: this.llmErrorText(response, err) };
1369
- return;
1356
+ else if (event.type === 'text.delta') {
1357
+ yield { type: 'text', text: event.delta, reasoningContent: currentReasoningContent || undefined };
1370
1358
  }
1371
- reader = response.body?.getReader() ?? null;
1372
- if (!reader) {
1373
- yield { type: 'text', text: '[Error] No response body' };
1374
- return;
1359
+ else if (event.type === 'tool_call.completed') {
1360
+ yield { type: 'tool_call', text: '', toolCall: { id: event.id, name: event.name, arguments: event.arguments }, reasoningContent: currentReasoningContent || undefined };
1375
1361
  }
1376
- const decoder = new TextDecoder();
1377
- let buffer = '';
1378
- const toolCalls = new Map();
1379
- const toolCallOrder = [];
1380
- let syntheticToolIndex = 0;
1381
- let lastToolIndex = 0;
1382
- let currentReasoningContent = '';
1383
- let contentPolicyBlocked = false;
1384
- let emittedContent = false;
1385
- let explicitCompletion = false;
1386
- const streamSignal = signal || new AbortController().signal;
1387
- while (true) {
1388
- const { done, value } = await (0, providers_1.readProviderStreamChunk)(reader, streamSignal);
1389
- if (done)
1390
- break;
1391
- buffer += decoder.decode(value, { stream: true });
1392
- const lines = buffer.split('\n');
1393
- buffer = lines.pop() || '';
1394
- for (const line of lines) {
1395
- const trimmed = line.trim();
1396
- if (!trimmed.startsWith('data: '))
1397
- continue;
1398
- const data = trimmed.slice(6);
1399
- if (data === '[DONE]') {
1400
- explicitCompletion = true;
1401
- continue;
1402
- }
1403
- try {
1404
- const json = JSON.parse(data);
1405
- if (json.usage)
1406
- yield { type: 'usage', text: '', usage: (0, agentKernelDiagnostics_1.extractProviderUsage)(json) };
1407
- if (this.contentPolicyBlocked(json))
1408
- contentPolicyBlocked = true;
1409
- const delta = json.choices?.[0]?.delta;
1410
- if (!delta)
1411
- continue;
1412
- if (delta.reasoning_content) {
1413
- currentReasoningContent += delta.reasoning_content;
1414
- yield { type: 'status', text: '', reasoningContent: currentReasoningContent };
1415
- }
1416
- const deltaText = this.extractTextValue(delta.content);
1417
- if (deltaText) {
1418
- emittedContent = true;
1419
- yield { type: 'text', text: deltaText, reasoningContent: currentReasoningContent || undefined };
1420
- }
1421
- if (delta.tool_calls) {
1422
- for (const tc of delta.tool_calls) {
1423
- const rawIndex = Number(tc.index);
1424
- const index = Number.isInteger(rawIndex) && rawIndex >= 0
1425
- ? rawIndex
1426
- : (tc.id ? syntheticToolIndex++ : lastToolIndex);
1427
- lastToolIndex = index;
1428
- let call = toolCalls.get(index);
1429
- if (!call && (tc.id || tc.function?.name)) {
1430
- call = { id: tc.id || '', name: tc.function?.name || '', argumentParts: [] };
1431
- toolCalls.set(index, call);
1432
- toolCallOrder.push(index);
1433
- }
1434
- if (!call)
1435
- continue;
1436
- if (tc.id && !call.id)
1437
- call.id = tc.id;
1438
- if (tc.function?.name && !call.name)
1439
- call.name = tc.function.name;
1440
- if (tc.function?.arguments)
1441
- call.argumentParts.push(tc.function.arguments);
1442
- }
1443
- }
1444
- }
1445
- catch { /* skip malformed JSON */ }
1446
- }
1447
- }
1448
- if (toolCallOrder.length) {
1449
- for (const index of toolCallOrder) {
1450
- const call = toolCalls.get(index);
1451
- if (!call)
1452
- continue;
1453
- yield {
1454
- type: 'tool_call',
1455
- text: '',
1456
- toolCall: {
1457
- id: call.id,
1458
- name: call.name,
1459
- arguments: (0, providers_1.assembleCompatibleToolArguments)(call.argumentParts),
1460
- },
1461
- reasoningContent: currentReasoningContent || undefined,
1462
- };
1463
- }
1464
- }
1465
- else if (!emittedContent && contentPolicyBlocked) {
1466
- yield { type: 'text', text: '[Error] Content policy refusal (content_filter).' };
1362
+ else if (event.type === 'usage.updated') {
1363
+ yield { type: 'usage', text: '', usage: this.toStreamUsage(event.usage) };
1467
1364
  }
1468
- else if (!explicitCompletion && !emittedContent && !currentReasoningContent) {
1469
- yield { type: 'text', text: '[LLM Error] GitHub Models stream ended before an explicit completion.' };
1365
+ else if (event.type === 'response.failed') {
1366
+ yield { type: 'text', text: event.error, providerError: true };
1367
+ return;
1470
1368
  }
1471
1369
  }
1472
- finally {
1473
- reader?.releaseLock();
1474
- clearTimeout(timeout);
1475
- signal?.removeEventListener('abort', forwardAbort);
1476
- }
1477
- }
1478
- /**
1479
- * GitHub Models non-streaming fallback used when the streaming fetch fails
1480
- * and the node-http transport is available. Mirrors the legacy chat-tools
1481
- * node fallback; the responses downgrade is intentionally not applied for
1482
- * GitHub Models (its inference endpoint is chat-completions only).
1483
- */
1484
- async *githubModelsChatNonStreaming(url, streamingBody, signal) {
1485
- const body = { ...streamingBody, stream: false };
1486
- const response = await this.postJsonWithFetchFallback(url, this.openAIHeaders(), body, 120000, signal);
1487
- if (!response.ok) {
1488
- const err = await response.text();
1489
- yield { type: 'text', text: this.llmErrorText(response, err) };
1490
- return;
1491
- }
1492
- const json = await response.json();
1493
- yield { type: 'usage', text: '', usage: (0, agentKernelDiagnostics_1.extractProviderUsage)(json) };
1494
- const choice = json?.choices?.[0];
1495
- const message = choice?.message || {};
1496
- if (message.reasoning_content) {
1497
- yield { type: 'status', text: '', reasoningContent: String(message.reasoning_content) };
1498
- }
1499
- const messageText = this.extractTextValue(message.content) || this.extractTextValue(choice?.text);
1500
- if (messageText) {
1501
- yield { type: 'text', text: messageText, reasoningContent: message.reasoning_content ? String(message.reasoning_content) : undefined };
1502
- }
1503
- for (const tc of message.tool_calls || []) {
1504
- yield {
1505
- type: 'tool_call',
1506
- text: '',
1507
- toolCall: {
1508
- id: String(tc.id || ''),
1509
- name: String(tc.function?.name || ''),
1510
- arguments: String(tc.function?.arguments || '{}'),
1511
- },
1512
- };
1513
- }
1514
- if (!messageText && !(message.tool_calls || []).length && this.contentPolicyBlocked(json)) {
1515
- yield { type: 'text', text: '[Error] Content policy refusal (content_filter).' };
1516
- }
1517
1370
  }
1518
1371
  async *anthropicChatWithTools(model, messages, systemPrompt, temperature, maxTokens, tools, signal) {
1519
1372
  const { system, messages: anthropicMessages } = this.anthropicMessages(messages, systemPrompt);
@@ -1528,20 +1381,21 @@ class LLMProvider {
1528
1381
  const convertedTools = this.anthropicTools(tools);
1529
1382
  if (convertedTools.length)
1530
1383
  body.tools = convertedTools;
1531
- const response = await this.providerFetch(`${this.cleanBaseUrl()}/messages`, {
1532
- method: 'POST',
1533
- headers: this.anthropicHeaders(),
1534
- body: JSON.stringify(body),
1535
- signal,
1536
- });
1384
+ // A Build reuses its cancellation signal across many tool turns. Keep the
1385
+ // native JSON exchange request-local so completed fetches cannot accumulate
1386
+ // their abort followers on that long-lived signal until a later GC.
1387
+ const response = await this.postJsonOnce((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/messages'), this.anthropicHeaders(), body, 0, signal);
1537
1388
  if (!response.ok) {
1538
1389
  const err = await response.text();
1539
1390
  yield { type: 'text', text: this.llmErrorText(response, err) };
1540
1391
  return;
1541
1392
  }
1542
1393
  const json = await response.json();
1543
- yield { type: 'usage', text: '', usage: (0, agentKernelDiagnostics_1.extractProviderUsage)(json) };
1544
- for (const block of json.content || []) {
1394
+ yield { type: 'usage', text: '', usage: (0, agentKernelDiagnostics_1.extractProviderUsage)(json, { protocol: 'anthropic' }) };
1395
+ const blocks = Array.isArray(json.content) ? json.content : [];
1396
+ const completeCalls = blocks.filter(block => block.type === 'tool_use');
1397
+ const failure = this.nativeJsonCompletionError(json, 'anthropic');
1398
+ for (const block of blocks) {
1545
1399
  const type = String(block.type || '');
1546
1400
  if (type === 'text' && block.text) {
1547
1401
  const text = this.extractTextValue(block.text);
@@ -1551,18 +1405,37 @@ class LLMProvider {
1551
1405
  else if (type === 'thinking' && block.thinking) {
1552
1406
  yield { type: 'status', text: '', reasoningContent: String(block.thinking) };
1553
1407
  }
1554
- else if (type === 'tool_use') {
1555
- yield {
1556
- type: 'tool_call',
1557
- text: '',
1558
- toolCall: {
1559
- id: String(block.id || ''),
1560
- name: String(block.name || ''),
1561
- arguments: JSON.stringify(block.input || {}),
1562
- },
1563
- };
1564
- }
1565
1408
  }
1409
+ if (failure) {
1410
+ yield { type: 'text', text: failure };
1411
+ return;
1412
+ }
1413
+ if (completeCalls.some(block => !block.id || !block.name || !block.input || typeof block.input !== 'object' || Array.isArray(block.input))) {
1414
+ yield { type: 'text', text: '[LLM Error] Provider returned incomplete or invalid tool-call arguments.', providerError: true };
1415
+ return;
1416
+ }
1417
+ for (const block of completeCalls) {
1418
+ yield { type: 'tool_call', text: '', toolCall: {
1419
+ id: String(block.id), name: String(block.name), arguments: JSON.stringify(block.input),
1420
+ } };
1421
+ }
1422
+ }
1423
+ nativeJsonCompletionError(json, protocol) {
1424
+ if (json.error || json.type === 'error') {
1425
+ const error = json.error && typeof json.error === 'object' ? json.error : {};
1426
+ return '[LLM Error] ' + (this.extractTextValue(error.message || error.type || json.error) || 'Provider returned an error response.');
1427
+ }
1428
+ const choice = Array.isArray(json.choices) ? json.choices[0] : undefined;
1429
+ const reason = protocol === 'anthropic' ? json.stop_reason : choice?.finish_reason;
1430
+ if (reason === 'max_tokens' || reason === 'length')
1431
+ return '[LLM Error] Provider response was truncated by the output token limit.';
1432
+ if (reason === 'content_filter')
1433
+ return '[Error] Content policy refusal (content_filter).';
1434
+ if (protocol === 'responses' && json.status && json.status !== 'completed') {
1435
+ const details = json.incomplete_details && typeof json.incomplete_details === 'object' ? json.incomplete_details : {};
1436
+ return `[LLM Error] Responses response was ${json.status}${details.reason ? ': ' + details.reason : ''}.`;
1437
+ }
1438
+ return '';
1566
1439
  }
1567
1440
  /**
1568
1441
  * Deterministic capability probe that sends the schema through the native
@@ -1584,7 +1457,7 @@ class LLMProvider {
1584
1457
  };
1585
1458
  if (system)
1586
1459
  body.system = system;
1587
- const response = await this.postJsonWithFetchFallback(`${this.cleanBaseUrl()}/messages`, this.anthropicHeaders(), body, 120000, signal);
1460
+ const response = await this.postJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/messages'), this.anthropicHeaders(), body, 120000, signal);
1588
1461
  if (!response.ok)
1589
1462
  throw new Error(this.llmErrorText(response, await response.text()));
1590
1463
  const json = await response.json();
@@ -1598,7 +1471,7 @@ class LLMProvider {
1598
1471
  body.text = {
1599
1472
  format: { type: 'json_schema', name: schemaName, strict: true, schema },
1600
1473
  };
1601
- const response = await this.postJsonWithFetchFallback(`${this.cleanBaseUrl()}/responses`, this.openAIHeaders(), body, 120000, signal);
1474
+ const response = await this.postJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/responses'), this.openAIHeaders(), body, 120000, signal);
1602
1475
  if (!response.ok)
1603
1476
  throw new Error(this.llmErrorText(response, await response.text()));
1604
1477
  return this.extractResponsesText(this.normalizeResponsesPayload(await response.json()));
@@ -1606,7 +1479,7 @@ class LLMProvider {
1606
1479
  const isGitHubModels = this.protocol() === 'github_models';
1607
1480
  const url = isGitHubModels
1608
1481
  ? this.githubModelsUrl('/inference/chat/completions')
1609
- : `${this.cleanBaseUrl()}/chat/completions`;
1482
+ : (0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/chat/completions');
1610
1483
  const body = {
1611
1484
  model,
1612
1485
  messages: [
@@ -1644,8 +1517,8 @@ class LLMProvider {
1644
1517
  let family;
1645
1518
  if (this.protocol() === 'anthropic') {
1646
1519
  const prepared = this.anthropicMessages(messages, systemPrompt);
1647
- url = `${this.cleanBaseUrl()}/messages`;
1648
- headers = this.anthropicHeaders();
1520
+ url = (0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/messages');
1521
+ headers = this.anthropicHeaders(true);
1649
1522
  body = {
1650
1523
  model,
1651
1524
  messages: prepared.messages,
@@ -1658,8 +1531,8 @@ class LLMProvider {
1658
1531
  family = 'anthropic';
1659
1532
  }
1660
1533
  else if (this.protocol() !== 'github_models' && this.openAITransportMode() === 'responses') {
1661
- url = `${this.cleanBaseUrl()}/responses`;
1662
- headers = this.openAIHeaders();
1534
+ url = (0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/responses');
1535
+ headers = this.openAIHeaders(true);
1663
1536
  body = { ...this.responsesBody(model, messages, systemPrompt, temperature, maxTokens), stream: true };
1664
1537
  family = 'openai_responses';
1665
1538
  }
@@ -1667,8 +1540,8 @@ class LLMProvider {
1667
1540
  const isGitHubModels = this.protocol() === 'github_models';
1668
1541
  url = isGitHubModels
1669
1542
  ? this.githubModelsUrl('/inference/chat/completions')
1670
- : `${this.cleanBaseUrl()}/chat/completions`;
1671
- headers = isGitHubModels ? this.githubModelsHeaders() : this.openAIHeaders();
1543
+ : (0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/chat/completions');
1544
+ headers = isGitHubModels ? this.githubModelsHeaders(true) : this.openAIHeaders(true);
1672
1545
  body = {
1673
1546
  model,
1674
1547
  messages: [
@@ -1736,7 +1609,7 @@ class LLMProvider {
1736
1609
  signal?.removeEventListener('abort', forwardAbort);
1737
1610
  }
1738
1611
  }
1739
- async chat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier) {
1612
+ async chat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier, onUsage) {
1740
1613
  if (this.protocol() === 'anthropic') {
1741
1614
  const { system, messages: anthropicMessages } = this.anthropicMessages(messages, systemPrompt);
1742
1615
  const body = {
@@ -1747,11 +1620,15 @@ class LLMProvider {
1747
1620
  };
1748
1621
  if (system)
1749
1622
  body.system = system;
1750
- const response = await this.postJsonWithFetchFallback(`${this.cleanBaseUrl()}/messages`, this.anthropicHeaders(), body, 120000, signal);
1623
+ const response = await this.postJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/messages'), this.anthropicHeaders(), body, 120000, signal);
1751
1624
  if (!response.ok) {
1752
1625
  throw new Error(this.llmErrorText(response, await response.text()));
1753
1626
  }
1754
1627
  const json = await response.json();
1628
+ onUsage?.((0, agentKernelDiagnostics_1.extractProviderUsage)(json, { protocol: 'anthropic' }));
1629
+ const failure = this.nativeJsonCompletionError(json, 'anthropic');
1630
+ if (failure)
1631
+ throw new Error(failure);
1755
1632
  return (json.content || [])
1756
1633
  .filter(block => block.type === 'text' && block.text)
1757
1634
  .map(block => this.extractTextValue(block.text))
@@ -1760,7 +1637,7 @@ class LLMProvider {
1760
1637
  const isGitHubModels = this.protocol() === 'github_models';
1761
1638
  const url = isGitHubModels
1762
1639
  ? this.githubModelsUrl('/inference/chat/completions')
1763
- : `${this.cleanBaseUrl()}/chat/completions`;
1640
+ : (0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/chat/completions');
1764
1641
  const body = {
1765
1642
  model,
1766
1643
  messages: [
@@ -1773,17 +1650,21 @@ class LLMProvider {
1773
1650
  if (!isGitHubModels)
1774
1651
  this.applyChatReasoningEffort(body, model, reasoningTier);
1775
1652
  if (!isGitHubModels && this.openAITransportMode() === 'responses') {
1776
- return await this.openAIResponsesChat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier);
1653
+ return await this.openAIResponsesChat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier, onUsage);
1777
1654
  }
1778
1655
  const response = await this.postJsonWithFetchFallback(url, this.openAIHeaders(), body, 120000, signal);
1779
1656
  if (!response.ok) {
1780
1657
  const err = await response.text();
1781
1658
  if (!isGitHubModels && this.shouldUseResponsesFallback(response.status, err)) {
1782
- return await this.openAIResponsesChat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier);
1659
+ return await this.openAIResponsesChat(model, messages, systemPrompt, temperature, maxTokens, signal, reasoningTier, onUsage);
1783
1660
  }
1784
1661
  throw new Error(this.llmErrorText(response, err));
1785
1662
  }
1786
1663
  const json = await response.json();
1664
+ onUsage?.((0, agentKernelDiagnostics_1.extractProviderUsage)(json));
1665
+ const failure = this.nativeJsonCompletionError(json, 'chat');
1666
+ if (failure)
1667
+ throw new Error(failure);
1787
1668
  return this.extractChatCompletionText(json);
1788
1669
  }
1789
1670
  async modelCatalog() {
@@ -1801,7 +1682,7 @@ class LLMProvider {
1801
1682
  raw: typeof entry === 'string' ? { id: entry } : entry,
1802
1683
  })).filter((entry, index, all) => !!entry.id && all.findIndex(candidate => candidate.id === entry.id) === index);
1803
1684
  }
1804
- const response = await this.getJsonWithFetchFallback(`${this.cleanBaseUrl()}/models`, this.protocol() === 'anthropic' ? this.anthropicHeaders() : { 'Authorization': `Bearer ${this.apiKey}` });
1685
+ const response = await this.getJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/models'), this.protocol() === 'anthropic' ? this.anthropicHeaders() : this.openAIHeaders());
1805
1686
  if (!response.ok) {
1806
1687
  throw new Error(`Model list error: ${response.status} ${await response.text()}`);
1807
1688
  }
@@ -1849,7 +1730,7 @@ class LLMProvider {
1849
1730
  return { ok: false, latency: 0 };
1850
1731
  const start = Date.now();
1851
1732
  try {
1852
- const response = await this.postJsonWithFetchFallback(`${this.cleanBaseUrl()}/images/generations`, this.openAIHeaders(), {
1733
+ const response = await this.postJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/images/generations'), this.openAIHeaders(), {
1853
1734
  model,
1854
1735
  prompt: 'A single solid blue square on a white background.',
1855
1736
  size: '256x256',
@@ -1870,7 +1751,7 @@ class LLMProvider {
1870
1751
  async generateImage(model, prompt, size = '1024x1024', signal) {
1871
1752
  if (this.protocol() !== 'openai')
1872
1753
  throw new Error('Image generation requires an OpenAI-compatible provider.');
1873
- const response = await this.postJsonWithFetchFallback(`${this.cleanBaseUrl()}/images/generations`, this.openAIHeaders(), {
1754
+ const response = await this.postJsonWithFetchFallback((0, provider_request_compat_1.providerEndpoint)(this.cleanBaseUrl(), '/images/generations'), this.openAIHeaders(), {
1874
1755
  model,
1875
1756
  prompt,
1876
1757
  size,