newmark-agent 0.5.14 → 0.5.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/dist/conversation-utility-host.bundle.cjs +3373 -1900
  2. package/dist/conversation-utility-host.js +1 -1
  3. package/dist/core/agent.d.ts +105 -14
  4. package/dist/core/agent.js +658 -104
  5. package/dist/core/agentKernel/agent.d.ts +1 -0
  6. package/dist/core/agentKernel/agent.js +9 -0
  7. package/dist/core/agentKernelDiagnostics.d.ts +4 -6
  8. package/dist/core/agentKernelDiagnostics.js +12 -9
  9. package/dist/core/agentKernelRunner.d.ts +5 -0
  10. package/dist/core/agentKernelRunner.js +236 -85
  11. package/dist/core/autoRouter.js +16 -3
  12. package/dist/core/conversationCommandState.d.ts +17 -0
  13. package/dist/core/conversationCommandState.js +140 -0
  14. package/dist/core/conversationKernel.d.ts +25 -4
  15. package/dist/core/conversationKernel.js +347 -83
  16. package/dist/core/conversationListEvent.d.ts +6 -0
  17. package/dist/core/conversationListEvent.js +14 -0
  18. package/dist/core/electronUtilityAgentClient.d.ts +4 -1
  19. package/dist/core/electronUtilityAgentClient.js +4 -4
  20. package/dist/core/electronUtilityRuntimePool.d.ts +8 -2
  21. package/dist/core/electronUtilityRuntimePool.js +11 -5
  22. package/dist/core/installUpdate.js +41 -29
  23. package/dist/core/providerUsageAccounting.d.ts +47 -0
  24. package/dist/core/providerUsageAccounting.js +71 -0
  25. package/dist/core/requestContextEstimate.d.ts +18 -0
  26. package/dist/core/requestContextEstimate.js +54 -0
  27. package/dist/core/subagent.d.ts +89 -5
  28. package/dist/core/subagent.js +264 -41
  29. package/dist/core/subagentCommunication.d.ts +43 -0
  30. package/dist/core/subagentCommunication.js +167 -0
  31. package/dist/core/types.d.ts +32 -1
  32. package/dist/core/utilityAgentProtocol.d.ts +4 -0
  33. package/dist/core/workEventCoalescer.js +4 -1
  34. package/dist/core/wslAgentClient.d.ts +18 -5
  35. package/dist/core/wslAgentClient.js +79 -27
  36. package/dist/core/wslAgentProtocol.d.ts +7 -0
  37. package/dist/core/wslAgentRuntimePool.d.ts +8 -2
  38. package/dist/core/wslAgentRuntimePool.js +19 -6
  39. package/dist/core/wslRuntimeProcessTree.d.ts +10 -0
  40. package/dist/core/wslRuntimeProcessTree.js +104 -0
  41. package/dist/llm/provider.d.ts +5 -12
  42. package/dist/llm/provider.js +148 -267
  43. package/dist/main.js +582 -356
  44. package/dist/preload.js +3 -3
  45. package/dist/providers/chat-completions.adapter.d.ts +2 -0
  46. package/dist/providers/chat-completions.adapter.js +84 -36
  47. package/dist/providers/provider-adapter.d.ts +2 -0
  48. package/dist/providers/provider-events.d.ts +23 -14
  49. package/dist/providers/provider-events.js +148 -43
  50. package/dist/providers/provider-headers.d.ts +7 -3
  51. package/dist/providers/provider-headers.js +109 -39
  52. package/dist/providers/provider-request-compat.d.ts +6 -0
  53. package/dist/providers/provider-request-compat.js +37 -0
  54. package/dist/providers/responses.adapter.d.ts +2 -0
  55. package/dist/providers/responses.adapter.js +192 -126
  56. package/dist/server.d.ts +9 -2
  57. package/dist/server.js +100 -24
  58. package/dist/tools/index.js +17 -1
  59. package/dist/ui/index.html +2374 -616
  60. package/dist/ui/lucide-sprite.svg +7 -0
  61. package/dist/ui/startup.html +6 -6
  62. package/dist/wsl-agent-host.bundle.cjs +3381 -1906
  63. package/dist/wsl-agent-host.js +9 -4
  64. package/package.json +23 -4
@@ -42,6 +42,7 @@ exports.toolResultObjectiveOutcome = toolResultObjectiveOutcome;
42
42
  const fs = __importStar(require("fs"));
43
43
  const path = __importStar(require("path"));
44
44
  const terminalTakeover_1 = require("../tools/terminalTakeover");
45
+ const autoRouter_1 = require("./autoRouter");
45
46
  const toolPolicy_1 = require("./toolPolicy");
46
47
  const performanceDiagnostics_1 = require("./performanceDiagnostics");
47
48
  const agentKernelDiagnostics_1 = require("./agentKernelDiagnostics");
@@ -280,13 +281,17 @@ function throwIfKernelAborted(signal) {
280
281
  throw error;
281
282
  }
282
283
  async function runAgentKernel(agent) {
284
+ agent.acknowledgeSubagentSettlementReceipts();
283
285
  const stopContextTimer = (0, performanceDiagnostics_1.performanceTimer)('context_prepare', { conversationId: agent.activeConversationId });
284
286
  const processSignal = agent.activeProcessSignal();
285
287
  if (processSignal?.aborted) {
286
288
  stopContextTimer();
287
289
  throwIfKernelAborted(processSignal);
288
290
  }
289
- if (!agent.engineModel()) {
291
+ // One configuration-checked provider slot per Build: retain compatibility
292
+ // learning and connection pools without sharing them with another Build.
293
+ const buildProviderCache = {};
294
+ if (!agent.engineModel(buildProviderCache)) {
290
295
  const message = 'No LLM configured. Add provider in Settings > Models.';
291
296
  agent.status = 'error';
292
297
  agent.saveWorkspaceConversationState();
@@ -332,15 +337,25 @@ async function runAgentKernel(agent) {
332
337
  const initialToolSurface = refreshToolSurface(true);
333
338
  const assembledContext = agent.assembleContextV2(initialToolSurface.systemPromptNotice);
334
339
  const systemPrompt = assembledContext.text;
340
+ const peerCacheIdentity = agent.peerRequestCacheIdentity(systemPrompt, agent.cachedToolDefinitions());
341
+ const peerRequestCache = agent.readPeerRequestCache(peerCacheIdentity);
342
+ if (peerRequestCache)
343
+ toolProvisioning.restore(peerRequestCache.initialTools, peerRequestCache.provisionedTools);
335
344
  throwIfKernelAborted(processSignal);
336
- let providerRequestCount = 0;
337
- let bootstrappedCompressionAt = agent.lastCompression?.at || '';
345
+ // Capture request-only metadata once for this Build. Removing it after the
346
+ // first tool call rewrites the system prefix before all retained messages.
347
+ let buildTaskFocusSnapshot = peerRequestCache?.taskFocus;
348
+ let peerInputCheckpointed = false;
349
+ let lastRequestReplaySafe = true;
350
+ // A completed useful provider turn starts a new request recovery budget.
351
+ // Empty/thinking-only replies and failures never reset that budget.
352
+ let requestProgressGeneration = 0;
338
353
  stopContextTimer();
339
354
  const kernel = new NativeAgent({
340
355
  streamFn: streamWithNewmarkProvider(agent, KernelStreamCompat),
341
356
  toolExecution: 'parallel',
342
357
  convertToLlm: (messages) => messages.filter(m => m.role === 'user' || m.role === 'assistant' || m.role === 'toolResult'),
343
- transformContext: async (messages, signal) => transformContext(agent, messages, signal),
358
+ transformContext: async (messages, signal) => transformContext(agent, messages, signal, buildProviderCache),
344
359
  resolveTools: () => {
345
360
  const definitions = refreshToolSurface().definitions;
346
361
  prepareAssistantToolVisibility(agent, definitions);
@@ -350,9 +365,10 @@ async function runAgentKernel(agent) {
350
365
  });
351
366
  kernel.state.systemPrompt = systemPrompt;
352
367
  kernel.state.model = toKernelModel(agent);
353
- kernel.state.tools = toKernelTools(agent, initialToolSurface.definitions, toolProvisioning);
368
+ kernel.state.tools = toKernelTools(agent, refreshToolSurface().definitions, toolProvisioning);
354
369
  kernel.state.messages = toKernelMessages(agent);
355
370
  agent.attachAgentKernelRuntime(kernel);
371
+ agent.flushDirectRootInbox();
356
372
  let detachProcessAbort = () => { };
357
373
  if (processSignal) {
358
374
  const abortKernel = () => kernel.abort();
@@ -381,6 +397,13 @@ async function runAgentKernel(agent) {
381
397
  let observedActivity = false;
382
398
  let observedThought = false;
383
399
  const unsubscribe = kernel.subscribe(async (event) => {
400
+ if (event.type === 'message_start' && event.message.role === 'assistant') {
401
+ // A single kernel.prompt can execute several tools before its final
402
+ // provider reply. Earlier tool progress cannot hide a later empty one.
403
+ observedActivity = false;
404
+ observedThought = false;
405
+ lastAssistant = null;
406
+ }
384
407
  await handleKernelEvent(agent, event, tokens);
385
408
  if (event.type === 'message_update') {
386
409
  const delta = event.assistantMessageEvent;
@@ -412,15 +435,17 @@ async function runAgentKernel(agent) {
412
435
  const assistant = lastAssistant;
413
436
  const text = assistant ? KernelMessageText(assistant) : '';
414
437
  const hasToolCall = !!assistant?.content?.some(content => content.type === 'toolCall');
438
+ const usableText = !!text.trim() && !agent.isLlmErrorText(text);
415
439
  const emptyResponse = !assistant
416
440
  || (!text.trim() && !hasToolCall && !observedActivity && String(assistant?.stopReason || '') !== 'aborted');
417
441
  return {
418
442
  text: emptyResponse ? '[Error] Provider returned an empty response.' : text,
419
443
  stopReason: String(assistant?.stopReason || ''),
420
444
  errorMessage: String(assistant?.errorMessage || (emptyResponse ? 'Provider returned an empty response.' : '')),
421
- activity: observedActivity || !!text.trim() || hasToolCall,
445
+ activity: observedActivity || usableText || hasToolCall,
422
446
  thoughtOnly: observedThought && !text.trim() && !hasToolCall
423
447
  && !['error', 'aborted'].includes(String(assistant?.stopReason || '')),
448
+ requestReplaySafe: lastRequestReplaySafe,
424
449
  };
425
450
  }
426
451
  finally {
@@ -475,29 +500,84 @@ async function runAgentKernel(agent) {
475
500
  });
476
501
  }
477
502
  let consecutiveEmptyResponses = 0;
503
+ let noProgressGeneration = requestProgressGeneration;
504
+ let routeRetries = 0;
505
+ const transientRetries = new Set();
478
506
  for (;;) {
479
- const emptyResponseState = (0, emptyResponseRetry_1.observeEmptyResponseOutcome)(consecutiveEmptyResponses, providerTurnIsEmpty(lastTurn));
480
- consecutiveEmptyResponses = emptyResponseState.consecutiveEmptyResponses;
481
- if (lastTurn.thoughtOnly) {
482
- removeTrailingThoughtOnlyAssistant(kernel.state.messages);
507
+ throwIfKernelAborted(processSignal);
508
+ if (noProgressGeneration !== requestProgressGeneration) {
509
+ consecutiveEmptyResponses = 0;
510
+ noProgressGeneration = requestProgressGeneration;
511
+ }
512
+ // Every outcome, including one produced by a route retry, passes through
513
+ // the same bounded no-progress recovery. Alternating empty/thought-only
514
+ // responses cannot reset the five-retry allowance.
515
+ if (providerTurnIsEmpty(lastTurn) || lastTurn.thoughtOnly) {
516
+ const emptyResponseState = (0, emptyResponseRetry_1.observeEmptyResponseOutcome)(consecutiveEmptyResponses, true);
517
+ consecutiveEmptyResponses = emptyResponseState.consecutiveEmptyResponses;
518
+ if (!emptyResponseState.retry) {
519
+ throw new ProviderRunError(`Provider returned an empty response or only hidden reasoning for ${consecutiveEmptyResponses} consecutive requests; the recovery limit was reached.`);
520
+ }
521
+ if (lastTurn.thoughtOnly)
522
+ removeTrailingThoughtOnlyAssistant(kernel.state.messages);
523
+ else
524
+ removeTrailingFailedAssistant(agent, kernel.state.messages);
525
+ const retryNumber = consecutiveEmptyResponses;
526
+ const reason = lastTurn.thoughtOnly ? 'only hidden reasoning without an answer or tool call' : 'an empty response';
527
+ const notice = `[Model retry] Provider returned ${reason}; retrying the same deployment (${retryNumber}/${emptyResponseRetry_1.MAX_EMPTY_RESPONSE_RETRIES}) after ${(0, emptyResponseRetry_1.emptyResponseRetryDelayMs)(consecutiveEmptyResponses)}ms.`;
528
+ tokens.push({ type: 'text', text: notice });
529
+ agent.recordWorkStatus(notice);
530
+ await agent.waitForPlannedRouteRetry((0, emptyResponseRetry_1.emptyResponseRetryDelayMs)(consecutiveEmptyResponses));
483
531
  lastTurn = await runWithCompressionResume([], false);
484
532
  continue;
485
533
  }
486
- if (!emptyResponseState.retry)
534
+ if (!kernelTurnFailed(agent, lastTurn) || lastTurn.requestReplaySafe === false)
487
535
  break;
488
- removeTrailingFailedAssistant(agent, kernel.state.messages);
489
- const retryNumber = consecutiveEmptyResponses;
490
- const notice = `[Model retry] Provider returned an empty response; retrying the same deployment (${retryNumber}/${emptyResponseRetry_1.MAX_EMPTY_RESPONSE_RETRIES}) after ${(0, emptyResponseRetry_1.emptyResponseRetryDelayMs)(consecutiveEmptyResponses)}ms.`;
491
- tokens.push({ type: 'text', text: notice });
492
- agent.recordWorkStatus(notice);
493
- await agent.waitForPlannedRouteRetry((0, emptyResponseRetry_1.emptyResponseRetryDelayMs)(consecutiveEmptyResponses));
494
- lastTurn = await runWithCompressionResume([], false);
495
- }
496
- let routeRetries = 0;
497
- while (kernelTurnFailed(agent, lastTurn) && routeRetries < 2) {
498
- const previous = agent.switchToFallbackModel(lastTurn.errorMessage || lastTurn.text);
499
- if (!previous)
536
+ const failureText = lastTurn.errorMessage || lastTurn.text;
537
+ const failure = (0, autoRouter_1.classifyRouteFailure)(failureText);
538
+ const deployment = agent.activeDeployment();
539
+ const retryKey = `${requestProgressGeneration}:${deployment?.providerId || ''}:${deployment?.modelId || agent.activeModelName()}`;
540
+ // Retry-After applies to 503/408 as well as 429; the routing classifier
541
+ // currently retains it only for rate limits. Never shorten the server's
542
+ // requested wait to fit our automatic retry budget.
543
+ const retryAfter = failureText.match(/retry[- ]after\s*[:=]?\s*(\d+(?:\.\d+)?)\s*(ms|s|seconds?)?/i);
544
+ const retryAfterMs = failure.retryAfterMs ?? (retryAfter
545
+ ? Number(retryAfter[1]) * ((retryAfter[2] || 's').toLowerCase() === 'ms' ? 1 : 1000) : undefined);
546
+ const retryDelayMs = retryAfterMs ?? 250;
547
+ const retryBudgetMs = Math.max(0, Math.min(5000, agent.lastRouteDecision?.retryBudgetMs ?? 5000));
548
+ const transientFailure = failure.retryable && ['transport', 'timeout', 'server_error', 'rate_limited'].includes(failure.type);
549
+ if (transientFailure && retryAfterMs !== undefined && retryAfterMs > retryBudgetMs)
550
+ break;
551
+ const safeTransient = transientFailure
552
+ && lastTurn.requestReplaySafe === true && retryDelayMs <= retryBudgetMs;
553
+ const retryCurrentRequest = async () => {
554
+ transientRetries.add(retryKey);
555
+ removeTrailingFailedAssistant(agent, kernel.state.messages);
556
+ const notice = `[Model retry] Temporary provider failure; retrying the current request once on the same deployment after ${retryDelayMs}ms.`;
557
+ tokens.push({ type: 'text', text: notice });
558
+ agent.recordWorkStatus(notice);
559
+ await agent.waitForPlannedRouteRetry(retryDelayMs);
560
+ lastTurn = await runWithCompressionResume([], false);
561
+ };
562
+ // Fixed selections have no Auto retry planner. A completed tool from an
563
+ // earlier subturn does not make the *next*, still-uncommitted request
564
+ // unsafe to retry; its exact existing messages/results remain in place.
565
+ if (agent.model !== 'auto' && safeTransient && !transientRetries.has(retryKey)) {
566
+ await retryCurrentRequest();
567
+ continue;
568
+ }
569
+ const previous = routeRetries < 2 ? agent.switchToFallbackModel(failureText) : null;
570
+ if (!previous) {
571
+ if (agent.model === 'auto' && safeTransient && !transientRetries.has(retryKey)) {
572
+ await retryCurrentRequest();
573
+ continue;
574
+ }
500
575
  break;
576
+ }
577
+ // The existing Auto same-deployment attempt consumes the same allowance;
578
+ // our local recovery must not add a second retry after it is exhausted.
579
+ if (agent.routeTransitionKind() === 'retry_same_deployment')
580
+ transientRetries.add(retryKey);
501
581
  removeTrailingFailedAssistant(agent, kernel.state.messages);
502
582
  routeRetries += 1;
503
583
  const notice = routeTransitionNotice(agent, previous);
@@ -509,8 +589,14 @@ async function runAgentKernel(agent) {
509
589
  fallback: { from: previous, to: agent.model, providerId: agent.activeDeployment()?.providerId },
510
590
  });
511
591
  kernel.state.model = toKernelModel(agent);
512
- const fallbackToolSurface = refreshToolSurface(true);
513
- kernel.state.systemPrompt = [agent.buildSystemPrompt(), fallbackToolSurface.systemPromptNotice].filter(Boolean).join('\n\n');
592
+ const toolSurfaceChanged = toolSurfaceIdentityForAgent(agent) !== activeToolSurfaceIdentity;
593
+ const fallbackToolSurface = refreshToolSurface();
594
+ // A same-deployment retry must retain the original Build system and
595
+ // provisioned schema sequence, even if a Guide changed the latest task.
596
+ // Actual deployment or capability transitions still refresh the surface.
597
+ if (toolSurfaceChanged) {
598
+ kernel.state.systemPrompt = [agent.buildSystemPrompt(), fallbackToolSurface.systemPromptNotice].filter(Boolean).join('\n\n');
599
+ }
514
600
  kernel.state.tools = toKernelTools(agent, fallbackToolSurface.definitions, toolProvisioning);
515
601
  await agent.waitForPlannedRouteRetry();
516
602
  lastTurn = await runWithCompressionResume([], false);
@@ -569,41 +655,53 @@ async function runAgentKernel(agent) {
569
655
  const tools = context.tools || [];
570
656
  const brokerOnlySurface = tools.length > 0 && tools.every(tool => tool.name === TOOL_PROVISION_NAME || tool.name === 'skill' || ALWAYS_AVAILABLE_AGENT_TOOL_NAMES.has(tool.name));
571
657
  currentAgent.beginRouteAttempt();
658
+ lastRequestReplaySafe = true;
572
659
  try {
573
- const currentProvider = currentAgent.engineModel();
660
+ const currentProvider = currentAgent.engineModel(buildProviderCache);
574
661
  const currentModelName = currentAgent.activeModelName();
575
662
  if (!currentProvider || !currentModelName)
576
663
  throw new Error('No resolved model deployment is available.');
664
+ if (!peerInputCheckpointed) {
665
+ currentAgent.checkpointPeerInput();
666
+ peerInputCheckpointed = true;
667
+ }
577
668
  const { temperature, maxTokens, reasoningEffort } = currentProvider.intelligenceConfig(currentAgent.intelligence);
578
- const newmarkMessages = fromKernelMessages(context.messages);
579
- const currentCompressionAt = currentAgent.lastCompression?.at || '';
580
- const compressionCompleted = !!currentCompressionAt && currentCompressionAt !== bootstrappedCompressionAt;
581
- const includeBootstrap = providerRequestCount === 0 || compressionCompleted;
582
- // Keep the stable base prompt identical across tool sub-turns. The
583
- // request-scoped ledger/bootstrap is needed on the first request
584
- // (and once after compression), but re-injecting it on every round
585
- // makes otherwise cacheable prompt prefixes look like new prompts.
669
+ const newmarkMessages = fromKernelMessages(context.messages).map(message => message.role === 'assistant'
670
+ // Use the same public-content boundary on its first provider
671
+ // submission and after persistence. Trimming only the durable
672
+ // copy changes earlier tool-call envelopes on mailbox recovery.
673
+ ? { ...message, content: currentAgent.sanitizeAssistantOutput(String(message.content || '')) }
674
+ : message);
675
+ // Tool results and Guides append to messages; neither mutable task
676
+ // status nor a growing tool catalog may regenerate this snapshot.
677
+ // Compression replaces messages explicitly, not the stable system.
678
+ buildTaskFocusSnapshot ??= buildRequestTaskFocus(currentAgent, context.messages, {
679
+ activeTools: context.tools || [],
680
+ toolCatalog: currentAgent.cachedToolDefinitions(),
681
+ });
682
+ if (peerCacheIdentity)
683
+ currentAgent.persistPeerRequestCache({
684
+ version: 1, identity: peerCacheIdentity, taskFocus: buildTaskFocusSnapshot,
685
+ ...toolProvisioning.snapshot(),
686
+ });
586
687
  const requestSystemPrompt = [
587
688
  context.systemPrompt || '',
588
- includeBootstrap || compressionCompleted
589
- ? buildRequestTaskFocus(currentAgent, context.messages, {
590
- includeBootstrap,
591
- compressionCompleted,
592
- activeTools: context.tools || [],
593
- toolCatalog: currentAgent.cachedToolDefinitions(),
594
- })
595
- : '',
689
+ buildTaskFocusSnapshot,
596
690
  ].filter(Boolean).join('\n\n');
597
- providerRequestCount += 1;
598
- if (compressionCompleted)
599
- bootstrappedCompressionAt = currentCompressionAt;
600
- (0, agentKernelDiagnostics_1.emitRequestContextDiagnostic)({
601
- conversationId: currentAgent.activeConversationId,
602
- systemPrompt: requestSystemPrompt,
603
- messages: newmarkMessages,
604
- tools: context.tools || [],
605
- });
606
- for await (const token of currentProvider.chatStreamWithTools(currentModelName, newmarkMessages, requestSystemPrompt, temperature, maxTokens, toProviderToolDefinitions(context.tools || []), options?.signal, reasoningEffort, currentAgent.config.getBool('context', 'provider_session_id')
691
+ const providerTools = toProviderToolDefinitions(context.tools || []);
692
+ const usageRequest = currentAgent.beginProviderUsageRequest();
693
+ currentAgent.recordRequestContext(usageRequest, newmarkMessages, requestSystemPrompt, providerTools, currentModelName);
694
+ // Sorting/serializing the complete history is diagnostic work, not
695
+ // required request preparation. A sink can still opt in mid-Build.
696
+ if ((0, agentKernelDiagnostics_1.agentKernelDiagnosticsRequested)()) {
697
+ (0, agentKernelDiagnostics_1.emitRequestContextDiagnostic)({
698
+ conversationId: currentAgent.activeConversationId,
699
+ systemPrompt: requestSystemPrompt,
700
+ messages: newmarkMessages,
701
+ tools: context.tools || [],
702
+ });
703
+ }
704
+ for await (const token of currentProvider.chatStreamWithTools(currentModelName, newmarkMessages, requestSystemPrompt, temperature, maxTokens, providerTools, options?.signal, reasoningEffort, currentAgent.config.getBool('context', 'provider_session_id')
607
705
  ? currentAgent.activeConversationId
608
706
  : undefined)) {
609
707
  if (!firstTokenRecorded && ((token.type === 'text' && token.text) || (token.type === 'tool_call' && token.toolCall))) {
@@ -612,15 +710,8 @@ async function runAgentKernel(agent) {
612
710
  }
613
711
  if (process.env.NEWMARK_PROVIDER_DIAGNOSTICS === '1')
614
712
  console.error(`[NewmarkKernel] provider-token type=${token.type}`);
615
- if (options?.signal?.aborted)
616
- break;
617
713
  if (token.type === 'usage' && token.usage) {
618
- currentAgent.recordProviderUsage({
619
- input: token.usage.input,
620
- output: token.usage.output,
621
- cacheRead: token.usage.cacheRead,
622
- cacheWrite: token.usage.cacheWrite,
623
- });
714
+ currentAgent.recordProviderUsage(token.usage, usageRequest);
624
715
  (0, agentKernelDiagnostics_1.emitProviderUsageDiagnostic)({
625
716
  conversationId: currentAgent.activeConversationId,
626
717
  inputTokens: token.usage.input,
@@ -628,8 +719,15 @@ async function runAgentKernel(agent) {
628
719
  cacheReadTokens: token.usage.cacheRead,
629
720
  cacheWriteTokens: token.usage.cacheWrite,
630
721
  });
722
+ if (options?.signal?.aborted)
723
+ break;
631
724
  continue;
632
725
  }
726
+ if (options?.signal?.aborted)
727
+ break;
728
+ if ((token.type === 'text' && token.text && !currentAgent.isLlmErrorText(token.text))
729
+ || (token.type === 'tool_call' && token.toolCall))
730
+ lastRequestReplaySafe = false;
633
731
  if (token.reasoningContent) {
634
732
  const delta = token.reasoningContent.slice(thinking.length);
635
733
  thinking = token.reasoningContent;
@@ -708,6 +806,8 @@ async function runAgentKernel(agent) {
708
806
  }
709
807
  const final = assistantMessage(model, finalContent, finalContent.some(c => c.type === 'toolCall') ? 'toolUse' : 'stop');
710
808
  if (!currentAgent.isLlmErrorText(text)) {
809
+ if (finalContent.some(content => content.type === 'toolCall' || (content.type === 'text' && typeof content.text === 'string' && content.text.trim())))
810
+ requestProgressGeneration++;
711
811
  if (brokerOnlySurface && !finalContent.some(content => content.type === 'toolCall') && text)
712
812
  currentAgent.markRouteStreamCommitted();
713
813
  const durationMs = Math.max(1, Date.now() - requestStartedAt);
@@ -738,7 +838,7 @@ async function runAgentKernel(agent) {
738
838
  };
739
839
  }
740
840
  }
741
- async function transformContext(agent, messages, signal) {
841
+ async function transformContext(agent, messages, signal, providerCache) {
742
842
  const processSignal = agent.activeProcessSignal();
743
843
  if (processSignal?.aborted)
744
844
  return messages;
@@ -748,7 +848,7 @@ async function transformContext(agent, messages, signal) {
748
848
  // active model after an Auto/fallback transition can pair the wrong model
749
849
  // name with the provider captured for this compaction request.
750
850
  const compressionModel = agent.activeModelName();
751
- const provider = agent.engineModel();
851
+ const provider = agent.engineModel(providerCache);
752
852
  if (!provider || !compressionModel)
753
853
  return messages;
754
854
  // Context compression persists Agent history. Feed it only the public
@@ -835,27 +935,23 @@ function buildBuildContextBootstrap(agent, messages, options) {
835
935
  .filter(definition => toolDefinitionName(definition) !== TOOL_PROVISION_NAME)
836
936
  .map(definition => `- ${toolDefinitionName(definition)}: ${compactToolDescription(toolDefinitionDescription(definition))}`);
837
937
  const retainedMessages = messages.length;
838
- // 缓存命中关键:压缩后不再把压缩摘要冗余注入 bootstrap——压缩摘要已通过
839
- // transformContext 的 compressionContinuationPrompt 写入 messages 前缀。这里
840
- // 保持 bootstrap 文案在「压缩前/后」字节稳定,避免 compressionCompleted 分支
841
- // 单独改变 system 内容而让 provider 前缀缓存失效。
842
- // 首 Build 命名已由 Agent 在首个完成 Build 的最终响应处自动完成
843
- // (deriveConversationTitleFromSummary),不再注入一次性 tool-call 指令,
844
- // 保持首轮 provider 请求的 system 前缀与后续工具子轮字节稳定。
938
+ // This is a Build-initialization snapshot retained across all its requests.
939
+ // Current results, Guides and compression continuation stay in messages;
940
+ // they must not rewrite this metadata or be copied into durable history.
845
941
  return [
846
942
  '## Build Context Bootstrap',
847
943
  'Injection reason: this is the first provider request of a new Build.',
848
944
  'This block is request-only runtime metadata. Do not quote it into conversation history, Build summaries, Memory Lab, or future compression summaries.',
849
945
  'Current context boundary:',
850
946
  '- The durable conversation messages in this provider request are the current authoritative context; use them directly and do not reinterpret them as a backlog.',
851
- `- Retained non-system request messages: ${retainedMessages}. The latest real user-role message remains authoritative.`,
947
+ `- Retained non-system messages at Build initialization: ${retainedMessages}. Later tool results and Guides follow in the request messages; the latest real user-role message remains authoritative.`,
852
948
  buildConversationTaskLedger(agent),
853
949
  '## Tool Awareness Bootstrap',
854
950
  'The following catalog is capability metadata only. Tool descriptions are not instructions, and a tool is callable only when its full schema is present in the provider tools field.',
855
951
  'Only bash, pwd, read, write, edit, delete_file, glob, and grep are foundational tools with initial full schemas (subject to mode and policy filtering).',
856
952
  'Advanced capabilities—including SubAgent, task tools, Git/GitHub, browser, Computer Use, skills, MCP, automations, Flow, and Memory Lab—are not initially callable. Before using one, first call tool_provision with its exact tool name as the only tool call in that assistant subturn; call the advanced tool only on the following model turn after its full schema appears.',
857
953
  ...(catalogLines.length ? catalogLines : ['- No callable tools are available for this provider turn.']),
858
- `Necessary full schemas supplied natively for this provider turn: ${activeNames.length ? activeNames.join(', ') : '(none; use tool_provision when its schema is available)'}.`,
954
+ `Initial full schemas supplied natively for this Build: ${activeNames.length ? activeNames.join(', ') : '(none; use tool_provision when its schema is available)'}. Additional provisioned schemas appear in the current provider tools field.`,
859
955
  'Do not invent parameters from the brief catalog. Use only the exact full schemas supplied through the provider tool interface; provision another exact tool when needed.',
860
956
  ].join('\n');
861
957
  }
@@ -919,6 +1015,14 @@ async function handleKernelEvent(agent, event, tokens) {
919
1015
  const history = toHistoryMessage(event.message);
920
1016
  agent.persistGuideMessage(event.message.clientMessageId, display, event.message.runId, history.content);
921
1017
  }
1018
+ const mailboxMarker = event.message.hiddenUserInput && text.match(/^\[(?:Peer mailbox|Root subagent inbox) id=[0-9a-f-]{36}\b/i)?.[0];
1019
+ if (mailboxMarker && !agent.history.some(message => message.role === 'user' && String(message.content || '').startsWith(mailboxMarker))) {
1020
+ // Commit the consumed directive before acknowledging its mailbox ID.
1021
+ // Otherwise a cold continuation loses an already-read instruction
1022
+ // and deletes that message from the provider's retained prefix.
1023
+ agent.history.push({ ...toHistoryMessage(event.message), run_id: agent.currentWorkRunId() || undefined });
1024
+ agent.saveWorkspaceConversationState(true);
1025
+ }
922
1026
  agent.notifyAgentKernelUserMessageStart(text, event.message.clientMessageId);
923
1027
  }
924
1028
  else if (event.message.role === 'assistant') {
@@ -1010,9 +1114,10 @@ async function handleKernelEvent(agent, event, tokens) {
1010
1114
  const publicMessage = internalProvision ? { ...event.message, content: publicContent } : event.message;
1011
1115
  const text = agent.sanitizeAssistantOutput(KernelMessageText(publicMessage));
1012
1116
  const failed = event.message.stopReason === 'error' || agent.isLlmErrorText(text);
1013
- if (!failed)
1117
+ const aborted = event.message.stopReason === 'aborted';
1118
+ if (!failed && !aborted)
1014
1119
  emitBufferedAssistantText(agent, tokens);
1015
- if (text && !failed)
1120
+ if (text && !failed && !aborted)
1016
1121
  agent.emitWorkEvent({ type: realToolCalls.length ? 'response' : 'final_response', content: text });
1017
1122
  if ((text || realToolCalls.length) && event.message.stopReason !== 'aborted' && !failed) {
1018
1123
  if (text && !realToolCalls.length)
@@ -1021,6 +1126,7 @@ async function handleKernelEvent(agent, event, tokens) {
1021
1126
  // Keep tool-call metadata, but never replay a hidden-reasoning line
1022
1127
  // that was deliberately removed from the public completed message.
1023
1128
  historyMessage.content = text;
1129
+ historyMessage.run_id = agent.currentWorkRunId() || undefined;
1024
1130
  agent.history.push(historyMessage);
1025
1131
  agent.saveWorkspaceConversationState();
1026
1132
  }
@@ -1028,8 +1134,24 @@ async function handleKernelEvent(agent, event, tokens) {
1028
1134
  resetAssistantToolVisibility(agent);
1029
1135
  }
1030
1136
  else if (event.message.role === 'toolResult' && event.message.toolName !== TOOL_PROVISION_NAME) {
1031
- agent.history.push(toHistoryMessage(event.message));
1032
- agent.saveWorkspaceConversationState();
1137
+ const receipt = event.message.details?.settlementReceipt;
1138
+ const historyMessage = { ...toHistoryMessage(event.message), run_id: agent.currentWorkRunId() || undefined,
1139
+ ...(!event.message.isError && receipt ? { subagent_settlement_receipt: { ...receipt } } : {}),
1140
+ };
1141
+ agent.history.push(historyMessage);
1142
+ try {
1143
+ agent.saveWorkspaceConversationState();
1144
+ }
1145
+ catch (error) {
1146
+ // A later notification must not mistake this unsaved in-memory
1147
+ // receipt for a committed result after the persistence error.
1148
+ delete historyMessage.subagent_settlement_receipt;
1149
+ throw error;
1150
+ }
1151
+ // Retire redundant automatic wakes only after the full tool result and
1152
+ // its exact version receipt are durable. The receipt is never content.
1153
+ if (receipt && !event.message.isError)
1154
+ agent.acknowledgeSubagentSettlementReceipts();
1033
1155
  }
1034
1156
  break;
1035
1157
  }
@@ -1123,12 +1245,33 @@ class ToolProvisionSession {
1123
1245
  this.broker = this.brokerDefinition();
1124
1246
  }
1125
1247
  currentDefinitions() {
1126
- const active = new Set([...this.initialNames, ...this.provisionedNames]);
1127
1248
  return [
1128
- ...this.catalog.filter(definition => active.has(toolDefinitionName(definition))),
1249
+ ...this.catalog.filter(definition => this.initialNames.has(toolDefinitionName(definition))),
1129
1250
  this.broker,
1251
+ // Preserve the complete previously exposed schema sequence. Loading a
1252
+ // new tool appends its schema instead of inserting it before the broker
1253
+ // or reordering earlier provisions by catalog position.
1254
+ ...[...this.provisionedNames]
1255
+ .filter(name => !this.initialNames.has(name))
1256
+ .map(name => this.definitionsByName.get(name)),
1130
1257
  ];
1131
1258
  }
1259
+ snapshot() {
1260
+ return { initialTools: [...this.initialNames], provisionedTools: [...this.provisionedNames] };
1261
+ }
1262
+ restore(initial, provisioned) {
1263
+ // Reuse ordering only through the current policy-filtered catalog. A
1264
+ // persisted name is never authority to restore a revoked capability.
1265
+ this.initialNames.clear();
1266
+ for (const name of initial)
1267
+ if (this.definitionsByName.has(name))
1268
+ this.initialNames.add(name);
1269
+ this.provisionedNames.clear();
1270
+ for (const name of provisioned)
1271
+ if (this.definitionsByName.has(name))
1272
+ this.provisionedNames.add(name);
1273
+ this.broker = this.brokerDefinition();
1274
+ }
1132
1275
  metrics() {
1133
1276
  const active = this.currentDefinitions();
1134
1277
  return {
@@ -1356,7 +1499,9 @@ function routeToolSurfaceV2(agent, definitions, toolchain, task) {
1356
1499
  const plan = planner.plan({
1357
1500
  agentRunId: agent.runtimeActorId,
1358
1501
  buildBlockId: agent.activeConversationId || 'build',
1359
- userInput: task,
1502
+ // Mailbox deltas do not change the peer's assigned capability plan.
1503
+ // Keep this diagnostic fingerprint out of the changing request suffix.
1504
+ userInput: agent.isSubagentRuntime ? (agent.subagents.get(agent.runtimeActorId)?.prompt || task) : task,
1360
1505
  objective: '',
1361
1506
  previousToolCalls: [],
1362
1507
  toolUsageFrequency: new Map(),
@@ -1441,7 +1586,8 @@ function toKernelTools(agent, definitions, provisioning) {
1441
1586
  };
1442
1587
  }
1443
1588
  const args = JSON.stringify(params || {});
1444
- const rawText = await executeNewmarkTool(agent, name, args, fn.parameters, signal);
1589
+ let settlementReceipt;
1590
+ const rawText = await executeNewmarkTool(agent, name, args, fn.parameters, signal, receipt => { settlementReceipt = receipt; });
1445
1591
  if (signal?.aborted) {
1446
1592
  discardComputerUseVisionImage(name, rawText);
1447
1593
  throw abortError();
@@ -1474,7 +1620,9 @@ function toKernelTools(agent, definitions, provisioning) {
1474
1620
  }
1475
1621
  catch { }
1476
1622
  }
1477
- return { content, details: { tool: name, ok: true, terminate, ...(launchReceipt ? { launchReceipt } : {}), visionImagePath: visionImage.imagePath || undefined, ephemeralVisionImage: !!visionImage.image, capturedAttachmentId: capturedInput?.id, displayImage }, terminate };
1623
+ return { content, details: { tool: name, ok: true, terminate, ...(launchReceipt ? { launchReceipt } : {}),
1624
+ ...(settlementReceipt && text === rawText ? { settlementReceipt } : {}),
1625
+ visionImagePath: visionImage.imagePath || undefined, ephemeralVisionImage: !!visionImage.image, capturedAttachmentId: capturedInput?.id, displayImage }, terminate };
1478
1626
  },
1479
1627
  };
1480
1628
  }).filter((tool) => !!tool.name);
@@ -1627,7 +1775,7 @@ function visualFallbackImageInput(agent, name, text) {
1627
1775
  return {};
1628
1776
  }
1629
1777
  }
1630
- async function executeNewmarkTool(agent, name, args, inputSchema, signal) {
1778
+ async function executeNewmarkTool(agent, name, args, inputSchema, signal, onSettlementReceipt) {
1631
1779
  const stopToolTimer = (0, performanceDiagnostics_1.performanceTimer)('tool_execution', { conversationId: agent.activeConversationId, detail: { tool: name } });
1632
1780
  try {
1633
1781
  const wsDir = agent.workspace.current?.path || agent.rootPath;
@@ -1663,10 +1811,13 @@ async function executeNewmarkTool(agent, name, args, inputSchema, signal) {
1663
1811
  return (await agent.handleSubagentContinueEnvelope(args)).output;
1664
1812
  if (name === 'subagent_list')
1665
1813
  return agent.handleSubagentListEnvelope(args).output;
1666
- if (name === 'subagent_read')
1667
- return agent.handleSubagentReadEnvelope(args).output;
1668
- if (name === 'subagent_result')
1669
- return agent.handleSubagentResultEnvelope(args).output;
1814
+ if (name === 'subagent_read' || name === 'subagent_result') {
1815
+ const result = name === 'subagent_read' ? agent.handleSubagentReadEnvelope(args) : agent.handleSubagentResultEnvelope(args);
1816
+ const receipt = result.metadata?.settlementReceipt;
1817
+ if (result.ok && receipt)
1818
+ onSettlementReceipt?.(receipt);
1819
+ return result.output;
1820
+ }
1670
1821
  if (name === 'subagent_close')
1671
1822
  return agent.handleSubagentCloseEnvelope(args).output;
1672
1823
  if (name === 'branch_list')
@@ -93,6 +93,18 @@ function classifyTaskClasses(taskText, requiredCapabilities, estimatedInputToken
93
93
  }
94
94
  function classifyRouteFailure(error) {
95
95
  const text = error instanceof Error ? `${error.name} ${error.message}` : String(error || '');
96
+ // Gateways may return HTTP 200 followed by a structured provider error.
97
+ // Match actual code/type fields, never incidental overload words in prose.
98
+ let structuredCodes = [];
99
+ const jsonStart = text.indexOf('{');
100
+ if (jsonStart >= 0) {
101
+ try {
102
+ const payload = JSON.parse(text.slice(jsonStart));
103
+ const detail = payload?.error || payload?.response?.error || payload;
104
+ structuredCodes = [detail?.code, detail?.type].filter(value => typeof value === 'string').map(value => value.toLowerCase());
105
+ }
106
+ catch { /* Ordinary non-JSON errors retain their existing classification. */ }
107
+ }
96
108
  const statusMatch = text.match(/(?:http|error|status)?\s*[:=]?\s*(408|429|4\d\d|5\d\d)\b/i);
97
109
  const statusCode = statusMatch ? Number(statusMatch[1]) : undefined;
98
110
  const retryAfterMatch = text.match(/retry[- ]after\s*[:=]?\s*(\d+(?:\.\d+)?)\s*(ms|s|seconds?)?/i);
@@ -122,10 +134,11 @@ function classifyRouteFailure(error) {
122
134
  return { type: 'rate_limited', retryable: true, switchAllowed: true, statusCode: 429, retryAfterMs };
123
135
  }
124
136
  if (statusCode === 408 || /timeout|timed out|aborterror/i.test(text)) {
125
- return { type: 'timeout', retryable: true, switchAllowed: true, statusCode };
137
+ return { type: 'timeout', retryable: true, switchAllowed: true, statusCode, ...(retryAfterMs === undefined ? {} : { retryAfterMs }) };
126
138
  }
127
- if (statusCode !== undefined && statusCode >= 500) {
128
- return { type: 'server_error', retryable: true, switchAllowed: true, statusCode };
139
+ if ((statusCode !== undefined && statusCode >= 500)
140
+ || structuredCodes.some(code => ['server_is_overloaded', 'overloaded_error', 'service_unavailable_error', 'server_error', 'internal_server_error'].includes(code))) {
141
+ return { type: 'server_error', retryable: true, switchAllowed: true, statusCode, ...(retryAfterMs === undefined ? {} : { retryAfterMs }) };
129
142
  }
130
143
  if (/empty response|no response body|empty completion/i.test(text))
131
144
  return { type: 'empty_response', retryable: true, switchAllowed: true };
@@ -0,0 +1,17 @@
1
+ export interface ConversationCommandState {
2
+ mode?: 'build' | 'plan' | 'chat' | 'goal' | 'flow';
3
+ inputMode?: 'guide' | 'next';
4
+ queuePaused?: boolean;
5
+ }
6
+ /** GUI command owner only; execution and continuation payloads remain owned by their Agent. */
7
+ export declare class ConversationCommandStateStore {
8
+ private readonly file;
9
+ private records;
10
+ constructor(root: string);
11
+ get(key: string): ConversationCommandState | undefined;
12
+ set(key: string, patch: ConversationCommandState): ConversationCommandState;
13
+ delete(key: string): void;
14
+ private key;
15
+ private persist;
16
+ }
17
+ //# sourceMappingURL=conversationCommandState.d.ts.map