@yeaft/webchat-agent 1.0.505 → 1.0.507

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import { providerProjectionDigest, providerStateBytes, MAX_PROVIDER_STATE_BYTES, ProviderStateError } from './llm/provider-state.js';
1
2
  /**
2
3
  * history-window.js — deterministic history shaping for provider requests.
3
4
  *
@@ -157,7 +158,19 @@ export function estimateContentTokens(content) {
157
158
  export function estimateMessageTokens(message) {
158
159
  if (!message || typeof message !== 'object') return 0;
159
160
  let total = 2 + estimateContentTokens(message.content);
160
- total += estimateThinkingBlocksTokens(message.thinkingBlocks);
161
+ // Responses ciphertext has a byte budget, not a text-token cost. Anthropic
162
+ // thinking is plaintext and must be charged, without recounting the native
163
+ // text/tool items already represented by content/toolCalls below.
164
+ if (message.providerState && providerStateBytes(message.providerState) > MAX_PROVIDER_STATE_BYTES) {
165
+ throw new ProviderStateError('state exceeds byte budget');
166
+ }
167
+ if (message.providerState?.protocol === 'anthropic') {
168
+ const thinking = (message.providerState.items || [])
169
+ .filter(item => ['thinking', 'redacted_thinking'].includes(item?.type));
170
+ if (thinking.length > 0) total += estimateContentTokens(thinking);
171
+ } else if (!message.providerState) {
172
+ total += estimateThinkingBlocksTokens(message.thinkingBlocks);
173
+ }
161
174
  if (Array.isArray(message.toolCalls)) {
162
175
  for (const toolCall of message.toolCalls) {
163
176
  total += 4;
@@ -242,6 +255,7 @@ export function stripToolNoiseFromOlderTurns(messages, options = {}) {
242
255
  const next = { ...message };
243
256
  if (Array.isArray(next.toolCalls)) delete next.toolCalls;
244
257
  if (Array.isArray(next.thinkingBlocks)) delete next.thinkingBlocks;
258
+ delete next.providerState;
245
259
  if (Array.isArray(next.content)) next.content = stripToolContentParts(next.content);
246
260
  if (next.role === 'assistant' && !hasContentAfterToolStrip(next.content)) continue;
247
261
  if (next.role === 'user' && Array.isArray(next.content) && next.content.length === 0) continue;
@@ -257,6 +271,7 @@ function stripAllToolNoise(messages) {
257
271
  const next = { ...message };
258
272
  if (Array.isArray(next.toolCalls)) delete next.toolCalls;
259
273
  if (Array.isArray(next.thinkingBlocks)) delete next.thinkingBlocks;
274
+ delete next.providerState;
260
275
  if (Array.isArray(next.content)) next.content = stripToolContentParts(next.content);
261
276
  if (next.role === 'assistant' && !hasContentAfterToolStrip(next.content)) continue;
262
277
  if (next.role === 'user' && Array.isArray(next.content) && next.content.length === 0) continue;
@@ -374,6 +389,7 @@ function dropEmptyAssistantRows(messages) {
374
389
  if (!message || message.role !== 'assistant') return true;
375
390
  return hasProviderContent(message.content)
376
391
  || (Array.isArray(message.toolCalls) && message.toolCalls.length > 0)
392
+ || Boolean(message.providerState)
377
393
  || (Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0);
378
394
  });
379
395
  }
@@ -421,6 +437,13 @@ function shrinkMessageToBudget(message, tokenBudget) {
421
437
  if (estimateMessageTokens(next) > tokenBudget && Array.isArray(next.toolCalls)) {
422
438
  delete next.toolCalls;
423
439
  }
440
+ if (next.providerState && providerProjectionDigest(next) !== next.providerState.projectionDigest) {
441
+ // Modified assistant projections must never resurrect removed text/calls.
442
+ const signed = next.providerState.protocol === 'anthropic'
443
+ && next.providerState.items?.some(item => ['thinking', 'redacted_thinking'].includes(item?.type));
444
+ delete next.providerState;
445
+ if (signed) delete next.toolCalls;
446
+ }
424
447
  return next;
425
448
  }
426
449
 
@@ -608,6 +631,16 @@ function enrichTextBaselineWithTools(textBaseline, toolSource, selectedCallIds)
608
631
  const selectedParts = source.content.filter(part => selectedIds.has(toolContentPartCallId(part)));
609
632
  owner.content = [...baselineContent, ...selectedParts];
610
633
  }
634
+ if (source.providerState) {
635
+ if (providerProjectionDigest(owner) === source.providerState.projectionDigest) {
636
+ owner.providerState = source.providerState;
637
+ } else {
638
+ delete owner.providerState;
639
+ // A changed text baseline or call subset cannot replay this signed
640
+ // turn. Leave its text-only baseline and let pairing drop the results.
641
+ if (hasSignedProviderThinking(source)) continue;
642
+ }
643
+ }
611
644
  messagesBySourceIndex.set(sourceIndex, owner);
612
645
  }
613
646
 
@@ -627,19 +660,41 @@ function preservesTextBaseline(textBaseline, enriched) {
627
660
  });
628
661
  }
629
662
 
663
+ function hasSignedProviderThinking(message) {
664
+ return message.providerState?.protocol === 'anthropic'
665
+ && message.providerState.items?.some(item => ['thinking', 'redacted_thinking'].includes(item?.type));
666
+ }
667
+
668
+ function completeToolCallGroups(messages) {
669
+ const completeIds = new Set(completeToolCallIds(messages));
670
+ const groups = [];
671
+ for (let index = messages.length - 1; index >= 0; index -= 1) {
672
+ const message = messages[index];
673
+ if (message?.role !== 'assistant' || !Array.isArray(message.toolCalls)) continue;
674
+ const ids = message.toolCalls.map(call => call?.id);
675
+ if (hasSignedProviderThinking(message)) {
676
+ // A signed assistant projection binds every call, including their order.
677
+ // A missing result makes the whole group ineligible, not a smaller turn.
678
+ if (ids.length > 0 && ids.every(id => completeIds.has(id))) groups.push(ids);
679
+ } else {
680
+ groups.push(...ids.reverse().filter(id => completeIds.has(id)).map(id => [id]));
681
+ }
682
+ }
683
+ return groups;
684
+ }
685
+
630
686
  function addOptionalRecentToolPairs(textBaseline, toolSource, options) {
631
687
  const selectedCallIds = new Set();
632
688
  let out = pairSanitize(textBaseline);
633
- for (const callId of completeToolCallIds(toolSource)) {
634
- const trialIds = new Set(selectedCallIds);
635
- trialIds.add(callId);
689
+ for (const callIds of completeToolCallGroups(toolSource)) {
690
+ const trialIds = new Set([...selectedCallIds, ...callIds]);
636
691
  const trial = enrichTextBaselineWithTools(textBaseline, toolSource, trialIds);
637
692
  if (trial.length > options.maxMessageCount) continue;
638
693
  const fitted = fitMessagesToBudget(trial, options.messageTokenBudget);
639
694
  const fittedCallIds = new Set(completeToolCallIds(fitted));
640
695
  if (![...trialIds].every(id => fittedCallIds.has(id))) continue;
641
696
  if (!preservesTextBaseline(textBaseline, fitted)) continue;
642
- selectedCallIds.add(callId);
697
+ for (const callId of callIds) selectedCallIds.add(callId);
643
698
  out = fitted;
644
699
  }
645
700
  return out;
@@ -44,11 +44,12 @@ import { utf8PrefixWithinBytes } from '../utf8.js';
44
44
  * @typedef {{ type: 'thinking_delta', text: string }} ThinkingDeltaEvent
45
45
  * @typedef {{ type: 'thinking_block_end', thinking: string, signature: string }} ThinkingBlockEndEvent
46
46
  * @typedef {{ type: 'tool_call', id: string, name: string, input: object }} ToolCallEvent
47
- * @typedef {{ type: 'usage', inputTokens: number, outputTokens: number, cacheReadTokens?: number, cacheWriteTokens?: number, cacheTokensAreIncludedInInput?: boolean }} UsageEvent
47
+ * @typedef {{ type: 'provider_state', providerState: object, providerStateBytes: number }} ProviderStateEvent Internal only; never forward to UI/search.
48
+ * @typedef {{ type: 'usage', inputTokens: number, outputTokens: number, reasoningTokens?: number, cacheReadTokens?: number, cacheWriteTokens?: number, cacheTokensAreIncludedInInput?: boolean }} UsageEvent
48
49
  * @typedef {{ type: 'stop', stopReason: 'end_turn' | 'tool_use' | 'max_tokens' }} StopEvent
49
50
  * @typedef {{ type: 'error', error: Error, retryable: boolean }} ErrorEvent
50
51
  *
51
- * @typedef {TextDeltaEvent | ThinkingDeltaEvent | ThinkingBlockEndEvent | ToolCallEvent | UsageEvent | StopEvent | ErrorEvent} StreamEvent
52
+ * @typedef {TextDeltaEvent | ThinkingDeltaEvent | ThinkingBlockEndEvent | ProviderStateEvent | ToolCallEvent | UsageEvent | StopEvent | ErrorEvent} StreamEvent
52
53
  */
53
54
 
54
55
  // ─── Unified Message Types ─────────────────────────────────────
@@ -24,6 +24,7 @@ import {
24
24
  createBoundedTextAccumulator,
25
25
  toWellFormedJson,
26
26
  } from './adapter.js';
27
+ import { enforceSubAgentEffortPayload, captureEffortDecision } from '../effort.js';
27
28
  import {
28
29
  normalizeEffort,
29
30
  thinkingBudgetForEffort,
@@ -61,6 +62,8 @@ function applyAnthropicThinking(body, model, effort, effortContext = {}) {
61
62
  }
62
63
  }
63
64
 
65
+ import { ProviderStateError, createProviderContext, createProviderState, replayProviderState, providerStateBytes, applyAnthropicCaching, reasoningUsage } from './provider-state.js';
66
+
64
67
  const DEFAULT_BASE_URL = 'https://api.anthropic.com';
65
68
  const API_VERSION = '2023-06-01';
66
69
 
@@ -144,7 +147,7 @@ export class AnthropicAdapter extends LLMAdapter {
144
147
  * @param {import('./adapter.js').UnifiedMessage[]} messages
145
148
  * @returns {object[]}
146
149
  */
147
- #translateMessages(messages) {
150
+ #translateMessages(messages, context, identity) {
148
151
  const result = [];
149
152
  for (const msg of messages) {
150
153
  if (msg.role === 'system') continue; // system goes separately
@@ -152,24 +155,13 @@ export class AnthropicAdapter extends LLMAdapter {
152
155
  const content = translateUserContent(msg.content);
153
156
  if (content) result.push({ role: 'user', content });
154
157
  } else if (msg.role === 'assistant') {
158
+ const native = replayProviderState(msg, context, identity);
159
+ if (native) { result.push({ role: 'assistant', content: native }); continue; }
155
160
  const content = [];
156
- // task-327d: Anthropic requires thinking blocks to appear BEFORE
157
- // any text / tool_use in the content array on echo-back. When the
158
- // previous turn produced thinking blocks (with server-signed
159
- // signature), we MUST replay them verbatim or the next request
160
- // 400s with "content[].thinking in the thinking mode must be
161
- // passed back to the API". Order is mandatory.
162
- if (Array.isArray(msg.thinkingBlocks)) {
163
- for (const tb of msg.thinkingBlocks) {
164
- if (!tb || typeof tb.signature !== 'string' || !tb.signature) continue;
165
- if (tb.redacted) {
166
- if (typeof tb.data !== 'string') continue;
167
- content.push({ type: 'redacted_thinking', data: tb.data, signature: tb.signature });
168
- } else {
169
- if (typeof tb.thinking !== 'string') continue;
170
- content.push({ type: 'thinking', thinking: tb.thinking, signature: tb.signature });
171
- }
172
- }
161
+ // Legacy records remain readable, but have no trustworthy wire origin.
162
+ // Never replay their signatures into an arbitrary current account/model.
163
+ if (!msg.providerState && msg.thinkingBlocks?.length && msg.toolCalls?.length) {
164
+ throw new ProviderStateError('legacy signed tool history has no origin; start a new context');
173
165
  }
174
166
  if (hasNonEmptyText(msg.content)) {
175
167
  content.push({ type: 'text', text: msg.content });
@@ -246,14 +238,16 @@ export class AnthropicAdapter extends LLMAdapter {
246
238
  * @param {{ model: string, system: string, messages: import('./adapter.js').UnifiedMessage[], tools?: import('./adapter.js').UnifiedToolDef[], maxTokens?: number, effort?: 'low'|'medium'|'high'|'xhigh'|'max', effortSource?: 'user'|'auto', effortContext?: object, signal?: AbortSignal, onRawExchange?: ({rawRequest, rawResponse}) => void }} params
247
239
  * @returns {AsyncGenerator<import('./adapter.js').StreamEvent>}
248
240
  */
249
- async *stream({ model, system, messages, tools, maxTokens = 16384, effort, effortSource, effortContext, signal, onRawExchange, rawExchangeMaxBytes = 512 * 1024, onRequestStart }) {
241
+ async *stream({ model, system, messages, tools, maxTokens = 16384, effort, effortSource, effortContext, extraBody, providerContext, requestIdentity, onProviderDiagnostics, effortConstraint = null, onEffortDecision = null, signal, onRawExchange, rawExchangeMaxBytes = 512 * 1024, onRequestStart }) {
250
242
  if (signal?.aborted) throw new LLMAbortError();
251
243
 
244
+ const context = providerContext || createProviderContext({ protocol: 'anthropic', baseUrl: this.#baseUrl, model });
245
+ const translatedMessages = this.#translateMessages(messages, context, requestIdentity);
252
246
  const body = {
253
247
  model,
254
248
  max_tokens: maxTokens,
255
249
  system,
256
- messages: this.#translateMessages(messages),
250
+ messages: translatedMessages,
257
251
  stream: true,
258
252
  };
259
253
 
@@ -267,6 +261,15 @@ export class AnthropicAdapter extends LLMAdapter {
267
261
 
268
262
  const translatedTools = this.#translateTools(tools);
269
263
  if (translatedTools) body.tools = translatedTools;
264
+ if (extraBody) Object.assign(body, extraBody);
265
+ body.model = model; // Do not let extraBody bypass origin/model ownership.
266
+ body.messages = translatedMessages;
267
+ body.system = system;
268
+ applyAnthropicCaching(body, context, onProviderDiagnostics);
269
+ const effortDecision = effortConstraint
270
+ ? enforceSubAgentEffortPayload(body, { model, protocol: 'anthropic', effortContext, effortConstraint })
271
+ : captureEffortDecision({ body, model, protocol: 'anthropic', effortContext, requested: effort, source: effortSource || 'scenario' });
272
+ onEffortDecision?.(effortDecision);
270
273
  const wireBody = toWellFormedJson(body);
271
274
 
272
275
  const url = `${this.#baseUrl}/v1/messages`;
@@ -308,6 +311,28 @@ export class AnthropicAdapter extends LLMAdapter {
308
311
  throw this.#classifyError(response.status, errorBody, response);
309
312
  }
310
313
 
314
+ // Some native-compatible gateways return a complete JSON response even
315
+ // when streaming was requested. Seal exactly the same native blocks.
316
+ if ((response.headers?.get('content-type') || '').includes('application/json')) {
317
+ const result = await response.json();
318
+ const state = createProviderState({ context, identity: requestIdentity, items: result.content, responseId: result.id });
319
+ for (const block of result.content || []) {
320
+ if (block.type === 'text') yield { type: 'text_delta', text: block.text };
321
+ if (block.type === 'tool_use') yield { type: 'tool_call', id: block.id, name: block.name, input: block.input };
322
+ if (block.type === 'thinking') yield { type: 'thinking_block_end', thinking: block.thinking, signature: block.signature };
323
+ if (block.type === 'redacted_thinking') yield { type: 'thinking_block_end', redacted: true, data: block.data };
324
+ }
325
+ if (state) yield { type: 'provider_state', providerState: state, providerStateBytes: providerStateBytes(state) };
326
+ yield { type: 'usage', inputTokens: result.usage?.input_tokens || 0, outputTokens: result.usage?.output_tokens || 0,
327
+ cacheReadTokens: result.usage?.cache_read_input_tokens || 0, cacheWriteTokens: result.usage?.cache_creation_input_tokens || 0,
328
+ ...reasoningUsage(result.usage, 'anthropic') };
329
+ yield { type: 'stop', stopReason: this.#mapStopReason(result.stop_reason) };
330
+ if (onRawExchange) {
331
+ try { onRawExchange({ rawRequest, rawResponse: { status: response.status, headers: safeHeaders(response), body: result } }); } catch { /* diagnostic only */ }
332
+ }
333
+ return;
334
+ }
335
+
311
336
  // Parse SSE stream
312
337
  const reader = response.body.getReader();
313
338
  const decoder = new TextDecoder();
@@ -327,6 +352,10 @@ export class AnthropicAdapter extends LLMAdapter {
327
352
  // signature → next turn 400s identically).
328
353
  /** @type {Map<number, { kind: string, [k: string]: any }>} */
329
354
  const blockByIndex = new Map();
355
+ const completedBlocks = new Map();
356
+ let responseId;
357
+ let stateFailed = false;
358
+ let cumulativeReasoningTokens = 0;
330
359
  // Keep raw SSE chunks only until the engine receives the bounded exchange
331
360
  // callback. The engine owns the configured byte budget; the adapter avoids
332
361
  // quadratic string concatenation by storing chunks separately.
@@ -370,9 +399,13 @@ export class AnthropicAdapter extends LLMAdapter {
370
399
  if (type === 'content_block_start') {
371
400
  const block = event.content_block;
372
401
  const idx = event.index;
373
- if (block?.type === 'tool_use') {
402
+ if (block?.type === 'text') {
403
+ blockByIndex.set(idx, { kind: 'text', text: block.text || '', native: block });
404
+ if (block.text) yield { type: 'text_delta', text: block.text };
405
+ } else if (block?.type === 'tool_use') {
374
406
  blockByIndex.set(idx, {
375
407
  kind: 'tool_use',
408
+ native: block,
376
409
  id: block.id,
377
410
  name: block.name,
378
411
  input: '',
@@ -380,6 +413,7 @@ export class AnthropicAdapter extends LLMAdapter {
380
413
  } else if (block?.type === 'thinking') {
381
414
  blockByIndex.set(idx, {
382
415
  kind: 'thinking',
416
+ native: block,
383
417
  thinking: typeof block.thinking === 'string' ? block.thinking : '',
384
418
  signature: typeof block.signature === 'string' ? block.signature : '',
385
419
  });
@@ -390,6 +424,7 @@ export class AnthropicAdapter extends LLMAdapter {
390
424
  // 400s with the same "must be passed back" error.
391
425
  blockByIndex.set(idx, {
392
426
  kind: 'redacted_thinking',
427
+ native: block,
393
428
  data: typeof block.data === 'string' ? block.data : '',
394
429
  signature: typeof block.signature === 'string' ? block.signature : '',
395
430
  });
@@ -399,6 +434,7 @@ export class AnthropicAdapter extends LLMAdapter {
399
434
  const idx = event.index;
400
435
  const st = blockByIndex.get(idx);
401
436
  if (delta?.type === 'text_delta') {
437
+ if (st?.kind === 'text') st.text += delta.text || '';
402
438
  yield { type: 'text_delta', text: delta.text };
403
439
  } else if (delta?.type === 'thinking_delta') {
404
440
  // Forward delta for live UI; ALSO accumulate for round-trip.
@@ -422,10 +458,12 @@ export class AnthropicAdapter extends LLMAdapter {
422
458
  } else if (st.kind === 'tool_use') {
423
459
  let parsedInput = {};
424
460
  try {
425
- parsedInput = st.input ? JSON.parse(st.input) : {};
461
+ parsedInput = st.input ? JSON.parse(st.input) : st.native.input || {};
426
462
  } catch {
427
463
  parsedInput = {};
464
+ stateFailed = true;
428
465
  }
466
+ completedBlocks.set(idx, { ...st.native, input: parsedInput });
429
467
  yield {
430
468
  type: 'tool_call',
431
469
  id: st.id,
@@ -452,11 +490,15 @@ export class AnthropicAdapter extends LLMAdapter {
452
490
  };
453
491
  }
454
492
  }
493
+ if (st?.kind === 'text') completedBlocks.set(idx, { ...st.native, text: st.text });
494
+ if (st?.kind === 'thinking') completedBlocks.set(idx, { ...st.native, thinking: st.thinking, signature: st.signature });
495
+ if (st?.kind === 'redacted_thinking') completedBlocks.set(idx, { type: 'redacted_thinking', data: st.data });
455
496
  blockByIndex.delete(idx);
456
497
  } else if (type === 'message_delta') {
457
498
  const stopReason = event.delta?.stop_reason;
458
499
  if (stopReason) {
459
- sawStop = true;
500
+ // Only message_stop seals native state. EOF after message_delta
501
+ // must not silently complete a signed tool turn without its state.
460
502
  yield {
461
503
  type: 'stop',
462
504
  stopReason: this.#mapStopReason(stopReason),
@@ -469,24 +511,40 @@ export class AnthropicAdapter extends LLMAdapter {
469
511
  const nextOutputTokens = Math.max(0, Number(event.usage.output_tokens) || 0);
470
512
  const outputTokens = Math.max(0, nextOutputTokens - cumulativeOutputTokens);
471
513
  cumulativeOutputTokens = Math.max(cumulativeOutputTokens, nextOutputTokens);
514
+ const reasoning = reasoningUsage(event.usage, 'anthropic');
515
+ if (reasoning.reasoningTokens !== undefined) {
516
+ const next = reasoning.reasoningTokens;
517
+ reasoning.reasoningTokens = Math.max(0, next - cumulativeReasoningTokens);
518
+ cumulativeReasoningTokens = Math.max(cumulativeReasoningTokens, next);
519
+ }
472
520
  yield {
473
521
  type: 'usage',
522
+ ...reasoning,
474
523
  inputTokens: 0, // Only in message_start
475
524
  outputTokens,
476
525
  };
477
526
  }
478
527
  } else if (type === 'message_stop') {
479
528
  sawStop = true;
529
+ if (!stateFailed && blockByIndex.size === 0) {
530
+ const state = createProviderState({ context, identity: requestIdentity, responseId,
531
+ items: [...completedBlocks].sort(([a], [b]) => a - b).map(([, block]) => block) });
532
+ if (state) yield { type: 'provider_state', providerState: state, providerStateBytes: providerStateBytes(state) };
533
+ }
480
534
  } else if (type === 'message_start') {
481
535
  sawMessageStart = true;
536
+ responseId = event.message?.id;
482
537
  // Usage from message_start
483
538
  if (event.message?.usage) {
484
539
  cumulativeOutputTokens = Math.max(
485
540
  cumulativeOutputTokens,
486
541
  Math.max(0, Number(event.message.usage.output_tokens) || 0),
487
542
  );
543
+ const reasoning = reasoningUsage(event.message.usage, 'anthropic');
544
+ cumulativeReasoningTokens = reasoning.reasoningTokens || 0;
488
545
  yield {
489
546
  type: 'usage',
547
+ ...reasoning,
490
548
  inputTokens: event.message.usage.input_tokens || 0,
491
549
  outputTokens: cumulativeOutputTokens,
492
550
  cacheReadTokens: event.message.usage.cache_read_input_tokens || 0,
@@ -494,6 +552,7 @@ export class AnthropicAdapter extends LLMAdapter {
494
552
  };
495
553
  }
496
554
  } else if (type === 'error') {
555
+ stateFailed = true;
497
556
  yield {
498
557
  type: 'error',
499
558
  error: new Error(event.error?.message || 'Unknown streaming error'),
@@ -538,14 +597,16 @@ export class AnthropicAdapter extends LLMAdapter {
538
597
  * models silently drop the param. max_tokens auto-widens to budget+1024
539
598
  * when needed.
540
599
  */
541
- async call({ model, system, messages, maxTokens = 4096, effort, effortSource, effortContext, signal, onRequestStart }) {
600
+ async call({ model, system, messages, maxTokens = 4096, effort, effortSource, effortContext, extraBody, providerContext, requestIdentity, onProviderDiagnostics, effortConstraint = null, onEffortDecision = null, signal, onRequestStart }) {
542
601
  if (signal?.aborted) throw new LLMAbortError();
543
602
 
603
+ const context = providerContext || createProviderContext({ protocol: 'anthropic', baseUrl: this.#baseUrl, model });
604
+ const translatedMessages = this.#translateMessages(messages, context, requestIdentity);
544
605
  const body = {
545
606
  model,
546
607
  max_tokens: maxTokens,
547
608
  system,
548
- messages: this.#translateMessages(messages),
609
+ messages: translatedMessages,
549
610
  };
550
611
 
551
612
  // task-327c: mirror stream()'s thinking injection for side queries.
@@ -553,6 +614,15 @@ export class AnthropicAdapter extends LLMAdapter {
553
614
  if ((thinkingV1Enabled() || effortSource === 'user') && normEffort) {
554
615
  applyAnthropicThinking(body, model, normEffort, effortContext);
555
616
  }
617
+ if (extraBody) Object.assign(body, extraBody);
618
+ body.model = model; // Do not let extraBody bypass origin/model ownership.
619
+ body.messages = translatedMessages;
620
+ body.system = system;
621
+ applyAnthropicCaching(body, context, onProviderDiagnostics);
622
+ const effortDecision = effortConstraint
623
+ ? enforceSubAgentEffortPayload(body, { model, protocol: 'anthropic', effortContext, effortConstraint })
624
+ : captureEffortDecision({ body, model, protocol: 'anthropic', effortContext, requested: effort, source: effortSource || 'scenario' });
625
+ onEffortDecision?.(effortDecision);
556
626
  const wireBody = toWellFormedJson(body);
557
627
 
558
628
  let response;
@@ -581,8 +651,10 @@ export class AnthropicAdapter extends LLMAdapter {
581
651
 
582
652
  return {
583
653
  text,
654
+ providerState: createProviderState({ context, identity: requestIdentity, items: result.content, responseId: result.id }),
584
655
  stopReason: this.#mapStopReason(result.stop_reason),
585
656
  usage: {
657
+ ...reasoningUsage(result.usage, 'anthropic'),
586
658
  inputTokens: result.usage?.input_tokens || 0,
587
659
  outputTokens: result.usage?.output_tokens || 0,
588
660
  cacheReadTokens: result.usage?.cache_read_input_tokens || 0,
@@ -25,6 +25,7 @@
25
25
  * - tool_result.toolCallId ← → function_call_output.call_id (direct passthrough)
26
26
  */
27
27
 
28
+ import { enforceSubAgentEffortPayload, captureEffortDecision } from '../effort.js';
28
29
  import {
29
30
  LLMAdapter,
30
31
  LLMRateLimitError,
@@ -50,6 +51,8 @@ import {
50
51
  mapEffortToOpenAIReasoning,
51
52
  } from '../models.js';
52
53
 
54
+ import { createProviderContext, createProviderState, replayProviderState, providerStateBytes, applyResponsesContinuity, reasoningUsage } from './provider-state.js';
55
+
53
56
  const DEFAULT_BASE_URL = 'https://api.openai.com/v1';
54
57
 
55
58
  /**
@@ -163,7 +166,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
163
166
  * Assistant tool_calls → separate { type:'function_call', call_id, name, arguments } items
164
167
  * Tool message → { type:'function_call_output', call_id, output }
165
168
  */
166
- #translateInput(messages) {
169
+ #translateInput(messages, context, identity) {
167
170
  const input = [];
168
171
  for (const msg of messages) {
169
172
  if (msg.role === 'system') {
@@ -177,6 +180,8 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
177
180
  content: this.#translateUserContent(msg.content),
178
181
  });
179
182
  } else if (msg.role === 'assistant') {
183
+ const native = replayProviderState(msg, context, identity);
184
+ if (native) { input.push(...native); continue; }
180
185
  // Emit a message item if there is text content
181
186
  if (msg.content && typeof msg.content === 'string' && msg.content.trim()) {
182
187
  input.push({
@@ -263,12 +268,14 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
263
268
  * `api-key` headers are auto-redacted (see `redactRawRequest` in
264
269
  * `adapter.js`); request-body fields are caller-controlled.
265
270
  */
266
- async *stream({ model, system, messages, tools, maxTokens = 16384, effort, effortSource, effortContext = {}, extraBody, signal, onRawExchange, rawExchangeMaxBytes = 512 * 1024, onRequestStart }) {
271
+ async *stream({ model, system, messages, tools, maxTokens = 16384, effort, effortSource, effortContext = {}, extraBody, providerContext, requestIdentity, onProviderDiagnostics, effortConstraint = null, onEffortDecision = null, signal, onRawExchange, rawExchangeMaxBytes = 512 * 1024, onRequestStart }) {
267
272
  if (signal?.aborted) throw new LLMAbortError();
268
273
 
274
+ const context = providerContext || createProviderContext({ protocol: 'openai-responses', baseUrl: this.#baseUrl, model });
275
+ const input = this.#translateInput(messages, context, requestIdentity);
269
276
  const body = {
270
277
  model,
271
- input: this.#translateInput(messages),
278
+ input,
272
279
  stream: true,
273
280
  max_output_tokens: maxTokens,
274
281
  };
@@ -291,6 +298,12 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
291
298
  }
292
299
 
293
300
  if (extraBody) Object.assign(body, extraBody);
301
+ body.model = model; // Origin binding must describe the actual dispatched model.
302
+ applyResponsesContinuity(body, { context, identity: requestIdentity, input, onProviderDiagnostics });
303
+ const effortDecision = effortConstraint
304
+ ? enforceSubAgentEffortPayload(body, { model, protocol: 'openai-responses', effortContext, effortConstraint })
305
+ : captureEffortDecision({ body, model, protocol: 'openai-responses', effortContext, requested: effort, source: effortSource || 'scenario' });
306
+ onEffortDecision?.(effortDecision);
294
307
  const wireBody = toWellFormedJson(body);
295
308
 
296
309
  const url = `${this.#baseUrl}/responses`;
@@ -334,6 +347,31 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
334
347
  throw this.#classifyError(response.status, errorBody, response);
335
348
  }
336
349
 
350
+ if (response.headers?.get('content-type')?.includes('application/json')) {
351
+ const result = await response.json();
352
+ const items = Array.isArray(result.output) ? result.output : [];
353
+ for (const item of items) {
354
+ if (item.type === 'message') {
355
+ for (const part of item.content || []) if (part.type === 'output_text') yield { type: 'text_delta', text: part.text };
356
+ } else if (item.type === 'function_call') {
357
+ let input = {};
358
+ try { input = JSON.parse(item.arguments); } catch { /* legacy projection */ }
359
+ yield { type: 'tool_call', id: item.call_id, name: item.name, input };
360
+ }
361
+ }
362
+ if (['completed', 'incomplete'].includes(result.status)) {
363
+ const state = createProviderState({ context, identity: requestIdentity, items, responseId: result.id });
364
+ if (state) yield { type: 'provider_state', providerState: state, providerStateBytes: providerStateBytes(state) };
365
+ }
366
+ const usage = result.usage || {};
367
+ yield { type: 'usage', inputTokens: usage.input_tokens || 0, outputTokens: usage.output_tokens || 0,
368
+ cacheReadTokens: usage.input_tokens_details?.cached_tokens || 0, cacheWriteTokens: 0,
369
+ cacheTokensAreIncludedInInput: true,
370
+ ...reasoningUsage(usage, 'openai-responses') };
371
+ if (result.status === 'failed') throw new LLMServerError('OpenAI JSON response failed', 0);
372
+ yield { type: 'stop', stopReason: this.#mapStopReason(result, false) };
373
+ return;
374
+ }
337
375
  const reader = response.body.getReader();
338
376
  const decoder = new TextDecoder();
339
377
  // Incremental O(n) line splitter (see SseLineBuffer): the old
@@ -349,6 +387,8 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
349
387
  /** call_ids already emitted as tool_call events (to avoid duplicating on completed fallback). */
350
388
  const emittedToolCallIds = new Set();
351
389
  let sawToolCall = false;
390
+ const completedItems = new Map();
391
+ let emittedText = false;
352
392
 
353
393
  // Keep raw SSE chunks only until the engine receives the bounded exchange
354
394
  // callback. The engine owns the configured byte budget; the adapter avoids
@@ -389,7 +429,10 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
389
429
 
390
430
  const type = event.type;
391
431
 
392
- if (type === 'response.output_item.added') {
432
+ if (sawTerminalEvent) continue;
433
+ if (type === 'response.output_item.done') {
434
+ if (Number.isInteger(event.output_index) && event.item) completedItems.set(event.output_index, event.item);
435
+ } else if (type === 'response.output_item.added') {
393
436
  const item = event.item;
394
437
  const idx = event.output_index;
395
438
  if (item?.type === 'function_call') {
@@ -401,6 +444,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
401
444
  }
402
445
  } else if (type === 'response.output_text.delta') {
403
446
  if (typeof event.delta === 'string' && event.delta.length > 0) {
447
+ emittedText = true;
404
448
  yield { type: 'text_delta', text: event.delta };
405
449
  }
406
450
  } else if (type === 'response.function_call_arguments.delta') {
@@ -439,7 +483,12 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
439
483
 
440
484
  // Fallback: flush any function_call items in the final output that we
441
485
  // didn't see a .done event for (defensive against partial streams).
442
- const outputArr = Array.isArray(respObj.output) ? respObj.output : [];
486
+ const outputArr = Array.isArray(respObj.output) ? respObj.output : [...completedItems].sort(([a], [b]) => a - b).map(([, item]) => item);
487
+ if (!emittedText) {
488
+ for (const item of outputArr) if (item?.type === 'message') {
489
+ for (const part of item.content || []) if (part?.type === 'output_text') yield { type: 'text_delta', text: part.text };
490
+ }
491
+ }
443
492
  for (const item of outputArr) {
444
493
  if (item?.type !== 'function_call') continue;
445
494
  const cid = item.call_id || item.id || '';
@@ -460,6 +509,9 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
460
509
  };
461
510
  }
462
511
 
512
+ const state = createProviderState({ context, identity: requestIdentity, items: outputArr, responseId: respObj.id });
513
+ if (state) yield { type: 'provider_state', providerState: state, providerStateBytes: providerStateBytes(state) };
514
+
463
515
  // Usage
464
516
  const usage = respObj.usage || {};
465
517
  yield {
@@ -469,6 +521,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
469
521
  cacheReadTokens: usage.input_tokens_details?.cached_tokens || 0,
470
522
  cacheWriteTokens: 0,
471
523
  cacheTokensAreIncludedInInput: true,
524
+ ...reasoningUsage(usage, 'openai-responses'),
472
525
  };
473
526
 
474
527
  yield {
@@ -527,12 +580,14 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
527
580
  * expose them, mirror the stream() instrumentation. Parity with
528
581
  * anthropic.js's `call()`.
529
582
  */
530
- async call({ model, system, messages, maxTokens = 4096, effort, effortSource, effortContext = {}, extraBody, signal, onRequestStart }) {
583
+ async call({ model, system, messages, maxTokens = 4096, effort, effortSource, effortContext = {}, extraBody, providerContext, requestIdentity, onProviderDiagnostics, effortConstraint = null, onEffortDecision = null, signal, onRequestStart }) {
531
584
  if (signal?.aborted) throw new LLMAbortError();
532
585
 
586
+ const context = providerContext || createProviderContext({ protocol: 'openai-responses', baseUrl: this.#baseUrl, model });
587
+ const input = this.#translateInput(messages, context, requestIdentity);
533
588
  const body = {
534
589
  model,
535
- input: this.#translateInput(messages),
590
+ input,
536
591
  max_output_tokens: maxTokens,
537
592
  };
538
593
  if (system) body.instructions = system;
@@ -549,6 +604,12 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
549
604
  }
550
605
 
551
606
  if (extraBody) Object.assign(body, extraBody);
607
+ body.model = model; // Origin binding must describe the actual dispatched model.
608
+ applyResponsesContinuity(body, { context, identity: requestIdentity, input, onProviderDiagnostics });
609
+ const effortDecision = effortConstraint
610
+ ? enforceSubAgentEffortPayload(body, { model, protocol: 'openai-responses', effortContext, effortConstraint })
611
+ : captureEffortDecision({ body, model, protocol: 'openai-responses', effortContext, requested: effort, source: effortSource || 'scenario' });
612
+ onEffortDecision?.(effortDecision);
552
613
  const wireBody = toWellFormedJson(body);
553
614
 
554
615
  let response;
@@ -594,6 +655,8 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
594
655
  const usage = result.usage || {};
595
656
  return {
596
657
  text,
658
+ providerState: ['completed', 'incomplete'].includes(result.status)
659
+ ? createProviderState({ context, identity: requestIdentity, items: result.output, responseId: result.id }) : null,
597
660
  stopReason: this.#mapStopReason(result, false),
598
661
  usage: {
599
662
  inputTokens: usage.input_tokens || 0,
@@ -601,6 +664,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
601
664
  cacheReadTokens: usage.input_tokens_details?.cached_tokens || 0,
602
665
  cacheWriteTokens: 0,
603
666
  cacheTokensAreIncludedInInput: true,
667
+ ...reasoningUsage(usage, 'openai-responses'),
604
668
  },
605
669
  };
606
670
  }