@yeaft/webchat-agent 1.0.505 → 1.0.506

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Binary file
@@ -17,6 +17,6 @@
17
17
  </head>
18
18
  <body>
19
19
  <div id="app"></div>
20
- <script type="module" src="app.bundle.js?v=38debc76"></script>
20
+ <script type="module" src="app.bundle.js?v=10d8663e"></script>
21
21
  </body>
22
22
  </html>
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yeaft/webchat-agent",
3
- "version": "1.0.505",
3
+ "version": "1.0.506",
4
4
  "description": "Remote worker agent for Yeaft Web Code Agent — connects the native Yeaft engine, CLI providers, and workbench tools",
5
5
  "main": "index.js",
6
6
  "type": "module",
package/yeaft/cli.js CHANGED
@@ -388,6 +388,7 @@ async function runREPL(config, args) {
388
388
  content: m.content,
389
389
  ...(m.toolCallId && { toolCallId: m.toolCallId }),
390
390
  ...(m.toolCalls && { toolCalls: m.toolCalls }),
391
+ ...(m.providerState && { providerState: m.providerState }),
391
392
  ...(Array.isArray(m.thinkingBlocks) && m.thinkingBlocks.length > 0
392
393
  ? { thinkingBlocks: m.thinkingBlocks.map(block => ({ ...block })) }
393
394
  : {}),
@@ -908,6 +909,7 @@ async function runStreamJson(config, args) {
908
909
  content: message.content,
909
910
  ...(message.toolCallId && { toolCallId: message.toolCallId }),
910
911
  ...(message.toolCalls && { toolCalls: message.toolCalls }),
912
+ ...(message.providerState && { providerState: message.providerState }),
911
913
  ...(Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0
912
914
  ? { thinkingBlocks: message.thinkingBlocks.map(block => ({ ...block })) }
913
915
  : {}),
@@ -1263,6 +1265,7 @@ async function runOnce(config, args) {
1263
1265
  content: m.content,
1264
1266
  ...(m.toolCallId && { toolCallId: m.toolCallId }),
1265
1267
  ...(m.toolCalls && { toolCalls: m.toolCalls }),
1268
+ ...(m.providerState && { providerState: m.providerState }),
1266
1269
  ...(Array.isArray(m.thinkingBlocks) && m.thinkingBlocks.length > 0
1267
1270
  ? { thinkingBlocks: m.thinkingBlocks.map(block => ({ ...block })) }
1268
1271
  : {}),
@@ -351,8 +351,10 @@ export function projectVisibleSessionMessages(messages) {
351
351
  }
352
352
 
353
353
  const visible = [];
354
- for (const row of rows) {
355
- if (!row || (row.role !== 'user' && row.role !== 'assistant')) continue;
354
+ for (const sourceRow of rows) {
355
+ if (!sourceRow || (sourceRow.role !== 'user' && sourceRow.role !== 'assistant')) continue;
356
+ // Public history never owns provider-private continuation payloads.
357
+ const { providerState, thinkingBlocks, ...row } = sourceRow;
356
358
  if (!isVisibleConversationRow(row)) continue;
357
359
  if (row.role !== 'assistant' || !Array.isArray(row.toolCalls) || row.toolCalls.length === 0) {
358
360
  if (row.role === 'assistant' && !row.content && !row.attachments && !row.images
@@ -502,19 +504,18 @@ function serializeMessage(msg) {
502
504
  // bytes that don't need to be human-readable. Without this round-trip
503
505
  // the next Anthropic request 400s with "content[].thinking in the
504
506
  // thinking mode must be passed back to the API".
505
- if (msg.thinkingBlocks && msg.thinkingBlocks.length > 0) {
507
+ if (msg.providerState) fm.push(`providerStateB64: ${Buffer.from(JSON.stringify(msg.providerState)).toString('base64')}`);
508
+ if (!msg.providerState && msg.thinkingBlocks && msg.thinkingBlocks.length > 0) {
506
509
  fm.push(`thinkingBlocks:`);
507
510
  for (const tb of msg.thinkingBlocks) {
508
- if (!tb || typeof tb.signature !== 'string' || !tb.signature) continue;
511
+ if (!tb) continue;
509
512
  if (tb.redacted) {
510
513
  if (typeof tb.data !== 'string') continue;
511
514
  const dataB64 = Buffer.from(tb.data, 'utf8').toString('base64');
512
- const signatureB64 = Buffer.from(tb.signature, 'utf8').toString('base64');
513
515
  fm.push(` - redacted: true`);
514
516
  fm.push(` dataB64: ${dataB64}`);
515
- fm.push(` signatureB64: ${signatureB64}`);
516
517
  } else {
517
- if (typeof tb.thinking !== 'string') continue;
518
+ if (typeof tb.thinking !== 'string' || typeof tb.signature !== 'string' || !tb.signature) continue;
518
519
  const thinkingB64 = Buffer.from(tb.thinking, 'utf8').toString('base64');
519
520
  const signatureB64 = Buffer.from(tb.signature, 'utf8').toString('base64');
520
521
  fm.push(` - thinkingB64: ${thinkingB64}`);
@@ -646,7 +647,11 @@ export function parseMessage(raw) {
646
647
  }
647
648
 
648
649
  // task-327d: parse thinkingBlocks (mirror of toolCalls parser above)
649
- if (frontmatter.includes('thinkingBlocks:')) {
650
+ const stateMatch = frontmatter.match(/^providerStateB64: (.+)$/m);
651
+ if (stateMatch) {
652
+ try { msg.providerState = JSON.parse(Buffer.from(stateMatch[1], 'base64').toString('utf8')); } catch { /* legacy invalid row */ }
653
+ }
654
+ if (!msg.providerState && frontmatter.includes('thinkingBlocks:')) {
650
655
  const thinkingBlocks = [];
651
656
  const tbMatch = frontmatter.match(/thinkingBlocks:\n((?:\s+-\s+[\s\S]*?)(?=\n\w|$))/);
652
657
  if (tbMatch) {
@@ -672,7 +677,7 @@ export function parseMessage(raw) {
672
677
  }
673
678
  // Both fields required — an unsigned block would 400 on replay.
674
679
  if (tb.redacted) {
675
- if (typeof tb.data === 'string' && typeof tb.signature === 'string' && tb.signature) {
680
+ if (typeof tb.data === 'string') {
676
681
  thinkingBlocks.push(tb);
677
682
  }
678
683
  } else if (typeof tb.thinking === 'string' && typeof tb.signature === 'string' && tb.signature) {
@@ -1621,6 +1626,7 @@ export class ConversationStore {
1621
1626
  const copy = { ...m };
1622
1627
  delete copy.toolCalls;
1623
1628
  delete copy.thinkingBlocks;
1629
+ delete copy.providerState;
1624
1630
  out.push(copy);
1625
1631
  }
1626
1632
  continue;
@@ -54,8 +54,9 @@ function recordMessageScan(telemetry) {
54
54
  }
55
55
 
56
56
  function withSource(msg, source) {
57
+ const { providerState, thinkingBlocks, ...publicMessage } = msg;
57
58
  return {
58
- ...msg,
59
+ ...publicMessage,
59
60
  content: searchableContent(msg),
60
61
  sessionId: msg.sessionId || source.sessionId || null,
61
62
  historySource: source.kind,
package/yeaft/effort.js CHANGED
@@ -14,14 +14,15 @@
14
14
  *
15
15
  * Red lines:
16
16
  * • Never error on unknown scenario — default to 'max'.
17
- * • Feature flag YEAFT_THINKING_V1 is enforced at the adapter/router
18
- * layer; this module just computes the intended value. If the flag
19
- * is off, adapters drop it anyway.
20
- * • Unsupported models silently drop effort at the router — this
21
- * module does NOT consult the capability matrix.
17
+ * • The ordinary picker preserves existing scenario defaults. Child effort
18
+ * is separately constrained by the capability-aware final payload helpers.
19
+ * • Child ceilings cannot be disabled by YEAFT_THINKING_V1, user overrides,
20
+ * routing, nesting, or extraBody. Unsupported models omit effort fields.
22
21
  */
23
22
 
24
- import { normalizeEffort } from './models.js';
23
+ import {
24
+ normalizeEffort, getThinkingCapability, getModelEffortOptions, thinkingBudgetForEffort,
25
+ } from './models.js';
25
26
 
26
27
  /**
27
28
  * Number of tool-loop turns past which a query is considered "complex"
@@ -116,3 +117,155 @@ export function parseEffortPrefix(prompt) {
116
117
  const cleanedPrompt = prompt.slice(m[0].length);
117
118
  return { effort, cleanedPrompt };
118
119
  }
120
+
121
+ // Ordinal levels, not lexical sorting or cross-model token-budget equivalence.
122
+ export const EFFORT_LEVELS = Object.freeze(['minimal', 'low', 'medium', 'high', 'xhigh', 'max', 'ultra']);
123
+ const effortRank = value => EFFORT_LEVELS.indexOf(normalizeEffort(value));
124
+ const atMost = (value, ceiling) => effortRank(value) >= 0 && effortRank(value) <= effortRank(ceiling);
125
+
126
+ /** Copy only durable decision fields; never retain a mutable request/config object. */
127
+ export function snapshotEffortDecision(decision = null) {
128
+ return Object.freeze({
129
+ requested: normalizeEffort(decision?.requested),
130
+ effective: normalizeEffort(decision?.effective),
131
+ source: typeof decision?.source === 'string' ? decision.source : 'unknown',
132
+ model: typeof decision?.model === 'string' ? decision.model : null,
133
+ wireMode: typeof decision?.wireMode === 'string' ? decision.wireMode : 'omitted',
134
+ thinkingEnabled: decision?.thinkingEnabled === true,
135
+ cap: normalizeEffort(decision?.cap),
136
+ ...(Number.isFinite(decision?.budgetTokens) ? { budgetTokens: decision.budgetTokens } : {}),
137
+ });
138
+ }
139
+
140
+ /** Capture the decision belonging to the provider response that generated a tool call. */
141
+ export function captureParentEffortDecision(ctx = {}) {
142
+ return snapshotEffortDecision(ctx.effortDecision ?? ctx.parentEngineDeps?.effortDecision);
143
+ }
144
+
145
+ function effortError(model, target, detail = '') {
146
+ const error = new Error(`Sub-agent model "${model}" cannot express effort <= ${target}${detail ? ` (${detail})` : ''}. Select a compatible child model.`);
147
+ error.code = 'SUB_AGENT_EFFORT_UNREPRESENTABLE';
148
+ return error;
149
+ }
150
+
151
+ function capabilityContext(protocol, effortContext) {
152
+ return { ...effortContext, protocol: protocol || effortContext?.protocol };
153
+ }
154
+
155
+ /**
156
+ * Resolve the non-disableable child ceiling using the actual model's capabilities.
157
+ * Unknown/omitted parent wire defaults are conservative medium, never chat/max.
158
+ */
159
+ export function resolveSubAgentEffort({ parentDecision = null, model, effortContext = {}, protocol } = {}) {
160
+ const parent = snapshotEffortDecision(parentDecision);
161
+ // An unsupported intermediate model has no wire effort, but must not erase
162
+ // an inherited lower ceiling when it delegates again.
163
+ let parentTarget = parent.effective || 'medium';
164
+ if (parent.cap && atMost(parent.cap, parentTarget)) parentTarget = parent.cap;
165
+ const target = atMost(parentTarget, 'high') ? parentTarget : 'high';
166
+ const context = capabilityContext(protocol, effortContext);
167
+ const capability = getThinkingCapability(model, context);
168
+ const base = {
169
+ requested: parentTarget, source: parent.effective ? 'inherited' : 'fallback',
170
+ model, cap: target, thinkingEnabled: false,
171
+ };
172
+ if (!capability.supportsThinking || capability.thinkingProtocol === 'none') {
173
+ return snapshotEffortDecision({ ...base, effective: null, wireMode: 'unsupported' });
174
+ }
175
+ const supported = getModelEffortOptions(model, context).filter(value => atMost(value, target));
176
+ const effective = EFFORT_LEVELS.filter(value => supported.includes(value)).at(-1);
177
+ if (!effective) throw effortError(model, target);
178
+ const wireMode = protocol === 'openai-responses' || capability.thinkingProtocol === 'openai-reasoning'
179
+ ? 'reasoning-effort'
180
+ : capability.thinkingProtocol === 'anthropic-adaptive' ? 'adaptive' : 'manual';
181
+ return snapshotEffortDecision({ ...base, effective, wireMode, thinkingEnabled: true });
182
+ }
183
+
184
+ function manualEffortForBudget(model, budget) {
185
+ if (!Number.isFinite(budget) || budget <= 0) return null;
186
+ // Round upward: a nonstandard budget must never masquerade as a lower tier.
187
+ return ['low', 'medium', 'high', 'max'].find(level => budget <= thinkingBudgetForEffort(model, level)) || 'ultra';
188
+ }
189
+
190
+ /**
191
+ * Read the FINAL wire payload. requested is observability only, not effective.
192
+ * Call after all extraBody/feature-flag/mapping changes, before serialization.
193
+ */
194
+ export function captureEffortDecision({ body = {}, model, protocol, effortContext = {}, requested = null, source = 'scenario' } = {}) {
195
+ const context = capabilityContext(protocol, effortContext);
196
+ const capability = getThinkingCapability(model, context);
197
+ const base = { requested, source, model, cap: null, thinkingEnabled: false };
198
+ if (!capability.supportsThinking || capability.thinkingProtocol === 'none') {
199
+ return snapshotEffortDecision({ ...base, effective: null, wireMode: 'unsupported' });
200
+ }
201
+ if (protocol === 'openai-responses') {
202
+ const effective = normalizeEffort(body.reasoning?.effort);
203
+ if (effective) return snapshotEffortDecision({ ...base, effective, wireMode: 'reasoning-effort', thinkingEnabled: true });
204
+ } else if (body.thinking?.type === 'enabled') {
205
+ const budgetTokens = body.thinking.budget_tokens;
206
+ return snapshotEffortDecision({ ...base, effective: manualEffortForBudget(model, budgetTokens), wireMode: 'manual', thinkingEnabled: true, budgetTokens });
207
+ } else if (body.thinking?.type === 'adaptive') {
208
+ const effective = normalizeEffort(body.output_config?.effort);
209
+ if (effective) return snapshotEffortDecision({ ...base, effective, wireMode: 'adaptive', thinkingEnabled: true });
210
+ }
211
+ const modelDefault = body.thinking?.type === 'disabled' ? null : normalizeEffort(capability.defaultEffort);
212
+ return snapshotEffortDecision({ ...base, effective: modelDefault, source: modelDefault ? 'model-default' : source, wireMode: 'omitted' });
213
+ }
214
+
215
+ function removeEffortField(body, key) {
216
+ if (!body[key] || typeof body[key] !== 'object' || Array.isArray(body[key])) {
217
+ delete body[key];
218
+ return;
219
+ }
220
+ const { effort: _effort, ...rest } = body[key];
221
+ if (Object.keys(rest).length) body[key] = rest;
222
+ else delete body[key];
223
+ }
224
+
225
+ /**
226
+ * Mutate the FINAL provider body in place and return its immutable child decision.
227
+ * The router must preserve effortConstraint even when YEAFT_THINKING_V1 is off.
228
+ * All adapter stream/call paths invoke this AFTER extraBody, BEFORE fetch, and
229
+ * must not subsequently rewrite reasoning/thinking/output_config/max_tokens.
230
+ * @param {object} body Final body owned by the adapter (never caller config).
231
+ * @param {{ model: string, protocol: string, effortContext?: object,
232
+ * effortConstraint: { parentDecision: object|null } }} options
233
+ */
234
+ export function enforceSubAgentEffortPayload(body, { model, protocol, effortContext = {}, effortConstraint } = {}) {
235
+ if (!effortConstraint) return captureEffortDecision({ body, model, protocol, effortContext });
236
+ const decision = resolveSubAgentEffort({ parentDecision: effortConstraint.parentDecision, model, protocol, effortContext });
237
+ if (decision.wireMode === 'unsupported') {
238
+ removeEffortField(body, 'reasoning');
239
+ removeEffortField(body, 'output_config');
240
+ delete body.thinking;
241
+ return decision;
242
+ }
243
+ const context = capabilityContext(protocol, effortContext);
244
+ const options = getModelEffortOptions(model, context);
245
+ const wireEffort = protocol === 'openai-responses' ? body.reasoning?.effort : body.output_config?.effort;
246
+ const effective = options.includes(wireEffort) && atMost(wireEffort, decision.effective)
247
+ ? wireEffort : decision.effective;
248
+ if (protocol === 'openai-responses') {
249
+ body.reasoning = { ...(body.reasoning && typeof body.reasoning === 'object' && !Array.isArray(body.reasoning) ? body.reasoning : {}), effort: effective };
250
+ delete body.thinking;
251
+ removeEffortField(body, 'output_config');
252
+ } else if (decision.wireMode === 'adaptive') {
253
+ body.thinking = { type: 'adaptive' };
254
+ body.output_config = { ...(body.output_config && typeof body.output_config === 'object' && !Array.isArray(body.output_config) ? body.output_config : {}), effort: effective };
255
+ removeEffortField(body, 'reasoning');
256
+ } else {
257
+ const capability = getThinkingCapability(model, context);
258
+ let budget = thinkingBudgetForEffort(model, effective);
259
+ if (!budget) throw effortError(model, decision.cap, 'no manual thinking budget');
260
+ if (Number.isFinite(capability.maxBudgetTokens)) budget = Math.min(budget, capability.maxBudgetTokens);
261
+ const supplied = body.thinking?.type === 'enabled' ? body.thinking.budget_tokens : null;
262
+ if (Number.isInteger(supplied) && supplied >= 1024) budget = Math.min(budget, supplied);
263
+ if (Number.isFinite(body.max_tokens)) budget = Math.min(budget, Math.floor(body.max_tokens) - 1);
264
+ if (budget < 1024) throw effortError(model, decision.cap, 'max_tokens must allow at least 1024 thinking tokens');
265
+ body.thinking = { type: 'enabled', budget_tokens: budget };
266
+ removeEffortField(body, 'output_config');
267
+ removeEffortField(body, 'reasoning');
268
+ return snapshotEffortDecision({ ...decision, effective: manualEffortForBudget(model, budget), budgetTokens: budget });
269
+ }
270
+ return snapshotEffortDecision({ ...decision, effective });
271
+ }
package/yeaft/engine.js CHANGED
@@ -44,7 +44,8 @@ import { perfNowMs, recordAgentPerfTrace } from './perf-trace.js';
44
44
  // Default thread marker for legacy / non-group flows. Group VP runtime may
45
45
  // pass a real threadId per (sessionId, vpId, threadId) engine instance.
46
46
  const MAIN_THREAD_ID = 'main';
47
- import { pickEffort, parseEffortPrefix } from './effort.js';
47
+ import { pickEffort, parseEffortPrefix, snapshotEffortDecision } from './effort.js';
48
+ import { bindProviderState } from './llm/provider-state.js';
48
49
  import { DEFAULT_CONTEXT_WINDOW, normalizeEffort, resolveContextWindow, resolveModel } from './models.js';
49
50
  import { lookupModelLimitSync } from './llm/models-dev.js';
50
51
  import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
@@ -92,7 +93,8 @@ const MAX_CONTINUE_TURNS = 3;
92
93
  * network, or subprocess reads. Only tools whose metadata explicitly declares
93
94
  * both read-only and concurrency-safe execution enter this lane.
94
95
  */
95
- const MAX_CONCURRENT_READ_ONLY_TOOLS = 4;
96
+ // Safe shared tools run together within the finite provider batch. Unsafe tools
97
+ // form exclusive barriers; resource-specific limits belong to the owning tool.
96
98
 
97
99
  /** Maximum silence while a visible turn waits for a result-producing task. */
98
100
  const DEFAULT_ASYNC_TASK_WAIT_TIMEOUT_MS = 120_000;
@@ -1392,6 +1394,8 @@ export class Engine {
1392
1394
  #buildToolContext(signal, vpCtx) {
1393
1395
  return {
1394
1396
  signal,
1397
+ effortDecision: snapshotEffortDecision(vpCtx?.effortDecision),
1398
+ requestIdentity: vpCtx?.requestIdentity,
1395
1399
  yeaftDir: this.#yeaftDir,
1396
1400
  managedCliReady: this.#managedCliReady,
1397
1401
  runtimePlatform: getRuntimePlatformInfo(),
@@ -1591,6 +1595,7 @@ export class Engine {
1591
1595
  if (message.toolCallId) record.toolCallId = message.toolCallId;
1592
1596
  if (Array.isArray(message.toolCalls) && message.toolCalls.length > 0) record.toolCalls = message.toolCalls;
1593
1597
  if (Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0) record.thinkingBlocks = message.thinkingBlocks;
1598
+ if (message.providerState) record.providerState = message.providerState;
1594
1599
  if (message.isError) record.isError = true;
1595
1600
  if (message.imageAssetAnchor) record.imageAssetAnchor = true;
1596
1601
  if (message._reflection) record._reflection = true;
@@ -1625,7 +1630,7 @@ export class Engine {
1625
1630
  : message.content != null;
1626
1631
  const hasToolCalls = Array.isArray(message.toolCalls) && message.toolCalls.length > 0;
1627
1632
  const hasThinking = Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0;
1628
- if (!hasContent && !hasToolCalls && !hasThinking && message.role !== 'tool') return null;
1633
+ if (!hasContent && !hasToolCalls && !hasThinking && !message.providerState && message.role !== 'tool') return null;
1629
1634
  return this.#conversationStore.append(this.#conversationRecord(message, context));
1630
1635
  }
1631
1636
 
@@ -1877,7 +1882,7 @@ export class Engine {
1877
1882
  }
1878
1883
  }
1879
1884
 
1880
- async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
1885
+ async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
1881
1886
  if (!prompt || typeof prompt !== 'string' || !prompt.trim()) {
1882
1887
  const error = new Error('prompt is required and must be a non-empty string');
1883
1888
  yield {
@@ -1969,7 +1974,7 @@ export class Engine {
1969
1974
  try {
1970
1975
  this.#currentThreadId = threadId || MAIN_THREAD_ID;
1971
1976
  this.#currentCausalRootId = effectiveCausalRootId;
1972
- yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
1977
+ yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
1973
1978
  } finally {
1974
1979
  // Closing the async generator at a visible retry boundary means the
1975
1980
  // continuation never reached a provider. Keep it out of history and
@@ -2024,7 +2029,7 @@ export class Engine {
2024
2029
  * in a try/finally without indenting the whole loop.
2025
2030
  * @private
2026
2031
  */
2027
- async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
2032
+ async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
2028
2033
 
2029
2034
  const effectiveCollabToolPolicy = collabToolPolicy === COLLAB_TOOL_POLICY.SINGLE_VP || collabToolPolicy === COLLAB_TOOL_POLICY.MULTI_VP
2030
2035
  ? collabToolPolicy
@@ -2049,6 +2054,15 @@ export class Engine {
2049
2054
  // standalone/CLI callers pass it per query.
2050
2055
  this.#sessionId = runtimeSessionId || null;
2051
2056
  this.#currentThreadId = runtimeThreadId;
2057
+ const requestIdentity = Object.freeze({
2058
+ instanceScope: this.#yeaftDir || '',
2059
+ ownerScope: this.#yeaftDir ? 'instance-local-owner' : '',
2060
+ sessionId: runtimeSessionId || this.#chatId || '',
2061
+ vpId: this.#vpId || senderVpId || vpPersona?.vpId || 'default',
2062
+ threadId: runtimeThreadId,
2063
+ });
2064
+ const effortConstraint = isSubAgent || scenario === 'sub_agent' || vpPersona?.subAgent
2065
+ ? Object.freeze({ parentDecision: snapshotEffortDecision(parentEffortDecision) }) : null;
2052
2066
  const queryStartedAt = Date.now();
2053
2067
  const userQuestionPreview = String(prompt || '').slice(0, 200);
2054
2068
  const queryVpId = vpPersona && typeof vpPersona === 'object'
@@ -2276,7 +2290,7 @@ export class Engine {
2276
2290
  envelope: inboundEnvelope || null,
2277
2291
  };
2278
2292
 
2279
- const projectDocSource = this.#getProjectDocBlock(workDir);
2293
+ let projectDocSource = this.#getProjectDocBlock(workDir);
2280
2294
  let projectDocLoadedPathHints = [];
2281
2295
  let projectDocContext = selectProjectDocContext(projectDocSource, {
2282
2296
  prompt,
@@ -2675,7 +2689,9 @@ export class Engine {
2675
2689
  });
2676
2690
  };
2677
2691
  const toolCalls = [];
2678
- const thinkingBlocks = []; // task-327d: collected from adapter for round-trip
2692
+ const thinkingBlocks = []; // Legacy adapter compatibility only.
2693
+ let providerState = null;
2694
+ let requestEffortDecision = snapshotEffortDecision();
2679
2695
  let stopReason = 'end_turn';
2680
2696
  const totalUsage = { inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0, cacheInputDeltaTokens: 0 };
2681
2697
  // Raw provider exchange is diagnostic source data, not model context or
@@ -2796,7 +2812,7 @@ export class Engine {
2796
2812
  // routerPlan, the VP's role default, and the global config all
2797
2813
  // outrank the scenario picker for `'high'|'max'`. UI/userEffort
2798
2814
  // is already honoured by pickEffort (highest precedence).
2799
- if (vpPersona && vpPersona.vpId) {
2815
+ if (!requestUserEffort && vpPersona && vpPersona.vpId) {
2800
2816
  const priorPlan = extractPriorPlan(conversationMessages, vpPersona.vpId);
2801
2817
  const thinkingCfg = (this.#config && this.#config.thinking) || {};
2802
2818
  // PR-I: live routerPlan.thinking — when the dispatcher passes
@@ -2969,6 +2985,10 @@ export class Engine {
2969
2985
  tools: toolDefs.length > 0 ? toolDefs : undefined,
2970
2986
  maxTokens: requestConfig.maxOutputTokens || 16384,
2971
2987
  effort: resolvedEffort,
2988
+ effortConstraint,
2989
+ requestIdentity,
2990
+ onEffortDecision: decision => { requestEffortDecision = snapshotEffortDecision(decision); },
2991
+ onProviderDiagnostics: info => traceRequest('llm.capabilities', { detail: info }),
2972
2992
  effortSource: requestUserEffort ? 'user' : 'auto',
2973
2993
  signal,
2974
2994
  onRawExchange: captureRawExchange,
@@ -3029,6 +3049,9 @@ export class Engine {
3029
3049
  responseText += event.text;
3030
3050
  yield event;
3031
3051
  break;
3052
+ case 'provider_state':
3053
+ providerState = event.providerState;
3054
+ break;
3032
3055
  case 'thinking_delta':
3033
3056
  yield event;
3034
3057
  break;
@@ -3066,6 +3089,9 @@ export class Engine {
3066
3089
  const cacheInputDeltaTokens = event.cacheTokensAreIncludedInInput ? 0 : cacheReadTokens + cacheWriteTokens;
3067
3090
  totalUsage.inputTokens += inputTokens;
3068
3091
  totalUsage.outputTokens += outputTokens;
3092
+ if (Number.isFinite(event.reasoningTokens) && event.reasoningTokens >= 0) {
3093
+ totalUsage.reasoningTokens = (totalUsage.reasoningTokens || 0) + event.reasoningTokens;
3094
+ }
3069
3095
  totalUsage.cacheReadTokens += cacheReadTokens;
3070
3096
  totalUsage.cacheWriteTokens += cacheWriteTokens;
3071
3097
  totalUsage.cacheInputDeltaTokens += cacheInputDeltaTokens;
@@ -3162,6 +3188,7 @@ export class Engine {
3162
3188
  usage: {
3163
3189
  inputTokens: totalUsage.inputTokens || 0,
3164
3190
  outputTokens: totalUsage.outputTokens || 0,
3191
+ ...(totalUsage.reasoningTokens !== undefined ? { reasoningTokens: totalUsage.reasoningTokens } : {}),
3165
3192
  cacheReadTokens: totalUsage.cacheReadTokens || 0,
3166
3193
  cacheWriteTokens: totalUsage.cacheWriteTokens || 0,
3167
3194
  totalInputTokens: (totalUsage.inputTokens || 0) + (totalUsage.cacheInputDeltaTokens || 0),
@@ -3408,6 +3435,7 @@ export class Engine {
3408
3435
  usage: {
3409
3436
  inputTokens: totalUsage.inputTokens || 0,
3410
3437
  outputTokens: totalUsage.outputTokens || 0,
3438
+ ...(totalUsage.reasoningTokens !== undefined ? { reasoningTokens: totalUsage.reasoningTokens } : {}),
3411
3439
  cacheReadTokens: totalUsage.cacheReadTokens || 0,
3412
3440
  cacheWriteTokens: totalUsage.cacheWriteTokens || 0,
3413
3441
  totalInputTokens: (totalUsage.inputTokens || 0) + (totalUsage.cacheInputDeltaTokens || 0),
@@ -3434,6 +3462,7 @@ export class Engine {
3434
3462
  usage: {
3435
3463
  inputTokens: totalUsage.inputTokens || 0,
3436
3464
  outputTokens: errLoopOutputTokens,
3465
+ ...(totalUsage.reasoningTokens !== undefined ? { reasoningTokens: totalUsage.reasoningTokens } : {}),
3437
3466
  cacheReadTokens: totalUsage.cacheReadTokens || 0,
3438
3467
  cacheWriteTokens: totalUsage.cacheWriteTokens || 0,
3439
3468
  totalInputTokens: errLoopInputTokens,
@@ -3511,6 +3540,7 @@ export class Engine {
3511
3540
  usage: {
3512
3541
  inputTokens: totalUsage.inputTokens || 0,
3513
3542
  outputTokens: totalUsage.outputTokens || 0,
3543
+ ...(totalUsage.reasoningTokens !== undefined ? { reasoningTokens: totalUsage.reasoningTokens } : {}),
3514
3544
  cacheReadTokens: totalUsage.cacheReadTokens || 0,
3515
3545
  cacheWriteTokens: totalUsage.cacheWriteTokens || 0,
3516
3546
  totalInputTokens: turnInputTokens,
@@ -3533,13 +3563,10 @@ export class Engine {
3533
3563
  input: tc.input,
3534
3564
  }));
3535
3565
  }
3536
- if (thinkingBlocks.length > 0) {
3537
- assistantMsg.thinkingBlocks = thinkingBlocks.map(tb => (
3538
- tb.redacted
3539
- ? { redacted: true, data: tb.data, signature: tb.signature }
3540
- : { thinking: tb.thinking, signature: tb.signature }
3541
- ));
3542
- }
3566
+ const boundProviderState = bindProviderState(providerState, assistantMsg);
3567
+ if (boundProviderState) assistantMsg.providerState = boundProviderState;
3568
+ // New private reasoning is persisted only with verified provider ownership;
3569
+ // never downgrade it to the origin-free legacy thinkingBlocks format.
3543
3570
  if (vpPersona && vpPersona.vpId) {
3544
3571
  const planForThisVp = (vpPlan && typeof vpPlan === 'object'
3545
3572
  && typeof vpPlan.vpId === 'string' && vpPlan.vpId === vpPersona.vpId)
@@ -3609,6 +3636,7 @@ export class Engine {
3609
3636
  usage: {
3610
3637
  inputTokens: totalUsage.inputTokens || 0,
3611
3638
  outputTokens: loopOutputTokens,
3639
+ ...(totalUsage.reasoningTokens !== undefined ? { reasoningTokens: totalUsage.reasoningTokens } : {}),
3612
3640
  cacheReadTokens: totalUsage.cacheReadTokens || 0,
3613
3641
  cacheWriteTokens: totalUsage.cacheWriteTokens || 0,
3614
3642
  totalInputTokens: loopInputTokens,
@@ -3914,6 +3942,8 @@ export class Engine {
3914
3942
  // We re-create the closure each iteration because endTurnRequested
3915
3943
  // is a per-query local (reset implicitly at the top of #runQuery).
3916
3944
  const toolCtx = this.#buildToolContext(signal, {
3945
+ effortDecision: snapshotEffortDecision(requestEffortDecision),
3946
+ requestIdentity,
3917
3947
  router,
3918
3948
  senderVpId,
3919
3949
  sessionId: runtimeSessionId,
@@ -4094,9 +4124,8 @@ export class Engine {
4094
4124
  && duplicatePolicyForCall(tc) !== 'suppress'
4095
4125
  && !mayMutateWorkspaceAfterReturn(this, tc.name, tc.input)) {
4096
4126
  const parallelCalls = [];
4097
- const segmentCacheKeys = new Set();
4098
4127
  for (let candidateIndex = toolCallIndex;
4099
- candidateIndex < toolCalls.length && parallelCalls.length < MAX_CONCURRENT_READ_ONLY_TOOLS;
4128
+ candidateIndex < toolCalls.length;
4100
4129
  candidateIndex += 1) {
4101
4130
  const candidate = toolCalls[candidateIndex];
4102
4131
  if (!toolAllowedForRequest(candidate)
@@ -4105,21 +4134,27 @@ export class Engine {
4105
4134
  || mayMutateWorkspaceAfterReturn(this, candidate.name, candidate.input)) break;
4106
4135
  const candidateKey = `${candidate.name}\u001f${argsHashOf(candidate.input)}`;
4107
4136
  const candidateCacheable = isCacheableTool(this, candidate.name, candidate.input);
4108
- // Keep identical cacheable reads on the serial commit path so the
4109
- // second call reuses the first result instead of duplicating I/O.
4110
- if (candidateCacheable
4111
- && (readOnlyToolResults.has(candidateKey) || segmentCacheKeys.has(candidateKey))) break;
4137
+ // Already committed reads reuse their result on the commit path.
4138
+ if (!readOnlyToolReuseDisabled && candidateCacheable
4139
+ && readOnlyToolResults.has(candidateKey)) break;
4112
4140
  parallelCalls.push(candidate);
4113
- if (candidateCacheable) segmentCacheKeys.add(candidateKey);
4114
4141
  }
4115
4142
 
4116
4143
  if (parallelCalls.length > 1) {
4117
4144
  const executions = [];
4145
+ const inFlightReads = new Map();
4118
4146
  for (const call of parallelCalls) {
4119
- if (signal?.aborted) {
4120
- abortedDuringTools = true;
4147
+ if (signal?.aborted || toolBatchBarrier) {
4148
+ if (signal?.aborted) abortedDuringTools = true;
4121
4149
  break;
4122
4150
  }
4151
+ const cacheKey = `${call.name}\u001f${argsHashOf(call.input)}`;
4152
+ const cacheable = !readOnlyToolReuseDisabled && isCacheableTool(this, call.name, call.input);
4153
+ const sharedExecution = cacheable ? inFlightReads.get(cacheKey) : null;
4154
+ if (sharedExecution) {
4155
+ executions.push(sharedExecution.then(result => ({ ...result, call, shared: true })));
4156
+ continue;
4157
+ }
4123
4158
  announcedParallelToolCalls.add(call.id);
4124
4159
  yield {
4125
4160
  type: 'tool_start',
@@ -4132,7 +4167,8 @@ export class Engine {
4132
4167
  abortedDuringTools = true;
4133
4168
  break;
4134
4169
  }
4135
- executions.push((async () => {
4170
+ if (toolBatchBarrier) break;
4171
+ const execution = (async () => {
4136
4172
  const startedAt = Date.now();
4137
4173
  const callContext = toolContextForCall(call);
4138
4174
  const toolErrorOutput = this.#toolRegistry
@@ -4146,7 +4182,9 @@ export class Engine {
4146
4182
  } catch (error) {
4147
4183
  return { call, startedAt, durationMs: Date.now() - startedAt, error, toolErrorOutput };
4148
4184
  }
4149
- })());
4185
+ })();
4186
+ executions.push(execution);
4187
+ if (cacheable) inFlightReads.set(cacheKey, execution);
4150
4188
  }
4151
4189
  const completed = await Promise.all(executions);
4152
4190
  for (const execution of completed) {
@@ -4203,7 +4241,12 @@ export class Engine {
4203
4241
  const missingProjectDocScopes = hasTool && !readOnlyTool
4204
4242
  ? projectDocWriteScopesNeedingReload(projectDocContext, toolProjectDocPathHints)
4205
4243
  : new Set();
4206
- const needsProjectDocReload = missingProjectDocScopes.size > 0;
4244
+ // A changed rule source was not present in the request that generated
4245
+ // this write, even when its scope label is unchanged. Never replay it.
4246
+ const freshProjectDocSource = hasTool && !readOnlyTool
4247
+ ? this.#getProjectDocBlock(workDir) : projectDocSource;
4248
+ const projectDocSourceChanged = freshProjectDocSource !== projectDocSource;
4249
+ const needsProjectDocReload = missingProjectDocScopes.size > 0 || projectDocSourceChanged;
4207
4250
 
4208
4251
  if (abortSkipped) {
4209
4252
  output = `Skipped ${tc.name} because the turn was aborted before this tool started.`;
@@ -4246,6 +4289,7 @@ export class Engine {
4246
4289
  isError = true;
4247
4290
  yield { type: 'tool_end', id: tc.id, name: tc.name, output, isError: true, threadId: this.currentThreadId };
4248
4291
  } else if (needsProjectDocReload) {
4292
+ projectDocSource = freshProjectDocSource;
4249
4293
  projectDocLoadedPathHints = [...new Set([
4250
4294
  ...projectDocLoadedPathHints,
4251
4295
  ...toolProjectDocPathHints,
@@ -4392,6 +4436,15 @@ export class Engine {
4392
4436
 
4393
4437
  currentToolCallForAsyncTask = null;
4394
4438
 
4439
+ // Record concrete inputs, never tool/file prose. Apply only after the
4440
+ // whole batch: a read cannot authorize a same-response write whose
4441
+ // model has not yet received the newly selected rules.
4442
+ if (hasTool && readOnlyTool && !skipped && !isError) {
4443
+ projectDocLoadedPathHints = [...new Set([
4444
+ ...projectDocLoadedPathHints, ...toolProjectDocPathHints,
4445
+ ])].slice(-128);
4446
+ }
4447
+
4395
4448
  if (!skipped && !duplicateCallSuppressed && !isError && !reusedReadOnlyResult
4396
4449
  && duplicateCallPolicy !== 'allow') {
4397
4450
  const nextDuplicateCount = successfulDuplicateCount + 1;
@@ -4519,6 +4572,19 @@ export class Engine {
4519
4572
  if (fatalToolError) throw fatalToolError;
4520
4573
  }
4521
4574
 
4575
+ // Preload rules for observed read paths before the next provider input.
4576
+ // No tool from the just-completed response can benefit retroactively.
4577
+ projectDocSource = this.#getProjectDocBlock(workDir);
4578
+ const nextProjectDocContext = selectProjectDocContext(projectDocSource, {
4579
+ prompt, messages, pathHints: projectDocLoadedPathHints,
4580
+ forcedScopes: [...projectDocContext.selectedScopes],
4581
+ language: this.#config.language || 'en',
4582
+ });
4583
+ if (nextProjectDocContext.text !== projectDocContext.text) {
4584
+ projectDocContext = nextProjectDocContext;
4585
+ systemPrompt = buildCurrentSystemPrompt();
4586
+ }
4587
+
4522
4588
  // PR-L: flush any duplicate-call reminders queued during the batch.
4523
4589
  // Pushed AFTER the for-loop so the tool_use → tool_result pairing
4524
4590
  // is intact; the next adapter.stream() will see the reminder as a