@yeaft/webchat-agent 1.0.454 → 1.0.456

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":"1.0.454"}
1
+ {"version":"1.0.456"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yeaft/webchat-agent",
3
- "version": "1.0.454",
3
+ "version": "1.0.456",
4
4
  "description": "Remote worker agent for Yeaft Web Code Agent — connects the native Yeaft engine, CLI providers, and workbench tools",
5
5
  "main": "index.js",
6
6
  "type": "module",
@@ -1025,6 +1025,7 @@ function traceToolToLegacy(trace, tool) {
1025
1025
  tool_call_id: tool.toolCallId || null,
1026
1026
  duration_ms: tool.durationMs || 0,
1027
1027
  is_error: tool.isError ? 1 : 0,
1028
+ suppressed: !!tool.suppressed,
1028
1029
  created_at: tool.createdAt || trace.openedAt || 0,
1029
1030
  };
1030
1031
  }
@@ -1305,7 +1306,15 @@ export class DebugTrace {
1305
1306
  this.#appendTraceRecord(trace, 'loop', loop, { writeMeta: !trace.active });
1306
1307
  }
1307
1308
 
1308
- logTool(turnId, { toolName, toolCallId = null, toolInput = null, toolOutput = null, durationMs = null, isError = false } = {}) {
1309
+ logTool(turnId, {
1310
+ toolName,
1311
+ toolCallId = null,
1312
+ toolInput = null,
1313
+ toolOutput = null,
1314
+ durationMs = null,
1315
+ isError = false,
1316
+ suppressed = false,
1317
+ } = {}) {
1309
1318
  const id = randomUUID();
1310
1319
  const ctx = this.#turnIndex.get(turnId);
1311
1320
  if (!ctx) return id;
@@ -1322,6 +1331,7 @@ export class DebugTrace {
1322
1331
  toolOutput: truncateText(toolOutput == null ? null : String(toolOutput), this.#textMaxBytes),
1323
1332
  durationMs: Number(durationMs || 0),
1324
1333
  isError: !!isError,
1334
+ suppressed: !!suppressed,
1325
1335
  createdAt: Date.now(),
1326
1336
  };
1327
1337
  trace.tools.push(tool);
package/yeaft/engine.js CHANGED
@@ -2420,6 +2420,11 @@ export class Engine {
2420
2420
  let lastT1AtToolCount = 0;
2421
2421
  let arcStartIdx = turnStartIdx + 1;
2422
2422
  let t1CollapsesDone = 0;
2423
+ // Duplicate policy is scoped to one user query. Only successful, real
2424
+ // executions increment these counters; errors and cache reuse do not.
2425
+ const queryDuplicateCounts = new Map();
2426
+ const queryDuplicateSuppressions = new Map();
2427
+ let duplicateReminderAwaitingResponse = false;
2423
2428
  const queryNumber = (this.#__queryCounter = (this.#__queryCounter || 0) + 1);
2424
2429
 
2425
2430
  // feat-6af5f9f1 PR B: a Turn = one user prompt + all AI responses.
@@ -2894,7 +2899,7 @@ export class Engine {
2894
2899
  commitDispatch();
2895
2900
  },
2896
2901
  });
2897
- yield { type: 'turn_start', turnNumber, threadId };
2902
+ yield { type: 'turn_start', turnId: queryTurnId, turnNumber, threadId };
2898
2903
 
2899
2904
  // Provider iteration begins after the visible boundary. Native adapters
2900
2905
  // commit in onRequestStart immediately before fetch. A plain legacy
@@ -3546,6 +3551,23 @@ export class Engine {
3546
3551
  fullResponseText += responseText;
3547
3552
 
3548
3553
  // ─── Handle max_tokens → auto-continue ────────────
3554
+ // A suppressed call leaves a synthetic reminder as the latest user
3555
+ // message. Some models answer it with an empty end_turn. Continue exactly
3556
+ // once so suppression cannot silently abandon the user's task.
3557
+ if (duplicateReminderAwaitingResponse) {
3558
+ if (responseText.trim() || toolCalls.length > 0) {
3559
+ duplicateReminderAwaitingResponse = false;
3560
+ } else {
3561
+ duplicateReminderAwaitingResponse = false;
3562
+ conversationMessages.push({
3563
+ role: 'user',
3564
+ content: '[system note] The duplicate tool call was suppressed, but the current user task is still active. Continue toward the requested outcome using the prior result or a different action; do not end the turn solely because the duplicate was blocked.',
3565
+ });
3566
+ yield { type: 'turn_end', turnNumber, stopReason: 'duplicate_tool_continue', threadId };
3567
+ continue;
3568
+ }
3569
+ }
3570
+
3549
3571
  if (stopReason === 'max_tokens' && continueTurns < MAX_CONTINUE_TURNS) {
3550
3572
  continueTurns++;
3551
3573
  // This synthetic continuation is part of the model-visible protocol.
@@ -3922,6 +3944,16 @@ export class Engine {
3922
3944
  let abortedDuringTools = false;
3923
3945
  /** @type {string[]} */
3924
3946
  const pendingDupReminders = [];
3947
+ let terminateAfterDuplicateBatch = false;
3948
+ const duplicatePolicyForCall = (toolCall) => {
3949
+ const toolDef = this.#toolRegistry
3950
+ ? this.#toolRegistry.get(toolCall.name)
3951
+ : this.#tools.get(toolCall.name);
3952
+ const requested = typeof toolDef?.duplicateCallPolicy === 'function'
3953
+ ? toolDef.duplicateCallPolicy(toolCall.input)
3954
+ : 'warn';
3955
+ return ['allow', 'warn', 'suppress'].includes(requested) ? requested : 'warn';
3956
+ };
3925
3957
  /**
3926
3958
  * Completed executions waiting for their original-order commit. Starting
3927
3959
  * a bounded read-only segment together removes wall-clock latency without
@@ -3977,6 +4009,7 @@ export class Engine {
3977
4009
 
3978
4010
  if (!preparedParallelExecution && !toolBatchBarrier && !signal?.aborted
3979
4011
  && toolAllowedForRequest(tc) && isConcurrencySafeTool(this, tc.name, tc.input)
4012
+ && duplicatePolicyForCall(tc) !== 'suppress'
3980
4013
  && !mayMutateWorkspaceAfterReturn(this, tc.name, tc.input)) {
3981
4014
  const parallelCalls = [];
3982
4015
  const segmentCacheKeys = new Set();
@@ -3986,6 +4019,7 @@ export class Engine {
3986
4019
  const candidate = toolCalls[candidateIndex];
3987
4020
  if (!toolAllowedForRequest(candidate)
3988
4021
  || !isConcurrencySafeTool(this, candidate.name, candidate.input)
4022
+ || duplicatePolicyForCall(candidate) === 'suppress'
3989
4023
  || mayMutateWorkspaceAfterReturn(this, candidate.name, candidate.input)) break;
3990
4024
  const candidateKey = `${candidate.name}\u001f${argsHashOf(candidate.input)}`;
3991
4025
  const candidateCacheable = isCacheableTool(this, candidate.name, candidate.input);
@@ -4048,36 +4082,14 @@ export class Engine {
4048
4082
  || (activeToolBatchBarrier != null && !readyParallelExecution);
4049
4083
  const toolStartTime = readyParallelExecution?.startedAt || Date.now();
4050
4084
 
4051
- // PR-L: duplicate-call detection. If this exact (toolName,
4052
- // argsHash) pair has already been executed DUP_TOOL_THRESHOLD
4053
- // (3) times within the current turn + last 2 turns, queue a
4054
- // system reminder. We push the reminder AFTER the tool batch
4055
- // completes (not now) so the
4056
- // assistant(tool_use) → user(tool_result, …) pairing demanded
4057
- // by the Anthropic / OpenAI Responses APIs stays intact. We
4058
- // don't block the call — the LLM still decides.
4059
- if (!skipped) {
4060
- const dupHash = argsHashOf(tc.input);
4061
- // PR-L follow-up: lookback is by user-conversation turn
4062
- // (`queryNumber`), NOT by inner adapter loop iteration. Each call
4063
- // to query() bumps queryNumber once, so "last 2 turns" means the
4064
- // current user turn + the previous two user turns — the natural
4065
- // semantic for "the model is stuck in a loop across the
4066
- // conversation."
4067
- const dupInfo = this.#execLog.dupInfo({
4068
- toolName: tc.name,
4069
- argsHash: dupHash,
4070
- currentTurn: queryNumber,
4071
- lookbackTurns: 2,
4072
- });
4073
- if (dupInfo.count + 1 >= DUP_TOOL_THRESHOLD) {
4074
- pendingDupReminders.push(buildDuplicateReminder({
4075
- toolName: tc.name,
4076
- count: dupInfo.count + 1,
4077
- lastResultBrief: dupInfo.lastResultBrief,
4078
- }));
4079
- }
4080
- }
4085
+ const dupHash = argsHashOf(tc.input);
4086
+ const duplicateCallKey = `${tc.name}:${dupHash}`;
4087
+ const duplicateCallPolicy = duplicatePolicyForCall(tc);
4088
+ const successfulDuplicateCount = queryDuplicateCounts.get(duplicateCallKey) || 0;
4089
+ const duplicateCallSuppressed = !skipped
4090
+ && duplicateCallPolicy === 'suppress'
4091
+ && successfulDuplicateCount >= DUP_TOOL_THRESHOLD - 1;
4092
+ let suppressionCount = queryDuplicateSuppressions.get(duplicateCallKey) || 0;
4081
4093
 
4082
4094
  let output;
4083
4095
  let displayImages = [];
@@ -4179,6 +4191,24 @@ export class Engine {
4179
4191
  sourceToolName: tc.name,
4180
4192
  };
4181
4193
  yield { type: 'tool_end', id: tc.id, name: tc.name, output, isError: true, skipped: true, threadId: this.currentThreadId };
4194
+ } else if (duplicateCallSuppressed) {
4195
+ suppressionCount += 1;
4196
+ queryDuplicateSuppressions.set(duplicateCallKey, suppressionCount);
4197
+ const dupInfo = this.#execLog.dupInfo({
4198
+ toolName: tc.name,
4199
+ argsHash: dupHash,
4200
+ currentTurn: queryNumber,
4201
+ lookbackTurns: 0,
4202
+ });
4203
+ pendingDupReminders.push(buildDuplicateReminder({
4204
+ toolName: tc.name,
4205
+ count: successfulDuplicateCount + suppressionCount,
4206
+ lastResultBrief: dupInfo.lastResultBrief,
4207
+ }));
4208
+ if (suppressionCount >= DUP_TOOL_THRESHOLD) terminateAfterDuplicateBatch = true;
4209
+ output = `Suppressed duplicate call: ${tc.name} returned a stable result twice with these arguments in this turn. Reuse the previous result or choose a different action.`;
4210
+ yield { type: 'tool_start', id: tc.id, name: tc.name, input: tc.input, threadId: this.currentThreadId };
4211
+ yield { type: 'tool_end', id: tc.id, name: tc.name, output, isError: false, suppressed: true, threadId: this.currentThreadId };
4182
4212
  } else {
4183
4213
  duplicateKey = `${tc.name}\u001f${argsHashOf(tc.input)}`;
4184
4214
  cacheableTool = isCacheableTool(this, tc.name, tc.input);
@@ -4280,6 +4310,24 @@ export class Engine {
4280
4310
 
4281
4311
  currentToolCallForAsyncTask = null;
4282
4312
 
4313
+ if (!skipped && !duplicateCallSuppressed && !isError && !reusedReadOnlyResult
4314
+ && duplicateCallPolicy !== 'allow') {
4315
+ const nextDuplicateCount = successfulDuplicateCount + 1;
4316
+ queryDuplicateCounts.set(duplicateCallKey, nextDuplicateCount);
4317
+ if (duplicateCallPolicy === 'warn' && nextDuplicateCount === DUP_TOOL_THRESHOLD) {
4318
+ const dupInfo = this.#execLog.dupInfo({
4319
+ toolName: tc.name,
4320
+ argsHash: dupHash,
4321
+ currentTurn: queryNumber,
4322
+ lookbackTurns: 0,
4323
+ });
4324
+ pendingDupReminders.push(buildDuplicateReminder({
4325
+ toolName: tc.name,
4326
+ count: nextDuplicateCount,
4327
+ lastResultBrief: dupInfo.lastResultBrief,
4328
+ }));
4329
+ }
4330
+ }
4283
4331
  const toolDurationMs = readyParallelExecution?.durationMs ?? (Date.now() - toolStartTime);
4284
4332
 
4285
4333
  // feat-6af5f9f1 PR B: emit a structured `tool_exec` event for the
@@ -4294,6 +4342,7 @@ export class Engine {
4294
4342
  name: tc.name,
4295
4343
  durationMs: toolDurationMs,
4296
4344
  isError,
4345
+ suppressed: duplicateCallSuppressed,
4297
4346
  toolOutput: output,
4298
4347
  ...(reusedReadOnlyResult ? { reused: true, reusedCallId: reusedReadOnlyCallId } : {}),
4299
4348
  ...(skipped ? { skipped: true } : {}),
@@ -4303,7 +4352,8 @@ export class Engine {
4303
4352
  // 2026-05-13: feed the per-tool counters. Stays best-effort — a
4304
4353
  // stats sink that throws shouldn't crash the engine. `record`
4305
4354
  // already swallows internal write errors.
4306
- if (!skipped && this.#toolStats && typeof this.#toolStats.record === 'function') {
4355
+ if (!skipped && !duplicateCallSuppressed
4356
+ && this.#toolStats && typeof this.#toolStats.record === 'function') {
4307
4357
  try {
4308
4358
  this.#toolStats.record({
4309
4359
  name: tc.name,
@@ -4322,15 +4372,17 @@ export class Engine {
4322
4372
  toolOutput: output,
4323
4373
  durationMs: toolDurationMs,
4324
4374
  isError,
4375
+ suppressed: duplicateCallSuppressed,
4325
4376
  skipped,
4326
4377
  reused: reusedReadOnlyResult,
4327
4378
  reusedCallId: reusedReadOnlyCallId,
4328
4379
  });
4329
4380
 
4330
- if (!skipped && !reusedReadOnlyResult && tc.name === 'StartPlan') {
4381
+ if (!skipped && !duplicateCallSuppressed && !reusedReadOnlyResult && tc.name === 'StartPlan') {
4331
4382
  planBootstrapPending = true;
4332
4383
  }
4333
- if (!skipped && !reusedReadOnlyResult && !readOnlyToolReuseDisabled && cacheableTool) {
4384
+ if (!skipped && !duplicateCallSuppressed && !reusedReadOnlyResult
4385
+ && !readOnlyToolReuseDisabled && cacheableTool) {
4334
4386
  readOnlyToolResults.set(duplicateKey, {
4335
4387
  output,
4336
4388
  isError,
@@ -4372,7 +4424,7 @@ export class Engine {
4372
4424
  // user-conversation turn), not the inner loop's turnNumber.
4373
4425
  // Aligns exec-log layout with dup detection lookback and the
4374
4426
  // T2 fallback-stub readTurn() call below.
4375
- if (!skipped) {
4427
+ if (!skipped && !duplicateCallSuppressed) {
4376
4428
  this.#execLog.append(queryNumber, buildExecLogEntry({
4377
4429
  loopIdx: queryToolCount,
4378
4430
  toolName: tc.name,
@@ -4392,6 +4444,23 @@ export class Engine {
4392
4444
  for (const reminder of pendingDupReminders) {
4393
4445
  conversationMessages.push({ role: 'user', content: reminder });
4394
4446
  }
4447
+ if (pendingDupReminders.length > 0) duplicateReminderAwaitingResponse = true;
4448
+ if (terminateAfterDuplicateBatch) {
4449
+ yield {
4450
+ type: 'error',
4451
+ error: 'Terminated repeated duplicate tool calls after bounded suppression attempts.',
4452
+ code: 'duplicate_tool_loop',
4453
+ retryable: false,
4454
+ };
4455
+ yield {
4456
+ type: 'turn_end',
4457
+ turnNumber,
4458
+ stopReason: 'duplicate_tool_loop',
4459
+ threadId,
4460
+ terminal: true,
4461
+ };
4462
+ break;
4463
+ }
4395
4464
 
4396
4465
  // A plan bootstrap that produced only a TodoWrite has no executable work
4397
4466
  // to feed back to the provider. Close it here. This is intentionally
package/yeaft/personas.js CHANGED
@@ -37,12 +37,12 @@ let cached = null;
37
37
  * @returns {{ meta: object, body: string }}
38
38
  */
39
39
  export function parseFrontmatter(source) {
40
- const match = source.match(/^---\n([\s\S]*?)\n---\n?([\s\S]*)$/);
40
+ const match = source.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/);
41
41
  if (!match) return { meta: {}, body: source };
42
42
 
43
43
  const [, yaml, body] = match;
44
44
  const meta = {};
45
- const lines = yaml.split('\n');
45
+ const lines = yaml.split(/\r?\n/);
46
46
  let currentKey = null;
47
47
  let currentList = null;
48
48
 
package/yeaft/prompts.js CHANGED
@@ -462,8 +462,8 @@ const TOOL_GUIDANCE_GROUPS = Object.freeze([
462
462
  },
463
463
  {
464
464
  tools: ['FileRead', 'FileWrite', 'FileEdit', 'Glob', 'Grep', 'ListDir', 'ApplyPatch', 'NotebookEdit'],
465
- en: 'Read existing files before editing. Use dedicated file/search tools instead of shell search or `sed -i`; make small, reviewable edits and batch independent reads.',
466
- zh: '编辑前先读现有文件。文件搜索和修改优先使用专用工具,不用 shell 搜索或 `sed -i`;改动保持小而可审查,独立读取应并行发出。',
465
+ en: 'Read existing files before editing. Use dedicated file/search tools instead of shell search or `sed -i`; make small, reviewable edits. Parallelize only proven-independent reads under the accuracy-first rule above.',
466
+ zh: '编辑前先读现有文件。文件搜索和修改优先使用专用工具,不用 shell 搜索或 `sed -i`;改动保持小而可审查。只有明确满足上述准确性优先判据的读取才可并行。',
467
467
  },
468
468
  {
469
469
  tools: ['Bash'],
@@ -472,13 +472,13 @@ const TOOL_GUIDANCE_GROUPS = Object.freeze([
472
472
  },
473
473
  {
474
474
  tools: ['TodoWrite'],
475
- en: 'For non-trivial multi-step work, write a brief visible plan and call `TodoWrite` in the same assistant response as the first independent work tools. Do not spend a separate model round entering planning mode, and do not stop after planning unless user input genuinely blocks the first step.',
476
- zh: '非平凡多步骤任务先写简短可见计划,并在同一个 assistant response 中把 `TodoWrite` 与第一批独立工作工具一起发出。不要用单独的模型回合进入规划模式;只有用户信息确实阻塞第一步时才在规划后停下。',
475
+ en: 'For non-trivial multi-step work, write a brief visible plan and call `TodoWrite` in the same assistant response as the first necessary work-tool call only when its arguments and safety do not depend on another result. Start with the smallest such call; do not speculative-batch the investigation or stop after planning unless user input genuinely blocks the first step.',
476
+ zh: '非平凡多步骤任务先写简短可见计划。只有第一个工作工具调用已经确定有必要,且其参数和安全性都不依赖其他结果时,才在同一个 assistant response 中把它与 `TodoWrite` 一起发出;先执行满足条件的最小调用,不要推测性批量展开调查。只有用户信息确实阻塞第一步时才在规划后停下。',
477
477
  },
478
478
  {
479
479
  tools: ['SpawnAgent', 'PromptAgent', 'WaitAgent', 'CloseAgent', 'ListAgents'],
480
- en: 'Delegate only independent, bounded work. Keep ownership in the parent, avoid polling loops, and close sub-agents after collecting their result.',
481
- zh: '只委派边界清晰且独立的工作。父级保留任务所有权,不要循环轮询,取得结果后关闭子 Agent。',
480
+ en: 'Delegate only independent, bounded work. Keep ownership in the parent. After PromptAgent queues follow-up work, call WaitAgent in the same parent turn and collect the reply before ending. If a bounded wait times out, wait again with a larger bound unless the agent is stale/stalled. Relay the reply or continue the dependent work, then close the sub-agent when it is no longer needed.',
481
+ zh: '只委派边界清晰且独立的工作。父级保留任务所有权。PromptAgent 排队后续工作后,必须在同一个父级 turn 调用 WaitAgent 并拿到回复再结束;有界等待超时后,除非 Agent 已 stale/stalled,否则使用更大上限继续等待;随后转述结果或继续依赖该结果的工作,不再需要时关闭子 Agent。',
482
482
  },
483
483
  {
484
484
  tools: ['ListTasks', 'ReadTaskLog', 'CancelTask'],
@@ -500,6 +500,11 @@ const TOOL_GUIDANCE_GROUPS = Object.freeze([
500
500
  function renderActiveToolGuidance(toolNames, language) {
501
501
  const active = new Set(Array.isArray(toolNames) ? toolNames : []);
502
502
  const lines = [];
503
+ if (active.size > 0) {
504
+ lines.push(language === 'zh'
505
+ ? '- 准确性优先:先用能解决当前未知的最小定向调用。只有每个调用都已经确定有必要,且其参数和安全性都不依赖同批其他结果时,才在一个响应中发出多个工具调用;否则串行执行。不要推测性扇出、重复成功的读取/搜索,也不要默认抓取多个来源;先检查证据,再决定是否扩展。'
506
+ : '- Accuracy first: start with the smallest targeted call that can resolve the current uncertainty. Issue multiple tool calls in one response only when every call is already necessary and its arguments and safety do not depend on another call\'s result. Otherwise run them sequentially. Do not fan out speculatively, repeat a successful read/search, or fetch multiple sources by default; inspect evidence before expanding.');
507
+ }
503
508
  for (const group of TOOL_GUIDANCE_GROUPS) {
504
509
  if (!group.tools.some(name => active.has(name))) continue;
505
510
  lines.push(`- ${language === 'zh' ? group.zh : group.en}`);
@@ -117,7 +117,14 @@ export function diagnoseAgentLiveness(agent, opts = {}) {
117
117
  : DEFAULT_STALL_THRESHOLD_MS;
118
118
  const liveness = snapshotLiveness(agent?.liveness, now);
119
119
  const fallbackAt = agent?.createdAt || agent?.usage?.startedAt || null;
120
- const activityAt = liveness.lastEventAt || fallbackAt;
120
+ // Queueing a PromptAgent continuation starts a new collection window. Do not
121
+ // diagnose that fresh follow-up as stale merely because the retained child
122
+ // last emitted an event during an older turn.
123
+ const activityAt = Math.max(
124
+ liveness.lastEventAt || 0,
125
+ fallbackAt || 0,
126
+ agent?.promptReplyPendingAt || 0,
127
+ ) || null;
121
128
  const msSinceActivity = activityAt ? Math.max(0, now - activityAt) : null;
122
129
  const stale = agent?.status === 'running'
123
130
  && msSinceActivity !== null
@@ -19,7 +19,7 @@ You are a fast, read-only **Explorer** sub-agent. Your job is to scout the codeb
19
19
  ## Operating Principles
20
20
 
21
21
  - **Read-only**: Never modify files, run bash, or spawn agents.
22
- - **Be fast**: Use `Grep` / `Glob` / `ListDir` to narrow the search, then `Read` only the needed ranges.
22
+ - **Be targeted**: Use one focused `Grep` / `Glob` / `ListDir` call to narrow the search, inspect it, then `Read` only the ranges still needed. Do not speculative-batch alternative searches.
23
23
  - **Be specific**: Return concrete file paths, line numbers, and short excerpts.
24
24
  - **Respect the contract**: Match your output to the `expected_output` schema exactly.
25
25
 
@@ -36,7 +36,7 @@ Structured. Bullet points. File paths as backticked references with `path:line`.
36
36
  ## 操作原则
37
37
 
38
38
  - **只读**:不要修改文件,不要运行 bash,不要派生 Agent。
39
- - **要快**:先用 `Grep` / `Glob` / `ListDir` 缩小范围,再只 `Read` 必要行段。
39
+ - **要定向**:先执行一个聚焦的 `Grep` / `Glob` / `ListDir` 调用缩小范围,检查结果后再只 `Read` 仍然必要的行段。不要推测性批量发出备选搜索。
40
40
  - **要具体**:返回明确的文件路径、行号和短摘录。
41
41
  - **遵守契约**:输出必须严格匹配 `expected_output` schema。
42
42
 
@@ -18,7 +18,7 @@ You are a **Researcher** sub-agent. Your job is to gather information from the w
18
18
  ## Operating Principles
19
19
 
20
20
  - **Cite sources**: Every factual claim must link back to a URL or doc path.
21
- - **Triangulate**: Prefer multiple independent sources over one.
21
+ - **Validate proportionally**: Start with the most authoritative relevant source. Add another source only when the claim is consequential, disputed, stale, or not established by the first source; do not fetch multiple sources by default.
22
22
  - **Summarize**: Return digest-form findings, not raw dumps.
23
23
  - **Track freshness**: Note publication dates when recency matters.
24
24
 
@@ -35,7 +35,7 @@ Short synthesis first, then bulleted sources with one-line summaries. No filler.
35
35
  ## 操作原则
36
36
 
37
37
  - **引用来源**:每个事实性判断都必须能回链到 URL 或文档路径。
38
- - **交叉验证**:优先使用多个独立来源,而不是只依赖一个来源。
38
+ - **按风险验证**:先查最权威、最相关的来源。只有结论影响重大、存在争议、可能过时,或首个来源不能证明时才增加来源;不要默认抓取多个来源。
39
39
  - **做综合**:返回摘要式发现,不要倾倒原始材料。
40
40
  - **关注时效**:当新旧会影响判断时,标明发布时间。
41
41
 
@@ -17,7 +17,7 @@ You have just entered **planning mode** for the topic below. Your job is to thin
17
17
 
18
18
  1. Write a short prose plan: problem, approach, risks.
19
19
  2. Call `TodoWrite` with the ordered steps. Mark exactly one item as `in_progress`.
20
- 3. If the first-step tools and arguments are already known, emit `TodoWrite` and those independent tool calls in the same assistant response. Otherwise wait only for a result that genuinely determines the next action.
20
+ 3. Emit `TodoWrite` with a first work-tool call in the same assistant response only when that call is already necessary and its arguments and safety do not depend on another result. Start with the smallest such call. If its result can change the next action, inspect it before issuing more calls; do not speculative-batch the investigation.
21
21
 
22
22
  If the first step is to ask the user a blocking question, ask it and stop. Otherwise keep moving.
23
23
 
@@ -40,6 +40,6 @@ If the first step is to ask the user a blocking question, ask it and stop. Other
40
40
 
41
41
  1. 写一段简短计划:问题、方案、风险。
42
42
  2. 调用 `TodoWrite` 写入有序步骤,并且只能把一个条目标记为 `in_progress`。
43
- 3. 如果第一步所需的工具和参数已经确定,应在同一个 assistant response 中发出 `TodoWrite` 和这些彼此独立的工具调用;否则只在确实需要某个结果来决定下一动作时等待。
43
+ 3. 只有第一个工作工具调用已经确定有必要,且其参数和安全性都不依赖其他结果时,才在同一个 assistant response 中把它与 `TodoWrite` 一起发出。先执行满足条件的最小调用;如果它的结果可能改变下一动作,应先检查结果,不要推测性批量展开调查。
44
44
 
45
45
  如果第一步是向用户询问阻塞问题,那就提问并停下。否则继续推进。
@@ -52,7 +52,7 @@ export const CONDITIONAL_BUILTIN_TOOL_NAMES = new Set([
52
52
  'RouteForward',
53
53
  'CreateWorkItem',
54
54
  // Legacy compatibility only. New planning is a single provider response:
55
- // visible prose + TodoWrite + the first independent work tools. Keeping the
55
+ // visible prose + TodoWrite + the first justified work-tool call. Keeping the
56
56
  // definition registered lets old direct callers resolve it without paying a
57
57
  // dedicated StartPlan -> provider -> TodoWrite round trip on every new task.
58
58
  'StartPlan',
@@ -239,11 +239,15 @@ Async orchestration:
239
239
  2. Continue — keep working in the parent VP; do not block just to poll.
240
240
  3. ListAgents — non-blocking status check when you need progress/liveness.
241
241
  4. PromptAgent — optional follow-up if the sub-agent is idle and needs guidance.
242
+ After queueing it, call WaitAgent in the same parent turn and collect the
243
+ reply before ending. If a bounded wait times out, wait again with a larger
244
+ bound unless the agent is stale/stalled.
242
245
  5. CloseAgent — stop or finalize a sub-agent when it is no longer needed.
243
246
 
244
247
  Completion/failure is delivered through sub-agent notifications on later parent
245
- turns. WaitAgent remains available only as a short compatibility poll; do not
246
- use it as the default workflow or call it repeatedly in a loop.`,
248
+ turns, so do not call WaitAgent merely to poll a newly spawned running agent.
249
+ After PromptAgent follow-up, however, the answer is required to complete that
250
+ workflow; use bounded WaitAgent calls and inspect liveness instead of blind loops.`,
247
251
  zh: `创建一个子 Agent 并行处理独立任务。
248
252
 
249
253
  子 Agent 在独立上下文中运行,可给定具体 mission 和可选的 expected_output schema。
@@ -263,11 +267,14 @@ use it as the default workflow or call it repeatedly in a loop.`,
263
267
  1. SpawnAgent — 启动子 Agent 作为后台任务并立即返回。
264
268
  2. Continue — 父 VP 继续工作;不要仅仅为了轮询而阻塞。
265
269
  3. ListAgents — 需要进度信息时的非阻塞状态检查。
266
- 4. WaitAgent — 短轮询(<5s)或仅在真正需要结果时才明确长等待。
267
- 5. CloseAgent — 完成后销毁子 Agent。
270
+ 4. PromptAgent — 子 Agent 空闲且需要指导时,可选发送后续提示。
271
+ 排队后必须在同一个父级 turn 调用 WaitAgent 并拿到回复再结束;有界等待超时后,除非 Agent
272
+ 已 stale/stalled,否则使用更大的有界 timeout 再次等待。
273
+ 5. CloseAgent — 不再需要时停止或结束子 Agent。
268
274
 
269
- 不要在循环中无终止条件地调用 WaitAgent。如果短检查后子 Agent 仍在运行,继续前进,
270
- 让 notification 在下个 turn 告知你。`
275
+ 完成或失败会通过之后父级 turn 的 notification 送达,因此不要为了轮询刚创建且仍运行的
276
+ Agent 调用 WaitAgent。但 PromptAgent 后续工作必须拿到答案才算完成;使用有界 WaitAgent 调用并检查
277
+ liveness,不要盲目循环。`
271
278
  },
272
279
  parameters: {
273
280
  type: 'object',
@@ -88,7 +88,8 @@ Supports offset and limit for reading specific portions of large files.
88
88
 
89
89
  Guidelines:
90
90
  - Use absolute paths when possible
91
- - A file is "large" only at >3000 lines. Read the whole file by default; only use offset/limit above that threshold or when you already know the exact line range you need.
91
+ - Read only the smallest range that answers the current question. A whole-file read is reasonable only when the file is at most 3000 lines and its full contents are actually needed.
92
+ - Do not repeat a successful read with the same range. Continue only when the truncation marker or inspected content shows that another range is necessary.
92
93
  - Binary files are detected by extension and rejected
93
94
  - Maximum file size: 10MB
94
95
  - Default limit: 3000 lines (matches the "large file = >3000 lines" threshold)`,
@@ -98,7 +99,8 @@ Guidelines:
98
99
 
99
100
  使用指南:
100
101
  - 尽量使用绝对路径
101
- - 超过 3000 行才算大文件。默认读完整文件;仅在超过此阈值或已知精确行范围时使用 offset/limit
102
+ - 只读取能回答当前问题的最小范围。只有文件不超过 3000 行且确实需要全部内容时,才适合整文件读取
103
+ - 不要用相同范围重复成功的读取。只有截断标记或已检查的内容表明仍需其他范围时才继续
102
104
  - 二进制文件通过扩展名识别并拒绝
103
105
  - 最大文件大小:10MB
104
106
  - 默认行数限制:3000 行`
@@ -123,6 +123,7 @@ Guidelines:
123
123
  - Use "**/" for recursive directory matching
124
124
  - Common directories (node_modules, .git, etc.) are skipped
125
125
  - Returns file paths relative to the search directory
126
+ - Use the narrowest pattern and smallest useful limit; inspect the result before broadening or issuing alternative searches
126
127
  - Limited to 500 results by default`,
127
128
  zh: `查找匹配 glob 模式的文件。
128
129
 
@@ -132,6 +133,7 @@ Guidelines:
132
133
  - 用 "**/" 进行递归目录匹配
133
134
  - 常见目录(node_modules、.git 等)被跳过
134
135
  - 返回相对于搜索目录的文件路径
136
+ - 使用最窄的模式和满足需要的最小 limit;先检查结果,再决定是否扩大或发出备选搜索
135
137
  - 默认限制 500 条结果`
136
138
  },
137
139
  parameters: {
@@ -808,9 +808,10 @@ Guidelines:
808
808
  - Uses the JavaScript RegExp syntax supported by the running Node.js version
809
809
  - Escape special characters such as \\. and \\{
810
810
  - Skips symlinks, binary files, invalid UTF-8, and text files larger than 16 MiB
811
- - Use glob or type filters to narrow the search
811
+ - Use glob or type filters and the smallest useful head_limit to narrow the search
812
+ - Inspect one focused search before broadening or trying alternatives; do not repeat a successful equivalent search
812
813
  - Skips common large directories such as node_modules and .git
813
- - Results are limited to 500 matches by default`,
814
+ - Results are limited to 250 matches by default`,
814
815
  zh: `用正则表达式搜索文件内容。
815
816
 
816
817
  优先使用 ripgrep (rg) 快速搜索,回退到 Node.js 实现。
@@ -824,9 +825,10 @@ Guidelines:
824
825
  - 使用当前 Node.js 版本支持的 JavaScript RegExp 语法
825
826
  - 特殊字符需转义,如 \\.、\\{
826
827
  - 跳过符号链接、二进制、无效 UTF-8 和超过 16 MiB 的文本文件
827
- - 用 glob 或 type 过滤缩小搜索范围
828
+ - 用 glob 或 type 过滤,并设置满足需要的最小 head_limit 来缩小搜索范围
829
+ - 先检查一个聚焦搜索的结果,再决定是否扩大或尝试其他搜索;不要重复成功的等价搜索
828
830
  - 跳过 node_modules、.git 等常见大目录
829
- - 默认结果限制 500 条`
831
+ - 默认结果限制 250 条`
830
832
  },
831
833
  parameters: {
832
834
  type: 'object',
@@ -58,6 +58,7 @@ stale/stalled 诊断、result 尾部和消息数量。将此作为异步子 Agen
58
58
  isConcurrencySafe: () => true,
59
59
  isReadOnly: () => true,
60
60
  cacheWithinQuery: false,
61
+ duplicateCallPolicy: () => 'allow',
61
62
  async execute(input, ctx) {
62
63
  const includeTerminal = Boolean(input?.include_closed || input?.include_terminal);
63
64
  const agents = getAgentRegistry();
@@ -24,6 +24,7 @@ export default defineTool({
24
24
  isConcurrencySafe: () => true,
25
25
  isReadOnly: () => true,
26
26
  cacheWithinQuery: false,
27
+ duplicateCallPolicy: () => 'allow',
27
28
  async execute(input = {}, ctx = {}) {
28
29
  if (!ctx.taskManager) return JSON.stringify({ error: 'task manager unavailable' });
29
30
  const taskId = input.taskId;
@@ -19,12 +19,13 @@ export default defineTool({
19
19
  Use this to give the sub-agent more work, additional instructions, or relay
20
20
  information. The prompt is queued for the agent to process on its next turn.
21
21
 
22
- IMPORTANT — PromptAgent only QUEUES the message; it does NOT block. After this
23
- returns you almost always want to call WaitAgent next to collect the reply.
24
- Do NOT end your turn after PromptAgent without either (a) calling WaitAgent,
25
- (b) explaining to the user what you just asked the sub-agent, or (c) calling
26
- CloseAgent. The orchestration loop is
27
- SpawnAgent → (PromptAgent ↔ WaitAgent)+ → CloseAgent → final reply to user.
22
+ IMPORTANT — PromptAgent only QUEUES the message; it does NOT block. A follow-up
23
+ is unfinished until you collect the reply. After PromptAgent returns, call
24
+ WaitAgent in the same parent turn. If WaitAgent reports running/timedOut, call
25
+ WaitAgent again with a larger bounded timeout unless the agent is stale/stalled.
26
+ When the reply arrives, relay the result to the user or
27
+ continue the work that depends on it. Do not end the parent turn immediately
28
+ after PromptAgent.
28
29
 
29
30
  PromptAgent is rejected if the sub-agent is in a terminal state
30
31
  (completed/failed/closed/abandoned). Use SpawnAgent to start a fresh one.`,
@@ -32,10 +33,10 @@ PromptAgent is rejected if the sub-agent is in a terminal state
32
33
 
33
34
  用于给子 Agent 更多工作、额外指令或传递信息。提示会排队等待子 Agent 在其下一个 turn 处理。
34
35
 
35
- 重要——PromptAgent 仅将消息排队,不阻塞。返回后你通常需要立即调用 WaitAgent 来收集回复。
36
- 不要在 PromptAgent 后直接结束 turn,除非:(a) 调用 WaitAgent,(b) 向用户说明你刚让子 Agent
37
- 做了什么,或 (c) 调用 CloseAgent。编排循环为:
38
- SpawnAgent -> (PromptAgent <-> WaitAgent)+ -> CloseAgent -> 最终回复给用户。
36
+ 重要——PromptAgent 仅将消息排队,不阻塞。后续任务只有拿到回复才算完成。PromptAgent 返回后,
37
+ 父级必须在同一个 turn 调用 WaitAgent。如果 WaitAgent 返回 running/timedOut,除非 Agent 已经
38
+ stale/stalled,否则必须使用更大的有界 timeout 再次调用 WaitAgent。回复到达后,必须向用户转述结果或继续执行
39
+ 依赖该结果的工作。禁止在 PromptAgent 后立刻结束父级 turn。
39
40
 
40
41
  如果子 Agent 处于终止状态(completed/failed/closed/abandoned),PromptAgent 会被拒绝。
41
42
  用 SpawnAgent 启动新的。`
@@ -125,6 +126,11 @@ SpawnAgent -> (PromptAgent <-> WaitAgent)+ -> CloseAgent -> 最终回复给用
125
126
  content: message,
126
127
  timestamp: Date.now(),
127
128
  });
129
+ // WaitAgent uses this marker to distinguish an explicitly queued follow-up
130
+ // from an ordinary asynchronous SpawnAgent run. A bounded timeout must not
131
+ // silently downgrade the same-parent-turn collection contract.
132
+ agent.promptReplyPending = true;
133
+ agent.promptReplyPendingAt = Date.now();
128
134
  if (agent.status === STATUS.IDLE || agent.status === STATUS.CREATED) {
129
135
  agent.status = STATUS.RUNNING;
130
136
  }
@@ -132,15 +138,15 @@ SpawnAgent -> (PromptAgent <-> WaitAgent)+ -> CloseAgent -> 最终回复给用
132
138
  return JSON.stringify({
133
139
  next_steps:
134
140
  'Message is queued — the sub-agent has NOT replied yet. Call WaitAgent ' +
135
- 'next to collect the reply, then relay it to the user. Do NOT end your ' +
136
- 'turn here without either waiting for the reply or telling the user ' +
137
- 'what you just asked.',
141
+ 'in this parent turn and collect the reply. If it is still running, call ' +
142
+ 'WaitAgent again with a larger bounded timeout unless it is stale/stalled. ' +
143
+ 'Relay the reply or continue the dependent work; do NOT end now.',
138
144
  success: true,
139
145
  agentId: agent_id,
140
146
  name: agent.name,
141
147
  messageCount: agent.messages.length,
142
148
  pending: agent.pendingPrompts.length,
143
- message: `Message sent to agent "${agent.name}". Use WaitAgent to collect its reply.`,
149
+ message: `Message sent to agent "${agent.name}". Call WaitAgent now; the reply is still pending.`,
144
150
  });
145
151
  },
146
152
  });
@@ -86,6 +86,7 @@ Actions:
86
86
  },
87
87
  isConcurrencySafe: () => true,
88
88
  isReadOnly: () => true,
89
+ duplicateCallPolicy: () => 'suppress',
89
90
  async execute(input, ctx) {
90
91
  const skillManager = ctx?.skillManager;
91
92
 
@@ -46,7 +46,7 @@ export default defineTool({
46
46
  This tool returns a planning instruction. Use it to land a structured plan, then keep working in the same turn. The expected flow is:
47
47
  1. Produce a short prose plan (problem, approach, risks).
48
48
  2. Call \`TodoWrite\` with the ordered steps. Mark the first concrete step "in_progress", the rest "pending".
49
- 3. When the first-step tools and arguments are already known, emit \`TodoWrite\` and those independent tool calls in the same assistant response. Wait only when a result genuinely determines the next action.
49
+ 3. Emit \`TodoWrite\` with a first work-tool call only when that call is already necessary and its arguments and safety do not depend on another result. Start with the smallest such call; inspect its result before issuing calls it could change or make unnecessary.
50
50
 
51
51
  WHEN TO USE:
52
52
  - Multi-step implementation (3+ steps), refactor, or open-ended investigation.
@@ -65,7 +65,7 @@ The tool takes the topic plus optional guiding fields (stuckAt, userProblem, exp
65
65
  此工具返回规划指令。用它产出一份结构化计划,然后在同一个 turn 中继续工作。预期流程是:
66
66
  1. 产出简短文字计划(问题、方法、风险)。
67
67
  2. 调用 TodoWrite 写出有序步骤。将第一个具体步骤标记为 "in_progress",其余标记为 "pending"。
68
- 3. 如果第一步所需的工具和参数已经确定,应在同一个 assistant response 中发出 TodoWrite 和这些彼此独立的工具调用;只有某个结果确实决定下一动作时才等待。
68
+ 3. 只有第一个工作工具调用已经确定有必要,且其参数和安全性都不依赖其他结果时,才在同一个 assistant response 中把它与 TodoWrite 一起发出。先执行满足条件的最小调用;如果结果可能改变或使后续调用不再必要,应先检查结果。
69
69
 
70
70
  何时使用:
71
71
  - 多步骤实现(3+ 步)、重构或开放式调查。
@@ -173,8 +173,8 @@ The tool takes the topic plus optional guiding fields (stuckAt, userProblem, exp
173
173
  }
174
174
  lines.push('');
175
175
  const nextInstruction = String(language).toLowerCase().startsWith('zh')
176
- ? '下一步:产出计划并调用 `TodoWrite`。如果第一步的工具和参数已经确定,在同一个 assistant response 中一并发出这些彼此独立的工具调用。只有第一步必须询问用户时才在计划后停下。'
177
- : 'Next: produce the plan and call `TodoWrite`. If the first-step tools and arguments are already known, emit those independent tool calls in the same assistant response. Stop after the plan only when the first step must ask the user.';
176
+ ? '下一步:产出计划并调用 `TodoWrite`。只有第一个工作工具调用已经确定有必要,且其参数和安全性都不依赖其他结果时,才在同一个 assistant response 中发出这个最小调用;如果结果可能改变后续调用,先检查结果。只有第一步必须询问用户时才在计划后停下。'
177
+ : 'Next: produce the plan and call `TodoWrite`. Emit only the smallest first work-tool call whose necessity, arguments, and safety do not depend on another result; inspect its result before calls it could change. Stop after the plan only when the first step must ask the user.';
178
178
  lines.push(nextInstruction);
179
179
 
180
180
  return lines.join('\n');
@@ -36,7 +36,7 @@ WHEN TO USE:
36
36
  FIRST CALL — PLAN WITHOUT AN EXTRA MODEL ROUND:
37
37
  - Write a short visible prose plan in the same assistant response: problem, approach, and risks.
38
38
  - Call TodoWrite directly; do not call a separate planning-mode tool first.
39
- - If the first work tools and arguments are known, emit them beside TodoWrite in that same response.
39
+ - Emit TodoWrite beside the first work-tool call only when that call is already necessary and its arguments and safety do not depend on another result. Start with the smallest such call.
40
40
 
41
41
  HOW TO USE:
42
42
  - First call: enumerate all the todos with status "pending", set exactly one to "in_progress".
@@ -45,8 +45,8 @@ HOW TO USE:
45
45
  - \`content\` is the imperative form ("Run tests"); \`activeForm\` is the present-continuous shown during execution ("Running tests").
46
46
 
47
47
  BATCH WITH WORK:
48
- - Avoid an intermediate TodoWrite-only model round when the next work tool and its arguments are already known. Emit TodoWrite and those independent work tool calls in the same assistant response.
49
- - This is batching, not speculative progress: mark work completed only after evidence. Keep calls separate when a pending result can change the next action, its arguments, or its safety.
48
+ - Avoid an intermediate TodoWrite-only model round only when the next work-tool call passes the necessity, argument-independence, and safety-independence test. Emit that minimal call beside TodoWrite.
49
+ - Do not speculative-batch an investigation. Mark work completed only after evidence, and inspect a pending result before issuing any call it could change, invalidate, or make unnecessary.
50
50
  - A standalone TodoWrite remains valid when no work tool should follow, including final completion or a blocking user question.
51
51
 
52
52
  WHEN NOT TO USE:
@@ -61,7 +61,7 @@ WHEN NOT TO USE:
61
61
  首次调用——不要浪费额外模型回合进入规划模式:
62
62
  - 在同一个 assistant response 中先写简短可见计划:问题、方案和风险。
63
63
  - 直接调用 TodoWrite,不要先调用单独的规划模式工具。
64
- - 如果第一批工作工具及参数已经确定,把它们和 TodoWrite 在同一响应中发出。
64
+ - 只有第一个工作工具调用已经确定有必要,且其参数和安全性都不依赖其他结果时,才把它与 TodoWrite 在同一响应中发出;先执行满足条件的最小调用。
65
65
 
66
66
  如何使用:
67
67
  - 首次调用:枚举所有 todo,状态为 "pending",将其中恰好一个设为 "in_progress"。
@@ -70,8 +70,8 @@ WHEN NOT TO USE:
70
70
  - content 是祈使形式(如 "Run tests");activeForm 是执行时显示的进行时态(如 "Running tests")。
71
71
 
72
72
  和工作工具合批:
73
- - 如果下一项工作所用的工具和参数已经确定,不要让中间状态的 TodoWrite 单独占一个模型回合;应在同一个 assistant response 中发出 TodoWrite 和这些彼此独立的工作工具调用。
74
- - 这是合批,不是提前宣告进度:只有已有证据时才能把工作标记为完成。如果待返回结果可能改变下一动作、参数或安全性,就必须分开调用。
73
+ - 只有下一个工作工具调用通过必要性、参数独立性和安全独立性检查时,才避免让中间状态的 TodoWrite 单独占一个模型回合;把这个最小调用与 TodoWrite 一起发出。
74
+ - 不要推测性批量展开调查。只有已有证据时才能把工作标记为完成;如果待返回结果可能改变、否定或使后续调用不再必要,应先检查该结果。
75
75
  - 没有工作工具应继续执行时(包括记录最终完成态或询问阻塞问题),TodoWrite 仍可单独调用。
76
76
 
77
77
  何时不使用:
@@ -72,6 +72,10 @@
72
72
  * @property {boolean | ((input?: object) => boolean)} [cacheWithinQuery] — explicitly safe to reuse for identical calls in one query
73
73
  * @property {boolean | ((input?: object) => boolean)} [mayMutateWorkspaceAfterReturn] — may keep changing the workspace after execute() resolves; disables same-query read reuse
74
74
  * @property {(input?: object) => boolean} [isDestructive] — destructive operation?
75
+ * @property {(input?: object) => 'allow' | 'warn' | 'suppress'} [duplicateCallPolicy]
76
+ * — repeated exact-call policy for one query. Use `allow` for polling/time-varying
77
+ * tools and `suppress` only when the result is stable for the whole query.
78
+ * Errors are never counted as successful duplicates.
75
79
  * @property {'json-error-envelope' | null} [errorOutput] — explicit returned-output error contract; null means only thrown errors fail
76
80
  * @property {string} [mcpServer] — owning MCP server for flattened MCP tools
77
81
  * @property {'external' | 'run'} [sideEffectScope] — whether mutations escape the current Run collector
@@ -90,6 +94,7 @@
90
94
  * cacheWithinQuery?: boolean | ((input?: object) => boolean),
91
95
  * mayMutateWorkspaceAfterReturn?: boolean | ((input?: object) => boolean),
92
96
  * isDestructive?: (input?: object) => boolean,
97
+ * duplicateCallPolicy?: (input?: object) => 'allow' | 'warn' | 'suppress',
93
98
  * errorOutput?: 'json-error-envelope' | null,
94
99
  * mcpServer?: string,
95
100
  * sideEffectScope?: 'external' | 'run',
@@ -108,6 +113,7 @@ export function defineTool({
108
113
  cacheWithinQuery = false,
109
114
  mayMutateWorkspaceAfterReturn = false,
110
115
  isDestructive = () => false,
116
+ duplicateCallPolicy = () => 'warn',
111
117
  errorOutput = 'json-error-envelope',
112
118
  mcpServer,
113
119
  sideEffectScope = 'external',
@@ -126,6 +132,7 @@ export function defineTool({
126
132
  cacheWithinQuery,
127
133
  mayMutateWorkspaceAfterReturn,
128
134
  isDestructive,
135
+ duplicateCallPolicy,
129
136
  errorOutput,
130
137
  sideEffectScope,
131
138
  };
@@ -19,7 +19,7 @@
19
19
  * processing. The envelope flags `runningInBackground:
20
20
  * true` (the sub-agent IS continuing — it does NOT need
21
21
  * another PromptAgent to keep going) and recommends either
22
- * another WaitAgent or CloseAgent. `result` carries the
22
+ * another bounded WaitAgent or CloseAgent. `result` carries the
23
23
  * mid-stream preview (the driver keeps lastResult fresh
24
24
  * from every text_delta).
25
25
  *
@@ -43,7 +43,7 @@ import { consumeNotificationForAgent } from '../sub-agent/notifications.js';
43
43
  * otherwise eat tail-positioned nudges when `result` is long).
44
44
  *
45
45
  * @param {string} status
46
- * @param {{ timedOut?: boolean, runningInBackground?: boolean, budgetExceeded?: boolean, stale?: boolean }} [opts]
46
+ * @param {{ timedOut?: boolean, budgetExceeded?: boolean, stale?: boolean, mustCollectReply?: boolean }} [opts]
47
47
  */
48
48
  function nextStepsFor(status, opts = {}) {
49
49
  if (opts.budgetExceeded) {
@@ -63,6 +63,15 @@ function nextStepsFor(status, opts = {}) {
63
63
  'task and start a fresh agent if needed.'
64
64
  );
65
65
  }
66
+ if (opts.timedOut && opts.mustCollectReply) {
67
+ return (
68
+ 'The PromptAgent follow-up reply is still pending and must be collected ' +
69
+ 'in this parent turn. Call WaitAgent again with a larger bounded timeout. ' +
70
+ 'Do not end the turn or switch to ListAgents/notifications. Stop re-waiting ' +
71
+ 'only if the agent becomes stale/stalled, the wait is cancelled, or the ' +
72
+ 'agent returns idle/terminal.'
73
+ );
74
+ }
66
75
  if (opts.timedOut) {
67
76
  return (
68
77
  'Sub-agent is running in the background; it does not need another ' +
@@ -129,6 +138,11 @@ function errorNextSteps() {
129
138
  */
130
139
  function buildEnvelope(agent, { timedOut = false } = {}) {
131
140
  const status = agent.status;
141
+ const mustCollectReply = agent.promptReplyPending === true;
142
+ if (!timedOut && (status === STATUS.IDLE || isTerminalAgentStatus(status))) {
143
+ agent.promptReplyPending = false;
144
+ agent.promptReplyPendingAt = null;
145
+ }
132
146
  const liveness = diagnoseAgentLiveness(agent);
133
147
  const budgetResult = agent.result && typeof agent.result === 'object'
134
148
  && agent.result.status === 'budget_exceeded'
@@ -140,7 +154,12 @@ function buildEnvelope(agent, { timedOut = false } = {}) {
140
154
  ? agent.result
141
155
  : (agent.lastResult || ''));
142
156
  const env = {
143
- next_steps: nextStepsFor(status, { timedOut, budgetExceeded: !!budgetResult, stale: liveness.stale }),
157
+ next_steps: nextStepsFor(status, {
158
+ timedOut,
159
+ budgetExceeded: !!budgetResult,
160
+ stale: liveness.stale,
161
+ mustCollectReply: mustCollectReply && !liveness.stale,
162
+ }),
144
163
  agentId: agent.id,
145
164
  name: agent.name,
146
165
  status,
@@ -154,6 +173,7 @@ function buildEnvelope(agent, { timedOut = false } = {}) {
154
173
  diagnostic: liveness.diagnostic,
155
174
  messages: Array.isArray(agent.messages) ? agent.messages.length : 0,
156
175
  turns: agent.usage?.turns || 0,
176
+ mustCollectReply: timedOut ? mustCollectReply : false,
157
177
  };
158
178
  if (timedOut) {
159
179
  env.timedOut = true;
@@ -192,15 +212,15 @@ Status semantics:
192
212
  another PromptAgent. Either WaitAgent again with a larger timeout,
193
213
  CloseAgent to cut it short, or tell the user it's still working.
194
214
 
195
- CRITICAL — after WaitAgent returns you MUST take one of these actions:
196
- • status terminal: relay/retry/report.
197
- • status idle: reply to user OR PromptAgent OR CloseAgent.
198
- • timedOut: re-wait, cut short, or report progress.
199
- NEVER end your turn silently right after WaitAgent — the user has not seen
200
- the sub-agent's reply yet; only you have. The orchestration loop is
201
- SpawnAgent → (PromptAgent ↔ WaitAgent)+ → CloseAgent → final reply to user.
215
+ After WaitAgent returns, act on the status. A non-stale timeout after PromptAgent
216
+ has 'mustCollectReply=true': call WaitAgent again with a larger bounded timeout in
217
+ the same parent turn until idle/terminal. A stale/stalled agent breaks that loop:
218
+ inspect/report/close it instead. For ordinary SpawnAgent background work, use
219
+ ListAgents or later completion notifications instead of repeatedly re-waiting.
202
220
 
203
- Compatibility tool. The default wait is a short 5000ms poll. Callers may request up to 300000ms (5 minutes), but this is no longer the primary sub-agent workflow; prefer SpawnAgent + ListAgents + completion notifications for async background work.`,
221
+ The default wait is a bounded 5000ms poll; callers may request up to 300000ms
222
+ (5 minutes). Never use an unbounded blind loop: every wait is capped, liveness is
223
+ checked after each timeout, and stale/stalled is the explicit stop condition.`,
204
224
  zh: `等待子 Agent 的下一次状态变更(turn 结束、终止或等待超时)并获取状态信封。
205
225
 
206
226
  返回 JSON,含明确的 status、最新的 result 文本、liveness 计数器(toolUseCount、tokenCount、
@@ -215,11 +235,13 @@ msSinceLastEvent、recentTools)、可随时 Read 的持久化 outputFile 路
215
235
  仍在运行。不需要再 PromptAgent。要么用更大 timeout 再次 WaitAgent,要么 CloseAgent 中断,
216
236
  要么告知用户它仍在工作。
217
237
 
238
+ PromptAgent 后若非 stale/stalled 的有界等待超时,必须在同一父级 turn 使用更大的有界 timeout
239
+ 再次调用 WaitAgent,直到 idle/terminal;不要改用 ListAgents/notification 丢下未收集的回复。
218
240
  关键——如果信封显示 stale/stalled,子 Agent 可能卡死或空转。不要反复调用 WaitAgent——向用户
219
241
  报告情况,决定是 CloseAgent(带 close_reason)还是重试。
220
242
 
221
- 此工具保留用于向后兼容。现代异步流程请用 ListAgents 做非阻塞状态检查,依赖 turn 开始时的
222
- notification 获取完成事件。`
243
+ 普通 SpawnAgent 异步流程仍用 ListAgents 做非阻塞状态检查,并依赖后续 completion
244
+ notification;不要对普通后台任务盲目循环等待。每次等待都有上限,stale/stalled 是停止条件。`
223
245
  },
224
246
  parameters: {
225
247
  type: 'object',
@@ -247,6 +269,7 @@ notification 获取完成事件。`
247
269
  isConcurrencySafe: () => true,
248
270
  isReadOnly: () => true,
249
271
  cacheWithinQuery: false,
272
+ duplicateCallPolicy: () => 'allow',
250
273
  async execute(input, ctx) {
251
274
  const { agent_id, timeout_ms = 5000 } = input;
252
275
  if (!agent_id) {
@@ -297,6 +320,13 @@ notification 获取完成事件。`
297
320
  await new Promise(r => setTimeout(r, 200));
298
321
  }
299
322
 
323
+ // Re-check after the final sleep. The status may have changed just before
324
+ // the deadline without another loop iteration.
325
+ if (isTerminalAgentStatus(agent.status) || agent.status === STATUS.IDLE) {
326
+ consumeNotificationForAgent(agent.id);
327
+ return JSON.stringify(buildEnvelope(agent));
328
+ }
329
+
300
330
  // Wait elapsed; the sub-agent is still running. Surface mid-stream
301
331
  // preview + liveness so the parent has actionable signal.
302
332
  return JSON.stringify(buildEnvelope(agent, { timedOut: true }));
@@ -42,7 +42,8 @@ Use this to read documentation, articles, or any web page.
42
42
 
43
43
  Guidelines:
44
44
  - Provide the full URL including protocol (https://)
45
- - Large pages will be truncated — use the offset parameter for pagination
45
+ - Fetch one authoritative relevant page first; inspect it before fetching alternatives
46
+ - Set max_length to the smallest useful content budget. Large pages are truncated; use a more targeted source rather than repeatedly fetching the same URL
46
47
  - For APIs, the raw response body is returned as-is
47
48
  - Respects the abort signal for cancellation`,
48
49
  zh: `获取并读取网页内容。
@@ -51,7 +52,8 @@ Guidelines:
51
52
 
52
53
  使用指南:
53
54
  - 提供完整 URL 含协议(https://)
54
- - 大页面会截断——用 offset 参数做分页
55
+ - 先抓取一个最权威、最相关的页面;检查结果后再决定是否抓取备选页面
56
+ - 将 max_length 设为满足需要的最小内容预算。大页面会截断;应改用更定向的来源,而不是重复抓取同一 URL
55
57
  - 对 API 请求,原始响应体原样返回
56
58
  - 尊重取消信号`
57
59
  },
@@ -32,17 +32,17 @@ Use this when you need up-to-date information that may not be in your training d
32
32
  Returns search results with titles, URLs, and snippets.
33
33
 
34
34
  Guidelines:
35
- - Use specific, targeted search queries
35
+ - Use one specific, targeted query with the smallest useful result limit; inspect it before trying alternatives
36
36
  - Include the current year for time-sensitive queries
37
- - Combine with WebFetch to read full page content from results`,
37
+ - Fetch the most authoritative relevant result first. Add another source only when the claim is consequential, disputed, stale, or not established by the first source`,
38
38
  zh: `搜索网页获取最新信息。
39
39
 
40
40
  当你需要训练数据中可能没有的最新信息时使用。返回搜索结果,含标题、URL 和摘要。
41
41
 
42
42
  使用指南:
43
- - 使用具体、有针对性的搜索关键词
43
+ - 先执行一个具体、定向的查询,并使用满足需要的最小结果数;检查结果后再决定是否尝试其他查询
44
44
  - 时间敏感的查询要包含当前年份
45
- - 配合 WebFetch 读取搜索结果中的完整页面内容`
45
+ - 优先抓取最权威、最相关的结果。只有结论影响重大、存在争议、可能过时,或首个来源不能证明时才增加来源`
46
46
  },
47
47
  parameters: {
48
48
  type: 'object',