@selesai/code 0.5.29 → 0.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (94) hide show
  1. package/CHANGELOG.md +34 -0
  2. package/README.md +1 -1
  3. package/dist/config.d.ts +16 -3
  4. package/dist/config.d.ts.map +1 -1
  5. package/dist/config.js +106 -5
  6. package/dist/config.js.map +1 -1
  7. package/dist/core/system-prompt.d.ts.map +1 -1
  8. package/dist/core/system-prompt.js +18 -0
  9. package/dist/core/system-prompt.js.map +1 -1
  10. package/dist/core/system-prompt.test.d.ts +2 -0
  11. package/dist/core/system-prompt.test.d.ts.map +1 -0
  12. package/dist/core/system-prompt.test.js +89 -0
  13. package/dist/core/system-prompt.test.js.map +1 -0
  14. package/dist/defaults/models.json +13 -45
  15. package/dist/defaults/settings.json +5 -7
  16. package/dist/extensions/copy-turn.test.ts +131 -0
  17. package/dist/extensions/copy-turn.ts +6 -1
  18. package/dist/extensions/node_modules/.vite/vitest/da39a3ee5e6b4b0d3255bfef95601890afd80709/results.json +1 -0
  19. package/dist/extensions/package.json +0 -1
  20. package/dist/extensions/pi-subagents/CHANGELOG.md +3 -0
  21. package/dist/extensions/pi-subagents/README.md +27 -32
  22. package/dist/extensions/pi-subagents/agents/architect.md +4 -4
  23. package/dist/extensions/pi-subagents/agents/builder.md +5 -4
  24. package/dist/extensions/pi-subagents/agents/commentator.md +3 -2
  25. package/dist/extensions/pi-subagents/agents/explorer.md +3 -2
  26. package/dist/extensions/pi-subagents/agents/recapper.md +3 -2
  27. package/dist/extensions/pi-subagents/agents/researcher.md +4 -3
  28. package/dist/extensions/pi-subagents/skills/pi-subagents/SKILL.md +2 -0
  29. package/dist/extensions/pi-subagents/skills/pi-subagents/references/constraints-and-recipes.md +10 -9
  30. package/dist/extensions/pi-subagents/skills/pi-subagents/references/execution-controls.md +12 -11
  31. package/dist/extensions/pi-subagents/skills/pi-subagents/references/prompting-and-roles.md +10 -11
  32. package/dist/extensions/pi-subagents/src/agents/agent-management.ts +56 -9
  33. package/dist/extensions/pi-subagents/src/agents/task-aware-routing.ts +125 -0
  34. package/dist/extensions/pi-subagents/src/api/preflight.ts +1 -1
  35. package/dist/extensions/pi-subagents/src/extension/index.ts +5 -1
  36. package/dist/extensions/pi-subagents/src/extension/schemas.ts +2 -2
  37. package/dist/extensions/pi-subagents/src/extension/tool-description.ts +24 -7
  38. package/dist/extensions/pi-subagents/src/runs/background/async-execution.ts +23 -5
  39. package/dist/extensions/pi-subagents/src/runs/background/notify.ts +27 -1
  40. package/dist/extensions/pi-subagents/src/runs/background/result-watcher.ts +64 -6
  41. package/dist/extensions/pi-subagents/src/runs/background/subagent-runner.ts +16 -1
  42. package/dist/extensions/pi-subagents/src/runs/foreground/chain-execution.ts +72 -18
  43. package/dist/extensions/pi-subagents/src/runs/foreground/execution.ts +19 -5
  44. package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +127 -31
  45. package/dist/extensions/pi-subagents/src/runs/shared/acceptance.ts +4 -6
  46. package/dist/extensions/pi-subagents/src/runs/shared/single-output.ts +63 -9
  47. package/dist/extensions/pi-subagents/src/runs/shared/task-intent.ts +21 -0
  48. package/dist/extensions/pi-subagents/src/shared/types.ts +41 -2
  49. package/dist/extensions/pi-subagents/src/shared/utils.ts +29 -1
  50. package/dist/extensions/pi-subagents/src/slash/delegation-adapters.ts +5 -1
  51. package/dist/extensions/pi-subagents/src/tui/render.ts +32 -6
  52. package/dist/extensions/pi-subagents/test/e2e/real-session-subagent.test.ts +111 -6
  53. package/dist/extensions/pi-subagents/test/integration/async-execution.test.ts +74 -43
  54. package/dist/extensions/pi-subagents/test/integration/chain-execution.test.ts +36 -21
  55. package/dist/extensions/pi-subagents/test/integration/fork-context-execution.test.ts +5 -3
  56. package/dist/extensions/pi-subagents/test/integration/intercom-result-delivery.test.ts +20 -8
  57. package/dist/extensions/pi-subagents/test/integration/parallel-execution.test.ts +14 -7
  58. package/dist/extensions/pi-subagents/test/integration/render-fork-badge.test.ts +227 -0
  59. package/dist/extensions/pi-subagents/test/integration/result-watcher.test.ts +81 -5
  60. package/dist/extensions/pi-subagents/test/integration/single-execution.test.ts +49 -10
  61. package/dist/extensions/pi-subagents/test/support/real-session-runner.ts +18 -2
  62. package/dist/extensions/pi-subagents/test/unit/agent-disabled.test.ts +1 -1
  63. package/dist/extensions/pi-subagents/test/unit/agent-frontmatter.test.ts +70 -6
  64. package/dist/extensions/pi-subagents/test/unit/agent-management.test.ts +161 -1
  65. package/dist/extensions/pi-subagents/test/unit/builtin-agent-documentation.test.ts +63 -0
  66. package/dist/extensions/pi-subagents/test/unit/capability-ceiling-agent-allowlist.test.ts +34 -0
  67. package/dist/extensions/pi-subagents/test/unit/delegation-api.test.ts +24 -0
  68. package/dist/extensions/pi-subagents/test/unit/index-child-registration.test.ts +6 -1
  69. package/dist/extensions/pi-subagents/test/unit/notify.test.ts +29 -0
  70. package/dist/extensions/pi-subagents/test/unit/preflight.test.ts +2 -0
  71. package/dist/extensions/pi-subagents/test/unit/schemas.test.ts +12 -0
  72. package/dist/extensions/pi-subagents/test/unit/single-output.test.ts +91 -1
  73. package/dist/extensions/pi-subagents/test/unit/task-aware-routing.test.ts +213 -0
  74. package/dist/extensions/pi-subagents/test/unit/task-intent.test.ts +23 -1
  75. package/dist/extensions/pi-subagents/test/unit/tool-description.test.ts +60 -9
  76. package/dist/extensions/pi-web-agent/package.json +1 -1
  77. package/dist/skills/pi-subagents/SKILL.md +43 -0
  78. package/dist/skills/pi-subagents/references/constraints-and-recipes.md +257 -0
  79. package/dist/skills/pi-subagents/references/execution-controls.md +431 -0
  80. package/dist/skills/pi-subagents/references/management-authoring-rpc.md +144 -0
  81. package/dist/skills/pi-subagents/references/prompting-and-roles.md +281 -0
  82. package/dist/skills/ponytail/SKILL.md +1 -3
  83. package/docs/plans/subagent-delegation/phase-0-correctness.md +265 -0
  84. package/docs/plans/subagent-delegation/phase-1-behavioral-contract.md +486 -0
  85. package/docs/plans/subagent-delegation/phase-2-context-controls.md +282 -0
  86. package/docs/plans/subagent-delegation/phase-3-advisory-routing.md +362 -0
  87. package/docs/plans/subagent-delegation/phase-4-optional-enforcement.md +381 -0
  88. package/package.json +2 -2
  89. package/dist/extensions/caveman/caveman-instructions.cjs +0 -11
  90. package/dist/extensions/caveman/index.js +0 -118
  91. package/dist/extensions/caveman/package.json +0 -8
  92. package/dist/extensions/caveman/test/extension.test.js +0 -203
  93. package/dist/extensions/caveman/test/helpers.test.js +0 -58
  94. package/dist/skills/caveman/SKILL.md +0 -50
@@ -27,6 +27,32 @@ import { contextModeBadge, contextModePrefix } from "../runs/shared/context-mode
27
27
 
28
28
  type Theme = ExtensionContext["ui"]["theme"];
29
29
 
30
+ /**
31
+ * UI-side output projection: completed terminal results strip `finalOutput`/
32
+ * `truncation` when an authoritative saved output path exists (see
33
+ * compactForegroundResult). The terminal renderer stays reference-first: it
34
+ * never re-reads the saved child output file. Settled file-only results show
35
+ * the saved-output reference (path/size/lines) instead of re-inlining child
36
+ * prose; explicit `outputMode: "inline"` results keep their full text via
37
+ * `finalOutput`. Legacy/foreign result data may carry `savedOutputPath` without
38
+ * an `outputReference`; a path-only reference is synthesized so the saved file
39
+ * stays discoverable without reading it.
40
+ */
41
+ function formatPathOnlyOutputReference(savedOutputPath: string): string {
42
+ return `Output saved to: ${path.resolve(savedOutputPath)}. Read this file if needed.`;
43
+ }
44
+
45
+ function resultOutputForUi(r: Details["results"][number]): string {
46
+ // Explicit `outputMode: "inline"` is the sole legacy full-text opt-out:
47
+ // keep the inline text even when a saved output path also exists.
48
+ if (r.outputMode === "inline") return r.truncation?.text || getSingleResultOutput(r) || "";
49
+ // Settled file-backed results are reference-first: never re-read or re-inline
50
+ // the saved child output, even when foreign/legacy data still carries a text
51
+ // projection. Synthesize a path-only reference when none was persisted.
52
+ if (r.savedOutputPath) return r.outputReference?.message || formatPathOnlyOutputReference(r.savedOutputPath);
53
+ return r.truncation?.text || getSingleResultOutput(r) || "";
54
+ }
55
+
30
56
  function liveDetailKeyText(): string {
31
57
  return keyText("app.tools.expand");
32
58
  }
@@ -1308,7 +1334,7 @@ export function renderWidget(ctx: ExtensionContext, jobs: AsyncJobState[]): void
1308
1334
  }
1309
1335
 
1310
1336
  function renderSingleCompact(d: Details, r: Details["results"][number], theme: Theme, frame?: number): Component {
1311
- const output = r.truncation?.text || getSingleResultOutput(r);
1337
+ const output = resultOutputForUi(r);
1312
1338
  const progress = r.progress || r.progressSummary;
1313
1339
  const isRunning = r.progress?.status === "running";
1314
1340
  const contextBadge = contextModeBadge(theme, r.context ?? d.context);
@@ -1418,7 +1444,7 @@ function renderMultiCompact(d: Details, theme: Theme, frame?: number): Component
1418
1444
  c.addChild(new Text(truncLine(theme.fg("dim", ` ◦ ${pendingLabel}: ${agentName} · pending`), width), 0, 0));
1419
1445
  continue;
1420
1446
  }
1421
- const output = getSingleResultOutput(r);
1447
+ const output = resultOutputForUi(r);
1422
1448
  const progressFromArray = d.progress?.find((p) => p.index === i) || d.progress?.find((p) => p.agent === r.agent && p.status === "running");
1423
1449
  const rProg = r.progress || progressFromArray || r.progressSummary;
1424
1450
  const rRunning = rProg && "status" in rProg && rProg.status === "running";
@@ -1500,7 +1526,7 @@ export function renderSubagentResult(
1500
1526
  ? theme.fg("success", "ok")
1501
1527
  : theme.fg("error", "failed");
1502
1528
  const contextBadge = contextModeBadge(theme, r.context ?? d.context);
1503
- const output = r.truncation?.text || getSingleResultOutput(r);
1529
+ const output = resultOutputForUi(r);
1504
1530
 
1505
1531
  const progressInfo = isRunning && r.progress
1506
1532
  ? ` | ${r.progress.toolCount} tools, ${formatTokens(r.progress.tokens)} tok, ${formatDuration(r.progress.durationMs)}`
@@ -1600,7 +1626,7 @@ export function renderSubagentResult(
1600
1626
  const hasEmptyWithoutTarget = d.results.some((r) =>
1601
1627
  r.exitCode === 0
1602
1628
  && r.progress?.status !== "running"
1603
- && hasEmptyTextOutputWithoutOutputTarget(r.task, getSingleResultOutput(r)),
1629
+ && hasEmptyTextOutputWithoutOutputTarget(r.task, resultOutputForUi(r)),
1604
1630
  );
1605
1631
  const hasWorkflowFailure = workflowGraphHasStatus(d, ["failed"]);
1606
1632
  const hasWorkflowStop = d.results.some((r) => r.stopped && r.progress?.status !== "running") || workflowGraphHasStatus(d, ["stopped"]);
@@ -1658,7 +1684,7 @@ export function renderSubagentResult(
1658
1684
  const isComplete = result && result.exitCode === 0 && result.progress?.status !== "running";
1659
1685
  const isEmptyWithoutTarget = Boolean(result)
1660
1686
  && Boolean(isComplete)
1661
- && hasEmptyTextOutputWithoutOutputTarget(result.task, getSingleResultOutput(result));
1687
+ && hasEmptyTextOutputWithoutOutputTarget(result.task, resultOutputForUi(result));
1662
1688
  const isCurrent = i === (d.currentStepIndex ?? d.results.length);
1663
1689
  const stepIcon = isFailed
1664
1690
  ? theme.fg("error", "failed")
@@ -1729,7 +1755,7 @@ export function renderSubagentResult(
1729
1755
  const rRunning = rProg?.status === "running";
1730
1756
  const stepNumber = typeof rProg?.index === "number" ? rProg.index + 1 : i + 1;
1731
1757
 
1732
- const resultOutput = getSingleResultOutput(r);
1758
+ const resultOutput = resultOutputForUi(r);
1733
1759
  const statusIcon = rRunning
1734
1760
  ? theme.fg("warning", "running")
1735
1761
  : r.exitCode !== 0
@@ -13,6 +13,7 @@
13
13
 
14
14
  import { afterEach, describe, it } from "node:test";
15
15
  import assert from "node:assert/strict";
16
+ import * as fs from "node:fs";
16
17
  import * as os from "node:os";
17
18
  import * as path from "node:path";
18
19
  import { tryImport } from "../support/helpers.ts";
@@ -23,6 +24,17 @@ const piAi = await tryImport<unknown>("@earendil-works/pi-ai");
23
24
  const available = Boolean(piCodingAgent && piAi);
24
25
 
25
26
  const CHILD_MARKER = "CHILD_REAL_SESSION_OK";
27
+
28
+ /**
29
+ * Reference-first tool results: completion content carries "Output saved to:
30
+ * <path> (…)". Read the durable file to inspect the full child output.
31
+ */
32
+ function readSavedOutput(text: string): string {
33
+ const rest = text.split("Output saved to: ")[1];
34
+ assert.ok(rest, `expected a saved-output reference in: ${text.slice(0, 160)}`);
35
+ const outputPath = rest.split(" (")[0]!;
36
+ return fs.readFileSync(outputPath, "utf-8");
37
+ }
26
38
  // Env vars the runner must clear so a parent that was itself spawned as a
27
39
  // subagent child can still launch fresh children. The values are deliberately
28
40
  // bogus sentinels (nonexistent paths) so a leaked value would break spawning.
@@ -122,10 +134,14 @@ Use the available tools.`;
122
134
  const chainDetails = JSON.stringify((toolMessages[1] as { details?: unknown } | undefined)?.details);
123
135
  const structuredDetails = JSON.stringify((toolMessages[2] as { details?: unknown } | undefined)?.details);
124
136
  assert.equal(results.length, 4);
125
- assert.match(results[0] ?? "", /ACTIVE_TOOLS:[^\n]*fixture_search/);
126
- assert.match(results[0] ?? "", /ACTIVE_TOOLS:[^\n]*read/);
127
- assert.match(chainDetails, /ACTIVE_TOOLS:[^\n]*fixture_search/);
128
- assert.match(chainDetails, /ACTIVE_TOOLS:[^\n]*read/);
137
+ const directOutput = readSavedOutput(results[0] ?? "");
138
+ assert.match(directOutput, /ACTIVE_TOOLS:[^\n]*fixture_search/);
139
+ assert.match(directOutput, /ACTIVE_TOOLS:[^\n]*read/);
140
+ const chainFirstChild = (toolMessages[1] as { details?: { results?: Array<{ savedOutputPath?: string }> } } | undefined)?.details?.results?.[0];
141
+ assert.ok(chainFirstChild?.savedOutputPath, "chain details should carry the saved output path");
142
+ const chainOutput = fs.readFileSync(chainFirstChild.savedOutputPath, "utf-8");
143
+ assert.match(chainOutput, /ACTIVE_TOOLS:[^\n]*fixture_search/);
144
+ assert.match(chainOutput, /ACTIVE_TOOLS:[^\n]*read/);
129
145
  assert.match(structuredDetails, /STRUCTURED_OUTPUT_OK/);
130
146
  assert.match(results[3] ?? "", /requested unavailable child tools: missing_search/);
131
147
  assert.match(results[3] ?? "", /subagentOnlyExtensions/);
@@ -172,7 +188,8 @@ Report active tools.`;
172
188
 
173
189
  const results = subagentToolResults(run.parentSession);
174
190
  assert.equal(results.length, 1);
175
- assert.match(results[0] ?? "", /ACTIVE_TOOLS:[^\n]*fixture_async_search/);
191
+ const asyncOutput = readSavedOutput(results[0] ?? "");
192
+ assert.match(asyncOutput, /ACTIVE_TOOLS:[^\n]*fixture_async_search/);
176
193
  assert.doesNotMatch(results[0] ?? "", /requested unavailable child tools/);
177
194
  });
178
195
 
@@ -206,7 +223,7 @@ Report active tools.`;
206
223
 
207
224
  const toolResults = subagentToolResults(run.parentSession);
208
225
  assert.equal(toolResults.length, 1);
209
- assert.match(toolResults[0]!, new RegExp(CHILD_MARKER));
226
+ assert.match(readSavedOutput(toolResults[0]!), new RegExp(CHILD_MARKER));
210
227
  assert.match(run.responseText, new RegExp(CHILD_MARKER));
211
228
  assert.doesNotMatch(run.responseText, /CHILD_MISSING/);
212
229
  assert.ok(run.modelCalls >= 2, `expected parent tool-call and final turns, got ${run.modelCalls}`);
@@ -219,4 +236,92 @@ Report active tools.`;
219
236
  }
220
237
  }
221
238
  });
239
+
240
+ function latestSubagentToolResultText(messages: Array<{ role?: string; toolName?: string; content?: unknown }>): string | undefined {
241
+ for (let i = messages.length - 1; i >= 0; i--) {
242
+ const message = messages[i]!;
243
+ if (message.role === "toolResult" && message.toolName === "subagent") {
244
+ return Array.isArray(message.content)
245
+ ? message.content
246
+ .map((part) => part && typeof part === "object" && (part as { type?: unknown }).type === "text"
247
+ ? String((part as { text?: unknown }).text ?? "")
248
+ : "")
249
+ .join("")
250
+ : "";
251
+ }
252
+ }
253
+ return undefined;
254
+ }
255
+
256
+ it("lists then delegates to a non-bundled discovered writer in a broad-mutation request", async () => {
257
+ const { runRealSubagentSession, subagentCall, subagentToolResults } = await import("../support/real-session-runner.ts");
258
+ const writerAgent = `---
259
+ name: fixture-writer
260
+ description: Scoped mutation-capable fixture writer
261
+ aliases: fw
262
+ tools: read, grep, find, ls, bash, edit, write
263
+ acceptanceRole: writer
264
+ defaultContext: fork
265
+ completionGuard: false
266
+ ---
267
+ Implement the scoped fixture change and return the marker.`;
268
+
269
+ run = await runRealSubagentSession({
270
+ prompt: "Implement the fixture change across the codebase.",
271
+ childText: CHILD_MARKER,
272
+ projectFiles: {
273
+ ".selesai/agents/fixture-writer.md": writerAgent,
274
+ },
275
+ respond(context) {
276
+ const messages = context.messages as Array<{ role?: string; toolName?: string; content?: unknown; details?: unknown }>;
277
+ const subagentResults = messages.filter((message) => message.role === "toolResult" && message.toolName === "subagent");
278
+ if (subagentResults.length === 0) {
279
+ return subagentCall({ action: "list", agentScope: "project" }, "call-list-writer");
280
+ }
281
+ if (subagentResults.length === 1) {
282
+ const listText = latestSubagentToolResultText(messages) ?? "";
283
+ assert.match(
284
+ listText,
285
+ /- fixture-writer \(project, context: fork, role: writer, aliases: fw, tools: read, grep, find, ls, bash, edit, write\)/,
286
+ "catalog must expose the custom writer with its runtime metadata",
287
+ );
288
+ const listedDetails = JSON.stringify(subagentResults.at(-1)?.details ?? {});
289
+ assert.match(listedDetails, /"catalog"/);
290
+ assert.match(listedDetails, /"fixture-writer"/);
291
+ assert.match(listedDetails, /"acceptanceRole":"writer"/);
292
+ return subagentCall(
293
+ { agent: "fixture-writer", task: "Implement the change and return the marker.", context: "fresh", agentScope: "project" },
294
+ "call-fixture-writer",
295
+ );
296
+ }
297
+ return "Broad mutation work complete.";
298
+ },
299
+ timeoutMs: 60_000,
300
+ });
301
+
302
+ const results = subagentToolResults(run.parentSession);
303
+ assert.equal(results.length, 2);
304
+ assert.match(results[0] ?? "", /fixture-writer \(project, context: fork, role: writer/);
305
+ assert.match(readSavedOutput(results[1] ?? ""), new RegExp(CHILD_MARKER));
306
+ });
307
+
308
+ it("keeps tiny targeted reads local without a subagent call", async () => {
309
+ const { runRealSubagentSession, subagentToolResults } = await import("../support/real-session-runner.ts");
310
+ run = await runRealSubagentSession({
311
+ prompt: "What does the README say about subagents?",
312
+ childText: CHILD_MARKER,
313
+ projectFiles: {
314
+ "README.md": "Subagents are delegated workers.",
315
+ },
316
+ respond() {
317
+ return "The README says: Subagents are delegated workers.";
318
+ },
319
+ timeoutMs: 60_000,
320
+ });
321
+
322
+ const results = subagentToolResults(run.parentSession);
323
+ assert.equal(results.length, 0);
324
+ assert.match(run.responseText, /Subagents are delegated workers/);
325
+ assert.ok(run.modelCalls >= 1, `expected at least one parent turn, got ${run.modelCalls}`);
326
+ });
222
327
  });
@@ -80,7 +80,7 @@ interface AsyncResultPayload {
80
80
  totalCost?: { inputTokens: number; outputTokens: number; costUsd: number };
81
81
  usageBudget?: UsageBudgetState;
82
82
  checkpoint?: { name?: string; status?: string };
83
- results: Array<{ agent?: string; launchContractDigest?: string; launchResolvedExtensions?: LaunchResolvedExtensions; runtimeAcknowledgedExtensions?: RuntimeAcknowledgedExtensions; output?: string; outputState?: "present" | "absent" | "unknown"; success?: boolean; error?: string; protocolError?: { code?: string; stream?: string; limitBytes?: number; observedBytes?: number }; timedOut?: boolean; stopped?: boolean; turnBudget?: { maxTurns: number; graceTurns: number; outcome: string; turnCount: number; wrapUpRequestedAtTurn?: number; terminationDeferredAtTurn?: number; exceededAtTurn?: number }; turnBudgetExceeded?: boolean; wrapUpRequested?: boolean; model?: string; attemptedModels?: string[]; modelAttempts?: Array<{ success?: boolean; error?: string }>; totalCost?: { inputTokens: number; outputTokens: number; costUsd: number }; structuredOutput?: unknown; agentContract?: { version: 1 }; execution?: { status?: string; success?: boolean; exitCode?: number }; effects?: { fileMutation?: { status?: string; expected?: boolean; attempted?: boolean } }; intercomTarget?: string; acceptance?: { status?: string; effectiveAcceptance?: { level?: string }; childReport?: unknown; runtimeChecks?: Array<{ id?: string; status?: string; message?: string }> }; artifactPaths?: { outputPath?: string; inputPath?: string; metadataPath?: string }; capabilityCeiling?: { version?: number; allowedTools?: string[]; denyExtensions?: boolean; sources?: string[] }; capabilityAudit?: { effectiveTools?: string[]; removedTools?: string[]; extensionsDenied?: boolean } }>;
83
+ results: Array<{ agent?: string; launchContractDigest?: string; launchResolvedExtensions?: LaunchResolvedExtensions; runtimeAcknowledgedExtensions?: RuntimeAcknowledgedExtensions; output?: string; outputPath?: string; outputState?: "present" | "absent" | "unknown"; success?: boolean; error?: string; protocolError?: { code?: string; stream?: string; limitBytes?: number; observedBytes?: number }; timedOut?: boolean; stopped?: boolean; turnBudget?: { maxTurns: number; graceTurns: number; outcome: string; turnCount: number; wrapUpRequestedAtTurn?: number; terminationDeferredAtTurn?: number; exceededAtTurn?: number }; turnBudgetExceeded?: boolean; wrapUpRequested?: boolean; model?: string; attemptedModels?: string[]; modelAttempts?: Array<{ success?: boolean; error?: string }>; totalCost?: { inputTokens: number; outputTokens: number; costUsd: number }; structuredOutput?: unknown; agentContract?: { version: 1 }; execution?: { status?: string; success?: boolean; exitCode?: number }; effects?: { fileMutation?: { status?: string; expected?: boolean; attempted?: boolean } }; intercomTarget?: string; acceptance?: { status?: string; effectiveAcceptance?: { level?: string }; childReport?: unknown; runtimeChecks?: Array<{ id?: string; status?: string; message?: string }> }; artifactPaths?: { outputPath?: string; inputPath?: string; metadataPath?: string }; capabilityCeiling?: { version?: number; allowedTools?: string[]; denyExtensions?: boolean; sources?: string[] }; capabilityAudit?: { effectiveTools?: string[]; removedTools?: string[]; extensionsDenied?: boolean } }>;
84
84
  outputs?: Record<string, { text?: string; structured?: unknown }>;
85
85
  workflowGraph?: { nodes?: Array<{ kind?: string; label?: string; phase?: string; status?: string; acceptanceStatus?: string; error?: string; outputName?: string; structured?: boolean; children?: Array<{ label?: string; outputName?: string; itemKey?: string; status?: string; acceptanceStatus?: string; error?: string }> }> };
86
86
  parallelHandoff?: { version?: number; path?: string; groupCount?: number; childCount?: number; changedPatches?: number; cleanupState?: string };
@@ -453,6 +453,18 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
453
453
  return JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
454
454
  }
455
455
 
456
+ /**
457
+ * Reference-first async child assertion: the result payload carries the
458
+ * saved-output reference; the full text is read from the durable output path.
459
+ */
460
+ function assertAsyncChildOutput(payload: AsyncResultPayload, index: number, expected: string): void {
461
+ const child = payload.results[index];
462
+ assert.ok(child, `expected async child ${index}`);
463
+ assert.match(child.output ?? "", /Output saved to: /);
464
+ assert.ok(child.outputPath, `expected a durable output path for child ${index}`);
465
+ assert.equal(fs.readFileSync(child.outputPath, "utf-8"), expected);
466
+ }
467
+
456
468
  function launchProtocolTest(id: string): void {
457
469
  executeAsyncSingle(id, {
458
470
  agent: "worker",
@@ -480,7 +492,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
480
492
  launchProtocolTest(id);
481
493
  const payload = await readAsyncPayload(id);
482
494
  assert.equal(payload.success, true);
483
- assert.equal(payload.results[0]?.output, "你好 from fragmented async JSON");
495
+ assertAsyncChildOutput(payload, 0, "你好 from fragmented async JSON");
484
496
  });
485
497
 
486
498
  it("persists absent output provenance when async lifecycle text is synthetic", { skip: !isAsyncAvailable() ? "jiti not available" : undefined }, async () => {
@@ -500,7 +512,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
500
512
  const payload = await readAsyncPayload(id);
501
513
  assert.equal(payload.success, false);
502
514
  assert.equal(payload.results[0]?.outputState, "present");
503
- assert.equal(payload.results[0]?.output, "usable partial answer");
515
+ assertAsyncChildOutput(payload, 0, "usable partial answer");
504
516
  });
505
517
 
506
518
  it("matches preflight launch digest in equivalent foreground and async execution", { skip: !isAsyncAvailable() ? "jiti not available" : undefined }, async () => {
@@ -512,11 +524,14 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
512
524
  fs.writeFileSync(agentPath, `---\nname: ${agentName}\ndescription: Contract comparison worker\n---\n`, "utf-8");
513
525
  const discovered = discoverAgents(tempDir).agents.find((agent) => agent.name === agentName);
514
526
  assert.ok(discovered, "expected temporary agent definition to be discovered");
515
- const preflight = await resolveSubagentLaunchContract({ agent: agentName, cwd: tempDir, task, turnBudget, runId: "contract-preflight" });
527
+ // Stable explicit output keeps the preflight launch contract equivalent to
528
+ // both execution paths (generated per-run paths would differ by design).
529
+ const contractOutputPath = path.join(tempDir, "contract-output.md");
530
+ const preflight = await resolveSubagentLaunchContract({ agent: agentName, cwd: tempDir, task, turnBudget, runId: "contract-preflight", output: contractOutputPath });
516
531
  assert.equal(preflight.ok, true);
517
532
 
518
533
  mockPi.onCall({ output: "foreground contract comparison" });
519
- const foreground = await runSync(tempDir, [discovered], agentName, task, { runId: "contract-foreground", acceptance: false, turnBudget });
534
+ const foreground = await runSync(tempDir, [discovered], agentName, task, { runId: "contract-foreground", acceptance: false, turnBudget, outputPath: contractOutputPath });
520
535
  assert.equal(foreground.exitCode, 0);
521
536
  assert.equal(foreground.launchContractDigest, preflight.contract.launchContractDigest);
522
537
 
@@ -525,6 +540,8 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
525
540
  const launch = executeAsyncSingle(asyncId, {
526
541
  agent: agentName,
527
542
  task,
543
+ output: contractOutputPath,
544
+ outputMode: "inline",
528
545
  agentConfig: discovered,
529
546
  ctx: { pi: { events: { emit() {} } }, cwd: tempDir, currentSessionId: "session-1" },
530
547
  artifactConfig: { enabled: false, includeInput: false, includeOutput: false, includeJsonl: false, includeMetadata: false, cleanupDays: 7 },
@@ -721,7 +738,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
721
738
 
722
739
  const launch = await executor.execute(
723
740
  "async-session-artifact-dir",
724
- { agent: "worker", task: "Write async session artifacts", async: true, runId: "async-session-artifacts", acceptance: false },
741
+ { agent: "worker", task: "Write async session artifacts", async: true, runId: "async-session-artifacts", acceptance: false, artifacts: true },
725
742
  new AbortController().signal,
726
743
  undefined,
727
744
  ctx,
@@ -829,7 +846,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
829
846
  launchProtocolTest(id);
830
847
  const payload = await readAsyncPayload(id);
831
848
  assert.equal(payload.success, true);
832
- assert.equal(payload.results[0]?.output, "settled async response");
849
+ assertAsyncChildOutput(payload, 0, "settled async response");
833
850
  assert.ok(Date.now() - startedAt >= 1200, "background runner must not terminate during the retry delay");
834
851
  });
835
852
 
@@ -841,7 +858,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
841
858
  const payload = await readAsyncPayload(id);
842
859
  assert.equal(payload.success, true);
843
860
  assert.equal(payload.results[0]?.error, undefined);
844
- assert.equal(payload.results[0]?.output, "settled async without a terminal assistant stop");
861
+ assertAsyncChildOutput(payload, 0, "settled async without a terminal assistant stop");
845
862
  assert.ok(Date.now() - startedAt < 4000, "agent_settled should trigger bounded child cleanup");
846
863
  });
847
864
 
@@ -872,7 +889,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
872
889
  assert.match(call.args.at(-1) ?? "", /\{outputs\.name\}/);
873
890
  const payload = await readAsyncPayload(id);
874
891
  assert.equal(payload.success, true);
875
- assert.equal(payload.results[0]?.output, "OK");
892
+ assertAsyncChildOutput(payload, 0, "OK");
876
893
  });
877
894
 
878
895
  it("spawns the async runner with node when process.execPath is not node", { skip: !isAsyncAvailable() ? "jiti not available" : undefined }, async () => {
@@ -903,7 +920,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
903
920
  const resultPath = await waitForAsyncResultFile(id, 30_000);
904
921
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
905
922
  assert.equal(payload.success, true);
906
- assert.equal(payload.results[0]?.output, "non-node exec async done");
923
+ assertAsyncChildOutput(payload, 0, "non-node exec async done");
907
924
  } finally {
908
925
  process.execPath = originalExecPath;
909
926
  }
@@ -937,7 +954,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
937
954
  const resultPath = await waitForAsyncResultFile(id, 10_000);
938
955
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
939
956
  assert.equal(payload.success, true);
940
- assert.equal(payload.results[0]?.output, "stale node exec async done");
957
+ assertAsyncChildOutput(payload, 0, "stale node exec async done");
941
958
  } finally {
942
959
  process.execPath = originalExecPath;
943
960
  }
@@ -1178,8 +1195,11 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1178
1195
  assert.equal(payload.turnBudget?.turnCount, 2);
1179
1196
  assert.equal(payload.results[0]?.wrapUpRequested, true);
1180
1197
  assert.equal(payload.results[0]?.turnBudget?.turnCount, 2);
1181
- assert.match(payload.results[0]?.output ?? "", /Turn budget wrap-up was requested after 1 assistant turn/);
1182
- assert.match(payload.results[0]?.output ?? "", /final wrapped output/);
1198
+ // Reference-first delivery: the saved-output reference replaces inline prose;
1199
+ // the wrap-up note and raw output stay visible through status/result fields
1200
+ // and the persisted result file.
1201
+ assert.match(payload.results[0]?.output ?? "", /Output saved to: /);
1202
+ assertAsyncChildOutput(payload, 0, "final wrapped output");
1183
1203
  assert.equal(status.wrapUpRequested, true);
1184
1204
  assert.equal(status.turnBudgetExceeded, undefined);
1185
1205
  assert.equal(status.steps?.[0]?.wrapUpRequested, true);
@@ -1218,8 +1238,9 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1218
1238
  assert.equal(payload.turnBudget?.turnCount, 3);
1219
1239
  assert.equal(payload.turnBudget?.exceededAtTurn, 3);
1220
1240
  assert.equal(payload.results[0]?.turnBudgetExceeded, true);
1221
- assert.match(payload.results[0]?.output ?? "", /Partial output before turn-budget abort:/);
1222
- assert.match(payload.results[0]?.output ?? "", /safe assistant boundary after tool work/);
1241
+ assert.match(payload.error ?? "", /Subagent exceeded turn budget|turn budget/i);
1242
+ assert.match(payload.results[0]?.output ?? "", /Output saved to: /);
1243
+ assertAsyncChildOutput(payload, 0, "safe assistant boundary after tool work");
1223
1244
  assert.equal(status.state, "failed");
1224
1245
  assert.equal(status.turnBudgetExceeded, true);
1225
1246
  assert.equal(status.steps?.[0]?.turnBudgetExceeded, true);
@@ -1272,7 +1293,8 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1272
1293
  assert.equal(payload.turnBudget?.outcome, "exceeded");
1273
1294
  assert.equal(payload.turnBudget?.turnCount, 2);
1274
1295
  assert.equal(payload.results[0]?.turnBudgetExceeded, true);
1275
- assert.match(payload.results[0]?.output ?? "", /safe assistant boundary reached/);
1296
+ assert.match(payload.results[0]?.output ?? "", /Output saved to: /);
1297
+ assertAsyncChildOutput(payload, 0, "safe assistant boundary reached");
1276
1298
  assert.equal(status.state, "failed");
1277
1299
  assert.equal(status.turnBudgetExceeded, true);
1278
1300
  assert.equal(status.steps?.[0]?.turnBudget?.outcome, "exceeded");
@@ -1564,6 +1586,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1564
1586
  tasks: [{ agent: "builder", task: "Do async work", output: "async-top-output.md", reads: ["input.md"] }],
1565
1587
  async: true,
1566
1588
  clarify: false,
1589
+ artifacts: true,
1567
1590
  },
1568
1591
  new AbortController().signal,
1569
1592
  undefined,
@@ -1622,7 +1645,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1622
1645
  ];
1623
1646
  const launch = await executor.execute(
1624
1647
  `async-inherited-output-${outputOverride === true ? "true" : "omitted"}`,
1625
- { tasks, async: true, clarify: false },
1648
+ { tasks, async: true, clarify: false, artifacts: true },
1626
1649
  new AbortController().signal,
1627
1650
  undefined,
1628
1651
  makeMinimalCtx(tempDir),
@@ -1631,8 +1654,8 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1631
1654
  assert.equal(launch.isError, undefined);
1632
1655
  const payload = await readAsyncPayload(launch.details?.asyncId as string);
1633
1656
  assert.equal(payload.success, true);
1634
- assert.equal(payload.results[0]?.output?.split("\n\nOutput saved to:")[0], "first async report");
1635
- assert.equal(payload.results[1]?.output?.split("\n\nOutput saved to:")[0], "second async report");
1657
+ assertAsyncChildOutput(payload, 0, "first async report");
1658
+ assertAsyncChildOutput(payload, 1, "second async report");
1636
1659
  const outputDir = path.join(tempDir, ".pi-subagents", "artifacts", "outputs", launch.details?.asyncId as string);
1637
1660
  const authoritativePaths = [
1638
1661
  path.join(outputDir, "parallel-0", "0-worker", "context.md"),
@@ -1694,6 +1717,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1694
1717
  ],
1695
1718
  async: true,
1696
1719
  clarify: false,
1720
+ artifacts: true,
1697
1721
  },
1698
1722
  new AbortController().signal,
1699
1723
  undefined,
@@ -1932,13 +1956,13 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1932
1956
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
1933
1957
  const status = JSON.parse(fs.readFileSync(path.join(ASYNC_DIR, id, "status.json"), "utf-8")) as AsyncStatusPayload;
1934
1958
  assert.equal(payload.success, true);
1935
- assert.deepEqual(payload.results.map((entry) => entry.output), [
1936
- "Scout A async findings",
1937
- "Scout B async findings",
1938
- "Async funnel synthesis",
1939
- "Async reviewer A done",
1940
- "Async reviewer B done",
1941
- ]);
1959
+ // Reference-first chain delivery: every child result carries the saved-output
1960
+ // reference; full text is read from each durable output path.
1961
+ assertAsyncChildOutput(payload, 0, "Scout A async findings");
1962
+ assertAsyncChildOutput(payload, 1, "Scout B async findings");
1963
+ assertAsyncChildOutput(payload, 2, "Async funnel synthesis");
1964
+ assertAsyncChildOutput(payload, 3, "Async reviewer A done");
1965
+ assertAsyncChildOutput(payload, 4, "Async reviewer B done");
1942
1966
  assert.deepEqual(status.steps?.map((step) => step.status), ["complete", "complete", "complete", "complete", "complete"]);
1943
1967
  assert.deepEqual(status.parallelGroups, [
1944
1968
  { start: 0, count: 2, stepIndex: 0 },
@@ -1946,11 +1970,14 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
1946
1970
  ]);
1947
1971
  const funnelTask = readMockPiArgsMatching(mockPi, "Synthesize:").at(-1) ?? "";
1948
1972
  assert.match(funnelTask, /=== Parallel Task 1 \(scout-a\) ===/);
1949
- assert.match(funnelTask, /Scout A async findings/);
1950
1973
  assert.match(funnelTask, /=== Parallel Task 2 \(scout-b\) ===/);
1951
- assert.match(funnelTask, /Scout B async findings/);
1952
- assert.match(readMockPiArgsMatching(mockPi, "Review funnel A:").at(-1) ?? "", /Review funnel A:\nAsync funnel synthesis/);
1953
- assert.match(readMockPiArgsMatching(mockPi, "Review funnel B:").at(-1) ?? "", /Review funnel B:\nAsync funnel synthesis/);
1974
+ // Chain handoff stays reference-first: the funnel consumes the saved-output
1975
+ // references and reads the named paths instead of re-inlined child prose.
1976
+ assert.match(funnelTask, /Output saved to: /);
1977
+ assert.doesNotMatch(funnelTask, /Scout A async findings/);
1978
+ assert.doesNotMatch(funnelTask, /Scout B async findings/);
1979
+ assert.match(readMockPiArgsMatching(mockPi, "Review funnel A:").at(-1) ?? "", /Review funnel A:\nOutput saved to: /);
1980
+ assert.match(readMockPiArgsMatching(mockPi, "Review funnel B:").at(-1) ?? "", /Review funnel B:\nOutput saved to: /);
1954
1981
  assert.equal(payload.workflowGraph?.nodes?.[0]?.kind, "parallel-group");
1955
1982
  assert.equal(payload.workflowGraph?.nodes?.[0]?.status, "completed");
1956
1983
  assert.equal(payload.workflowGraph?.nodes?.[1]?.kind, "step");
@@ -2341,7 +2368,10 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
2341
2368
  const expectedConsumerTarget = `subagent-consumer-${id}-4`;
2342
2369
  assert.equal(payload.success, true);
2343
2370
  assert.equal(payload.results[3]?.intercomTarget, expectedConsumerTarget);
2344
- assert.deepEqual(JSON.parse(payload.results[3]?.output ?? "{}"), { SELESAI_SUBAGENT_INTERCOM_SESSION_NAME: expectedConsumerTarget });
2371
+ const consumerChild = payload.results[3];
2372
+ assert.ok(consumerChild?.outputPath, "expected a durable output path for the consumer child");
2373
+ assert.match(consumerChild.output ?? "", /Output saved to: /);
2374
+ assert.deepEqual(JSON.parse(fs.readFileSync(consumerChild.outputPath, "utf-8")), { SELESAI_SUBAGENT_INTERCOM_SESSION_NAME: expectedConsumerTarget });
2345
2375
  });
2346
2376
 
2347
2377
  it("async dynamic pre-spawn failures persist failed graph status and error", { skip: !isAsyncAvailable() ? "jiti not available" : undefined }, async () => {
@@ -2428,6 +2458,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
2428
2458
  async: true,
2429
2459
  clarify: false,
2430
2460
  worktree: true,
2461
+ artifacts: true,
2431
2462
  },
2432
2463
  new AbortController().signal,
2433
2464
  undefined,
@@ -2632,7 +2663,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
2632
2663
  assert.equal(payload.results[0]?.model, "openai/gpt-5-mini:high");
2633
2664
  assert.deepEqual(payload.results[0]?.attemptedModels, ["openai/gpt-5-mini:high"]);
2634
2665
  assert.deepEqual(payload.results[0]?.modelAttempts?.map((attempt) => attempt.success), [false, true]);
2635
- assert.match(payload.results[0]?.output ?? "", /\[startup-retry\].*Recovered asynchronously after startup race/s);
2666
+ assertAsyncChildOutput(payload, 0, "Recovered asynchronously after startup race");
2636
2667
  assert.equal(mockPi.callCount(), 2);
2637
2668
  });
2638
2669
 
@@ -2679,7 +2710,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
2679
2710
  const payload = JSON.parse(fs.readFileSync(await waitForAsyncResultFile(id), "utf-8"));
2680
2711
  assert.equal(payload.success, true);
2681
2712
  assert.deepEqual(payload.results[0].attemptedModels, ["openai/gpt-5-mini:high", "anthropic/claude-sonnet-4:low"]);
2682
- assert.match(payload.results[0].output ?? "", /Recovered after stream failure/);
2713
+ assertAsyncChildOutput(payload, 0, "Recovered after stream failure");
2683
2714
  assert.equal(mockPi.callCount(), 2);
2684
2715
  });
2685
2716
 
@@ -2820,7 +2851,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
2820
2851
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
2821
2852
  assert.equal(payload.success, true);
2822
2853
  assert.equal(payload.results[0]?.model, "anthropic/claude-sonnet-4");
2823
- assert.match(payload.results[0]?.output ?? "", /Recovered asynchronously from empty output/);
2854
+ assertAsyncChildOutput(payload, 0, "Recovered asynchronously from empty output");
2824
2855
  assert.match(payload.results[0]?.modelAttempts?.[0]?.error ?? "", /no output/i);
2825
2856
  assert.deepEqual(payload.results[0]?.modelAttempts?.map((attempt) => attempt.success), [false, true]);
2826
2857
  assert.equal(mockPi.callCount(), 2);
@@ -2960,7 +2991,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
2960
2991
  assert.equal(payload.exitCode, 0);
2961
2992
  assert.equal(payload.results[0]?.success, true);
2962
2993
  assert.equal(payload.results[0]?.error, undefined);
2963
- assert.equal(payload.results[0]?.output, "Recovered asynchronously");
2994
+ assertAsyncChildOutput(payload, 0, "Recovered asynchronously");
2964
2995
  const statusPayload = JSON.parse(fs.readFileSync(path.join(asyncDir, "status.json"), "utf-8")) as AsyncStatusPayload;
2965
2996
  assert.equal(statusPayload.state, "complete");
2966
2997
  assert.equal(statusPayload.steps?.[0]?.status, "complete");
@@ -3364,7 +3395,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
3364
3395
  assert.equal(payload.success, true);
3365
3396
  assert.equal(payload.exitCode, 0);
3366
3397
  assert.equal(payload.results[0].success, true);
3367
- assert.equal(payload.results[0].output, "cold start test after patch");
3398
+ assertAsyncChildOutput(payload, 0, "cold start test after patch");
3368
3399
 
3369
3400
  const eventsPath = path.join(ASYNC_DIR, id, "events.jsonl");
3370
3401
  const eventsText = fs.readFileSync(eventsPath, "utf-8");
@@ -4070,7 +4101,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4070
4101
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
4071
4102
  assert.ok(elapsed < 6000, `unconfigured watchdog status should not delay async final drain, took ${elapsed}ms`);
4072
4103
  assert.equal(payload.success, true);
4073
- assert.equal(payload.results[0]?.output, "async-done-without-watchdog-config");
4104
+ assertAsyncChildOutput(payload, 0, "async-done-without-watchdog-config");
4074
4105
  assert.equal((payload.results[0] as { watchdog?: unknown }).watchdog, undefined);
4075
4106
  });
4076
4107
  });
@@ -4105,7 +4136,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4105
4136
  assert.ok(elapsed >= 1200, `watchdog settlement should delay async final drain, took ${elapsed}ms`);
4106
4137
  assert.ok(elapsed < 9000, `settled watchdog should still allow async cleanup, took ${elapsed}ms`);
4107
4138
  assert.equal(payload.success, true);
4108
- assert.equal(payload.results[0]?.output, "async-done-before-watchdog");
4139
+ assertAsyncChildOutput(payload, 0, "async-done-before-watchdog");
4109
4140
  assert.equal((payload.results[0] as { watchdog?: { phase?: string } }).watchdog?.phase, "idle");
4110
4141
  });
4111
4142
  });
@@ -4136,7 +4167,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4136
4167
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
4137
4168
  assert.ok(elapsed < 6000, `watchdog tail fallback should not hang async final drain, took ${elapsed}ms`);
4138
4169
  assert.equal(payload.success, true);
4139
- assert.equal(payload.results[0]?.output, "async-done-before-watchdog-timeout");
4170
+ assertAsyncChildOutput(payload, 0, "async-done-before-watchdog-timeout");
4140
4171
  const watchdog = (payload.results[0] as { watchdog?: { phase?: string; timedOut?: boolean } }).watchdog;
4141
4172
  assert.equal(watchdog?.phase, "stale");
4142
4173
  assert.equal(watchdog?.timedOut, true);
@@ -4187,7 +4218,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4187
4218
  assert.equal(payload.success, true);
4188
4219
  assert.equal(payload.exitCode, 0);
4189
4220
  assert.equal(payload.results[0].success, true);
4190
- assert.equal(payload.results[0].output, "async-done-before-drain");
4221
+ assertAsyncChildOutput(payload, 0, "async-done-before-drain");
4191
4222
  });
4192
4223
 
4193
4224
  it("background forced drain after empty terminal assistant output is cleanup success", { skip: !isAsyncAvailable() ? "jiti not available" : undefined }, async () => {
@@ -4223,7 +4254,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4223
4254
  assert.equal(payload.success, true);
4224
4255
  assert.equal(payload.exitCode, 0);
4225
4256
  assert.equal(payload.results[0].success, true);
4226
- assert.equal(payload.results[0].output, "");
4257
+ assert.match(payload.results[0].output ?? "", /Output saved to: /);
4227
4258
  });
4228
4259
 
4229
4260
  it("background final-drain cleanup preserves explicit assistant errors", { skip: !isAsyncAvailable() ? "jiti not available" : undefined }, async () => {
@@ -4596,7 +4627,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4596
4627
  const resultPath = await waitForAsyncResultFile(id, 10_000);
4597
4628
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8")) as AsyncResultPayload;
4598
4629
  assert.equal(payload.success, true);
4599
- assert.equal(payload.results[0]?.output, "Done after noisy stream");
4630
+ assertAsyncChildOutput(payload, 0, "Done after noisy stream");
4600
4631
 
4601
4632
  const eventsText = fs.readFileSync(path.join(asyncDir, "events.jsonl"), "utf-8");
4602
4633
  assert.doesNotMatch(eventsText, /"type":"message_update"/);
@@ -4675,7 +4706,7 @@ describe("async execution utilities", { skip: !available ? "pi packages not avai
4675
4706
 
4676
4707
  const payload = JSON.parse(fs.readFileSync(resultPath, "utf-8"));
4677
4708
  assert.equal(payload.success, true);
4678
- assert.equal(payload.results[0].output, "Done streaming");
4709
+ assertAsyncChildOutput(payload, 0, "Done streaming");
4679
4710
 
4680
4711
  const status = JSON.parse(fs.readFileSync(path.join(asyncDir, "status.json"), "utf-8"));
4681
4712
  assert.deepEqual(status.steps[0].recentTools.map((tool: { tool: string; args: string }) => ({ tool: tool.tool, args: tool.args })), [{ tool: "bash", args: "ls" }]);