@selesai/code 0.13.31 → 0.13.33

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/dist/core/remote-catalog-provider.js +5 -1
  3. package/dist/core/remote-catalog-provider.test.d.ts +1 -0
  4. package/dist/core/remote-catalog-provider.test.js +29 -0
  5. package/dist/extensions/capability-gateway/index.ts +9 -4
  6. package/dist/extensions/capability-gateway/integration.test.ts +69 -3
  7. package/dist/extensions/pi-subagents/agents/worker.md +3 -2
  8. package/dist/extensions/pi-subagents/docs/agents.md +2 -0
  9. package/dist/extensions/pi-subagents/docs/extension-api.md +3 -1
  10. package/dist/extensions/pi-subagents/docs/observability.md +2 -0
  11. package/dist/extensions/pi-subagents/docs/tool-reference.md +2 -2
  12. package/dist/extensions/pi-subagents/docs/workflows.md +1 -1
  13. package/dist/extensions/pi-subagents/src/extension/rpc.ts +10 -1
  14. package/dist/extensions/pi-subagents/src/runs/background/active-async-capacity.ts +3 -21
  15. package/dist/extensions/pi-subagents/src/runs/background/async-execution.ts +2 -1
  16. package/dist/extensions/pi-subagents/src/runs/background/run-child-session.ts +1 -0
  17. package/dist/extensions/pi-subagents/src/runs/background/run-status.ts +5 -1
  18. package/dist/extensions/pi-subagents/src/runs/background/subagent-runner.ts +10 -3
  19. package/dist/extensions/pi-subagents/src/runs/background/workflow-terminal-proof.ts +67 -0
  20. package/dist/extensions/pi-subagents/src/runs/foreground/execution.ts +6 -2
  21. package/dist/extensions/pi-subagents/src/runs/shared/child-tool-plan.ts +67 -5
  22. package/dist/extensions/pi-subagents/src/runs/shared/completion-guard.ts +6 -0
  23. package/dist/extensions/pi-subagents/src/runs/shared/external-cli-runner.ts +2 -1
  24. package/dist/extensions/pi-subagents/src/runs/shared/git-environment.ts +29 -0
  25. package/dist/extensions/pi-subagents/src/runs/shared/structured-output.ts +69 -0
  26. package/dist/extensions/pi-subagents/src/shared/types.ts +20 -0
  27. package/dist/extensions/pi-subagents/src/slash/slash-commands.ts +4 -238
  28. package/dist/extensions/pi-subagents/src/slash/subagent-cost.ts +280 -0
  29. package/dist/extensions/pi-subagents/test/integration/async-execution.part-3.test.ts +67 -0
  30. package/dist/extensions/pi-subagents/test/integration/in-process-child.test.ts +29 -1
  31. package/dist/extensions/pi-subagents/test/integration/intercom-result-delivery.test.ts +6 -3
  32. package/dist/extensions/pi-subagents/test/integration/single-execution.part-2.test.ts +25 -0
  33. package/dist/extensions/pi-subagents/test/unit/agent-frontmatter.test.ts +5 -5
  34. package/dist/extensions/pi-subagents/test/unit/async-spawn-preload.test.ts +21 -0
  35. package/dist/extensions/pi-subagents/test/unit/child-tool-plan-permission-system.test.ts +98 -0
  36. package/dist/extensions/pi-subagents/test/unit/child-tool-plan.test.ts +11 -0
  37. package/dist/extensions/pi-subagents/test/unit/completion-guard.test.ts +12 -0
  38. package/dist/extensions/pi-subagents/test/unit/external-cli-runner.test.ts +25 -0
  39. package/dist/extensions/pi-subagents/test/unit/git-environment.test.ts +38 -0
  40. package/dist/extensions/pi-subagents/test/unit/preflight.test.ts +3 -1
  41. package/dist/extensions/pi-subagents/test/unit/rpc.test.ts +105 -1
  42. package/dist/extensions/pi-subagents/test/unit/run-status.test.ts +36 -0
  43. package/dist/extensions/pi-subagents/test/unit/structured-output-rejection.test.ts +67 -0
  44. package/dist/extensions/pi-subagents/test/unit/workflow-terminal-proof.test.ts +98 -0
  45. package/dist/extensions/pi-web-agent/src/extension.ts +2 -1
  46. package/dist/skills/pi-subagents/references/constraints-and-recipes.md +8 -8
  47. package/dist/skills/pi-subagents/references/execution-controls.md +1 -1
  48. package/dist/skills/pi-subagents/references/prompting-and-roles.md +1 -1
  49. package/package.json +3 -3
@@ -11,6 +11,7 @@ import * as path from "node:path";
11
11
  import { afterEach, describe, it } from "node:test";
12
12
  import { buildInProcessChildLaunch, type BuildInProcessChildLaunchInput } from "../../src/runs/shared/child-launch.ts";
13
13
  import {
14
+ CAPABILITY_GATEWAY_EXTENSION_PATH,
14
15
  resolvePermissionSystemExtension,
15
16
  resolvePiLaunchToolPlan,
16
17
  } from "../../src/runs/shared/child-tool-plan.ts";
@@ -19,6 +20,7 @@ const originalEnv = {
19
20
  HOME: process.env.HOME,
20
21
  USERPROFILE: process.env.USERPROFILE,
21
22
  SELESAI_CODING_AGENT_DIR: process.env.SELESAI_CODING_AGENT_DIR,
23
+ SELESAI_CAPABILITY_GATEWAY: process.env.SELESAI_CAPABILITY_GATEWAY,
22
24
  };
23
25
  const tempRoots: string[] = [];
24
26
 
@@ -33,6 +35,7 @@ function createFixture() {
33
35
  process.env.HOME = home;
34
36
  process.env.USERPROFILE = home;
35
37
  process.env.SELESAI_CODING_AGENT_DIR = agentDir;
38
+ delete process.env.SELESAI_CAPABILITY_GATEWAY;
36
39
  process.chdir(projectDir);
37
40
  return { root, agentDir, projectDir };
38
41
  }
@@ -397,3 +400,98 @@ describe("child launch <active_agent> tag injection", () => {
397
400
  );
398
401
  });
399
402
  });
403
+
404
+ describe("capability gateway child runtime policy", () => {
405
+ const gatewayTools = ["capability_catalog", "capability_discover", "capability_skill_show"];
406
+
407
+ for (const host of ["parent", "runner"] as const) {
408
+ it(`adds only gateway controls for extension-permitted ${host} children`, () => {
409
+ const { agentDir } = createFixture();
410
+ process.env.SELESAI_CODING_AGENT_DIR = agentDir;
411
+
412
+ const launch = buildInProcessChildLaunch(childLaunch({ host, tools: ["read", "web_explore"] }));
413
+ assert.equal(launch.toolPlan.capabilityGatewayEnabled, true);
414
+ assert.ok(launch.session.extensionPaths.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
415
+ assert.deepEqual(launch.session.tools, ["read", "web_explore", ...gatewayTools]);
416
+ assert.ok(gatewayTools.every((tool) => launch.toolPlan.requiredChildTools.includes(tool)));
417
+ if (host === "runner") {
418
+ assert.equal(launch.session.ambientExtensions, true);
419
+ assert.ok(launch.session.extensionPaths.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
420
+ } else {
421
+ assert.equal(launch.session.ambientExtensions, false);
422
+ assert.equal(launch.session.processEnv, undefined);
423
+ }
424
+ });
425
+ }
426
+
427
+ it("keeps omitted tools unrestricted without synthesizing an explicit allowlist", () => {
428
+ const { agentDir } = createFixture();
429
+ process.env.SELESAI_CODING_AGENT_DIR = agentDir;
430
+ const launch = buildInProcessChildLaunch(childLaunch({ host: "runner" }));
431
+ assert.equal(launch.toolPlan.capabilityGatewayEnabled, true);
432
+ assert.equal(launch.session.tools, undefined);
433
+ assert.ok(launch.session.extensionPaths.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
434
+ });
435
+
436
+ for (const host of ["parent", "runner"] as const) {
437
+ it(`skips gateway for ${host} children with an explicit empty tool set`, () => {
438
+ const { agentDir } = createFixture();
439
+ process.env.SELESAI_CODING_AGENT_DIR = agentDir;
440
+
441
+ const launch = buildInProcessChildLaunch(childLaunch({ host, tools: [] }));
442
+ assert.equal(launch.toolPlan.capabilityGatewayEnabled, false);
443
+ assert.deepEqual(launch.session.tools, []);
444
+ assert.ok(!launch.toolPlan.extensionArgs.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
445
+ assert.ok(!launch.session.extensionPaths.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
446
+ });
447
+ }
448
+
449
+ it("does not widen a capability ceiling and honors explicit extension policy", () => {
450
+ const { agentDir, projectDir } = createFixture();
451
+ process.env.SELESAI_CODING_AGENT_DIR = agentDir;
452
+ const limited = resolvePiLaunchToolPlan({
453
+ tools: ["read", "web_explore"],
454
+ capabilityCeiling: { version: 1, allowedTools: ["read", "web_explore"], denyExtensions: false, sources: ["test"] },
455
+ });
456
+ assert.equal(limited.capabilityGatewayEnabled, false);
457
+ assert.deepEqual(limited.effectiveToolAllowlist, ["read", "web_explore"]);
458
+ assert.ok(!limited.extensionArgs.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
459
+
460
+ const explicitBlock = resolvePiLaunchToolPlan({
461
+ tools: ["read"],
462
+ extensions: [path.join(projectDir, "other-extension.ts")],
463
+ });
464
+ assert.equal(explicitBlock.capabilityGatewayEnabled, false);
465
+ assert.ok(!explicitBlock.extensionArgs.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
466
+
467
+ const explicitOptIn = resolvePiLaunchToolPlan({
468
+ tools: ["read"],
469
+ extensions: [],
470
+ subagentOnlyExtensions: [path.dirname(CAPABILITY_GATEWAY_EXTENSION_PATH)],
471
+ });
472
+ assert.equal(explicitOptIn.capabilityGatewayEnabled, true);
473
+ assert.ok(explicitOptIn.extensionArgs.includes(path.dirname(CAPABILITY_GATEWAY_EXTENSION_PATH)));
474
+ assert.ok(explicitOptIn.effectiveToolAllowlist.includes("capability_discover"));
475
+ });
476
+
477
+ for (const host of ["parent", "runner"] as const) {
478
+ it(`does not load or authorize the gateway for ${host} children with denyExtensions`, () => {
479
+ const { agentDir } = createFixture();
480
+ process.env.SELESAI_CODING_AGENT_DIR = agentDir;
481
+
482
+ const launch = buildInProcessChildLaunch(childLaunch({
483
+ host,
484
+ tools: ["read"],
485
+ subagentOnlyExtensions: [CAPABILITY_GATEWAY_EXTENSION_PATH],
486
+ capabilityCeiling: { version: 1, allowedTools: ["read"], denyExtensions: true, sources: ["test"] },
487
+ }));
488
+ assert.equal(launch.toolPlan.capabilityGatewayEnabled, false);
489
+ assert.deepEqual(launch.session.tools, ["read"]);
490
+ assert.ok(!launch.session.extensionPaths.includes(CAPABILITY_GATEWAY_EXTENSION_PATH));
491
+ assert.ok(!launch.toolPlan.requiredChildTools.some((tool) => tool.startsWith("capability_")));
492
+ if (host === "runner") {
493
+ assert.equal(launch.session.ambientExtensions, false);
494
+ }
495
+ });
496
+ }
497
+ });
@@ -19,6 +19,17 @@ function runtimeSnapshotHost(serverName: string): McpRuntimeSnapshotHost {
19
19
  }
20
20
 
21
21
  describe("child tool plan", () => {
22
+ it("requires fanout authorization for the child supervisor reply tool", () => {
23
+ assert.throws(() => resolvePiLaunchToolPlan({ tools: ["read", "subagent_supervisor"] }), /subagent_supervisor.*requires fanout authorization/);
24
+ assert.throws(() => resolvePiLaunchToolPlan({ tools: ["read", "subagent", "subagent_supervisor"], excludeTools: ["subagent"] }), /subagent_supervisor.*requires fanout authorization/);
25
+ assert.throws(() => resolvePiLaunchToolPlan({
26
+ tools: ["read", "subagent", "subagent_supervisor"],
27
+ capabilityCeiling: { version: 1, allowedTools: ["read", "subagent_supervisor"], sources: ["test"] },
28
+ }), /subagent_supervisor.*requires fanout authorization/);
29
+ assert.equal(resolvePiLaunchToolPlan({ tools: ["read", "subagent", "subagent_supervisor"] }).fanoutAuthorized, true);
30
+ assert.equal(resolvePiLaunchToolPlan({ tools: ["read", "subagent_supervisor"], allowNestedSubagents: true }).fanoutAuthorized, true);
31
+ });
32
+
22
33
  it("fails a launch that selects MCP tools from the adapter's runtime snapshot", () => {
23
34
  const cwd = fs.mkdtempSync(path.join(os.tmpdir(), "pi-subagents-runtime-mcp-"));
24
35
  try {
@@ -7,6 +7,7 @@ import {
7
7
  evaluateCompletionMutationGuard,
8
8
  expectsImplementationMutation,
9
9
  hasMutationToolCall,
10
+ hasMutationToolCapability,
10
11
  validateImplementationToolContract,
11
12
  } from "../../src/runs/shared/completion-guard.ts";
12
13
  import { isMutatingTool } from "../../src/runs/shared/long-running-guard.ts";
@@ -500,6 +501,17 @@ test("edit and write tool calls count as mutation attempts", () => {
500
501
  assert.equal(hasMutationToolCall([assistantToolCall("write", { path: "a.ts" })]), true);
501
502
  });
502
503
 
504
+ test("capability gateway controls never count as mutation capability", () => {
505
+ const gatewayTools = ["capability_catalog", "capability_discover", "capability_skill_show"];
506
+ assert.equal(hasMutationToolCapability(["read", "grep", ...gatewayTools], undefined), false);
507
+ assert.equal(validateImplementationToolContract({
508
+ agent: "worker",
509
+ task: "Implement the requested source fix.",
510
+ tools: ["read", "grep", "find", "ls", "contact_supervisor", ...gatewayTools],
511
+ }), "Agent 'worker' was given an implementation task, but its tool allowlist has no mutation-capable tools. Add bash, edit, write, or another mutation-capable tool to the agent, or use a read-only task/agent.");
512
+ assert.equal(hasMutationToolCapability(["read", "web_explore"], undefined), true);
513
+ });
514
+
503
515
  test("declared extension mutation tools count without weakening unknown tools", () => {
504
516
  const messages = [assistantToolCall("replace", { remove_from: "Liv" })];
505
517
  assert.equal(hasMutationToolCall(messages), false);
@@ -43,6 +43,31 @@ describe("external CLI runner", () => {
43
43
  }
44
44
  });
45
45
 
46
+ it("strips inherited Git routing vars but honors an explicit environment allowlist", async () => {
47
+ const dir = tempDir();
48
+ const inherited = { GIT_DIR: "/outer/repo/.git", GIT_AUTHOR_NAME: "Kept Author" };
49
+ const previous = Object.fromEntries(Object.keys(inherited).map((key) => [key, process.env[key]]));
50
+ Object.assign(process.env, inherited);
51
+ const script = "process.stdout.write(JSON.stringify({gitDir:process.env.GIT_DIR??null,author:process.env.GIT_AUTHOR_NAME??null}))";
52
+ try {
53
+ const defaultEnv = await runExternalCli({ command: process.execPath, args: ["-e", script], cwd: dir, prompt: "x", asyncDir: dir, stepIndex: 1 });
54
+ assert.equal(defaultEnv.exitCode, 0);
55
+ assert.deepEqual(JSON.parse(defaultEnv.output), { gitDir: null, author: "Kept Author" });
56
+
57
+ const explicitEnv = await runExternalCli({
58
+ command: process.execPath, args: ["-e", script], cwd: dir, prompt: "x", asyncDir: dir, stepIndex: 2,
59
+ environment: { allowlist: ["GIT_DIR"] },
60
+ });
61
+ assert.equal(explicitEnv.exitCode, 0);
62
+ assert.deepEqual(JSON.parse(explicitEnv.output), { gitDir: "/outer/repo/.git", author: null });
63
+ } finally {
64
+ for (const [key, value] of Object.entries(previous)) {
65
+ if (value === undefined) delete process.env[key];
66
+ else process.env[key] = value;
67
+ }
68
+ }
69
+ });
70
+
46
71
  it("delivers the combined prompt only through stdin and preserves argv", async () => {
47
72
  const dir = tempDir();
48
73
  const prompt = buildExternalCliPrompt("Follow exactly.", "Review $HOME; echo nope");
@@ -0,0 +1,38 @@
1
+ import assert from "node:assert/strict";
2
+ import { describe, it } from "node:test";
3
+ import { omitGitRoutingEnv } from "../../src/runs/shared/git-environment.ts";
4
+
5
+ const GIT_ROUTING_ENV = {
6
+ GIT_ALTERNATE_OBJECT_DIRECTORIES: "/repo/objects",
7
+ GIT_COMMON_DIR: "/repo/common",
8
+ GIT_CONFIG: "/repo/config",
9
+ GIT_CONFIG_COUNT: "1",
10
+ GIT_CONFIG_KEY_0: "core.worktree",
11
+ GIT_CONFIG_PARAMETERS: "'core.worktree=/repo'",
12
+ GIT_CONFIG_VALUE_0: "/repo",
13
+ GIT_DIR: "/repo/.git",
14
+ GIT_GRAFT_FILE: "/repo/grafts",
15
+ GIT_IMPLICIT_WORK_TREE: "1",
16
+ GIT_INDEX_FILE: "/repo/index",
17
+ GIT_NAMESPACE: "tenant",
18
+ GIT_NO_REPLACE_OBJECTS: "1",
19
+ GIT_OBJECT_DIRECTORY: "/repo/object-dir",
20
+ GIT_PREFIX: "src/",
21
+ GIT_REPLACE_REF_BASE: "refs/replace/",
22
+ GIT_SHALLOW_FILE: "/repo/shallow",
23
+ GIT_WORK_TREE: "/repo",
24
+ };
25
+
26
+ describe("omitGitRoutingEnv", () => {
27
+ it("removes inherited repository routing and keeps unrelated values", () => {
28
+ assert.deepEqual(omitGitRoutingEnv({ ...GIT_ROUTING_ENV, KEEP_ME: "yes" }), { KEEP_ME: "yes" });
29
+ });
30
+
31
+ it("removes every indexed Git config entry", () => {
32
+ assert.deepEqual(omitGitRoutingEnv({ GIT_CONFIG_KEY_17: "user.name", GIT_CONFIG_VALUE_17: "Test", KEEP_ME: "yes" }), { KEEP_ME: "yes" });
33
+ });
34
+
35
+ it("matches names case-insensitively, as Windows and Git for Windows do", () => {
36
+ assert.deepEqual(omitGitRoutingEnv({ git_dir: "/repo/.git", Git_Work_Tree: "/repo", git_config_key_0: "core.bare", KEEP_ME: "yes" }), { KEEP_ME: "yes" });
37
+ });
38
+ });
@@ -857,7 +857,9 @@ Project prompt.
857
857
  assert.equal(result.ok, true);
858
858
  if (!result.ok) return;
859
859
  assert.deepEqual(result.contract.tools.excludeTools, ["write", "unknown_tool"]);
860
- assert.deepEqual(result.contract.tools.effectiveAllowlist, ["read"]);
860
+ // An explicit tool surface also gets the scoped capability gateway tools, which
861
+ // this launch does not exclude (see the capability-gateway child policy tests).
862
+ assert.deepEqual(result.contract.tools.effectiveAllowlist, ["read", "capability_catalog", "capability_discover", "capability_skill_show"]);
861
863
  assert.match(result.contract.launchContractDigest, /^[a-f0-9]{64}$/);
862
864
  });
863
865
 
@@ -5,6 +5,7 @@ import * as path from "node:path";
5
5
  import { describe, it } from "node:test";
6
6
  import type { AgentToolResult } from "@earendil-works/pi-agent-core";
7
7
  import { consumeStopRequestPayload, stopRequestPath, stopRequestsDir } from "../../src/runs/background/control-channel.ts";
8
+ import { getArtifactPaths, getArtifactsDir } from "../../src/shared/artifacts.ts";
8
9
  import {
9
10
  SUBAGENT_RPC_PROTOCOL_VERSION,
10
11
  SUBAGENT_RPC_READY_EVENT,
@@ -13,7 +14,7 @@ import {
13
14
  subagentRpcReplyEvent,
14
15
  type SubagentRpcReplyEnvelope,
15
16
  } from "../../src/extension/rpc.ts";
16
- import { SUBAGENT_CHILD_STATUS_EVENT, type Details, type SubagentChildStatusEvent, type SubagentState } from "../../src/shared/types.ts";
17
+ import { DIRS, SUBAGENT_CHILD_STATUS_EVENT, type Details, type SubagentChildStatusEvent, type SubagentState } from "../../src/shared/types.ts";
17
18
 
18
19
  class FakeEvents {
19
20
  readonly emitted: Array<{ event: string; data: unknown }> = [];
@@ -117,6 +118,8 @@ describe("subagent extension RPC bridge", () => {
117
118
  (reply as { data: { capabilities?: { statusProjection?: unknown } } }).data.capabilities?.statusProjection,
118
119
  { version: 1, untargeted: "in-memory-when-ready", targeted: "executor" },
119
120
  );
121
+ assert.deepEqual((reply as { data: { capabilities?: { cost?: unknown } } }).data.capabilities?.cost, { version: 1 });
122
+ assert.ok((reply as { data: { methods?: string[] } }).data.methods?.includes("cost"));
120
123
 
121
124
  bridge.dispose();
122
125
  });
@@ -1147,4 +1150,105 @@ describe("subagent extension RPC bridge", () => {
1147
1150
  fs.rmSync(root, { recursive: true, force: true });
1148
1151
  }
1149
1152
  });
1153
+
1154
+ it("answers cost with structured parent-plus-child accounting from the session branch", async () => {
1155
+ const events = new FakeEvents();
1156
+ const childUsage = { input: 8, output: 3, cacheRead: 5, cacheWrite: 0, cost: 0.25, turns: 1 };
1157
+ const branch = [
1158
+ { type: "message", message: { role: "assistant", usage: { input: 15, output: 3, cacheRead: 30, cacheWrite: 0, cost: { total: 0.25 } } } },
1159
+ { type: "message", message: { role: "toolResult", toolName: "subagent", details: { mode: "single", results: [{ agent: "reviewer", runId: "run-a", usage: childUsage, sessionFile: "/sessions/child-a.jsonl" }] } } },
1160
+ { type: "message", message: { role: "toolResult", toolName: "bg_wait", details: { mode: "single", results: [], completions: [{ mode: "single", runId: "run-a", results: [{ agent: "reviewer", runId: "run-a", usage: childUsage }] }] } } },
1161
+ ];
1162
+ const bridge = registerSubagentRpcBridge({
1163
+ events,
1164
+ getContext: () => ({
1165
+ cwd: "/repo",
1166
+ sessionManager: { getSessionId: () => "session-123", getSessionFile: () => "/sessions/parent.jsonl", getBranch: () => branch },
1167
+ }) as any,
1168
+ execute: async () => assert.fail("cost should not call executor"),
1169
+ state: { baseCwd: "/repo" } as SubagentState,
1170
+ });
1171
+
1172
+ const reply = await request(events, "cost-1", "cost");
1173
+ assert.equal(reply.success, true);
1174
+ assert.equal(reply.method, "cost");
1175
+ const data = (reply as { data: { version: number; parent: Record<string, number>; children: Array<{ agent?: string; runId?: string }>; childTotal: Record<string, number>; total: Record<string, number>; unresolvedAsyncChildren: number } }).data;
1176
+ assert.equal(data.version, 1);
1177
+ assert.deepEqual(data.parent, { input: 15, output: 3, cacheRead: 30, cacheWrite: 0, cost: 0.25, turns: 1 });
1178
+ assert.equal(data.children.length, 1, "run identity deduplicates the bg_wait completion");
1179
+ assert.equal(data.children[0]?.agent, "reviewer");
1180
+ assert.equal(data.children[0]?.runId, "run-a");
1181
+ assert.deepEqual(data.childTotal, childUsage);
1182
+ assert.deepEqual(data.total, { input: 23, output: 6, cacheRead: 35, cacheWrite: 0, cost: 0.5, turns: 2 });
1183
+ assert.equal(data.unresolvedAsyncChildren, 0);
1184
+
1185
+ const rejected = await request(events, "cost-2", "cost", "not-an-object");
1186
+ assert.equal(rejected.success, false);
1187
+ assert.equal((rejected as { error: { code: string } }).error.code, "invalid_params");
1188
+ bridge.dispose();
1189
+ });
1190
+
1191
+ it("answers cost with receipt-recovered async usage and an unresolved lower bound", async () => {
1192
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-rpc-cost-async-"));
1193
+ const workflowRunId = `workflow-rpc-cost-${process.pid}-${Date.now()}`;
1194
+ const recoveredRunId = `child-rpc-cost-${process.pid}-${Date.now()}`;
1195
+ const unresolvedRunId = `child-rpc-cost-missing-${process.pid}-${Date.now()}`;
1196
+ const unresolvedWorkflowRunId = `workflow-rpc-cost-missing-${process.pid}-${Date.now()}`;
1197
+ const asyncDir = path.join(DIRS.async, workflowRunId);
1198
+ try {
1199
+ const sessionFile = path.join(root, "sessions", "parent.jsonl");
1200
+ const artifactsDir = getArtifactsDir(sessionFile, root, "session");
1201
+ const metadataPath = getArtifactPaths(artifactsDir, recoveredRunId, "reviewer", 0).metadataPath;
1202
+ fs.mkdirSync(path.dirname(sessionFile), { recursive: true });
1203
+ fs.writeFileSync(sessionFile, "", "utf-8");
1204
+ fs.mkdirSync(asyncDir, { recursive: true });
1205
+ fs.mkdirSync(path.dirname(metadataPath), { recursive: true });
1206
+ fs.writeFileSync(path.join(asyncDir, "workflow-receipt.json"), JSON.stringify({
1207
+ version: 1, workflowRunId, state: "complete", createdAt: Date.now(),
1208
+ entries: {
1209
+ recovered: { key: "recovered", agent: "reviewer", latestRunId: recoveredRunId, continuation: { runIds: [recoveredRunId] }, resumability: { state: "resumable" } },
1210
+ unresolved: { key: "unresolved", agent: "worker", latestRunId: unresolvedRunId, continuation: { runIds: [unresolvedRunId] }, resumability: { state: "resumable" } },
1211
+ nestedWorkflow: { key: "nestedWorkflow", latestRunId: unresolvedWorkflowRunId, continuation: { runIds: [unresolvedWorkflowRunId] }, resumability: { state: "resumable" } },
1212
+ },
1213
+ }, null, 2), "utf-8");
1214
+ const recoveredUsage = { input: 20, output: 4, cacheRead: 80, cacheWrite: 0, cost: 0.5, turns: 2 };
1215
+ fs.writeFileSync(metadataPath, JSON.stringify({ runId: recoveredRunId, agent: "reviewer", usage: recoveredUsage }), "utf-8");
1216
+
1217
+ const events = new FakeEvents();
1218
+ const bridge = registerSubagentRpcBridge({
1219
+ events,
1220
+ getContext: () => ({
1221
+ cwd: root,
1222
+ sessionManager: {
1223
+ getSessionId: () => "session-async-cost",
1224
+ getSessionFile: () => sessionFile,
1225
+ getBranch: () => [{ type: "message", message: { role: "toolResult", toolName: "subagent", details: { mode: "workflow", runId: workflowRunId, results: [] } } }],
1226
+ },
1227
+ }) as any,
1228
+ execute: async () => assert.fail("cost should not call executor"),
1229
+ state: { baseCwd: root, artifactDirPreference: "session" } as SubagentState,
1230
+ });
1231
+
1232
+ const reply = await request(events, "cost-async", "cost");
1233
+ assert.equal(reply.success, true);
1234
+ const data = (reply as { data: { children: Array<{ agent?: string; runId?: string }>; childTotal: Record<string, number>; total: Record<string, number>; unresolvedAsyncChildren: number } }).data;
1235
+ assert.deepEqual(data.children.map(({ agent, runId }) => ({ agent, runId })), [{ agent: "reviewer", runId: recoveredRunId }]);
1236
+ assert.deepEqual(data.childTotal, recoveredUsage);
1237
+ assert.deepEqual(data.total, recoveredUsage, "totals remain a lower bound when metadata is unavailable");
1238
+ assert.equal(data.unresolvedAsyncChildren, 2);
1239
+ bridge.dispose();
1240
+ } finally {
1241
+ fs.rmSync(asyncDir, { recursive: true, force: true });
1242
+ fs.rmSync(root, { recursive: true, force: true });
1243
+ }
1244
+ });
1245
+
1246
+ it("cost requires an active session context like every non-ping method", async () => {
1247
+ const events = new FakeEvents();
1248
+ const bridge = registerSubagentRpcBridge({ events, getContext: () => null, execute: async () => assert.fail("cost should not call executor") });
1249
+ const reply = await request(events, "cost-3", "cost");
1250
+ assert.equal(reply.success, false);
1251
+ assert.equal((reply as { error: { code: string } }).error.code, "no_active_session");
1252
+ bridge.dispose();
1253
+ });
1150
1254
  });
@@ -1630,4 +1630,40 @@ describe("async run status inspection", () => {
1630
1630
  fs.rmSync(root, { recursive: true, force: true });
1631
1631
  }
1632
1632
  });
1633
+
1634
+ it("returns a workflow terminal proof after every async child exits", () => {
1635
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-run-status-workflow-terminal-"));
1636
+ try {
1637
+ const asyncRoot = path.join(root, "runs");
1638
+ const asyncDir = path.join(asyncRoot, "workflow-parent");
1639
+ const childDir = path.join(asyncRoot, "child-run");
1640
+ fs.mkdirSync(asyncDir, { recursive: true });
1641
+ fs.mkdirSync(childDir);
1642
+ const childProof = {
1643
+ version: 1, state: "observed", runId: "child-run", runnerProcessInstanceId: "runner-1", observedAt: 300,
1644
+ instances: [{ kind: "runner", processInstanceId: "runner-1", closeObservedAt: 300, exitCode: 0, signal: null }],
1645
+ };
1646
+ fs.writeFileSync(path.join(childDir, "process-terminal.json"), JSON.stringify(childProof));
1647
+ fs.writeFileSync(path.join(childDir, "status.json"), JSON.stringify({
1648
+ runId: "child-run", mode: "single", state: "complete", startedAt: 100, lastUpdate: 300,
1649
+ processTerminal: { version: 1, state: "pending", runId: "child-run", runnerProcessInstanceId: "runner-1" },
1650
+ }));
1651
+ fs.writeFileSync(path.join(asyncDir, "status.json"), JSON.stringify({
1652
+ runId: "workflow-parent", mode: "workflow", state: "complete", startedAt: 100, lastUpdate: 200, endedAt: 250,
1653
+ steps: [{ agent: "worker", workflowKey: "main", runId: "child-run", async: true, status: "completed" }],
1654
+ workflowChildren: {
1655
+ version: 1, parentToolCallId: "tool-call", workflowRunId: "workflow-parent", inventoryComplete: true,
1656
+ workflowState: "completed", children: [{ childId: "main", runId: "child-run", state: "completed" }],
1657
+ },
1658
+ }, null, 2), "utf-8");
1659
+
1660
+ const result = inspectSubagentStatus({ id: "workflow-parent" }, { asyncDirRoot: asyncRoot, resultsDir: path.join(root, "results") });
1661
+ assert.deepEqual(result.details.workflowTerminalProof, {
1662
+ version: 1, kind: "workflow", runId: "workflow-parent", state: "observed", dispatchClosed: true,
1663
+ observedAt: 300, children: [childProof],
1664
+ });
1665
+ } finally {
1666
+ fs.rmSync(root, { recursive: true, force: true });
1667
+ }
1668
+ });
1633
1669
  });
@@ -0,0 +1,67 @@
1
+ import assert from "node:assert/strict";
2
+ import { describe, it } from "node:test";
3
+ import type { Message } from "@earendil-works/pi-ai";
4
+ import {
5
+ formatStructuredOutputRejectionError,
6
+ INVALID_STRUCTURED_OUTPUT_SCHEMA_ERROR,
7
+ MAX_STRUCTURED_OUTPUT_REJECTION_ERROR_BYTES,
8
+ STRUCTURED_OUTPUT_REJECTION_ERROR,
9
+ STRUCTURED_OUTPUT_VALIDATOR_UNAVAILABLE_ERROR,
10
+ } from "../../src/runs/shared/structured-output.ts";
11
+
12
+ function messages(...values: unknown[]): Message[] {
13
+ return values as Message[];
14
+ }
15
+
16
+ describe("structured output rejection evidence", () => {
17
+ it("uses the latest failed result and correlates results without names by toolCallId", () => {
18
+ const result = formatStructuredOutputRejectionError(messages(
19
+ { role: "assistant", content: [{ type: "toolCall", id: "structured-1", name: "structured_output", arguments: { value: {} } }] },
20
+ { role: "toolResult", toolCallId: "structured-1", isError: true, content: [{ type: "text", text: "Structured output validation failed: first: is required" }] },
21
+ { role: "toolResult", toolName: "structured_output", toolCallId: "structured-2", isError: true, content: [{ type: "text", text: "Structured output validation failed: second: is required" }] },
22
+ ));
23
+
24
+ assert.equal(result, "Structured output validation failed: second: is required");
25
+ });
26
+
27
+ it("ignores unrelated failures and returns a truthful fallback", () => {
28
+ const result = formatStructuredOutputRejectionError(messages(
29
+ { role: "toolResult", toolName: "read", isError: true, content: [{ type: "text", text: "EISDIR" }] },
30
+ ));
31
+
32
+ assert.equal(result, STRUCTURED_OUTPUT_REJECTION_ERROR);
33
+ });
34
+
35
+ it("does not expose schema compiler diagnostics", () => {
36
+ const sentinel = "PRIVATE_SCHEMA_SENTINEL";
37
+ const result = formatStructuredOutputRejectionError(messages(
38
+ { role: "toolResult", toolName: "structured_output", isError: true, content: [{ type: "text", text: `Structured output validation failed: invalid outputSchema: Invalid regular expression: /${sentinel}_[invalid/u` }] },
39
+ ));
40
+
41
+ assert.equal(result, INVALID_STRUCTURED_OUTPUT_SCHEMA_ERROR);
42
+ assert.equal(result.includes(sentinel), false);
43
+ });
44
+
45
+ it("categorizes unavailable validation without exposing setup details", () => {
46
+ const sentinel = "PRIVATE_SETUP_SENTINEL";
47
+ const result = formatStructuredOutputRejectionError(messages(
48
+ { role: "toolResult", toolName: "structured_output", isError: true, content: [{ type: "text", text: `Cannot load typebox/compile for structured output validation (direct import failed: ${sentinel} at /private/compiler.ts)` }] },
49
+ ));
50
+
51
+ assert.equal(result, STRUCTURED_OUTPUT_VALIDATOR_UNAVAILABLE_ERROR);
52
+ assert.equal(result.includes(sentinel), false);
53
+ });
54
+
55
+ it("redacts payload and stack details and clamps complete UTF-8 characters", () => {
56
+ const secret = "do-not-leak";
57
+ const oversized = `Structured output validation failed: ${"界".repeat(2_000)}\nsubmitted value: ${secret}\n at /private/project/file.ts:1:1`;
58
+ const result = formatStructuredOutputRejectionError(messages(
59
+ { role: "toolResult", toolName: "structured_output", isError: true, content: [{ type: "text", text: oversized }] },
60
+ ));
61
+
62
+ assert.ok(Buffer.byteLength(result, "utf8") <= MAX_STRUCTURED_OUTPUT_REJECTION_ERROR_BYTES);
63
+ assert.equal(result.includes("�"), false);
64
+ assert.equal(result.includes(secret), false);
65
+ assert.equal(result.includes("/private/project"), false);
66
+ });
67
+ });
@@ -0,0 +1,98 @@
1
+ import assert from "node:assert/strict";
2
+ import * as fs from "node:fs";
3
+ import * as os from "node:os";
4
+ import * as path from "node:path";
5
+ import { describe, it } from "node:test";
6
+ import type { AsyncStatus, ProcessTerminal, WorkflowChildSummary } from "../../src/shared/types.ts";
7
+ import { readWorkflowTerminalProof } from "../../src/runs/background/workflow-terminal-proof.ts";
8
+
9
+ type Steps = NonNullable<AsyncStatus["steps"]>;
10
+
11
+ function summary(overrides: Partial<WorkflowChildSummary> = {}): WorkflowChildSummary {
12
+ return {
13
+ version: 1,
14
+ parentToolCallId: "parent-tool-call",
15
+ workflowRunId: "workflow-run",
16
+ inventoryComplete: true,
17
+ workflowState: "completed",
18
+ children: [{ childId: "main", runId: "child-run", state: "completed" }],
19
+ ...overrides,
20
+ };
21
+ }
22
+
23
+ const asyncChild = [{ agent: "worker", workflowKey: "main", runId: "child-run", async: true, status: "completed" }] as Steps;
24
+
25
+ function observedChild(): ProcessTerminal {
26
+ return {
27
+ version: 1,
28
+ state: "observed",
29
+ runId: "child-run",
30
+ runnerProcessInstanceId: "runner-1",
31
+ observedAt: 1_234,
32
+ instances: [{ kind: "runner", processInstanceId: "runner-1", closeObservedAt: 1_234, exitCode: 0, signal: null }],
33
+ };
34
+ }
35
+
36
+ function withFixture(run: (dirs: { root: string; asyncDir: string; writeChild: (status: object, proof?: ProcessTerminal) => void }) => void): void {
37
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "workflow-terminal-proof-"));
38
+ const asyncDir = path.join(root, "workflow-run");
39
+ fs.mkdirSync(asyncDir, { recursive: true });
40
+ const writeChild = (status: object, proof?: ProcessTerminal) => {
41
+ const childDir = path.join(root, "child-run");
42
+ fs.mkdirSync(childDir, { recursive: true });
43
+ fs.writeFileSync(path.join(childDir, "status.json"), JSON.stringify({ runId: "child-run", mode: "single", startedAt: 1, lastUpdate: 2, ...status }));
44
+ if (proof) fs.writeFileSync(path.join(childDir, "process-terminal.json"), JSON.stringify(proof));
45
+ };
46
+ try {
47
+ run({ root, asyncDir, writeChild });
48
+ } finally {
49
+ fs.rmSync(root, { recursive: true, force: true });
50
+ }
51
+ }
52
+
53
+ const runnerIdentity = { version: 1, runId: "child-run", runnerProcessInstanceId: "runner-1" };
54
+
55
+ describe("readWorkflowTerminalProof", () => {
56
+ it("keeps an open inventory pending", () => withFixture(({ asyncDir }) => {
57
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary({ inventoryComplete: false, workflowState: "running" }), 0, 2_000), {
58
+ version: 1, kind: "workflow", runId: "workflow-run", state: "pending", dispatchClosed: false, reason: "Workflow dispatch is still open.",
59
+ });
60
+ }));
61
+
62
+ it("keeps closed dispatch pending until the child's exit is observed", () => withFixture(({ asyncDir, writeChild }) => {
63
+ writeChild({ state: "complete", processTerminal: { ...runnerIdentity, state: "pending" } });
64
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary(), 0, 2_000), {
65
+ version: 1, kind: "workflow", runId: "workflow-run", state: "pending", dispatchClosed: true,
66
+ reason: "async workflow child main process-terminal proof is missing",
67
+ });
68
+ }));
69
+
70
+ it("returns observed after every async child has writer-exit evidence", () => withFixture(({ asyncDir, writeChild }) => {
71
+ const proof = observedChild();
72
+ writeChild({ state: "complete", processTerminal: { ...runnerIdentity, state: "pending" } }, proof);
73
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary({ workflowState: "stopped" }), 0, 1_000), {
74
+ version: 1, kind: "workflow", runId: "workflow-run", state: "observed", dispatchClosed: true, observedAt: 1_234, children: [proof],
75
+ });
76
+ }));
77
+
78
+ it("accepts a child whose runner failed before starting", () => withFixture(({ asyncDir, writeChild }) => {
79
+ const notStarted = { ...runnerIdentity, state: "not-started" } as ProcessTerminal;
80
+ writeChild({ state: "failed", error: "runner failed to start", processTerminal: notStarted });
81
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary({ workflowState: "failed" }), 0, 2_000), {
82
+ version: 1, kind: "workflow", runId: "workflow-run", state: "observed", dispatchClosed: true, observedAt: 2_000, children: [notStarted],
83
+ });
84
+ }));
85
+
86
+ it("needs no process evidence for synchronous children that run inside the host", () => withFixture(({ asyncDir }) => {
87
+ const syncChild = [{ agent: "worker", workflowKey: "main", async: false, status: "completed" }] as Steps;
88
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, syncChild, summary({ children: [{ childId: "main", state: "completed" }] }), 0, 2_000), {
89
+ version: 1, kind: "workflow", runId: "workflow-run", state: "observed", dispatchClosed: true, observedAt: 2_000, children: [],
90
+ });
91
+ }));
92
+
93
+ it("reports host commands as unknown", () => withFixture(({ asyncDir }) => {
94
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, [], summary({ children: [] }), 1, 2_000), {
95
+ version: 1, kind: "workflow", runId: "workflow-run", state: "unknown", dispatchClosed: true, reason: "Workflow host commands have no process-terminal proof.",
96
+ });
97
+ }));
98
+ });
@@ -104,7 +104,8 @@ export default function extension(pi: ExtensionAPI) {
104
104
  pi.on('before_agent_start', async (event) => ({
105
105
  systemPrompt:
106
106
  `${event.systemPrompt}\n\n` +
107
- 'For web research questions that require finding and comparing sources, use web_explore. ' +
107
+ 'For requests explicitly mentioning GitHub or open source, use grep_app_search instead of web_explore. ' +
108
+ 'For all other web research questions that require finding and comparing sources, use web_explore. ' +
108
109
  'web_explore handles search, fetch, source ranking, and headless escalation internally. ' +
109
110
  'If more web evidence is needed after web_explore, call web_explore again with a narrower query; do not use shell/network commands such as curl, Invoke-WebRequest, npm view/search/pack, or direct HTTP URLs for web research.'
110
111
  }));