@selesai/code 0.13.32 → 0.13.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/dist/extensions/capability-gateway/catalog.ts +4 -1
  3. package/dist/extensions/capability-gateway/index.test.ts +93 -4
  4. package/dist/extensions/capability-gateway/index.ts +294 -59
  5. package/dist/extensions/capability-gateway/integration.test.ts +71 -0
  6. package/dist/extensions/capability-gateway/routing.test.ts +154 -1
  7. package/dist/extensions/capability-gateway/routing.ts +227 -45
  8. package/dist/extensions/jev/decisions.test.ts +37 -0
  9. package/dist/extensions/jev/decisions.ts +148 -43
  10. package/dist/extensions/jev-ask-tool.test.ts +436 -0
  11. package/dist/extensions/jev-ask-tool.ts +587 -0
  12. package/dist/extensions/package.json +1 -0
  13. package/dist/extensions/pi-hermes-memory/README.md +11 -36
  14. package/dist/extensions/pi-hermes-memory/src/config.ts +36 -8
  15. package/dist/extensions/pi-hermes-memory/src/constants.ts +5 -5
  16. package/dist/extensions/pi-hermes-memory/src/handlers/auto-consolidate.ts +60 -46
  17. package/dist/extensions/pi-hermes-memory/tests/config.test.ts +26 -2
  18. package/dist/extensions/pi-hermes-memory/tests/handlers/auto-consolidate.test.ts +7 -1
  19. package/dist/extensions/pi-intercom/index.ts +5 -1
  20. package/dist/extensions/pi-subagents/agents/worker.md +3 -2
  21. package/dist/extensions/pi-subagents/docs/agents.md +2 -0
  22. package/dist/extensions/pi-subagents/docs/extension-api.md +3 -1
  23. package/dist/extensions/pi-subagents/docs/observability.md +2 -0
  24. package/dist/extensions/pi-subagents/docs/tool-reference.md +2 -2
  25. package/dist/extensions/pi-subagents/docs/workflows.md +1 -1
  26. package/dist/extensions/pi-subagents/src/extension/rpc.ts +10 -1
  27. package/dist/extensions/pi-subagents/src/runs/background/active-async-capacity.ts +3 -21
  28. package/dist/extensions/pi-subagents/src/runs/background/async-execution.ts +2 -1
  29. package/dist/extensions/pi-subagents/src/runs/background/run-child-session.ts +1 -0
  30. package/dist/extensions/pi-subagents/src/runs/background/run-status.ts +5 -1
  31. package/dist/extensions/pi-subagents/src/runs/background/subagent-runner.ts +10 -3
  32. package/dist/extensions/pi-subagents/src/runs/background/workflow-terminal-proof.ts +67 -0
  33. package/dist/extensions/pi-subagents/src/runs/foreground/execution.ts +6 -2
  34. package/dist/extensions/pi-subagents/src/runs/shared/child-tool-plan.ts +67 -5
  35. package/dist/extensions/pi-subagents/src/runs/shared/completion-guard.ts +6 -0
  36. package/dist/extensions/pi-subagents/src/runs/shared/external-cli-runner.ts +2 -1
  37. package/dist/extensions/pi-subagents/src/runs/shared/git-environment.ts +29 -0
  38. package/dist/extensions/pi-subagents/src/runs/shared/structured-output.ts +69 -0
  39. package/dist/extensions/pi-subagents/src/shared/types.ts +20 -0
  40. package/dist/extensions/pi-subagents/src/slash/slash-commands.ts +4 -238
  41. package/dist/extensions/pi-subagents/src/slash/subagent-cost.ts +280 -0
  42. package/dist/extensions/pi-subagents/test/integration/async-execution.part-3.test.ts +67 -0
  43. package/dist/extensions/pi-subagents/test/integration/in-process-child.test.ts +29 -1
  44. package/dist/extensions/pi-subagents/test/integration/intercom-result-delivery.test.ts +6 -3
  45. package/dist/extensions/pi-subagents/test/integration/single-execution.part-2.test.ts +25 -0
  46. package/dist/extensions/pi-subagents/test/unit/agent-frontmatter.test.ts +5 -5
  47. package/dist/extensions/pi-subagents/test/unit/async-spawn-preload.test.ts +21 -0
  48. package/dist/extensions/pi-subagents/test/unit/child-tool-plan-permission-system.test.ts +98 -0
  49. package/dist/extensions/pi-subagents/test/unit/child-tool-plan.test.ts +11 -0
  50. package/dist/extensions/pi-subagents/test/unit/completion-guard.test.ts +12 -0
  51. package/dist/extensions/pi-subagents/test/unit/external-cli-runner.test.ts +25 -0
  52. package/dist/extensions/pi-subagents/test/unit/git-environment.test.ts +38 -0
  53. package/dist/extensions/pi-subagents/test/unit/preflight.test.ts +3 -1
  54. package/dist/extensions/pi-subagents/test/unit/rpc.test.ts +105 -1
  55. package/dist/extensions/pi-subagents/test/unit/run-status.test.ts +36 -0
  56. package/dist/extensions/pi-subagents/test/unit/structured-output-rejection.test.ts +67 -0
  57. package/dist/extensions/pi-subagents/test/unit/workflow-terminal-proof.test.ts +98 -0
  58. package/dist/extensions/rtk.test.ts +21 -13
  59. package/dist/extensions/tps.test.ts +32 -1
  60. package/dist/extensions/tps.ts +3 -1
  61. package/dist/skills/pi-subagents/references/constraints-and-recipes.md +8 -8
  62. package/dist/skills/pi-subagents/references/execution-controls.md +1 -1
  63. package/dist/skills/pi-subagents/references/prompting-and-roles.md +1 -1
  64. package/docs/settings.md +46 -7
  65. package/package.json +3 -3
@@ -0,0 +1,67 @@
1
+ import assert from "node:assert/strict";
2
+ import { describe, it } from "node:test";
3
+ import type { Message } from "@earendil-works/pi-ai";
4
+ import {
5
+ formatStructuredOutputRejectionError,
6
+ INVALID_STRUCTURED_OUTPUT_SCHEMA_ERROR,
7
+ MAX_STRUCTURED_OUTPUT_REJECTION_ERROR_BYTES,
8
+ STRUCTURED_OUTPUT_REJECTION_ERROR,
9
+ STRUCTURED_OUTPUT_VALIDATOR_UNAVAILABLE_ERROR,
10
+ } from "../../src/runs/shared/structured-output.ts";
11
+
12
+ function messages(...values: unknown[]): Message[] {
13
+ return values as Message[];
14
+ }
15
+
16
+ describe("structured output rejection evidence", () => {
17
+ it("uses the latest failed result and correlates results without names by toolCallId", () => {
18
+ const result = formatStructuredOutputRejectionError(messages(
19
+ { role: "assistant", content: [{ type: "toolCall", id: "structured-1", name: "structured_output", arguments: { value: {} } }] },
20
+ { role: "toolResult", toolCallId: "structured-1", isError: true, content: [{ type: "text", text: "Structured output validation failed: first: is required" }] },
21
+ { role: "toolResult", toolName: "structured_output", toolCallId: "structured-2", isError: true, content: [{ type: "text", text: "Structured output validation failed: second: is required" }] },
22
+ ));
23
+
24
+ assert.equal(result, "Structured output validation failed: second: is required");
25
+ });
26
+
27
+ it("ignores unrelated failures and returns a truthful fallback", () => {
28
+ const result = formatStructuredOutputRejectionError(messages(
29
+ { role: "toolResult", toolName: "read", isError: true, content: [{ type: "text", text: "EISDIR" }] },
30
+ ));
31
+
32
+ assert.equal(result, STRUCTURED_OUTPUT_REJECTION_ERROR);
33
+ });
34
+
35
+ it("does not expose schema compiler diagnostics", () => {
36
+ const sentinel = "PRIVATE_SCHEMA_SENTINEL";
37
+ const result = formatStructuredOutputRejectionError(messages(
38
+ { role: "toolResult", toolName: "structured_output", isError: true, content: [{ type: "text", text: `Structured output validation failed: invalid outputSchema: Invalid regular expression: /${sentinel}_[invalid/u` }] },
39
+ ));
40
+
41
+ assert.equal(result, INVALID_STRUCTURED_OUTPUT_SCHEMA_ERROR);
42
+ assert.equal(result.includes(sentinel), false);
43
+ });
44
+
45
+ it("categorizes unavailable validation without exposing setup details", () => {
46
+ const sentinel = "PRIVATE_SETUP_SENTINEL";
47
+ const result = formatStructuredOutputRejectionError(messages(
48
+ { role: "toolResult", toolName: "structured_output", isError: true, content: [{ type: "text", text: `Cannot load typebox/compile for structured output validation (direct import failed: ${sentinel} at /private/compiler.ts)` }] },
49
+ ));
50
+
51
+ assert.equal(result, STRUCTURED_OUTPUT_VALIDATOR_UNAVAILABLE_ERROR);
52
+ assert.equal(result.includes(sentinel), false);
53
+ });
54
+
55
+ it("redacts payload and stack details and clamps complete UTF-8 characters", () => {
56
+ const secret = "do-not-leak";
57
+ const oversized = `Structured output validation failed: ${"界".repeat(2_000)}\nsubmitted value: ${secret}\n at /private/project/file.ts:1:1`;
58
+ const result = formatStructuredOutputRejectionError(messages(
59
+ { role: "toolResult", toolName: "structured_output", isError: true, content: [{ type: "text", text: oversized }] },
60
+ ));
61
+
62
+ assert.ok(Buffer.byteLength(result, "utf8") <= MAX_STRUCTURED_OUTPUT_REJECTION_ERROR_BYTES);
63
+ assert.equal(result.includes("�"), false);
64
+ assert.equal(result.includes(secret), false);
65
+ assert.equal(result.includes("/private/project"), false);
66
+ });
67
+ });
@@ -0,0 +1,98 @@
1
+ import assert from "node:assert/strict";
2
+ import * as fs from "node:fs";
3
+ import * as os from "node:os";
4
+ import * as path from "node:path";
5
+ import { describe, it } from "node:test";
6
+ import type { AsyncStatus, ProcessTerminal, WorkflowChildSummary } from "../../src/shared/types.ts";
7
+ import { readWorkflowTerminalProof } from "../../src/runs/background/workflow-terminal-proof.ts";
8
+
9
+ type Steps = NonNullable<AsyncStatus["steps"]>;
10
+
11
+ function summary(overrides: Partial<WorkflowChildSummary> = {}): WorkflowChildSummary {
12
+ return {
13
+ version: 1,
14
+ parentToolCallId: "parent-tool-call",
15
+ workflowRunId: "workflow-run",
16
+ inventoryComplete: true,
17
+ workflowState: "completed",
18
+ children: [{ childId: "main", runId: "child-run", state: "completed" }],
19
+ ...overrides,
20
+ };
21
+ }
22
+
23
+ const asyncChild = [{ agent: "worker", workflowKey: "main", runId: "child-run", async: true, status: "completed" }] as Steps;
24
+
25
+ function observedChild(): ProcessTerminal {
26
+ return {
27
+ version: 1,
28
+ state: "observed",
29
+ runId: "child-run",
30
+ runnerProcessInstanceId: "runner-1",
31
+ observedAt: 1_234,
32
+ instances: [{ kind: "runner", processInstanceId: "runner-1", closeObservedAt: 1_234, exitCode: 0, signal: null }],
33
+ };
34
+ }
35
+
36
+ function withFixture(run: (dirs: { root: string; asyncDir: string; writeChild: (status: object, proof?: ProcessTerminal) => void }) => void): void {
37
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "workflow-terminal-proof-"));
38
+ const asyncDir = path.join(root, "workflow-run");
39
+ fs.mkdirSync(asyncDir, { recursive: true });
40
+ const writeChild = (status: object, proof?: ProcessTerminal) => {
41
+ const childDir = path.join(root, "child-run");
42
+ fs.mkdirSync(childDir, { recursive: true });
43
+ fs.writeFileSync(path.join(childDir, "status.json"), JSON.stringify({ runId: "child-run", mode: "single", startedAt: 1, lastUpdate: 2, ...status }));
44
+ if (proof) fs.writeFileSync(path.join(childDir, "process-terminal.json"), JSON.stringify(proof));
45
+ };
46
+ try {
47
+ run({ root, asyncDir, writeChild });
48
+ } finally {
49
+ fs.rmSync(root, { recursive: true, force: true });
50
+ }
51
+ }
52
+
53
+ const runnerIdentity = { version: 1, runId: "child-run", runnerProcessInstanceId: "runner-1" };
54
+
55
+ describe("readWorkflowTerminalProof", () => {
56
+ it("keeps an open inventory pending", () => withFixture(({ asyncDir }) => {
57
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary({ inventoryComplete: false, workflowState: "running" }), 0, 2_000), {
58
+ version: 1, kind: "workflow", runId: "workflow-run", state: "pending", dispatchClosed: false, reason: "Workflow dispatch is still open.",
59
+ });
60
+ }));
61
+
62
+ it("keeps closed dispatch pending until the child's exit is observed", () => withFixture(({ asyncDir, writeChild }) => {
63
+ writeChild({ state: "complete", processTerminal: { ...runnerIdentity, state: "pending" } });
64
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary(), 0, 2_000), {
65
+ version: 1, kind: "workflow", runId: "workflow-run", state: "pending", dispatchClosed: true,
66
+ reason: "async workflow child main process-terminal proof is missing",
67
+ });
68
+ }));
69
+
70
+ it("returns observed after every async child has writer-exit evidence", () => withFixture(({ asyncDir, writeChild }) => {
71
+ const proof = observedChild();
72
+ writeChild({ state: "complete", processTerminal: { ...runnerIdentity, state: "pending" } }, proof);
73
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary({ workflowState: "stopped" }), 0, 1_000), {
74
+ version: 1, kind: "workflow", runId: "workflow-run", state: "observed", dispatchClosed: true, observedAt: 1_234, children: [proof],
75
+ });
76
+ }));
77
+
78
+ it("accepts a child whose runner failed before starting", () => withFixture(({ asyncDir, writeChild }) => {
79
+ const notStarted = { ...runnerIdentity, state: "not-started" } as ProcessTerminal;
80
+ writeChild({ state: "failed", error: "runner failed to start", processTerminal: notStarted });
81
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, asyncChild, summary({ workflowState: "failed" }), 0, 2_000), {
82
+ version: 1, kind: "workflow", runId: "workflow-run", state: "observed", dispatchClosed: true, observedAt: 2_000, children: [notStarted],
83
+ });
84
+ }));
85
+
86
+ it("needs no process evidence for synchronous children that run inside the host", () => withFixture(({ asyncDir }) => {
87
+ const syncChild = [{ agent: "worker", workflowKey: "main", async: false, status: "completed" }] as Steps;
88
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, syncChild, summary({ children: [{ childId: "main", state: "completed" }] }), 0, 2_000), {
89
+ version: 1, kind: "workflow", runId: "workflow-run", state: "observed", dispatchClosed: true, observedAt: 2_000, children: [],
90
+ });
91
+ }));
92
+
93
+ it("reports host commands as unknown", () => withFixture(({ asyncDir }) => {
94
+ assert.deepEqual(readWorkflowTerminalProof(asyncDir, [], summary({ children: [] }), 1, 2_000), {
95
+ version: 1, kind: "workflow", runId: "workflow-run", state: "unknown", dispatchClosed: true, reason: "Workflow host commands have no process-terminal proof.",
96
+ });
97
+ }));
98
+ });
@@ -41,9 +41,11 @@ esac
41
41
  function makePi() {
42
42
  const handlers = new Map<string, Function[]>();
43
43
  const pi = {
44
- exec: async (cmd: string, args: string[], opts?: { timeout?: number }) =>
44
+ exec: async (cmd: string, args: string[]) =>
45
45
  new Promise((resolve) => {
46
- execFile(cmd, args, { timeout: opts?.timeout }, (error, stdout, stderr) => {
46
+ // The shim runs under the suite's own load: enforcing the extension's production
47
+ // 2s budget here kills it, which the extension reports as a failed probe.
48
+ execFile(cmd, args, (error, stdout, stderr) => {
47
49
  if (error) {
48
50
  // execFile reports non-zero exits as errors; keep the real exit code.
49
51
  const code =
@@ -63,8 +65,14 @@ function makePi() {
63
65
  return { pi: pi as any, handlers };
64
66
  }
65
67
 
68
+ // Registration is deferred off startup (the managed-binary probe may spawn a process),
69
+ // so hook installation can outlast vi.waitFor's 1s default on a loaded machine.
70
+ function waitFor<T>(assertion: () => T | Promise<T>): Promise<T> {
71
+ return vi.waitFor(assertion, { timeout: 5_000 });
72
+ }
73
+
66
74
  async function getToolCall(handlers: Map<string, Function[]>): Promise<Function> {
67
- await vi.waitFor(() => expect(handlers.has("tool_call")).toBe(true));
75
+ await waitFor(() => expect(handlers.has("tool_call")).toBe(true));
68
76
  return handlers.get("tool_call")![0]!;
69
77
  }
70
78
 
@@ -131,7 +139,7 @@ posixOnly("rtk extension", () => {
131
139
  const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
132
140
  const { pi, handlers } = makePi();
133
141
  rtkExtension(pi);
134
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("rtk gain failed")));
142
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("rtk gain failed")));
135
143
  expect(handlers.has("tool_call")).toBe(false);
136
144
  } finally {
137
145
  if (previousPath) process.env.PATH = previousPath;
@@ -156,7 +164,7 @@ posixOnly("rtk extension", () => {
156
164
  // Override exec so --version fails.
157
165
  pi.exec = vi.fn(async () => ({ code: 1, stdout: "", stderr: "not found", killed: false }));
158
166
  rtkExtension(pi);
159
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("failed --version")));
167
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("failed --version")));
160
168
  expect(handlers.has("tool_call")).toBe(false);
161
169
  });
162
170
 
@@ -166,7 +174,7 @@ posixOnly("rtk extension", () => {
166
174
  ensureToolMock.mockResolvedValue("rtk");
167
175
  pi.exec = vi.fn(async () => ({ code: 0, stdout: "garbage output", stderr: "", killed: false }));
168
176
  rtkExtension(pi);
169
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("could not parse version")));
177
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("could not parse version")));
170
178
  expect(handlers.has("tool_call")).toBe(false);
171
179
  });
172
180
 
@@ -176,7 +184,7 @@ posixOnly("rtk extension", () => {
176
184
  ensureToolMock.mockResolvedValue("rtk");
177
185
  pi.exec = vi.fn(async () => ({ code: 0, stdout: " ", stderr: "", killed: false }));
178
186
  rtkExtension(pi);
179
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("<empty output>")));
187
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("<empty output>")));
180
188
  expect(handlers.has("tool_call")).toBe(false);
181
189
  });
182
190
 
@@ -186,7 +194,7 @@ posixOnly("rtk extension", () => {
186
194
  ensureToolMock.mockResolvedValue("rtk");
187
195
  pi.exec = vi.fn(async () => ({ code: 0, stdout: "rtk 0.22.0", stderr: "", killed: false }));
188
196
  rtkExtension(pi);
189
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("too old")));
197
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("too old")));
190
198
  expect(handlers.has("tool_call")).toBe(false);
191
199
  });
192
200
 
@@ -195,7 +203,7 @@ posixOnly("rtk extension", () => {
195
203
  ensureToolMock.mockRejectedValue(new Error("download failed"));
196
204
  const { pi, handlers } = makePi();
197
205
  rtkExtension(pi);
198
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("managed installation failed")));
206
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("managed installation failed")));
199
207
  expect(handlers.has("tool_call")).toBe(false);
200
208
  });
201
209
 
@@ -204,7 +212,7 @@ posixOnly("rtk extension", () => {
204
212
  ensureToolMock.mockRejectedValue("plain string failure");
205
213
  const { pi, handlers } = makePi();
206
214
  rtkExtension(pi);
207
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("plain string failure")));
215
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("plain string failure")));
208
216
  expect(handlers.has("tool_call")).toBe(false);
209
217
  });
210
218
 
@@ -213,7 +221,7 @@ posixOnly("rtk extension", () => {
213
221
  ensureToolMock.mockResolvedValue(undefined);
214
222
  const { pi, handlers } = makePi();
215
223
  rtkExtension(pi);
216
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("unavailable")));
224
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("unavailable")));
217
225
  expect(handlers.has("tool_call")).toBe(false);
218
226
  });
219
227
 
@@ -225,7 +233,7 @@ posixOnly("rtk extension", () => {
225
233
  throw new Error("spawn ENOENT");
226
234
  });
227
235
  rtkExtension(pi);
228
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("verification failed")));
236
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("verification failed")));
229
237
  expect(handlers.has("tool_call")).toBe(false);
230
238
  });
231
239
 
@@ -237,7 +245,7 @@ posixOnly("rtk extension", () => {
237
245
  throw "string failure";
238
246
  });
239
247
  rtkExtension(pi);
240
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("string failure")));
248
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("string failure")));
241
249
  expect(handlers.has("tool_call")).toBe(false);
242
250
  });
243
251
 
@@ -1,5 +1,5 @@
1
1
  import { describe, expect, it } from "vitest";
2
- import { calculateLiveTps, calculateReliableTps, type TpsTiming } from "./tps.ts";
2
+ import { calculateLiveTps, calculateReliableTps, setupTpsTracker, type TpsTiming } from "./tps.ts";
3
3
 
4
4
  function timing(overrides: Partial<TpsTiming> = {}): TpsTiming {
5
5
  return {
@@ -23,6 +23,37 @@ describe("calculateLiveTps", () => {
23
23
  });
24
24
  });
25
25
 
26
+ describe("setupTpsTracker with anthropic-style usage", () => {
27
+ it("ignores the tiny usage.output seeded at message_start while streaming", async () => {
28
+ const handlers = new Map<string, (event: any, ctx: any) => Promise<void>>();
29
+ const statuses: string[] = [];
30
+ const ctx = { ui: { setStatus: (_k: string, v: string) => statuses.push(v), notify: () => {}, theme: { fg: (_c: string, t: string) => t } } };
31
+ setupTpsTracker({ on: (name: string, fn: any) => handlers.set(name, fn) } as any);
32
+
33
+ let now = 0;
34
+ const realNow = performance.now;
35
+ performance.now = () => now;
36
+ try {
37
+ const message = { role: "assistant", provider: "anthropic", model: "claude", usage: { output: 1 } };
38
+ await handlers.get("agent_start")!({}, ctx);
39
+ await handlers.get("message_start")!({ message }, ctx);
40
+ for (let i = 0; i < 20; i++) {
41
+ now += 50;
42
+ await handlers.get("message_update")!({ message, assistantMessageEvent: { type: "text_delta", delta: "x".repeat(40) } }, ctx);
43
+ }
44
+ // 20 deltas * 10 est. tokens over ~1s => ~200 tok/s, not ~1 tok/s
45
+ expect(statuses.at(-1)).toBe("211 tok/s");
46
+
47
+ message.usage.output = 200;
48
+ await handlers.get("message_end")!({ message }, ctx);
49
+ await handlers.get("agent_end")!({}, ctx);
50
+ expect(statuses.at(-1)).toMatch(/^done [1-9]\d* t\/s \(main [1-9]\d* t\/s\)$/);
51
+ } finally {
52
+ performance.now = realNow;
53
+ }
54
+ });
55
+ });
56
+
26
57
  describe("calculateReliableTps", () => {
27
58
  it("uses active stream time for sufficiently sampled output", () => {
28
59
  expect(calculateReliableTps(100, timing())).toEqual({
@@ -230,7 +230,9 @@ export function setupTpsTracker(pi: ExtensionAPI): void {
230
230
  streamStart ??= now;
231
231
  estimatedStreamedTokens += Math.max(0, streamEvent.delta.length / 4);
232
232
  const officialTokens = generatedTokensFromUsage(asRecord(event.message.usage));
233
- const currentTokens = officialTokens > 0 ? officialTokens : estimatedStreamedTokens;
233
+ // Anthropic seeds usage.output (~1) at message_start and only finalizes it at message_delta,
234
+ // so a nonzero official count mid-stream is not authoritative; take whichever is larger.
235
+ const currentTokens = Math.max(officialTokens, estimatedStreamedTokens);
234
236
  const tps = calculateLiveTps(currentTokens, now - streamStart);
235
237
  if (tps !== null) {
236
238
  ctx.ui.setStatus("tps", ctx.ui.theme.fg("accent", `${tps} tok/s`));
@@ -6,10 +6,10 @@ This file is a detailed reference loaded from `skills/pi-subagents/SKILL.md`.
6
6
 
7
7
  - **Explicit forking requires a persisted parent session.** If the current session
8
8
  does not have a persisted session file or current leaf, explicit `context: "fork"`
9
- fails. An agent-level `defaultContext: fork` is a preference: packaged `worker`,
10
- `oracle`, and `advisor` fall back to `fresh` when those fork preconditions are not
11
- met yet. Use `context: "fresh"` when you do not want a fork even after the parent
12
- session exists.
9
+ fails. Packaged `worker` defaults to `fresh`; `oracle` and `advisor` default to
10
+ `fork`, which falls back to `fresh` when fork preconditions are not met yet. A
11
+ configured global `defaultSubagentContext` can override agent defaults. Use explicit
12
+ `context: "fresh"` or `context: "fork"` when the mode matters.
13
13
  - **Forked runs inherit parent history.** They are branched threads, not fresh
14
14
  filtered contexts. Use fresh context for adversarial reviewers unless the user explicitly asks for forked context.
15
15
  - **Default subagent nesting depth is 2.** Deeper recursive delegation is blocked
@@ -122,7 +122,7 @@ Run the work through seven gated phases:
122
122
 
123
123
  For straightforward non-trivial work, this sequence is the lightweight version of the parent-owned loop. When the task is complex, use Fable mode above. In either case, factor in the packaged prompt workflows without literally invoking slash commands. Use the same patterns through tools and subagents.
124
124
 
125
- Keep builtin agent defaults unless the user explicitly asks for a different model, thinking level, skills, output behavior, context mode, or other override. Do not add overrides just because you are orchestrating; the defaults encode the intended role behavior. In particular, packaged `worker`, `oracle`, and `advisor` default to forked context.
125
+ Keep builtin agent defaults unless the user explicitly asks for a different model, thinking level, skills, output behavior, context mode, or other override. Do not add overrides just because you are orchestrating; the defaults encode the intended role behavior. Packaged `worker` defaults to fresh context; `oracle` and `advisor` default to forked context.
126
126
 
127
127
  When the user approves launching a subagent to carry out a plan or workflow, treat that as approval to generate a proper role-specific meta prompt for that subagent. Include the approved plan path or summary, clarified requirements, non-goals, relevant context, role boundaries, files or areas to inspect, acceptance criteria, expected output, and validation expectations. Do not pass vague instructions like “implement the plan fully” or “review this” by themselves.
128
128
 
@@ -142,7 +142,7 @@ The validation contract defines acceptance before code is written: expected beha
142
142
 
143
143
  For one host-run verification command, `gate: "npm test"` on the child is shorthand for verified acceptance with that command; it cannot be combined with `acceptance` and is rejected on retained resume items. Use the structured `acceptance` field when the run should carry an explicit acceptance contract. If omitted, subagents infer an effective policy from role, mode, and risk. Evidence levels end at `verified`: use `level: "checked"` for ordinary writer evidence and `level: "verified"` when the runtime should run explicit validation commands. Independent review is orthogonal; use `review: { required: true, agent: "reviewer" }` and orchestrate the reviewer separately. `review-required` means evidence passed but review is pending, while `reviewed` means a real independent result found no blockers. For reviewer/read-only calls, omit `acceptance`. Never explicitly request `level: "reviewed"`; that value remains recognized only so preflight can return an actionable correction. To disable gates, use `{ level: "none", reason: "..." }`; the bare string `"none"` is rejected, and `false` is accepted only as a deprecated shorthand. Child-reported command success is evidence, not runtime verification.
144
144
 
145
- The first `worker` implements the approved plan. The parent continues with independent inspection or validation prep while it runs, not parallel edits to the same worktree. When the async worker completes, treat its handoff as the transition into review, not as final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Parallel reviewers inspect the resulting diff from fresh context. Validators check behavior with the best available evidence: commands, tests, browser/CLI interaction, screenshots, logs, or manual reproduction notes. The final `worker` applies synthesized review fixes in forked context, then the parent looks over the final diff before completing. The parent may launch these steps as an initial async `workflowScript` when the workflow is already clear, or as follow-up workflowScript runs after each async completion. Initial workflows should pass `async: true` so the main chat is unblocked. Do not stop after parallel review unless the user explicitly asked for review-only output or the review surfaced a decision that needs approval first.
145
+ The first `worker` implements the approved plan. The parent continues with independent inspection or validation prep while it runs, not parallel edits to the same worktree. When the async worker completes, treat its handoff as the transition into review, not as final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Parallel reviewers inspect the resulting diff from fresh context. Validators check behavior with the best available evidence: commands, tests, browser/CLI interaction, screenshots, logs, or manual reproduction notes. The final `worker` applies synthesized review fixes with explicit `context: "fork"` when it should reuse the parent thread, then the parent looks over the final diff before completing. The parent may launch these steps as an initial async `workflowScript` when the workflow is already clear, or as follow-up workflowScript runs after each async completion. Initial workflows should pass `async: true` so the main chat is unblocked. Do not stop after parallel review unless the user explicitly asked for review-only output or the review surfaced a decision that needs approval first.
146
146
 
147
147
  For complex work, risky changes, broad refactors, or many changed lines, increase review and validation fanout rather than trusting one reviewer. Use distinct angles such as correctness/regressions, tests/validation, simplicity/maintainability, security/privacy, performance, docs/API contracts, and user-flow behavior. When reviewers find non-trivial issues or the fix worker touches many lines, run another focused review round before final validation.
148
148
 
@@ -155,10 +155,10 @@ Keep orchestration authority in the parent session. Child subagents should not l
155
155
  1. Clarify first. This is mandatory. Gather code context with `scout`, add `researcher` only when external evidence matters, then ask the user clarifying questions with `interview` until scope, acceptance criteria, constraints, and non-goals are clear.
156
156
  2. Define the validation contract. State acceptance before implementation: expected behavior, checks to run, user flows to exercise, and evidence required in the worker handoff. For UI, CLI, integration, or workflow changes, include at least one validator angle that uses the product the way a user would rather than only reading code.
157
157
  3. Plan when useful. For complex work, write a plan doc yourself and get approval before implementation. For simple work, confirm shared understanding and explicitly note why planning is skipped.
158
- 4. Implement with one writer. After approval, launch `worker` asynchronously with a proper meta prompt that includes clarified requirements, relevant context, plan path or summary, the validation contract, and output expectations. Packaged `worker` defaults to forked context; pass `context: "fresh"` only when you intentionally want a fresh child. While it runs, prepare validation or inspect adjacent code instead of editing the same worktree.
158
+ 4. Implement with one writer. After approval, launch `worker` asynchronously with a proper meta prompt that includes clarified requirements, relevant context, plan path or summary, the validation contract, and output expectations. Packaged `worker` defaults to fresh context; pass `context: "fork"` only when inherited parent history is intentionally needed. While it runs, prepare validation or inspect adjacent code instead of editing the same worktree.
159
159
  5. Require a useful worker handoff. Ask the worker to report changed files, what was implemented, what was left undone, commands run with exit codes, validation evidence, surprises or new risks, decisions made inside approved scope, and decisions needing parent approval.
160
160
  6. Review after implementation. After the worker completes, launch parallel async fresh-context `reviewer` agents for correctness/regressions, tests/validation, and simplicity/maintainability. Add security, performance, docs/API, domain-specific, or user-flow validators for complex work, risky changes, broad refactors, or many changed lines. Use `output: false` unless review artifacts are explicitly needed.
161
- 7. Synthesize, then run the fix worker. Separate blockers, fixes worth doing now, optional improvements, and feedback to ignore/defer, then launch an async forked `worker` to apply fixes worth doing now when the workflow is implementation-authorized. If reviewers found scope/product/architecture choices that were not approved, ask the user first instead of applying them.
161
+ 7. Synthesize, then run the fix worker. Separate blockers, fixes worth doing now, optional improvements, and feedback to ignore/defer, then launch an async `worker` with `context: "fork"` when an authorized fix pass should inherit the parent thread. If reviewers found scope/product/architecture choices that were not approved, ask the user first instead of applying them.
162
162
  8. Review again when warranted. If the fix worker made substantial changes or addressed non-trivial findings, run another focused parallel review round before final validation.
163
163
  9. Validate and complete. After the fix worker and any follow-up review return, inspect the final diff yourself, run or confirm focused validation, update docs/changelog when relevant, and summarize what changed and why.
164
164
 
@@ -398,7 +398,7 @@ subagent({
398
398
  workflowScript: `return runs.run("oracle-check", { agent: "oracle", task: "Review my current direction, challenge assumptions, and propose the best next move." })`
399
399
  })
400
400
 
401
- // Implementation only after explicit approval. Worker defaults to forked context.
401
+ // Implementation only after explicit approval. Worker defaults fresh; set context: "fork" when inherited history is needed.
402
402
  subagent({
403
403
  workflowScript: `return runs.run("implementation", { agent: "worker", task: "Implement the approved approach: ..." })`
404
404
  })
@@ -88,7 +88,7 @@ subagent({
88
88
 
89
89
  ### Review-loop technique
90
90
 
91
- Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async forked `worker` applies them. The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no P0 blockers or P1 fixes worth doing now, remaining P2 feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
91
+ Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async `worker` applies them (pass `context: "fork"` when that fix pass should inherit the parent thread). The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no P0 blockers or P1 fixes worth doing now, remaining P2 feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
92
92
 
93
93
  As a conservative orchestration policy, do not pass `turnBudget` or a hard `toolBudget` to an implementation worker, fix worker, reviewer with edit authority, or other mutation-capable child. The default tool budget blocks read/search tools rather than mutation tools, but count limits still do not measure delivery safety. Use a narrow task plus an outer elapsed deadline with enough margin, then request a checkpoint after the current tool returns. The checkpoint should report changed files, build/test state, remaining work, and commit or PR state. An elapsed timeout is not a mutation-safe boundary and must not be used as the checkpoint trigger.
94
94
 
package/docs/settings.md CHANGED
@@ -49,8 +49,8 @@ Use `/trust` in interactive mode to save a project trust decision for future ses
49
49
  ### Jev Advisory Routing
50
50
 
51
51
  The bundled `jev-advisory-routing` extension uses the Jev decisions model for bounded opt-in
52
- routing. Every route stays off until it is enabled in `jevAdvisory`; the two routes share one
53
- Jev provider/model pair:
52
+ routing. The host-side routes stay off until they are enabled in `jevAdvisory`; the agent's own
53
+ `ask` route is on by default. All three share one Jev provider/model pair:
54
54
 
55
55
  - `memory` — only after an explicit durable-memory cue (for example “the convention we
56
56
  decided” or “don't repeat the past failure”), Jev chooses one read-only local
@@ -58,6 +58,14 @@ Jev provider/model pair:
58
58
  - `recommendations` — one discovered skill or prompt workflow that fits, or a proportionate
59
59
  verification level. Nothing is loaded, started, or executed: the recommendation is context for the
60
60
  agent, and required project/workflow gates are unchanged.
61
+ - `ask` — the agent-callable `ask_jev` tool (on by default). The agent decides when to call it: it
62
+ passes its own prose, file `paths`, and one `command`, and gets typed choice/score/noul answers back —
63
+ never the file contents or command output, so a verdict about code costs a few lines of context
64
+ instead of the files. Files must be inside the working directory, secret-named files (`.env`,
65
+ `*.pem`, `id_rsa*`, `auth.json`, …) are refused, and the command runs through the bash tool's local
66
+ shell. Whatever the agent passes is sent to Jev. Without Token-In credentials the tool is taken out
67
+ of the agent's loadout before each run (it returns after `/tokenin add`, no reload needed), and a call
68
+ that slips through reads no file and runs no command.
61
69
 
62
70
  | Setting | Type | Default | Description |
63
71
  |---------|------|---------|-------------|
@@ -66,11 +74,15 @@ Jev provider/model pair:
66
74
  | `jevAdvisory.baseUrl` | string | inherited | Base URL override; defaults to any registered model of `provider` |
67
75
  | `jevAdvisory.routes.memory.enabled` | boolean | `false` | Enable the memory-lookup route |
68
76
  | `jevAdvisory.routes.recommendations.enabled` | boolean | `false` | Enable skill/workflow and verification recommendations |
69
- | `<route>.timeoutMs` | number | `8000` | Route request timeout (ms); memory is capped at 750ms |
77
+ | `jevAdvisory.routes.ask.enabled` | boolean | `true` | Offer the agent the `ask_jev` tool; set `false` to turn it off |
78
+ | `<route>.timeoutMs` | number | `8000` | Route request timeout (ms); memory is capped at 750ms; `ask` defaults to `15000` |
70
79
  | `<route>.minConfidence` | number | `0.6` | Below this Jev confidence the route abstains |
71
80
  | `<route>.contextTurns` | number | `4` | Prior user turns sent as recommendation context; memory sends only the current bounded prompt |
72
81
  | `<route>.contextChars` | number | `4000` | Character budget for recommendation context |
73
- | `<route>.payloadBytes` | number | `8192` | Hard cap on the serialized decision request |
82
+ | `<route>.payloadBytes` | number | `8192` | Hard cap on the serialized decision request; `ask` defaults to `32768` |
83
+
84
+ `ask` ignores `minConfidence`, `contextTurns`, and `contextChars`: it returns every confidence to the
85
+ agent, and its context is whatever the agent put in the request.
74
86
 
75
87
  Only idle, top-level, interactive prompts are routed: queued steering/follow-up input, slash
76
88
  commands, and extension-injected turns are skipped, and each turn is routed at most once. The memory
@@ -99,16 +111,28 @@ needed: a compact `capability_catalog` lists them, `capability_discover` activat
99
111
  current run, and `capability_skill_show` loads one skill's full instructions. Set
100
112
  `SELESAI_CAPABILITY_GATEWAY=0` to disable the gateway and keep every tool visible.
101
113
 
102
- Routing has two rungs:
114
+ Routing has three rungs:
103
115
 
104
116
  1. A deterministic router activates a tool when the prompt uniquely matches its name, alias, or
105
117
  discovery summary. Skills are never auto-loaded or auto-selected.
106
118
  2. Default-on Jev tie-breaking: when the deterministic router returns an ambiguous lexical hint
107
119
  among optional tools, the Jev decisions model is asked which of two or three hinted tools (or
108
120
  `none`) should be exposed. Jev sees only the bounded current prompt and each hinted tool's compact
109
- discovery line — never conversation history, tool schemas, or the full catalog. A prompt with no
121
+ discovery line — never conversation history or tool schemas. A prompt with no
110
122
  lexical signal, a unique activation, and a skill-only match never reach Jev. Token-In credentials
111
123
  are required; without them, no Jev request is sent and Selesai prompts you to run `/tokenin add`.
124
+ 3. The agent-driven route: `capability_discover` also takes a `job` instead of a name. Tools and
125
+ skills are separate decisions — a tool is callable code from an extension or MCP server that the
126
+ agent may use many times, a skill is a written procedure it reads once — so one Jev request asks
127
+ two questions, each from catalog metadata only (never a schema or skill body), each allowed to
128
+ answer `none`. A chosen tool is activated for the run and its parameters are returned; a chosen
129
+ skill is named for `capability_skill_show`. Tools are offered first, so only skills can be left out
130
+ of an oversized catalog, and a `none` then says how many were not considered. A skill found in
131
+ several sources is offered once. While Jev has no credential, the `job` route is left out of the
132
+ agent's instructions and the agent is pointed at `capability_catalog` plus an exact `name`.
133
+
134
+ All Jev features share one warning per session, shown the first time one of them actually needs the
135
+ missing credential; a user without a subscription is not warned just for starting a session.
112
136
 
113
137
  | Setting | Type | Default | Description |
114
138
  |---------|------|---------|-------------|
@@ -117,8 +141,9 @@ Routing has two rungs:
117
141
  | `capabilityGateway.routing.jev.model` | string | `"jev-1.13"` | Decisions model the tie-breaker calls |
118
142
  | `capabilityGateway.routing.jev.baseUrl` | string | inherited | Base URL override; defaults to any registered model of `provider` |
119
143
  | `capabilityGateway.routing.jev.timeoutMs` | number | `1000` | Pre-turn request timeout (ms); hard-capped at `2000` |
144
+ | `capabilityGateway.routing.jev.discoverTimeoutMs` | number | `5000` | Timeout (ms) for `capability_discover({ job })`; hard-capped at `15000` |
120
145
  | `capabilityGateway.routing.jev.minConfidence` | number | `0.6` | Below this Jev confidence the tie-breaker abstains |
121
- | `capabilityGateway.routing.jev.payloadBytes` | number | `8192` | Hard cap on the serialized decision request |
146
+ | `capabilityGateway.routing.jev.payloadBytes` | number | `32768` | Hard cap on the serialized decision request; the pre-turn tie-break uses a few hundred bytes of it |
122
147
 
123
148
  This area is independent of `jevAdvisory`: gateway routing reads only
124
149
  `capabilityGateway.routing.jev` and shares just the Jev provider/model deployment identity. A
@@ -362,6 +387,20 @@ When selesai reads extensions from both `~/.selesai/agent/extensions/` and `~/.p
362
387
 
363
388
  Keys are top-level entry names (dir name for packaged extensions, file name for loose `.ts`). Values are `"selesai"` or `"pi"`. See [Shared Host Extensions](shared-host-extensions.md) for the full flow.
364
389
 
390
+ ## Bundled extension settings
391
+
392
+ The `pi-hermes-memory` extension reads its options from the `hermesMemory` object in global `~/.selesai/agent/settings.json`:
393
+
394
+ ```json
395
+ {
396
+ "hermesMemory": {
397
+ "llmThinkingOverride": "off",
398
+ "consolidationTimeoutMs": 300000
399
+ }
400
+ }
401
+ ```
402
+
403
+ These settings load at startup and take precedence over the legacy `hermes-memory-config.json` fallback. `llmModelOverride` is optional and uses `provider/model` format (for example, `tokenin/deepseek-v4.1-flash` if your account has access); leave it unset to use the active model. See the [memory extension configuration reference](../src/extensions/pi-hermes-memory/README.md#configuration) for all options.
365
404
 
366
405
  ## Example
367
406
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@selesai/code",
3
- "version": "0.13.32",
3
+ "version": "0.13.34",
4
4
  "description": "Maintained, extension-first Pi coding agent with built-in workflows, subagents, web research, questions, skills, and an enhanced terminal UI.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -55,8 +55,8 @@
55
55
  "clean": "shx rm -rf dist",
56
56
  "dev": "tsx src/cli.ts",
57
57
  "dev:print": "tsx src/cli.ts --print",
58
- "test": "vitest run src/__tests__/model-registry-completion.test.ts src/extensions/jev/decisions.test.ts src/extensions/capability-gateway/routing.test.ts src/core/settings-manager-auto-handoff.test.ts src/core/remote-catalog-provider.test.ts src/extensions/jev-advisory-memory.test.ts src/extensions/jev-advisory-recommendations.test.ts src/extensions/jev-advisory-lifecycle.test.ts src/extensions/undo.test.ts src/extensions/tps.test.ts src/extensions/pi-graft src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
59
- "test:coverage": "vitest run --coverage src/__tests__/model-registry-completion.test.ts src/extensions/jev/decisions.test.ts src/extensions/capability-gateway/routing.test.ts src/core/settings-manager-auto-handoff.test.ts src/core/remote-catalog-provider.test.ts src/extensions/jev-advisory-memory.test.ts src/extensions/jev-advisory-recommendations.test.ts src/extensions/jev-advisory-lifecycle.test.ts src/extensions/undo.test.ts src/extensions/tps.test.ts src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
58
+ "test": "vitest run src/__tests__/model-registry-completion.test.ts src/extensions/jev/decisions.test.ts src/extensions/capability-gateway/routing.test.ts src/core/settings-manager-auto-handoff.test.ts src/core/remote-catalog-provider.test.ts src/extensions/jev-advisory-memory.test.ts src/extensions/jev-advisory-recommendations.test.ts src/extensions/jev-ask-tool.test.ts src/extensions/jev-advisory-lifecycle.test.ts src/extensions/undo.test.ts src/extensions/tps.test.ts src/extensions/pi-graft src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
59
+ "test:coverage": "vitest run --coverage src/__tests__/model-registry-completion.test.ts src/extensions/jev/decisions.test.ts src/extensions/capability-gateway/routing.test.ts src/core/settings-manager-auto-handoff.test.ts src/core/remote-catalog-provider.test.ts src/extensions/jev-advisory-memory.test.ts src/extensions/jev-advisory-recommendations.test.ts src/extensions/jev-ask-tool.test.ts src/extensions/jev-advisory-lifecycle.test.ts src/extensions/undo.test.ts src/extensions/tps.test.ts src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
60
60
  "prepare": "npm run build",
61
61
  "build": "npm run clean && tsgo -p tsconfig.build.json && shx chmod +x dist/cli.js dist/rpc-entry.js && npm run copy-assets",
62
62
  "copy-assets": "shx mkdir -p dist/modes/interactive/theme && shx cp src/modes/interactive/theme/*.json dist/modes/interactive/theme/ && shx mkdir -p dist/modes/interactive/assets && shx cp src/modes/interactive/assets/*.png dist/modes/interactive/assets/ && shx mkdir -p dist/core/export-html/vendor && shx cp src/core/export-html/template.html src/core/export-html/template.css src/core/export-html/template.js dist/core/export-html/ && shx cp src/core/export-html/vendor/*.js dist/core/export-html/vendor/ && shx mkdir -p dist/defaults && shx cp src/defaults/* dist/defaults/ && shx mkdir -p dist/extensions && node scripts/copy-extensions.mjs && shx mkdir -p dist/themes && shx cp -r src/themes/. dist/themes/ && shx mkdir -p dist/skills && shx cp -r src/skills/. dist/skills/"