@selesai/code 0.13.33 → 0.13.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/dist/extensions/capability-gateway/catalog.ts +4 -1
  3. package/dist/extensions/capability-gateway/index.test.ts +93 -4
  4. package/dist/extensions/capability-gateway/index.ts +295 -59
  5. package/dist/extensions/capability-gateway/integration.test.ts +35 -0
  6. package/dist/extensions/capability-gateway/routing.test.ts +154 -1
  7. package/dist/extensions/capability-gateway/routing.ts +227 -45
  8. package/dist/extensions/jev/decisions.test.ts +37 -0
  9. package/dist/extensions/jev/decisions.ts +161 -43
  10. package/dist/extensions/jev-ask-tool.test.ts +501 -0
  11. package/dist/extensions/jev-ask-tool.ts +952 -0
  12. package/dist/extensions/package.json +1 -0
  13. package/dist/extensions/pi-hermes-memory/README.md +11 -36
  14. package/dist/extensions/pi-hermes-memory/src/config.ts +36 -8
  15. package/dist/extensions/pi-hermes-memory/src/constants.ts +5 -5
  16. package/dist/extensions/pi-hermes-memory/src/handlers/auto-consolidate.ts +60 -46
  17. package/dist/extensions/pi-hermes-memory/tests/config.test.ts +26 -2
  18. package/dist/extensions/pi-hermes-memory/tests/handlers/auto-consolidate.test.ts +7 -1
  19. package/dist/extensions/pi-intercom/index.ts +5 -1
  20. package/dist/extensions/pi-subagents/src/extension/public-execution.ts +6 -4
  21. package/dist/extensions/pi-subagents/src/extension/schemas.ts +1 -1
  22. package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +26 -1
  23. package/dist/extensions/pi-subagents/src/runs/shared/jev-subagent-routing.ts +255 -0
  24. package/dist/extensions/pi-subagents/test/unit/jev-subagent-routing.test.ts +117 -0
  25. package/dist/extensions/pi-subagents/test/unit/public-execution.test.ts +3 -1
  26. package/dist/extensions/rtk.test.ts +21 -13
  27. package/dist/extensions/tps.test.ts +32 -1
  28. package/dist/extensions/tps.ts +3 -1
  29. package/docs/settings.md +64 -7
  30. package/package.json +3 -3
@@ -0,0 +1,255 @@
1
+ /**
2
+ * Jev-assisted setup for a public single-child launch. Opt-in:
3
+ * `jevAdvisory.routes.subagent.enabled: true` in the agent settings.json.
4
+ *
5
+ * Jev only fills what the parent left open, and every answer can only narrow:
6
+ * 1. Agent: chosen by function when `agent` is omitted or generic (default "delegate"),
7
+ * from native, enabled agents the capability ceiling allows.
8
+ * 2. Tools: Jev may drop tools from the agent's declared allowlist for this task; it never adds one.
9
+ * 3. Model tier: simple | complex | reasoning, mapped through the user's
10
+ * `jevAdvisory.routes.subagent.tiers` (provider/id values), only when the parent passed no `model`.
11
+ *
12
+ * Every failure (no credential, timeout, malformed or low-confidence answer) leaves the launch
13
+ * exactly as requested. Launch validation, ceilings, and preflight remain authoritative.
14
+ */
15
+ import * as fs from "node:fs";
16
+ import * as path from "node:path";
17
+ import type { AgentConfig } from "../../agents/agents.ts";
18
+ import { getAgentDir } from "../../shared/utils.ts";
19
+ import { isAgentAllowedByCapabilityCeiling, type ResolvedSubagentCapabilityCeiling } from "./capability-ceiling.ts";
20
+ import {
21
+ askJev,
22
+ buildConversation,
23
+ buildJevPayload,
24
+ emitJevTelemetry,
25
+ jevConnection,
26
+ readJevAdvisoryConfig,
27
+ type JevAbstainReason,
28
+ type JevConnection,
29
+ type JevQuestion,
30
+ type JevRuntime,
31
+ } from "../../../../jev/decisions.ts";
32
+
33
+ export { warnJevUnavailableOnce } from "../../../../jev/decisions.ts";
34
+
35
+ export const MODEL_TIERS = ["simple", "complex", "reasoning"] as const;
36
+ export type ModelTier = (typeof MODEL_TIERS)[number];
37
+
38
+ const KEEP = "keep";
39
+ const DEFAULT_GENERIC_AGENTS = ["delegate"];
40
+ /** Never offered for removal: skills load through read, and the rest are supervision plumbing. */
41
+ const PROTECTED_TOOLS = new Set(["read", "contact_supervisor", "intercom", "structured_output", "subagent", "subagent_supervisor"]);
42
+ // ponytail: tools past this cap are simply kept; raise it if agents declare long allowlists.
43
+ const MAX_TOOL_QUESTIONS = 10;
44
+ const DESCRIPTION_CHARS = 300;
45
+
46
+ const TIER_CRITERIA: Record<ModelTier, string> = {
47
+ simple: "Mechanical or lookup work: find, list, summarize, rename, run a command and report.",
48
+ complex: "Multi-step implementation or investigation across several files with ordinary judgment.",
49
+ reasoning: "Hard judgment: subtle debugging, architecture or security review, tricky algorithms, ambiguous trade-offs.",
50
+ };
51
+
52
+ export interface SubagentRoutingSettings {
53
+ connection: JevConnection;
54
+ maxBytes: number;
55
+ contextChars: number;
56
+ genericAgents: string[];
57
+ tiers: Partial<Record<ModelTier, string>>;
58
+ }
59
+
60
+ export interface SubagentRoutingInput {
61
+ task: string;
62
+ agent?: string;
63
+ model?: string;
64
+ agents: AgentConfig[];
65
+ ceiling?: ResolvedSubagentCapabilityCeiling;
66
+ }
67
+
68
+ export interface SubagentRoutingResult {
69
+ /** The agent to launch: Jev's pick, or the requested one. */
70
+ agent?: string;
71
+ /** The run's agent list, with the launched agent narrowed (tools) or re-modelled (tier). */
72
+ agents: AgentConfig[];
73
+ routedAgent?: string;
74
+ tier?: ModelTier;
75
+ droppedTools: string[];
76
+ /** First reason a question went unanswered, for telemetry and the missing-agent error. */
77
+ abstained?: JevAbstainReason;
78
+ elapsedMs: number;
79
+ questions: number;
80
+ }
81
+
82
+ type JevAsk = typeof askJev;
83
+
84
+ /** The route's settings, or undefined while it is off. Never throws. */
85
+ export function readSubagentRoutingSettings(settingsPath = path.join(getAgentDir(), "settings.json")): SubagentRoutingSettings | undefined {
86
+ const config = readJevAdvisoryConfig(settingsPath);
87
+ const route = config.routes.subagent;
88
+ if (!route.enabled) return undefined;
89
+ let raw: Record<string, unknown> = {};
90
+ try {
91
+ const parsed = JSON.parse(fs.readFileSync(settingsPath, "utf-8")) as { jevAdvisory?: { routes?: { subagent?: Record<string, unknown> } } };
92
+ raw = parsed?.jevAdvisory?.routes?.subagent ?? {};
93
+ } catch {
94
+ // readJevAdvisoryConfig already defaulted; extra fields just stay unset.
95
+ }
96
+ const tiers: Partial<Record<ModelTier, string>> = {};
97
+ const rawTiers = raw.tiers && typeof raw.tiers === "object" ? (raw.tiers as Record<string, unknown>) : {};
98
+ for (const tier of MODEL_TIERS) {
99
+ const model = rawTiers[tier];
100
+ if (typeof model === "string" && model.trim()) tiers[tier] = model.trim();
101
+ }
102
+ const generic = Array.isArray(raw.genericAgents) ? raw.genericAgents.filter((name): name is string => typeof name === "string" && name.trim() !== "") : DEFAULT_GENERIC_AGENTS;
103
+ return {
104
+ connection: jevConnection(config, route),
105
+ maxBytes: route.payloadBytes,
106
+ contextChars: route.contextChars,
107
+ genericAgents: generic,
108
+ tiers,
109
+ };
110
+ }
111
+
112
+ /** Agents Jev may pick: native, enabled, file-defined, and inside the capability ceiling. */
113
+ export function routableAgents(agents: readonly AgentConfig[], ceiling: ResolvedSubagentCapabilityCeiling | undefined): AgentConfig[] {
114
+ return agents.filter(
115
+ (agent) =>
116
+ agent.source !== "runtime"
117
+ && agent.disabled !== true
118
+ && agent.runner?.type !== "external-cli"
119
+ && agent.runner?.type !== "external-job"
120
+ && isAgentAllowedByCapabilityCeiling(agent.name, ceiling),
121
+ );
122
+ }
123
+
124
+ function droppableTools(agent: AgentConfig): string[] {
125
+ const excluded = new Set(agent.excludeTools ?? []);
126
+ return (agent.tools ?? [])
127
+ .filter((tool) => !PROTECTED_TOOLS.has(tool) && !excluded.has(tool) && !/[/\\]|\.(?:ts|js)$/.test(tool))
128
+ .slice(0, MAX_TOOL_QUESTIONS);
129
+ }
130
+
131
+ function toolKey(index: number): string {
132
+ return `tool_${index}`;
133
+ }
134
+
135
+ export async function routeSubagentLaunch(
136
+ input: SubagentRoutingInput,
137
+ ctx: JevRuntime,
138
+ settings: SubagentRoutingSettings,
139
+ ask: JevAsk = askJev,
140
+ ): Promise<SubagentRoutingResult> {
141
+ const result: SubagentRoutingResult = { agent: input.agent, agents: input.agents, droppedTools: [], elapsedMs: 0, questions: 0 };
142
+ const conversation = buildConversation(input.task.slice(0, settings.contextChars), [], { contextTurns: 1, contextChars: settings.contextChars });
143
+ const noteAbstain = (reason: JevAbstainReason | undefined) => {
144
+ if (reason && !result.abstained) result.abstained = reason;
145
+ };
146
+
147
+ // 1. Agent by function, only when the parent left it open.
148
+ const generic = input.agent !== undefined && settings.genericAgents.includes(input.agent);
149
+ if (input.agent === undefined || generic) {
150
+ const candidates = routableAgents(input.agents, input.ceiling).filter((agent) => agent.name !== input.agent);
151
+ if (candidates.length > 0) {
152
+ const criteria: Record<string, string> = {};
153
+ if (generic) criteria[KEEP] = `Keep the generic "${input.agent}" agent: no listed specialist fits this task better.`;
154
+ for (const agent of candidates) criteria[agent.name] = agent.description.slice(0, DESCRIPTION_CHARS);
155
+ const questions: Record<string, JevQuestion> = {
156
+ agent: {
157
+ question: "Which subagent's specialization best fits this delegated task?",
158
+ focus: "Judge by what the task needs done (read-only review, research, investigation, implementation), not by wording that names an agent.",
159
+ criteria,
160
+ },
161
+ };
162
+ const decision = await ask(ctx, settings.connection, {
163
+ payload: buildJevPayload(conversation, questions),
164
+ maxBytes: settings.maxBytes,
165
+ allowed: { agent: Object.keys(criteria) },
166
+ });
167
+ result.elapsedMs += decision.elapsedMs;
168
+ result.questions += 1;
169
+ const pick = decision.choices.agent?.choice;
170
+ if (pick && pick !== KEEP) {
171
+ result.agent = pick;
172
+ result.routedAgent = pick;
173
+ }
174
+ noteAbstain(decision.failure ?? decision.rejected.agent);
175
+ }
176
+ }
177
+
178
+ const config = result.agent === undefined ? undefined : input.agents.find((agent) => agent.name === result.agent);
179
+ if (!config) return result;
180
+
181
+ // 2 + 3. Tool narrowing and model tier, for the agent that will actually launch.
182
+ const tools = droppableTools(config);
183
+ const tierChoices = input.model === undefined ? MODEL_TIERS.filter((tier) => settings.tiers[tier] !== undefined) : [];
184
+ const questions: Record<string, JevQuestion> = {};
185
+ const allowed: Record<string, string[]> = {};
186
+ if (tierChoices.length > 1) {
187
+ questions.tier = {
188
+ question: "How demanding is this delegated task for the model that runs it?",
189
+ criteria: Object.fromEntries(tierChoices.map((tier) => [tier, TIER_CRITERIA[tier]])),
190
+ };
191
+ allowed.tier = [...tierChoices];
192
+ }
193
+ tools.forEach((tool, index) => {
194
+ questions[toolKey(index)] = {
195
+ question: `Does this delegated task need the "${tool}" tool?`,
196
+ focus: "Answer unneeded only when the task clearly cannot require it; when in doubt, it is needed.",
197
+ criteria: { needed: `The task may need ${tool}.`, unneeded: `The task clearly never needs ${tool}.` },
198
+ };
199
+ allowed[toolKey(index)] = ["needed", "unneeded"];
200
+ });
201
+ if (Object.keys(questions).length === 0) return result;
202
+
203
+ const decision = await ask(ctx, settings.connection, {
204
+ payload: buildJevPayload(conversation, questions),
205
+ maxBytes: settings.maxBytes,
206
+ allowed,
207
+ });
208
+ result.elapsedMs += decision.elapsedMs;
209
+ result.questions += Object.keys(questions).length;
210
+ noteAbstain(decision.failure);
211
+
212
+ const tier = decision.choices.tier?.choice as ModelTier | undefined;
213
+ const tierModel = tier ? settings.tiers[tier] : undefined;
214
+ if (tierModel) result.tier = tier;
215
+ // Dropping a tool needs a stated confidence: an unquantified "unneeded" keeps the tool.
216
+ result.droppedTools = tools.filter((_tool, index) => {
217
+ const choice = decision.choices[toolKey(index)];
218
+ return choice?.choice === "unneeded" && typeof choice.confidence === "number";
219
+ });
220
+ if (!tierModel && result.droppedTools.length === 0) return result;
221
+
222
+ const { modelProvider: _modelProvider, modelSource: _modelSource, ...base } = config;
223
+ const routed: AgentConfig = {
224
+ ...(tierModel ? base : config),
225
+ ...(tierModel ? { model: tierModel } : {}),
226
+ ...(result.droppedTools.length > 0 ? { excludeTools: [...(config.excludeTools ?? []), ...result.droppedTools] } : {}),
227
+ };
228
+ result.agents = input.agents.map((agent) => (agent === config ? routed : agent));
229
+ return result;
230
+ }
231
+
232
+ /** Content-free: shape and outcome only, never the task, descriptions, or answers. */
233
+ export function emitSubagentRoutingTelemetry(events: { emit(channel: string, data: unknown): void } | undefined, result: SubagentRoutingResult): void {
234
+ if (result.questions === 0) return;
235
+ const changed = Boolean(result.routedAgent || result.tier || result.droppedTools.length > 0);
236
+ emitJevTelemetry(events, "decision", {
237
+ route: "subagent",
238
+ outcome: changed ? "jev" : "fallback",
239
+ candidates: result.questions,
240
+ elapsedMs: result.elapsedMs,
241
+ ...(result.routedAgent ? { agent: result.routedAgent } : {}),
242
+ ...(result.tier ? { tier: result.tier } : {}),
243
+ dropped: result.droppedTools.length,
244
+ ...(result.abstained ? { reason: result.abstained } : {}),
245
+ });
246
+ }
247
+
248
+ /** One line for the user and the parent: what Jev set up, or nothing when it changed nothing. */
249
+ export function describeSubagentRouting(result: SubagentRoutingResult, tiers: Partial<Record<ModelTier, string>>): string | undefined {
250
+ const parts: string[] = [];
251
+ if (result.routedAgent) parts.push(`agent ${result.routedAgent}`);
252
+ if (result.tier) parts.push(`${result.tier} tier (${tiers[result.tier]})`);
253
+ if (result.droppedTools.length > 0) parts.push(`without ${result.droppedTools.join(", ")}`);
254
+ return parts.length > 0 ? `Jev set up this subagent: ${parts.join("; ")}.` : undefined;
255
+ }
@@ -0,0 +1,117 @@
1
+ import assert from "node:assert/strict";
2
+ import * as fs from "node:fs";
3
+ import * as os from "node:os";
4
+ import * as path from "node:path";
5
+ import { describe, it } from "node:test";
6
+ import type { AgentConfig } from "../../src/agents/agents.ts";
7
+ import {
8
+ readSubagentRoutingSettings,
9
+ routeSubagentLaunch,
10
+ type SubagentRoutingSettings,
11
+ } from "../../src/runs/shared/jev-subagent-routing.ts";
12
+
13
+ function agent(name: string, extra: Partial<AgentConfig> = {}): AgentConfig {
14
+ return { name, description: `${name} agent`, systemPromptMode: "append", inheritProjectContext: true, inheritGlobalContext: true, inheritSkills: true, systemPrompt: "", source: "builtin", filePath: `${name}.md`, ...extra } as AgentConfig;
15
+ }
16
+
17
+ const agents = [
18
+ agent("delegate"),
19
+ agent("reviewer", { tools: ["read", "grep"] }),
20
+ agent("worker", { tools: ["read", "bash", "edit", "write"], model: "anthropic/opus", modelProvider: "anthropic" }),
21
+ agent("secret", { disabled: true }),
22
+ agent("external", { runner: { type: "external-cli" } as AgentConfig["runner"] }),
23
+ ];
24
+
25
+ const settings: SubagentRoutingSettings = {
26
+ connection: { provider: "tokenin", model: "jev", timeoutMs: 1000, minConfidence: 0.6 },
27
+ maxBytes: 16_384,
28
+ contextChars: 4000,
29
+ genericAgents: ["delegate"],
30
+ tiers: { simple: "p/small", reasoning: "p/big" },
31
+ };
32
+
33
+ type Asked = { allowed: Record<string, readonly string[]> };
34
+ /** A fake Jev: answers each asked question from `answers`, records what was offered. */
35
+ function fakeAsk(answers: Record<string, { choice: string; confidence?: number }>, asked: Asked[] = []) {
36
+ return async (_ctx: unknown, _connection: unknown, request: Asked) => {
37
+ asked.push({ allowed: request.allowed });
38
+ const choices = Object.fromEntries(Object.keys(request.allowed).filter((q) => answers[q]).map((q) => [q, answers[q]!]));
39
+ return { choices, rejected: {}, elapsedMs: 5 };
40
+ };
41
+ }
42
+ const ctx = {} as never;
43
+
44
+ describe("Jev subagent routing", () => {
45
+ it("routes an agentless task, offering only enabled native agents inside the ceiling", async () => {
46
+ const asked: Asked[] = [];
47
+ const result = await routeSubagentLaunch(
48
+ { task: "fix the bug", agents, ceiling: { allowedAgents: ["delegate", "worker", "secret"], sources: ["test"] } as never },
49
+ ctx, settings, fakeAsk({ agent: { choice: "worker", confidence: 0.9 } }, asked) as never,
50
+ );
51
+ assert.equal(result.agent, "worker");
52
+ assert.deepEqual(asked[0]!.allowed.agent, ["delegate", "worker"]);
53
+ });
54
+
55
+ it("offers keep for a generic agent and never re-routes an explicit specialist", async () => {
56
+ const asked: Asked[] = [];
57
+ const kept = await routeSubagentLaunch({ task: "t", agent: "delegate", agents }, ctx, settings, fakeAsk({ agent: { choice: "keep", confidence: 0.9 } }, asked) as never);
58
+ assert.equal(kept.agent, "delegate");
59
+ assert.ok(asked[0]!.allowed.agent!.includes("keep"));
60
+ assert.ok(!asked[0]!.allowed.agent!.includes("delegate"));
61
+
62
+ const explicitAsked: Asked[] = [];
63
+ const explicit = await routeSubagentLaunch({ task: "t", agent: "reviewer", model: "x/y", agents }, ctx, settings, fakeAsk({}, explicitAsked) as never);
64
+ assert.equal(explicit.agent, "reviewer");
65
+ assert.ok(explicitAsked.every((ask) => !("agent" in ask.allowed)));
66
+ });
67
+
68
+ it("abstention leaves the launch as requested", async () => {
69
+ const result = await routeSubagentLaunch({ task: "t", agents }, ctx, settings, (async () => ({ choices: {}, rejected: { agent: "no-credential" }, failure: "no-credential", elapsedMs: 0 })) as never);
70
+ assert.equal(result.agent, undefined);
71
+ assert.equal(result.agents, agents);
72
+ assert.equal(result.abstained, "no-credential");
73
+ });
74
+
75
+ it("only drops declared, unprotected tools on a confident unneeded, and applies the tier model", async () => {
76
+ const asked: Asked[] = [];
77
+ const result = await routeSubagentLaunch(
78
+ { task: "summarize the README", agent: "worker", agents },
79
+ ctx, settings,
80
+ fakeAsk({ tier: { choice: "simple", confidence: 0.8 }, tool_0: { choice: "unneeded", confidence: 0.9 }, tool_1: { choice: "unneeded" }, tool_2: { choice: "needed", confidence: 0.9 } }, asked) as never,
81
+ );
82
+ // read is never offered: tool_0..2 are bash, edit, write.
83
+ assert.deepEqual(Object.keys(asked[0]!.allowed).sort(), ["tier", "tool_0", "tool_1", "tool_2"]);
84
+ assert.deepEqual(asked[0]!.allowed.tier, ["simple", "reasoning"]);
85
+ assert.deepEqual(result.droppedTools, ["bash"]);
86
+ const worker = result.agents.find((a) => a.name === "worker")!;
87
+ assert.deepEqual(worker.excludeTools, ["bash"]);
88
+ assert.equal(worker.model, "p/small");
89
+ assert.equal(worker.modelProvider, undefined);
90
+ assert.deepEqual(worker.tools, ["read", "bash", "edit", "write"]);
91
+ // The discovered list itself is untouched.
92
+ assert.equal(agents.find((a) => a.name === "worker")!.model, "anthropic/opus");
93
+ });
94
+
95
+ it("skips the tier question when the parent chose a model", async () => {
96
+ const asked: Asked[] = [];
97
+ const result = await routeSubagentLaunch({ task: "t", agent: "worker", model: "x/y", agents }, ctx, settings, fakeAsk({ tier: { choice: "simple", confidence: 0.9 } }, asked) as never);
98
+ assert.ok(!("tier" in asked[0]!.allowed));
99
+ assert.equal(result.tier, undefined);
100
+ });
101
+
102
+ it("reads settings: off by default, tiers and generic agents when enabled", () => {
103
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "jev-subagent-"));
104
+ try {
105
+ const file = path.join(dir, "settings.json");
106
+ fs.writeFileSync(file, JSON.stringify({ jevAdvisory: { routes: {} } }));
107
+ assert.equal(readSubagentRoutingSettings(file), undefined);
108
+ fs.writeFileSync(file, JSON.stringify({ jevAdvisory: { routes: { subagent: { enabled: true, tiers: { simple: " p/s ", bogus: "x" }, genericAgents: ["delegate", "general"] } } } }));
109
+ const read = readSubagentRoutingSettings(file)!;
110
+ assert.deepEqual(read.tiers, { simple: "p/s" });
111
+ assert.deepEqual(read.genericAgents, ["delegate", "general"]);
112
+ assert.equal(read.connection.timeoutMs, 5000);
113
+ } finally {
114
+ fs.rmSync(dir, { recursive: true, force: true });
115
+ }
116
+ });
117
+ });
@@ -19,6 +19,8 @@ describe("public subagent execution normalization", () => {
19
19
  output: true,
20
20
  },
21
21
  });
22
+ // Agentless tasks are left for Jev subagent routing; the executor rejects them when it cannot choose.
23
+ assert.deepEqual(normalizePublicSubagentExecution({ task: "work" }), { ok: true, params: { task: "work", output: true } });
22
24
  assert.deepEqual(normalizePublicSubagentExecution({ agent: "worker" }), {
23
25
  ok: true,
24
26
  params: {
@@ -171,7 +173,7 @@ describe("public subagent execution normalization", () => {
171
173
  { action: "reject-checkpoint", id: "run" },
172
174
  { agent: "" },
173
175
  { agent: 42 },
174
- { task: "work" },
176
+ { task: " " },
175
177
  { agent: "worker", task: 42 },
176
178
  { agent: "worker", workflowScript: "return 1" },
177
179
  { action: "status", task: "work" },
@@ -41,9 +41,11 @@ esac
41
41
  function makePi() {
42
42
  const handlers = new Map<string, Function[]>();
43
43
  const pi = {
44
- exec: async (cmd: string, args: string[], opts?: { timeout?: number }) =>
44
+ exec: async (cmd: string, args: string[]) =>
45
45
  new Promise((resolve) => {
46
- execFile(cmd, args, { timeout: opts?.timeout }, (error, stdout, stderr) => {
46
+ // The shim runs under the suite's own load: enforcing the extension's production
47
+ // 2s budget here kills it, which the extension reports as a failed probe.
48
+ execFile(cmd, args, (error, stdout, stderr) => {
47
49
  if (error) {
48
50
  // execFile reports non-zero exits as errors; keep the real exit code.
49
51
  const code =
@@ -63,8 +65,14 @@ function makePi() {
63
65
  return { pi: pi as any, handlers };
64
66
  }
65
67
 
68
+ // Registration is deferred off startup (the managed-binary probe may spawn a process),
69
+ // so hook installation can outlast vi.waitFor's 1s default on a loaded machine.
70
+ function waitFor<T>(assertion: () => T | Promise<T>): Promise<T> {
71
+ return vi.waitFor(assertion, { timeout: 5_000 });
72
+ }
73
+
66
74
  async function getToolCall(handlers: Map<string, Function[]>): Promise<Function> {
67
- await vi.waitFor(() => expect(handlers.has("tool_call")).toBe(true));
75
+ await waitFor(() => expect(handlers.has("tool_call")).toBe(true));
68
76
  return handlers.get("tool_call")![0]!;
69
77
  }
70
78
 
@@ -131,7 +139,7 @@ posixOnly("rtk extension", () => {
131
139
  const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
132
140
  const { pi, handlers } = makePi();
133
141
  rtkExtension(pi);
134
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("rtk gain failed")));
142
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("rtk gain failed")));
135
143
  expect(handlers.has("tool_call")).toBe(false);
136
144
  } finally {
137
145
  if (previousPath) process.env.PATH = previousPath;
@@ -156,7 +164,7 @@ posixOnly("rtk extension", () => {
156
164
  // Override exec so --version fails.
157
165
  pi.exec = vi.fn(async () => ({ code: 1, stdout: "", stderr: "not found", killed: false }));
158
166
  rtkExtension(pi);
159
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("failed --version")));
167
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("failed --version")));
160
168
  expect(handlers.has("tool_call")).toBe(false);
161
169
  });
162
170
 
@@ -166,7 +174,7 @@ posixOnly("rtk extension", () => {
166
174
  ensureToolMock.mockResolvedValue("rtk");
167
175
  pi.exec = vi.fn(async () => ({ code: 0, stdout: "garbage output", stderr: "", killed: false }));
168
176
  rtkExtension(pi);
169
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("could not parse version")));
177
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("could not parse version")));
170
178
  expect(handlers.has("tool_call")).toBe(false);
171
179
  });
172
180
 
@@ -176,7 +184,7 @@ posixOnly("rtk extension", () => {
176
184
  ensureToolMock.mockResolvedValue("rtk");
177
185
  pi.exec = vi.fn(async () => ({ code: 0, stdout: " ", stderr: "", killed: false }));
178
186
  rtkExtension(pi);
179
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("<empty output>")));
187
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("<empty output>")));
180
188
  expect(handlers.has("tool_call")).toBe(false);
181
189
  });
182
190
 
@@ -186,7 +194,7 @@ posixOnly("rtk extension", () => {
186
194
  ensureToolMock.mockResolvedValue("rtk");
187
195
  pi.exec = vi.fn(async () => ({ code: 0, stdout: "rtk 0.22.0", stderr: "", killed: false }));
188
196
  rtkExtension(pi);
189
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("too old")));
197
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("too old")));
190
198
  expect(handlers.has("tool_call")).toBe(false);
191
199
  });
192
200
 
@@ -195,7 +203,7 @@ posixOnly("rtk extension", () => {
195
203
  ensureToolMock.mockRejectedValue(new Error("download failed"));
196
204
  const { pi, handlers } = makePi();
197
205
  rtkExtension(pi);
198
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("managed installation failed")));
206
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("managed installation failed")));
199
207
  expect(handlers.has("tool_call")).toBe(false);
200
208
  });
201
209
 
@@ -204,7 +212,7 @@ posixOnly("rtk extension", () => {
204
212
  ensureToolMock.mockRejectedValue("plain string failure");
205
213
  const { pi, handlers } = makePi();
206
214
  rtkExtension(pi);
207
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("plain string failure")));
215
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("plain string failure")));
208
216
  expect(handlers.has("tool_call")).toBe(false);
209
217
  });
210
218
 
@@ -213,7 +221,7 @@ posixOnly("rtk extension", () => {
213
221
  ensureToolMock.mockResolvedValue(undefined);
214
222
  const { pi, handlers } = makePi();
215
223
  rtkExtension(pi);
216
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("unavailable")));
224
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("unavailable")));
217
225
  expect(handlers.has("tool_call")).toBe(false);
218
226
  });
219
227
 
@@ -225,7 +233,7 @@ posixOnly("rtk extension", () => {
225
233
  throw new Error("spawn ENOENT");
226
234
  });
227
235
  rtkExtension(pi);
228
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("verification failed")));
236
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("verification failed")));
229
237
  expect(handlers.has("tool_call")).toBe(false);
230
238
  });
231
239
 
@@ -237,7 +245,7 @@ posixOnly("rtk extension", () => {
237
245
  throw "string failure";
238
246
  });
239
247
  rtkExtension(pi);
240
- await vi.waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("string failure")));
248
+ await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("string failure")));
241
249
  expect(handlers.has("tool_call")).toBe(false);
242
250
  });
243
251
 
@@ -1,5 +1,5 @@
1
1
  import { describe, expect, it } from "vitest";
2
- import { calculateLiveTps, calculateReliableTps, type TpsTiming } from "./tps.ts";
2
+ import { calculateLiveTps, calculateReliableTps, setupTpsTracker, type TpsTiming } from "./tps.ts";
3
3
 
4
4
  function timing(overrides: Partial<TpsTiming> = {}): TpsTiming {
5
5
  return {
@@ -23,6 +23,37 @@ describe("calculateLiveTps", () => {
23
23
  });
24
24
  });
25
25
 
26
+ describe("setupTpsTracker with anthropic-style usage", () => {
27
+ it("ignores the tiny usage.output seeded at message_start while streaming", async () => {
28
+ const handlers = new Map<string, (event: any, ctx: any) => Promise<void>>();
29
+ const statuses: string[] = [];
30
+ const ctx = { ui: { setStatus: (_k: string, v: string) => statuses.push(v), notify: () => {}, theme: { fg: (_c: string, t: string) => t } } };
31
+ setupTpsTracker({ on: (name: string, fn: any) => handlers.set(name, fn) } as any);
32
+
33
+ let now = 0;
34
+ const realNow = performance.now;
35
+ performance.now = () => now;
36
+ try {
37
+ const message = { role: "assistant", provider: "anthropic", model: "claude", usage: { output: 1 } };
38
+ await handlers.get("agent_start")!({}, ctx);
39
+ await handlers.get("message_start")!({ message }, ctx);
40
+ for (let i = 0; i < 20; i++) {
41
+ now += 50;
42
+ await handlers.get("message_update")!({ message, assistantMessageEvent: { type: "text_delta", delta: "x".repeat(40) } }, ctx);
43
+ }
44
+ // 20 deltas * 10 est. tokens over ~1s => ~200 tok/s, not ~1 tok/s
45
+ expect(statuses.at(-1)).toBe("211 tok/s");
46
+
47
+ message.usage.output = 200;
48
+ await handlers.get("message_end")!({ message }, ctx);
49
+ await handlers.get("agent_end")!({}, ctx);
50
+ expect(statuses.at(-1)).toMatch(/^done [1-9]\d* t\/s \(main [1-9]\d* t\/s\)$/);
51
+ } finally {
52
+ performance.now = realNow;
53
+ }
54
+ });
55
+ });
56
+
26
57
  describe("calculateReliableTps", () => {
27
58
  it("uses active stream time for sufficiently sampled output", () => {
28
59
  expect(calculateReliableTps(100, timing())).toEqual({
@@ -230,7 +230,9 @@ export function setupTpsTracker(pi: ExtensionAPI): void {
230
230
  streamStart ??= now;
231
231
  estimatedStreamedTokens += Math.max(0, streamEvent.delta.length / 4);
232
232
  const officialTokens = generatedTokensFromUsage(asRecord(event.message.usage));
233
- const currentTokens = officialTokens > 0 ? officialTokens : estimatedStreamedTokens;
233
+ // Anthropic seeds usage.output (~1) at message_start and only finalizes it at message_delta,
234
+ // so a nonzero official count mid-stream is not authoritative; take whichever is larger.
235
+ const currentTokens = Math.max(officialTokens, estimatedStreamedTokens);
234
236
  const tps = calculateLiveTps(currentTokens, now - streamStart);
235
237
  if (tps !== null) {
236
238
  ctx.ui.setStatus("tps", ctx.ui.theme.fg("accent", `${tps} tok/s`));