@selesai/code 0.13.33 → 0.13.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/dist/extensions/capability-gateway/catalog.ts +4 -1
- package/dist/extensions/capability-gateway/index.test.ts +93 -4
- package/dist/extensions/capability-gateway/index.ts +295 -59
- package/dist/extensions/capability-gateway/integration.test.ts +35 -0
- package/dist/extensions/capability-gateway/routing.test.ts +154 -1
- package/dist/extensions/capability-gateway/routing.ts +227 -45
- package/dist/extensions/jev/decisions.test.ts +37 -0
- package/dist/extensions/jev/decisions.ts +161 -43
- package/dist/extensions/jev-ask-tool.test.ts +501 -0
- package/dist/extensions/jev-ask-tool.ts +952 -0
- package/dist/extensions/package.json +1 -0
- package/dist/extensions/pi-hermes-memory/README.md +11 -36
- package/dist/extensions/pi-hermes-memory/src/config.ts +36 -8
- package/dist/extensions/pi-hermes-memory/src/constants.ts +5 -5
- package/dist/extensions/pi-hermes-memory/src/handlers/auto-consolidate.ts +60 -46
- package/dist/extensions/pi-hermes-memory/tests/config.test.ts +26 -2
- package/dist/extensions/pi-hermes-memory/tests/handlers/auto-consolidate.test.ts +7 -1
- package/dist/extensions/pi-intercom/index.ts +5 -1
- package/dist/extensions/pi-subagents/src/extension/public-execution.ts +6 -4
- package/dist/extensions/pi-subagents/src/extension/schemas.ts +1 -1
- package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +26 -1
- package/dist/extensions/pi-subagents/src/runs/shared/jev-subagent-routing.ts +255 -0
- package/dist/extensions/pi-subagents/test/unit/jev-subagent-routing.test.ts +117 -0
- package/dist/extensions/pi-subagents/test/unit/public-execution.test.ts +3 -1
- package/dist/extensions/rtk.test.ts +21 -13
- package/dist/extensions/tps.test.ts +32 -1
- package/dist/extensions/tps.ts +3 -1
- package/docs/settings.md +64 -7
- package/package.json +3 -3
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Jev-assisted setup for a public single-child launch. Opt-in:
|
|
3
|
+
* `jevAdvisory.routes.subagent.enabled: true` in the agent settings.json.
|
|
4
|
+
*
|
|
5
|
+
* Jev only fills what the parent left open, and every answer can only narrow:
|
|
6
|
+
* 1. Agent: chosen by function when `agent` is omitted or generic (default "delegate"),
|
|
7
|
+
* from native, enabled agents the capability ceiling allows.
|
|
8
|
+
* 2. Tools: Jev may drop tools from the agent's declared allowlist for this task; it never adds one.
|
|
9
|
+
* 3. Model tier: simple | complex | reasoning, mapped through the user's
|
|
10
|
+
* `jevAdvisory.routes.subagent.tiers` (provider/id values), only when the parent passed no `model`.
|
|
11
|
+
*
|
|
12
|
+
* Every failure (no credential, timeout, malformed or low-confidence answer) leaves the launch
|
|
13
|
+
* exactly as requested. Launch validation, ceilings, and preflight remain authoritative.
|
|
14
|
+
*/
|
|
15
|
+
import * as fs from "node:fs";
|
|
16
|
+
import * as path from "node:path";
|
|
17
|
+
import type { AgentConfig } from "../../agents/agents.ts";
|
|
18
|
+
import { getAgentDir } from "../../shared/utils.ts";
|
|
19
|
+
import { isAgentAllowedByCapabilityCeiling, type ResolvedSubagentCapabilityCeiling } from "./capability-ceiling.ts";
|
|
20
|
+
import {
|
|
21
|
+
askJev,
|
|
22
|
+
buildConversation,
|
|
23
|
+
buildJevPayload,
|
|
24
|
+
emitJevTelemetry,
|
|
25
|
+
jevConnection,
|
|
26
|
+
readJevAdvisoryConfig,
|
|
27
|
+
type JevAbstainReason,
|
|
28
|
+
type JevConnection,
|
|
29
|
+
type JevQuestion,
|
|
30
|
+
type JevRuntime,
|
|
31
|
+
} from "../../../../jev/decisions.ts";
|
|
32
|
+
|
|
33
|
+
export { warnJevUnavailableOnce } from "../../../../jev/decisions.ts";
|
|
34
|
+
|
|
35
|
+
export const MODEL_TIERS = ["simple", "complex", "reasoning"] as const;
|
|
36
|
+
export type ModelTier = (typeof MODEL_TIERS)[number];
|
|
37
|
+
|
|
38
|
+
const KEEP = "keep";
|
|
39
|
+
const DEFAULT_GENERIC_AGENTS = ["delegate"];
|
|
40
|
+
/** Never offered for removal: skills load through read, and the rest are supervision plumbing. */
|
|
41
|
+
const PROTECTED_TOOLS = new Set(["read", "contact_supervisor", "intercom", "structured_output", "subagent", "subagent_supervisor"]);
|
|
42
|
+
// ponytail: tools past this cap are simply kept; raise it if agents declare long allowlists.
|
|
43
|
+
const MAX_TOOL_QUESTIONS = 10;
|
|
44
|
+
const DESCRIPTION_CHARS = 300;
|
|
45
|
+
|
|
46
|
+
const TIER_CRITERIA: Record<ModelTier, string> = {
|
|
47
|
+
simple: "Mechanical or lookup work: find, list, summarize, rename, run a command and report.",
|
|
48
|
+
complex: "Multi-step implementation or investigation across several files with ordinary judgment.",
|
|
49
|
+
reasoning: "Hard judgment: subtle debugging, architecture or security review, tricky algorithms, ambiguous trade-offs.",
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
export interface SubagentRoutingSettings {
|
|
53
|
+
connection: JevConnection;
|
|
54
|
+
maxBytes: number;
|
|
55
|
+
contextChars: number;
|
|
56
|
+
genericAgents: string[];
|
|
57
|
+
tiers: Partial<Record<ModelTier, string>>;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export interface SubagentRoutingInput {
|
|
61
|
+
task: string;
|
|
62
|
+
agent?: string;
|
|
63
|
+
model?: string;
|
|
64
|
+
agents: AgentConfig[];
|
|
65
|
+
ceiling?: ResolvedSubagentCapabilityCeiling;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export interface SubagentRoutingResult {
|
|
69
|
+
/** The agent to launch: Jev's pick, or the requested one. */
|
|
70
|
+
agent?: string;
|
|
71
|
+
/** The run's agent list, with the launched agent narrowed (tools) or re-modelled (tier). */
|
|
72
|
+
agents: AgentConfig[];
|
|
73
|
+
routedAgent?: string;
|
|
74
|
+
tier?: ModelTier;
|
|
75
|
+
droppedTools: string[];
|
|
76
|
+
/** First reason a question went unanswered, for telemetry and the missing-agent error. */
|
|
77
|
+
abstained?: JevAbstainReason;
|
|
78
|
+
elapsedMs: number;
|
|
79
|
+
questions: number;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
type JevAsk = typeof askJev;
|
|
83
|
+
|
|
84
|
+
/** The route's settings, or undefined while it is off. Never throws. */
|
|
85
|
+
export function readSubagentRoutingSettings(settingsPath = path.join(getAgentDir(), "settings.json")): SubagentRoutingSettings | undefined {
|
|
86
|
+
const config = readJevAdvisoryConfig(settingsPath);
|
|
87
|
+
const route = config.routes.subagent;
|
|
88
|
+
if (!route.enabled) return undefined;
|
|
89
|
+
let raw: Record<string, unknown> = {};
|
|
90
|
+
try {
|
|
91
|
+
const parsed = JSON.parse(fs.readFileSync(settingsPath, "utf-8")) as { jevAdvisory?: { routes?: { subagent?: Record<string, unknown> } } };
|
|
92
|
+
raw = parsed?.jevAdvisory?.routes?.subagent ?? {};
|
|
93
|
+
} catch {
|
|
94
|
+
// readJevAdvisoryConfig already defaulted; extra fields just stay unset.
|
|
95
|
+
}
|
|
96
|
+
const tiers: Partial<Record<ModelTier, string>> = {};
|
|
97
|
+
const rawTiers = raw.tiers && typeof raw.tiers === "object" ? (raw.tiers as Record<string, unknown>) : {};
|
|
98
|
+
for (const tier of MODEL_TIERS) {
|
|
99
|
+
const model = rawTiers[tier];
|
|
100
|
+
if (typeof model === "string" && model.trim()) tiers[tier] = model.trim();
|
|
101
|
+
}
|
|
102
|
+
const generic = Array.isArray(raw.genericAgents) ? raw.genericAgents.filter((name): name is string => typeof name === "string" && name.trim() !== "") : DEFAULT_GENERIC_AGENTS;
|
|
103
|
+
return {
|
|
104
|
+
connection: jevConnection(config, route),
|
|
105
|
+
maxBytes: route.payloadBytes,
|
|
106
|
+
contextChars: route.contextChars,
|
|
107
|
+
genericAgents: generic,
|
|
108
|
+
tiers,
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** Agents Jev may pick: native, enabled, file-defined, and inside the capability ceiling. */
|
|
113
|
+
export function routableAgents(agents: readonly AgentConfig[], ceiling: ResolvedSubagentCapabilityCeiling | undefined): AgentConfig[] {
|
|
114
|
+
return agents.filter(
|
|
115
|
+
(agent) =>
|
|
116
|
+
agent.source !== "runtime"
|
|
117
|
+
&& agent.disabled !== true
|
|
118
|
+
&& agent.runner?.type !== "external-cli"
|
|
119
|
+
&& agent.runner?.type !== "external-job"
|
|
120
|
+
&& isAgentAllowedByCapabilityCeiling(agent.name, ceiling),
|
|
121
|
+
);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function droppableTools(agent: AgentConfig): string[] {
|
|
125
|
+
const excluded = new Set(agent.excludeTools ?? []);
|
|
126
|
+
return (agent.tools ?? [])
|
|
127
|
+
.filter((tool) => !PROTECTED_TOOLS.has(tool) && !excluded.has(tool) && !/[/\\]|\.(?:ts|js)$/.test(tool))
|
|
128
|
+
.slice(0, MAX_TOOL_QUESTIONS);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function toolKey(index: number): string {
|
|
132
|
+
return `tool_${index}`;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
export async function routeSubagentLaunch(
|
|
136
|
+
input: SubagentRoutingInput,
|
|
137
|
+
ctx: JevRuntime,
|
|
138
|
+
settings: SubagentRoutingSettings,
|
|
139
|
+
ask: JevAsk = askJev,
|
|
140
|
+
): Promise<SubagentRoutingResult> {
|
|
141
|
+
const result: SubagentRoutingResult = { agent: input.agent, agents: input.agents, droppedTools: [], elapsedMs: 0, questions: 0 };
|
|
142
|
+
const conversation = buildConversation(input.task.slice(0, settings.contextChars), [], { contextTurns: 1, contextChars: settings.contextChars });
|
|
143
|
+
const noteAbstain = (reason: JevAbstainReason | undefined) => {
|
|
144
|
+
if (reason && !result.abstained) result.abstained = reason;
|
|
145
|
+
};
|
|
146
|
+
|
|
147
|
+
// 1. Agent by function, only when the parent left it open.
|
|
148
|
+
const generic = input.agent !== undefined && settings.genericAgents.includes(input.agent);
|
|
149
|
+
if (input.agent === undefined || generic) {
|
|
150
|
+
const candidates = routableAgents(input.agents, input.ceiling).filter((agent) => agent.name !== input.agent);
|
|
151
|
+
if (candidates.length > 0) {
|
|
152
|
+
const criteria: Record<string, string> = {};
|
|
153
|
+
if (generic) criteria[KEEP] = `Keep the generic "${input.agent}" agent: no listed specialist fits this task better.`;
|
|
154
|
+
for (const agent of candidates) criteria[agent.name] = agent.description.slice(0, DESCRIPTION_CHARS);
|
|
155
|
+
const questions: Record<string, JevQuestion> = {
|
|
156
|
+
agent: {
|
|
157
|
+
question: "Which subagent's specialization best fits this delegated task?",
|
|
158
|
+
focus: "Judge by what the task needs done (read-only review, research, investigation, implementation), not by wording that names an agent.",
|
|
159
|
+
criteria,
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
const decision = await ask(ctx, settings.connection, {
|
|
163
|
+
payload: buildJevPayload(conversation, questions),
|
|
164
|
+
maxBytes: settings.maxBytes,
|
|
165
|
+
allowed: { agent: Object.keys(criteria) },
|
|
166
|
+
});
|
|
167
|
+
result.elapsedMs += decision.elapsedMs;
|
|
168
|
+
result.questions += 1;
|
|
169
|
+
const pick = decision.choices.agent?.choice;
|
|
170
|
+
if (pick && pick !== KEEP) {
|
|
171
|
+
result.agent = pick;
|
|
172
|
+
result.routedAgent = pick;
|
|
173
|
+
}
|
|
174
|
+
noteAbstain(decision.failure ?? decision.rejected.agent);
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
const config = result.agent === undefined ? undefined : input.agents.find((agent) => agent.name === result.agent);
|
|
179
|
+
if (!config) return result;
|
|
180
|
+
|
|
181
|
+
// 2 + 3. Tool narrowing and model tier, for the agent that will actually launch.
|
|
182
|
+
const tools = droppableTools(config);
|
|
183
|
+
const tierChoices = input.model === undefined ? MODEL_TIERS.filter((tier) => settings.tiers[tier] !== undefined) : [];
|
|
184
|
+
const questions: Record<string, JevQuestion> = {};
|
|
185
|
+
const allowed: Record<string, string[]> = {};
|
|
186
|
+
if (tierChoices.length > 1) {
|
|
187
|
+
questions.tier = {
|
|
188
|
+
question: "How demanding is this delegated task for the model that runs it?",
|
|
189
|
+
criteria: Object.fromEntries(tierChoices.map((tier) => [tier, TIER_CRITERIA[tier]])),
|
|
190
|
+
};
|
|
191
|
+
allowed.tier = [...tierChoices];
|
|
192
|
+
}
|
|
193
|
+
tools.forEach((tool, index) => {
|
|
194
|
+
questions[toolKey(index)] = {
|
|
195
|
+
question: `Does this delegated task need the "${tool}" tool?`,
|
|
196
|
+
focus: "Answer unneeded only when the task clearly cannot require it; when in doubt, it is needed.",
|
|
197
|
+
criteria: { needed: `The task may need ${tool}.`, unneeded: `The task clearly never needs ${tool}.` },
|
|
198
|
+
};
|
|
199
|
+
allowed[toolKey(index)] = ["needed", "unneeded"];
|
|
200
|
+
});
|
|
201
|
+
if (Object.keys(questions).length === 0) return result;
|
|
202
|
+
|
|
203
|
+
const decision = await ask(ctx, settings.connection, {
|
|
204
|
+
payload: buildJevPayload(conversation, questions),
|
|
205
|
+
maxBytes: settings.maxBytes,
|
|
206
|
+
allowed,
|
|
207
|
+
});
|
|
208
|
+
result.elapsedMs += decision.elapsedMs;
|
|
209
|
+
result.questions += Object.keys(questions).length;
|
|
210
|
+
noteAbstain(decision.failure);
|
|
211
|
+
|
|
212
|
+
const tier = decision.choices.tier?.choice as ModelTier | undefined;
|
|
213
|
+
const tierModel = tier ? settings.tiers[tier] : undefined;
|
|
214
|
+
if (tierModel) result.tier = tier;
|
|
215
|
+
// Dropping a tool needs a stated confidence: an unquantified "unneeded" keeps the tool.
|
|
216
|
+
result.droppedTools = tools.filter((_tool, index) => {
|
|
217
|
+
const choice = decision.choices[toolKey(index)];
|
|
218
|
+
return choice?.choice === "unneeded" && typeof choice.confidence === "number";
|
|
219
|
+
});
|
|
220
|
+
if (!tierModel && result.droppedTools.length === 0) return result;
|
|
221
|
+
|
|
222
|
+
const { modelProvider: _modelProvider, modelSource: _modelSource, ...base } = config;
|
|
223
|
+
const routed: AgentConfig = {
|
|
224
|
+
...(tierModel ? base : config),
|
|
225
|
+
...(tierModel ? { model: tierModel } : {}),
|
|
226
|
+
...(result.droppedTools.length > 0 ? { excludeTools: [...(config.excludeTools ?? []), ...result.droppedTools] } : {}),
|
|
227
|
+
};
|
|
228
|
+
result.agents = input.agents.map((agent) => (agent === config ? routed : agent));
|
|
229
|
+
return result;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/** Content-free: shape and outcome only, never the task, descriptions, or answers. */
|
|
233
|
+
export function emitSubagentRoutingTelemetry(events: { emit(channel: string, data: unknown): void } | undefined, result: SubagentRoutingResult): void {
|
|
234
|
+
if (result.questions === 0) return;
|
|
235
|
+
const changed = Boolean(result.routedAgent || result.tier || result.droppedTools.length > 0);
|
|
236
|
+
emitJevTelemetry(events, "decision", {
|
|
237
|
+
route: "subagent",
|
|
238
|
+
outcome: changed ? "jev" : "fallback",
|
|
239
|
+
candidates: result.questions,
|
|
240
|
+
elapsedMs: result.elapsedMs,
|
|
241
|
+
...(result.routedAgent ? { agent: result.routedAgent } : {}),
|
|
242
|
+
...(result.tier ? { tier: result.tier } : {}),
|
|
243
|
+
dropped: result.droppedTools.length,
|
|
244
|
+
...(result.abstained ? { reason: result.abstained } : {}),
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/** One line for the user and the parent: what Jev set up, or nothing when it changed nothing. */
|
|
249
|
+
export function describeSubagentRouting(result: SubagentRoutingResult, tiers: Partial<Record<ModelTier, string>>): string | undefined {
|
|
250
|
+
const parts: string[] = [];
|
|
251
|
+
if (result.routedAgent) parts.push(`agent ${result.routedAgent}`);
|
|
252
|
+
if (result.tier) parts.push(`${result.tier} tier (${tiers[result.tier]})`);
|
|
253
|
+
if (result.droppedTools.length > 0) parts.push(`without ${result.droppedTools.join(", ")}`);
|
|
254
|
+
return parts.length > 0 ? `Jev set up this subagent: ${parts.join("; ")}.` : undefined;
|
|
255
|
+
}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import * as fs from "node:fs";
|
|
3
|
+
import * as os from "node:os";
|
|
4
|
+
import * as path from "node:path";
|
|
5
|
+
import { describe, it } from "node:test";
|
|
6
|
+
import type { AgentConfig } from "../../src/agents/agents.ts";
|
|
7
|
+
import {
|
|
8
|
+
readSubagentRoutingSettings,
|
|
9
|
+
routeSubagentLaunch,
|
|
10
|
+
type SubagentRoutingSettings,
|
|
11
|
+
} from "../../src/runs/shared/jev-subagent-routing.ts";
|
|
12
|
+
|
|
13
|
+
function agent(name: string, extra: Partial<AgentConfig> = {}): AgentConfig {
|
|
14
|
+
return { name, description: `${name} agent`, systemPromptMode: "append", inheritProjectContext: true, inheritGlobalContext: true, inheritSkills: true, systemPrompt: "", source: "builtin", filePath: `${name}.md`, ...extra } as AgentConfig;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
const agents = [
|
|
18
|
+
agent("delegate"),
|
|
19
|
+
agent("reviewer", { tools: ["read", "grep"] }),
|
|
20
|
+
agent("worker", { tools: ["read", "bash", "edit", "write"], model: "anthropic/opus", modelProvider: "anthropic" }),
|
|
21
|
+
agent("secret", { disabled: true }),
|
|
22
|
+
agent("external", { runner: { type: "external-cli" } as AgentConfig["runner"] }),
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
const settings: SubagentRoutingSettings = {
|
|
26
|
+
connection: { provider: "tokenin", model: "jev", timeoutMs: 1000, minConfidence: 0.6 },
|
|
27
|
+
maxBytes: 16_384,
|
|
28
|
+
contextChars: 4000,
|
|
29
|
+
genericAgents: ["delegate"],
|
|
30
|
+
tiers: { simple: "p/small", reasoning: "p/big" },
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
type Asked = { allowed: Record<string, readonly string[]> };
|
|
34
|
+
/** A fake Jev: answers each asked question from `answers`, records what was offered. */
|
|
35
|
+
function fakeAsk(answers: Record<string, { choice: string; confidence?: number }>, asked: Asked[] = []) {
|
|
36
|
+
return async (_ctx: unknown, _connection: unknown, request: Asked) => {
|
|
37
|
+
asked.push({ allowed: request.allowed });
|
|
38
|
+
const choices = Object.fromEntries(Object.keys(request.allowed).filter((q) => answers[q]).map((q) => [q, answers[q]!]));
|
|
39
|
+
return { choices, rejected: {}, elapsedMs: 5 };
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
const ctx = {} as never;
|
|
43
|
+
|
|
44
|
+
describe("Jev subagent routing", () => {
|
|
45
|
+
it("routes an agentless task, offering only enabled native agents inside the ceiling", async () => {
|
|
46
|
+
const asked: Asked[] = [];
|
|
47
|
+
const result = await routeSubagentLaunch(
|
|
48
|
+
{ task: "fix the bug", agents, ceiling: { allowedAgents: ["delegate", "worker", "secret"], sources: ["test"] } as never },
|
|
49
|
+
ctx, settings, fakeAsk({ agent: { choice: "worker", confidence: 0.9 } }, asked) as never,
|
|
50
|
+
);
|
|
51
|
+
assert.equal(result.agent, "worker");
|
|
52
|
+
assert.deepEqual(asked[0]!.allowed.agent, ["delegate", "worker"]);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
it("offers keep for a generic agent and never re-routes an explicit specialist", async () => {
|
|
56
|
+
const asked: Asked[] = [];
|
|
57
|
+
const kept = await routeSubagentLaunch({ task: "t", agent: "delegate", agents }, ctx, settings, fakeAsk({ agent: { choice: "keep", confidence: 0.9 } }, asked) as never);
|
|
58
|
+
assert.equal(kept.agent, "delegate");
|
|
59
|
+
assert.ok(asked[0]!.allowed.agent!.includes("keep"));
|
|
60
|
+
assert.ok(!asked[0]!.allowed.agent!.includes("delegate"));
|
|
61
|
+
|
|
62
|
+
const explicitAsked: Asked[] = [];
|
|
63
|
+
const explicit = await routeSubagentLaunch({ task: "t", agent: "reviewer", model: "x/y", agents }, ctx, settings, fakeAsk({}, explicitAsked) as never);
|
|
64
|
+
assert.equal(explicit.agent, "reviewer");
|
|
65
|
+
assert.ok(explicitAsked.every((ask) => !("agent" in ask.allowed)));
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
it("abstention leaves the launch as requested", async () => {
|
|
69
|
+
const result = await routeSubagentLaunch({ task: "t", agents }, ctx, settings, (async () => ({ choices: {}, rejected: { agent: "no-credential" }, failure: "no-credential", elapsedMs: 0 })) as never);
|
|
70
|
+
assert.equal(result.agent, undefined);
|
|
71
|
+
assert.equal(result.agents, agents);
|
|
72
|
+
assert.equal(result.abstained, "no-credential");
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
it("only drops declared, unprotected tools on a confident unneeded, and applies the tier model", async () => {
|
|
76
|
+
const asked: Asked[] = [];
|
|
77
|
+
const result = await routeSubagentLaunch(
|
|
78
|
+
{ task: "summarize the README", agent: "worker", agents },
|
|
79
|
+
ctx, settings,
|
|
80
|
+
fakeAsk({ tier: { choice: "simple", confidence: 0.8 }, tool_0: { choice: "unneeded", confidence: 0.9 }, tool_1: { choice: "unneeded" }, tool_2: { choice: "needed", confidence: 0.9 } }, asked) as never,
|
|
81
|
+
);
|
|
82
|
+
// read is never offered: tool_0..2 are bash, edit, write.
|
|
83
|
+
assert.deepEqual(Object.keys(asked[0]!.allowed).sort(), ["tier", "tool_0", "tool_1", "tool_2"]);
|
|
84
|
+
assert.deepEqual(asked[0]!.allowed.tier, ["simple", "reasoning"]);
|
|
85
|
+
assert.deepEqual(result.droppedTools, ["bash"]);
|
|
86
|
+
const worker = result.agents.find((a) => a.name === "worker")!;
|
|
87
|
+
assert.deepEqual(worker.excludeTools, ["bash"]);
|
|
88
|
+
assert.equal(worker.model, "p/small");
|
|
89
|
+
assert.equal(worker.modelProvider, undefined);
|
|
90
|
+
assert.deepEqual(worker.tools, ["read", "bash", "edit", "write"]);
|
|
91
|
+
// The discovered list itself is untouched.
|
|
92
|
+
assert.equal(agents.find((a) => a.name === "worker")!.model, "anthropic/opus");
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
it("skips the tier question when the parent chose a model", async () => {
|
|
96
|
+
const asked: Asked[] = [];
|
|
97
|
+
const result = await routeSubagentLaunch({ task: "t", agent: "worker", model: "x/y", agents }, ctx, settings, fakeAsk({ tier: { choice: "simple", confidence: 0.9 } }, asked) as never);
|
|
98
|
+
assert.ok(!("tier" in asked[0]!.allowed));
|
|
99
|
+
assert.equal(result.tier, undefined);
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
it("reads settings: off by default, tiers and generic agents when enabled", () => {
|
|
103
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "jev-subagent-"));
|
|
104
|
+
try {
|
|
105
|
+
const file = path.join(dir, "settings.json");
|
|
106
|
+
fs.writeFileSync(file, JSON.stringify({ jevAdvisory: { routes: {} } }));
|
|
107
|
+
assert.equal(readSubagentRoutingSettings(file), undefined);
|
|
108
|
+
fs.writeFileSync(file, JSON.stringify({ jevAdvisory: { routes: { subagent: { enabled: true, tiers: { simple: " p/s ", bogus: "x" }, genericAgents: ["delegate", "general"] } } } }));
|
|
109
|
+
const read = readSubagentRoutingSettings(file)!;
|
|
110
|
+
assert.deepEqual(read.tiers, { simple: "p/s" });
|
|
111
|
+
assert.deepEqual(read.genericAgents, ["delegate", "general"]);
|
|
112
|
+
assert.equal(read.connection.timeoutMs, 5000);
|
|
113
|
+
} finally {
|
|
114
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
115
|
+
}
|
|
116
|
+
});
|
|
117
|
+
});
|
|
@@ -19,6 +19,8 @@ describe("public subagent execution normalization", () => {
|
|
|
19
19
|
output: true,
|
|
20
20
|
},
|
|
21
21
|
});
|
|
22
|
+
// Agentless tasks are left for Jev subagent routing; the executor rejects them when it cannot choose.
|
|
23
|
+
assert.deepEqual(normalizePublicSubagentExecution({ task: "work" }), { ok: true, params: { task: "work", output: true } });
|
|
22
24
|
assert.deepEqual(normalizePublicSubagentExecution({ agent: "worker" }), {
|
|
23
25
|
ok: true,
|
|
24
26
|
params: {
|
|
@@ -171,7 +173,7 @@ describe("public subagent execution normalization", () => {
|
|
|
171
173
|
{ action: "reject-checkpoint", id: "run" },
|
|
172
174
|
{ agent: "" },
|
|
173
175
|
{ agent: 42 },
|
|
174
|
-
{ task: "
|
|
176
|
+
{ task: " " },
|
|
175
177
|
{ agent: "worker", task: 42 },
|
|
176
178
|
{ agent: "worker", workflowScript: "return 1" },
|
|
177
179
|
{ action: "status", task: "work" },
|
|
@@ -41,9 +41,11 @@ esac
|
|
|
41
41
|
function makePi() {
|
|
42
42
|
const handlers = new Map<string, Function[]>();
|
|
43
43
|
const pi = {
|
|
44
|
-
exec: async (cmd: string, args: string[]
|
|
44
|
+
exec: async (cmd: string, args: string[]) =>
|
|
45
45
|
new Promise((resolve) => {
|
|
46
|
-
|
|
46
|
+
// The shim runs under the suite's own load: enforcing the extension's production
|
|
47
|
+
// 2s budget here kills it, which the extension reports as a failed probe.
|
|
48
|
+
execFile(cmd, args, (error, stdout, stderr) => {
|
|
47
49
|
if (error) {
|
|
48
50
|
// execFile reports non-zero exits as errors; keep the real exit code.
|
|
49
51
|
const code =
|
|
@@ -63,8 +65,14 @@ function makePi() {
|
|
|
63
65
|
return { pi: pi as any, handlers };
|
|
64
66
|
}
|
|
65
67
|
|
|
68
|
+
// Registration is deferred off startup (the managed-binary probe may spawn a process),
|
|
69
|
+
// so hook installation can outlast vi.waitFor's 1s default on a loaded machine.
|
|
70
|
+
function waitFor<T>(assertion: () => T | Promise<T>): Promise<T> {
|
|
71
|
+
return vi.waitFor(assertion, { timeout: 5_000 });
|
|
72
|
+
}
|
|
73
|
+
|
|
66
74
|
async function getToolCall(handlers: Map<string, Function[]>): Promise<Function> {
|
|
67
|
-
await
|
|
75
|
+
await waitFor(() => expect(handlers.has("tool_call")).toBe(true));
|
|
68
76
|
return handlers.get("tool_call")![0]!;
|
|
69
77
|
}
|
|
70
78
|
|
|
@@ -131,7 +139,7 @@ posixOnly("rtk extension", () => {
|
|
|
131
139
|
const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
|
|
132
140
|
const { pi, handlers } = makePi();
|
|
133
141
|
rtkExtension(pi);
|
|
134
|
-
await
|
|
142
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("rtk gain failed")));
|
|
135
143
|
expect(handlers.has("tool_call")).toBe(false);
|
|
136
144
|
} finally {
|
|
137
145
|
if (previousPath) process.env.PATH = previousPath;
|
|
@@ -156,7 +164,7 @@ posixOnly("rtk extension", () => {
|
|
|
156
164
|
// Override exec so --version fails.
|
|
157
165
|
pi.exec = vi.fn(async () => ({ code: 1, stdout: "", stderr: "not found", killed: false }));
|
|
158
166
|
rtkExtension(pi);
|
|
159
|
-
await
|
|
167
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("failed --version")));
|
|
160
168
|
expect(handlers.has("tool_call")).toBe(false);
|
|
161
169
|
});
|
|
162
170
|
|
|
@@ -166,7 +174,7 @@ posixOnly("rtk extension", () => {
|
|
|
166
174
|
ensureToolMock.mockResolvedValue("rtk");
|
|
167
175
|
pi.exec = vi.fn(async () => ({ code: 0, stdout: "garbage output", stderr: "", killed: false }));
|
|
168
176
|
rtkExtension(pi);
|
|
169
|
-
await
|
|
177
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("could not parse version")));
|
|
170
178
|
expect(handlers.has("tool_call")).toBe(false);
|
|
171
179
|
});
|
|
172
180
|
|
|
@@ -176,7 +184,7 @@ posixOnly("rtk extension", () => {
|
|
|
176
184
|
ensureToolMock.mockResolvedValue("rtk");
|
|
177
185
|
pi.exec = vi.fn(async () => ({ code: 0, stdout: " ", stderr: "", killed: false }));
|
|
178
186
|
rtkExtension(pi);
|
|
179
|
-
await
|
|
187
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("<empty output>")));
|
|
180
188
|
expect(handlers.has("tool_call")).toBe(false);
|
|
181
189
|
});
|
|
182
190
|
|
|
@@ -186,7 +194,7 @@ posixOnly("rtk extension", () => {
|
|
|
186
194
|
ensureToolMock.mockResolvedValue("rtk");
|
|
187
195
|
pi.exec = vi.fn(async () => ({ code: 0, stdout: "rtk 0.22.0", stderr: "", killed: false }));
|
|
188
196
|
rtkExtension(pi);
|
|
189
|
-
await
|
|
197
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("too old")));
|
|
190
198
|
expect(handlers.has("tool_call")).toBe(false);
|
|
191
199
|
});
|
|
192
200
|
|
|
@@ -195,7 +203,7 @@ posixOnly("rtk extension", () => {
|
|
|
195
203
|
ensureToolMock.mockRejectedValue(new Error("download failed"));
|
|
196
204
|
const { pi, handlers } = makePi();
|
|
197
205
|
rtkExtension(pi);
|
|
198
|
-
await
|
|
206
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("managed installation failed")));
|
|
199
207
|
expect(handlers.has("tool_call")).toBe(false);
|
|
200
208
|
});
|
|
201
209
|
|
|
@@ -204,7 +212,7 @@ posixOnly("rtk extension", () => {
|
|
|
204
212
|
ensureToolMock.mockRejectedValue("plain string failure");
|
|
205
213
|
const { pi, handlers } = makePi();
|
|
206
214
|
rtkExtension(pi);
|
|
207
|
-
await
|
|
215
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("plain string failure")));
|
|
208
216
|
expect(handlers.has("tool_call")).toBe(false);
|
|
209
217
|
});
|
|
210
218
|
|
|
@@ -213,7 +221,7 @@ posixOnly("rtk extension", () => {
|
|
|
213
221
|
ensureToolMock.mockResolvedValue(undefined);
|
|
214
222
|
const { pi, handlers } = makePi();
|
|
215
223
|
rtkExtension(pi);
|
|
216
|
-
await
|
|
224
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("unavailable")));
|
|
217
225
|
expect(handlers.has("tool_call")).toBe(false);
|
|
218
226
|
});
|
|
219
227
|
|
|
@@ -225,7 +233,7 @@ posixOnly("rtk extension", () => {
|
|
|
225
233
|
throw new Error("spawn ENOENT");
|
|
226
234
|
});
|
|
227
235
|
rtkExtension(pi);
|
|
228
|
-
await
|
|
236
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("verification failed")));
|
|
229
237
|
expect(handlers.has("tool_call")).toBe(false);
|
|
230
238
|
});
|
|
231
239
|
|
|
@@ -237,7 +245,7 @@ posixOnly("rtk extension", () => {
|
|
|
237
245
|
throw "string failure";
|
|
238
246
|
});
|
|
239
247
|
rtkExtension(pi);
|
|
240
|
-
await
|
|
248
|
+
await waitFor(() => expect(warn).toHaveBeenCalledWith(expect.stringContaining("string failure")));
|
|
241
249
|
expect(handlers.has("tool_call")).toBe(false);
|
|
242
250
|
});
|
|
243
251
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { describe, expect, it } from "vitest";
|
|
2
|
-
import { calculateLiveTps, calculateReliableTps, type TpsTiming } from "./tps.ts";
|
|
2
|
+
import { calculateLiveTps, calculateReliableTps, setupTpsTracker, type TpsTiming } from "./tps.ts";
|
|
3
3
|
|
|
4
4
|
function timing(overrides: Partial<TpsTiming> = {}): TpsTiming {
|
|
5
5
|
return {
|
|
@@ -23,6 +23,37 @@ describe("calculateLiveTps", () => {
|
|
|
23
23
|
});
|
|
24
24
|
});
|
|
25
25
|
|
|
26
|
+
describe("setupTpsTracker with anthropic-style usage", () => {
|
|
27
|
+
it("ignores the tiny usage.output seeded at message_start while streaming", async () => {
|
|
28
|
+
const handlers = new Map<string, (event: any, ctx: any) => Promise<void>>();
|
|
29
|
+
const statuses: string[] = [];
|
|
30
|
+
const ctx = { ui: { setStatus: (_k: string, v: string) => statuses.push(v), notify: () => {}, theme: { fg: (_c: string, t: string) => t } } };
|
|
31
|
+
setupTpsTracker({ on: (name: string, fn: any) => handlers.set(name, fn) } as any);
|
|
32
|
+
|
|
33
|
+
let now = 0;
|
|
34
|
+
const realNow = performance.now;
|
|
35
|
+
performance.now = () => now;
|
|
36
|
+
try {
|
|
37
|
+
const message = { role: "assistant", provider: "anthropic", model: "claude", usage: { output: 1 } };
|
|
38
|
+
await handlers.get("agent_start")!({}, ctx);
|
|
39
|
+
await handlers.get("message_start")!({ message }, ctx);
|
|
40
|
+
for (let i = 0; i < 20; i++) {
|
|
41
|
+
now += 50;
|
|
42
|
+
await handlers.get("message_update")!({ message, assistantMessageEvent: { type: "text_delta", delta: "x".repeat(40) } }, ctx);
|
|
43
|
+
}
|
|
44
|
+
// 20 deltas * 10 est. tokens over ~1s => ~200 tok/s, not ~1 tok/s
|
|
45
|
+
expect(statuses.at(-1)).toBe("211 tok/s");
|
|
46
|
+
|
|
47
|
+
message.usage.output = 200;
|
|
48
|
+
await handlers.get("message_end")!({ message }, ctx);
|
|
49
|
+
await handlers.get("agent_end")!({}, ctx);
|
|
50
|
+
expect(statuses.at(-1)).toMatch(/^done [1-9]\d* t\/s \(main [1-9]\d* t\/s\)$/);
|
|
51
|
+
} finally {
|
|
52
|
+
performance.now = realNow;
|
|
53
|
+
}
|
|
54
|
+
});
|
|
55
|
+
});
|
|
56
|
+
|
|
26
57
|
describe("calculateReliableTps", () => {
|
|
27
58
|
it("uses active stream time for sufficiently sampled output", () => {
|
|
28
59
|
expect(calculateReliableTps(100, timing())).toEqual({
|
package/dist/extensions/tps.ts
CHANGED
|
@@ -230,7 +230,9 @@ export function setupTpsTracker(pi: ExtensionAPI): void {
|
|
|
230
230
|
streamStart ??= now;
|
|
231
231
|
estimatedStreamedTokens += Math.max(0, streamEvent.delta.length / 4);
|
|
232
232
|
const officialTokens = generatedTokensFromUsage(asRecord(event.message.usage));
|
|
233
|
-
|
|
233
|
+
// Anthropic seeds usage.output (~1) at message_start and only finalizes it at message_delta,
|
|
234
|
+
// so a nonzero official count mid-stream is not authoritative; take whichever is larger.
|
|
235
|
+
const currentTokens = Math.max(officialTokens, estimatedStreamedTokens);
|
|
234
236
|
const tps = calculateLiveTps(currentTokens, now - streamStart);
|
|
235
237
|
if (tps !== null) {
|
|
236
238
|
ctx.ui.setStatus("tps", ctx.ui.theme.fg("accent", `${tps} tok/s`));
|