@diousk/pi-subagents-fast 0.20.0 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -19,10 +19,12 @@ import { statSync } from "node:fs";
19
19
  import { isAbsolute } from "node:path";
20
20
  import type { Model } from "@earendil-works/pi-ai";
21
21
  import type { AgentSession, ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
22
- import { resumeAgent, runAgent, type ToolActivity } from "./agent-runner.js";
22
+ import { resolveDefaultModel, resumeAgent, runAgent, type ToolActivity } from "./agent-runner.js";
23
+ import { getAgentConfig } from "./agent-types.js";
23
24
  import { assignHandle, handleBase } from "./mention.js";
24
25
  import { describeModel } from "./model-resolver.js";
25
- import type { AgentInvocation, AgentRecord, AgentTombstone, IsolationMode, MentionResolution, SubagentType, ThinkingLevel } from "./types.js";
26
+ import { loadRoutingPolicy, ModelRouter, type RoutingInput } from "./model-routing.js";
27
+ import type { AgentConfig, AgentInvocation, AgentRecord, AgentTombstone, IsolationMode, MentionResolution, SubagentType, ThinkingLevel } from "./types.js";
26
28
  import { addUsage, type LifetimeUsage } from "./usage.js";
27
29
  import type { CompiledSchema } from "./workflow/json-schema.js";
28
30
  import { cleanupWorktree, createWorktree, isWorktreeIsolationEnabled, pruneWorktrees, } from "./worktree.js";
@@ -167,6 +169,10 @@ interface SpawnArgs {
167
169
  }
168
170
 
169
171
  interface SpawnOptions {
172
+ /** Internal only; stripped at the programmatic/RPC boundary. */
173
+ routing?: RoutingInput;
174
+ /** Branch-local definition selected by a trusted invocation resolver. */
175
+ agentConfig?: AgentConfig;
170
176
  description: string;
171
177
  /**
172
178
  * Optional memorable name for this instance, becoming a second handle
@@ -361,6 +367,7 @@ async function shutdownChildSession(session: AgentSession | undefined): Promise<
361
367
  }
362
368
 
363
369
  export class AgentManager {
370
+ private router = new ModelRouter();
364
371
  private agents = new Map<string, AgentRecord>();
365
372
  private cleanupInterval: ReturnType<typeof setInterval>;
366
373
  private onComplete?: OnAgentComplete;
@@ -550,7 +557,7 @@ export class AgentManager {
550
557
  record.alias = assignHandle(handleBase(options.name), this.takenHandles());
551
558
  }
552
559
 
553
- const args: SpawnArgs = { pi, ctx, type, prompt, options };
560
+ const args: SpawnArgs = { pi, ctx, type, prompt, options: { ...options, agentConfig: options.agentConfig ?? getAgentConfig(type) } };
554
561
 
555
562
  const pool = this.poolFor(record);
556
563
  if (pool !== undefined && !options.bypassQueue && !this.poolHasRoom(pool)) {
@@ -696,12 +703,65 @@ export class AgentManager {
696
703
  if (pool === "background") this.runningBackground--;
697
704
  else if (pool === "foreground") this.runningForeground--;
698
705
  };
706
+ const wasQueued = record.startGate !== undefined;
699
707
  record.status = "running";
700
708
  record.startedAt = Date.now();
701
709
  record.startGate = undefined;
702
710
  if (pool === "background") this.runningBackground++;
703
711
  else if (pool === "foreground") this.runningForeground++;
704
712
 
713
+ const config = options.agentConfig;
714
+ const provenance = options.routing;
715
+ const explicit = provenance
716
+ ? provenance.modelExplicit || provenance.thinkingExplicit
717
+ : options.model != null || options.thinkingLevel != null;
718
+ const currentPolicy = wasQueued || !provenance?.policy ? loadRoutingPolicy(options.configCwd ?? ctx.cwd) : undefined;
719
+ const policy = currentPolicy ?? provenance!.policy!;
720
+ record.routing = { mode: policy.mode, source: policy.source, fallbackSource: policy.fallbackSource,
721
+ code: policy.mode === "off" ? "off" : policy.diagnostic ? "guideline_unavailable" : "baseline",
722
+ reason: policy.mode === "off" ? "Model routing and routing guidance are off" : policy.diagnostic ?? "Using the existing model", guidelinePath: policy.guidelinePath, guidelineHash: policy.guidelineHash };
723
+ if (policy.diagnostic) console.warn(`[pi-subagents] ${policy.diagnostic}`);
724
+ const route = policy.mode === "jev" || policy.mode === "shadow" ||
725
+ (policy.mode === "auto" && !explicit && !config?.model && !config?.thinking && policy.source === "jev");
726
+ if (!options.resumeSessionFile && provenance?.entrypoint !== "internal" && route) {
727
+ const stop = () => this.abort(id);
728
+ options.signal?.addEventListener("abort", stop, { once: true });
729
+ if (options.signal?.aborted) stop();
730
+ try {
731
+ const routed = await this.router.choose(ctx, policy, prompt, options.description ?? type,
732
+ options.model ?? resolveDefaultModel(ctx.model, ctx.modelRegistry, config?.model),
733
+ record.abortController!.signal, usage => {
734
+ record.routingUsage = usage;
735
+ // Classifier tokens stay separate from coding/context/output budgets.
736
+ const delta = { input: usage.input, output: usage.output, cacheWrite: usage.cacheWrite, cacheRead: usage.cacheRead, cost: usage.cost.total };
737
+ this.onUsage?.(record, delta);
738
+ for (let current: AgentRecord | undefined = record; current; ) {
739
+ current.lifetimeUsage.cost = (current.lifetimeUsage.cost ?? 0) + usage.cost.total;
740
+ current = current.parentAgentId ? this.agents.get(current.parentAgentId) : undefined;
741
+ }
742
+ });
743
+ record.routing = { ...routed.decision, guidelinePath: policy.guidelinePath, guidelineHash: policy.guidelineHash };
744
+ if (routed.model && policy.mode === "shadow") {
745
+ record.routing = { ...record.routing, code: "shadow", model: undefined, suggestedModel: routed.decision.model,
746
+ fallbackSource: policy.source, reason: "Jev suggested a model; shadow mode kept the default-priority model" };
747
+ } else if (routed.model) {
748
+ options.model = routed.model;
749
+ }
750
+ } catch {
751
+ record.routing.code = "classifier_error";
752
+ record.routing.reason = "Jev could not choose a model; using the existing model";
753
+ } finally {
754
+ options.signal?.removeEventListener("abort", stop);
755
+ }
756
+ if (record.status !== "running" || record.abortController!.signal.aborted) {
757
+ this.settleRun(record, true, pool);
758
+ return;
759
+ }
760
+ } else if (policy.mode === "auto" && (explicit || config?.model || config?.thinking)) {
761
+ record.routing.code = "explicit";
762
+ record.routing.reason = "Explicit model or thinking; automatic routing skipped";
763
+ }
764
+
705
765
  // Worktree isolation: try to create a temporary git worktree. Strict —
706
766
  // fail loud if not possible (no silent fallback to main tree). Done BEFORE
707
767
  // the run is kicked off so a failure doesn't leave a half-running agent.
@@ -761,6 +821,7 @@ export class AgentManager {
761
821
  const promise = runAgent(ctx, type, prompt, {
762
822
  pi,
763
823
  agentId: id,
824
+ agentConfig: options.agentConfig,
764
825
  model: options.model,
765
826
  maxTurns: options.maxTurns,
766
827
  isolated: options.isolated,
@@ -1556,6 +1617,7 @@ export class AgentManager {
1556
1617
  */
1557
1618
  async dispose(pi?: ExtensionAPI): Promise<void> {
1558
1619
  clearInterval(this.cleanupInterval);
1620
+ this.abortAll();
1559
1621
  // Clear queue — via dequeue, so anyone blocked in spawnAndWait is woken
1560
1622
  // rather than left awaiting a gate nothing will ever resolve.
1561
1623
  this.dequeue(() => true);
@@ -14,6 +14,7 @@ import {
14
14
  DefaultResourceLoader,
15
15
  type ExtensionAPI,
16
16
  getAgentDir,
17
+ type ModelRuntime,
17
18
  SessionManager,
18
19
  SettingsManager,
19
20
  } from "@earendil-works/pi-coding-agent";
@@ -27,7 +28,7 @@ import { createNestedSubagentTools, getMaxSubagentDepth, type NestedAgentManager
27
28
  import { buildAgentPrompt, type PromptExtras } from "./prompts.js";
28
29
  import { preloadSkills } from "./skill-loader.js";
29
30
  import { createStructuredCapture, createStructuredOutputTool, structuredRetryPrompt } from "./structured-output.js";
30
- import type { ServiceTier, SubagentType, ThinkingLevel } from "./types.js";
31
+ import type { AgentConfig, ServiceTier, SubagentType, ThinkingLevel } from "./types.js";
31
32
  import type { LifetimeUsage } from "./usage.js";
32
33
  import type { CompiledSchema } from "./workflow/json-schema.js";
33
34
 
@@ -94,6 +95,8 @@ export function installServiceTierPayload(
94
95
  * single-file extensions to the basename minus `.ts`/`.js`.
95
96
  */
96
97
  export function extensionCanonicalName(extPath: string): string {
98
+ if (extPath.startsWith("builtin:")) return extPath.slice("builtin:".length).toLowerCase();
99
+ if (extPath.startsWith("<inline:") && extPath.endsWith(">")) return extPath.slice(8, -1).toLowerCase();
97
100
  const base = basename(extPath);
98
101
  const name = base === "index.ts" || base === "index.js"
99
102
  ? basename(dirname(extPath))
@@ -159,6 +162,7 @@ function extensionPackageName(extPath: string): string | undefined {
159
162
  */
160
163
  export function extensionCanonicalNames(extPath: string): string[] {
161
164
  const canonical = extensionCanonicalName(extPath);
165
+ if (extPath.startsWith("builtin:") || extPath.startsWith("<inline:")) return [canonical];
162
166
  const pkg = extensionPackageName(extPath);
163
167
  return pkg && pkg !== canonical ? [canonical, pkg] : [canonical];
164
168
  }
@@ -250,12 +254,13 @@ export function parseExtSelectors(entries: string[]): {
250
254
  * snapshotted. `registerTool` writes into the very `extension.tools` maps this reads,
251
255
  * so `inScope()` sees late arrivals on the next call.
252
256
  *
253
- * Two enforcement points, because neither covers the whole picture:
257
+ * The active set and call-time checks cover different parts of scope:
254
258
  *
255
259
  * - `turn_end` re-narrows the ACTIVE set. pi emits `turn_end` immediately before
256
260
  * `prepareNextTurn` re-snapshots `agent.state.tools`, and session listeners run
257
261
  * synchronously, so the narrow lands in time for turns 2..N.
258
- * - `beforeToolCall` blocks out-of-scope calls. Turn 1 cannot be narrowed at all:
262
+ * - The loader's bound `tool_call` handler blocks direct and nested calls.
263
+ * Turn 1 cannot be narrowed at all:
259
264
  * `before_agent_start` fires INSIDE `prompt()` and may widen the tool set, but
260
265
  * `createContextSnapshot()` freezes that turn's tools immediately after — there
261
266
  * is no hook in between. A call-time check is the only correct guard there.
@@ -285,7 +290,7 @@ export function installExtensionToolScope(
285
290
  */
286
291
  readmitToolNames: Set<string>;
287
292
  },
288
- ): void {
293
+ ): (toolName: string) => boolean {
289
294
  const { loader, toolNames, disallowedSet, extNames, narrowing, readmitToolNames } = ctx;
290
295
 
291
296
  // The names allowed right now. Mirrors the `ext:` opt-in flip: when any `ext:`
@@ -309,7 +314,7 @@ export function installExtensionToolScope(
309
314
  }
310
315
  for (const name of EXCLUDED_TOOL_NAMES) keep.delete(name);
311
316
  // Injected tools are legitimately active for this agent — re-admit them so
312
- // the renarrow keeps them in the active set and beforeToolCall doesn't
317
+ // the renarrow keeps them in the active set and the call-time guard doesn't
313
318
  // block them. Already vetted against `disallowed_tools` by the caller,
314
319
  // which is the only place that knows which kind may be taken back.
315
320
  for (const name of readmitToolNames) keep.add(name);
@@ -318,8 +323,14 @@ export function installExtensionToolScope(
318
323
 
319
324
  const renarrow = () => {
320
325
  const allowed = inScope();
321
- const next = session.getAllTools().map((t) => t.name).filter((n) => allowed.has(n));
322
326
  const current = session.getActiveToolNames();
327
+ // Keep deferred/codemode tools callable without promoting them into model
328
+ // declarations. Explicit activation by an extension is preserved in scope.
329
+ const next = session.getAllTools().filter((tool) =>
330
+ allowed.has(tool.name) && (
331
+ tool.exposure === "direct" || tool.exposure === "model-only" || current.includes(tool.name)
332
+ ),
333
+ ).map((tool) => tool.name);
323
334
  // setActiveToolsByName unconditionally rebuilds the system prompt, so skip
324
335
  // the no-op that steady-state turns would otherwise pay for every turn.
325
336
  if (next.length !== current.length || next.some((n, i) => n !== current[i])) {
@@ -335,16 +346,7 @@ export function installExtensionToolScope(
335
346
  if (event.type === "turn_end") renarrow();
336
347
  });
337
348
 
338
- const priorBeforeToolCall = session.agent.beforeToolCall;
339
- session.agent.beforeToolCall = async (context, signal) => {
340
- if (!inScope().has(context.toolCall.name)) {
341
- return {
342
- block: true,
343
- reason: `Tool "${context.toolCall.name}" is not available to this subagent.`,
344
- };
345
- }
346
- return priorBeforeToolCall?.(context, signal);
347
- };
349
+ return (toolName) => inScope().has(toolName);
348
350
  }
349
351
 
350
352
  /** Default max turns. undefined = unlimited (no turn limit). */
@@ -434,6 +436,8 @@ export interface ToolActivity {
434
436
  }
435
437
 
436
438
  export interface RunOptions {
439
+ /** Snapshot of the selected definition for this branch. */
440
+ agentConfig?: AgentConfig;
437
441
  /** ExtensionAPI instance — used for pi.exec() instead of execSync. */
438
442
  pi: ExtensionAPI;
439
443
  /** Manager-assigned id; suffixes session name to disambiguate parallel spawns (e.g. `Explore#a1b2c3d4`). */
@@ -652,8 +656,8 @@ export async function runAgent(
652
656
  prompt: string,
653
657
  options: RunOptions,
654
658
  ): Promise<RunResult> {
655
- const config = getConfig(type);
656
- const agentConfig = getAgentConfig(type);
659
+ const agentConfig = options.agentConfig ?? getAgentConfig(type);
660
+ const config = agentConfig ?? getConfig(type);
657
661
 
658
662
  // Resolve working directory: worktree override > parent cwd
659
663
  const effectiveCwd = options.cwd ?? ctx.cwd;
@@ -686,7 +690,7 @@ export async function runAgent(
686
690
  }
687
691
  }
688
692
 
689
- let toolNames = getToolNamesForType(type);
693
+ let toolNames = options.agentConfig ? options.agentConfig.builtinToolNames ?? [...BUILTIN_TOOL_NAMES] : getToolNamesForType(type);
690
694
 
691
695
  // Persistent memory: detect write capability and branch accordingly.
692
696
  // Account for disallowedTools — a tool in the base set but on the denylist is not truly available.
@@ -768,6 +772,7 @@ export async function runAgent(
768
772
  // must compare against this, not the surviving set (absence from survivors is
769
773
  // an exclude *succeeding*).
770
774
  let discoveredNames: Set<string> | undefined;
775
+ let toolInScope: ((toolName: string) => boolean) | undefined;
771
776
  const extensionsOverride: ((base: LoadExtensionsResult) => LoadExtensionsResult) | undefined =
772
777
  noExtensions || (loadAll && !hasExcludes)
773
778
  ? undefined
@@ -776,6 +781,7 @@ export async function runAgent(
776
781
  return {
777
782
  ...base,
778
783
  extensions: base.extensions.filter((e) => {
784
+ if (e.path === "<inline:subagent-tool-scope>") return true;
779
785
  const canons = extensionCanonicalNames(e.path);
780
786
  if (canons.some((n) => excludeNames.has(n))) return false; // exclude wins
781
787
  return loadAll || canons.some((n) => keepNames.has(n));
@@ -788,6 +794,19 @@ export async function runAgent(
788
794
  agentDir,
789
795
  noExtensions,
790
796
  additionalExtensionPaths,
797
+ // Pi's nested ctx.executeTool calls dispatch tool_call directly, and prompt
798
+ // setup can replace agent.beforeToolCall. Enforce scope in that dispatch too.
799
+ extensionFactories: noExtensions ? [] : [{
800
+ name: "subagent-tool-scope",
801
+ hidden: true,
802
+ factory: (pi) => {
803
+ pi.on("tool_call", (event) => {
804
+ if (toolInScope && !toolInScope(event.toolName)) {
805
+ return { block: true, reason: `Tool "${event.toolName}" is not available to this subagent.` };
806
+ }
807
+ });
808
+ },
809
+ }],
791
810
  extensionsOverride,
792
811
  noSkills,
793
812
  noPromptTemplates: true,
@@ -1014,24 +1033,14 @@ export async function runAgent(
1014
1033
  })
1015
1034
  : SessionManager.inMemory(effectiveCwd);
1016
1035
 
1017
- // Pi 0.80.8 replaced createAgentSession's modelRegistry option with
1018
- // modelRuntime, but ExtensionContext still exposes only the registry facade.
1019
- // Pass both so the full supported Pi range retains the parent's providers.
1020
- const parentModelRuntime = (ctx.modelRegistry as unknown as { runtime?: unknown }).runtime;
1021
- const sessionOpts: Parameters<typeof createAgentSession>[0] & {
1022
- modelRegistry: ExtensionContext["modelRegistry"];
1023
- modelRuntime?: unknown;
1024
- } = {
1036
+ // ExtensionContext exposes the registry facade; its runtime owns provider auth.
1037
+ const parentModelRuntime = (ctx.modelRegistry as unknown as { runtime?: ModelRuntime }).runtime;
1038
+ const sessionOpts: Parameters<typeof createAgentSession>[0] = {
1025
1039
  cwd: effectiveCwd,
1026
1040
  agentDir,
1027
1041
  sessionManager,
1028
1042
  settingsManager,
1029
- modelRegistry: ctx.modelRegistry,
1030
- // `as never` is what keeps this assignable across the supported Pi range:
1031
- // pre-0.80.8 the field exists only via the `modelRuntime?: unknown` shim
1032
- // above, while newer Pi types it as `ModelRuntime` — a shape an opaque
1033
- // `unknown` read off the private facade field can never satisfy.
1034
- ...(parentModelRuntime !== undefined && { modelRuntime: parentModelRuntime as never }),
1043
+ modelRuntime: parentModelRuntime,
1035
1044
  model,
1036
1045
  tools: sessionTools,
1037
1046
  customTools: [...nestedTools, ...structuredTools],
@@ -1073,7 +1082,7 @@ export async function runAgent(
1073
1082
  // handled below by re-deriving scope from the loader's live extension maps —
1074
1083
  // `registerTool` writes into those same maps, so late arrivals are judged too.
1075
1084
  if (!noExtensions) {
1076
- installExtensionToolScope(session, {
1085
+ toolInScope = installExtensionToolScope(session, {
1077
1086
  loader,
1078
1087
  toolNames,
1079
1088
  disallowedSet,
@@ -3,9 +3,10 @@
3
3
  */
4
4
 
5
5
  import { existsSync, readdirSync, readFileSync } from "node:fs";
6
- import { basename, join } from "node:path";
6
+ import { basename, join, resolve } from "node:path";
7
7
  import { getAgentDir, parseFrontmatter } from "@earendil-works/pi-coding-agent";
8
8
  import { BUILTIN_TOOL_NAMES } from "./agent-types.js";
9
+ import { loadRoutingSettings } from "./settings.js";
9
10
  import type { AgentConfig, IsolationMode, MemoryScope, ServiceTier, ThinkingLevel } from "./types.js";
10
11
 
11
12
  /**
@@ -47,9 +48,10 @@ export function loadCustomAgents(cwd: string, strict = false): Map<string, Agent
47
48
  const projectDir = join(cwd, ".pi", "agents");
48
49
 
49
50
  const agents = new Map<string, AgentConfig>();
50
- loadFromDir(globalDir, agents, "global", strict); // lowest priority
51
- loadFromDir(workspaceProjectDir, agents, "project", strict); // shared workspace
52
- loadFromDir(projectDir, agents, "project", strict); // highest priority (overwrites)
51
+ const excluded = loadRoutingSettings(cwd).guidelineFile;
52
+ loadFromDir(globalDir, agents, "global", strict, excluded);
53
+ loadFromDir(workspaceProjectDir, agents, "project", strict, excluded);
54
+ loadFromDir(projectDir, agents, "project", strict, excluded);
53
55
 
54
56
  warnedLastLoad = warnedThisLoad;
55
57
  warnedThisLoad = new Set();
@@ -57,7 +59,7 @@ export function loadCustomAgents(cwd: string, strict = false): Map<string, Agent
57
59
  }
58
60
 
59
61
  /** Load agent configs from a directory into the map. */
60
- function loadFromDir(dir: string, agents: Map<string, AgentConfig>, source: "project" | "global", strict: boolean): void {
62
+ function loadFromDir(dir: string, agents: Map<string, AgentConfig>, source: "project" | "global", strict: boolean, excluded?: string): void {
61
63
  if (!existsSync(dir)) return;
62
64
 
63
65
  let files: string[];
@@ -68,6 +70,7 @@ function loadFromDir(dir: string, agents: Map<string, AgentConfig>, source: "pro
68
70
  }
69
71
 
70
72
  for (const file of files) {
73
+ if (file.toLowerCase() === "custom-route.md" || resolve(dir, file) === excluded) continue;
71
74
  const filenameType = basename(file, ".md");
72
75
 
73
76
  const path = join(dir, file);
@@ -297,7 +300,7 @@ function parseMemory(val: unknown): MemoryScope | undefined {
297
300
 
298
301
  /** Parse the OpenAI Responses/Codex `service_tier` frontmatter field. */
299
302
  function parseServiceTier(val: unknown): ServiceTier | undefined {
300
- if (val === "auto" || val === "default" || val === "flex" || val === "priority" || val === "scale") {
303
+ if (val === "auto" || val === "default" || val === "flex" || val === "fast" || val === "priority" || val === "scale") {
301
304
  return val;
302
305
  }
303
306
  return undefined;
package/src/index.ts CHANGED
@@ -29,12 +29,13 @@ import { isolationParam, resolveAgentInvocationConfig, resolveJoinMode } from ".
29
29
  import { describeMention, handleBase, isReservedHandle, parseMention, resolveHandleToType, stripAgentPrefix } from "./mention.js";
30
30
  import { runMentionClone } from "./mention-clone.js";
31
31
  import { describeModel, type ModelRegistry, resolveModel } from "./model-resolver.js";
32
+ import { loadRoutingPolicy, routingGuidance } from "./model-routing.js";
32
33
  import { checkModelScope, isScopeModelsEnabled, setScopeModelsEnabled } from "./model-scope.js";
33
34
  import { getMaxSubagentDepth, setMaxSubagentDepth } from "./nested-tools.js";
34
35
  import { createOutputFilePath, ensureOutputFile, getOutputTranscriptDefault, sessionTaskDir, setOutputTranscriptDefault, streamToOutputFile, writeInitialEntry } from "./output-file.js";
35
36
  import { SubagentScheduler } from "./schedule.js";
36
37
  import { resolveStorePath, ScheduleStore } from "./schedule-store.js";
37
- import { applyAndEmitLoaded, loadSettings, type SubagentsSettings, saveAndEmitChanged, type ToolDescriptionMode } from "./settings.js";
38
+ import { applyAndEmitLoaded, loadSettings, projectRoutingSettings, type SubagentsSettings, saveAndEmitChanged, type ToolDescriptionMode } from "./settings.js";
38
39
  import { getForegroundOutcomeNote, getStatusNote, partialOutputSuffix } from "./status-note.js";
39
40
  import { type AgentConfig, type AgentInvocation, type AgentMentionMode, type AgentRecord, type JoinMode, type NotificationDetails, type SubagentType, type ViewerMarkdownMode, type WidgetMode } from "./types.js";
40
41
  import { createMentionProvider, mentionRoster, type TypeInfo } from "./ui/agent-mention.js";
@@ -57,6 +58,7 @@ import {
57
58
  type UICtx,
58
59
  } from "./ui/agent-widget.js";
59
60
  import { FleetList, type FleetUICtx, type FleetWorkflow } from "./ui/fleet-list.js";
61
+ import { showRoutingMenu } from "./ui/model-routing-menu.js";
60
62
  import { showSchedulesMenu } from "./ui/schedule-menu.js";
61
63
  import { selectItem } from "./ui/select-item.js";
62
64
  import { renderWorkflowCard, renderWorkflowEntryCard } from "./ui/workflow-card.js";
@@ -396,6 +398,7 @@ export default function (pi: ExtensionAPI) {
396
398
  const reloadCustomAgents = (strict = false) => {
397
399
  const userAgents = loadCustomAgents(process.cwd(), strict);
398
400
  registerAgents(userAgents);
401
+ return userAgents;
399
402
  };
400
403
 
401
404
  // Initial load — the only strict one. A bad edit mid-session must not kill the
@@ -561,6 +564,8 @@ export default function (pi: ExtensionAPI) {
561
564
  durationMs,
562
565
  tokens,
563
566
  usage,
567
+ routing: record.routing,
568
+ routingUsage: record.routingUsage,
564
569
  };
565
570
  }
566
571
 
@@ -586,6 +591,7 @@ export default function (pi: ExtensionAPI) {
586
591
  id: record.id, type: record.type, description: record.description,
587
592
  status: record.status, result: record.result, error: record.error,
588
593
  startedAt: record.startedAt, completedAt: record.completedAt,
594
+ routing: record.routing, routingUsage: record.routingUsage,
589
595
  });
590
596
 
591
597
  // Skip notification if result was already consumed via get_subagent_result
@@ -697,6 +703,8 @@ export default function (pi: ExtensionAPI) {
697
703
 
698
704
  const spawnTopLevel = (piRef: any, ctxRef: any, type: string, prompt: string, options: any) => {
699
705
  const safeOptions = { ...(options ?? {}) };
706
+ delete safeOptions.routing;
707
+ delete safeOptions.agentConfig;
700
708
  delete safeOptions.parentAgentId;
701
709
  // Internal too: a forged value would hide an RPC-spawned agent inside
702
710
  // someone else's workflow, and take it out of the concurrency pool with it.
@@ -1364,6 +1372,16 @@ export default function (pi: ExtensionAPI) {
1364
1372
  widget.onTurnStart();
1365
1373
  });
1366
1374
 
1375
+ pi.on("before_agent_start", (event, ctx) => {
1376
+ const guidance = routingGuidance(loadRoutingPolicy(ctx.cwd));
1377
+ const description = agentToolDescription + (guidance ? "\n\n" + guidance : "");
1378
+ if (agentTool.description !== description) {
1379
+ agentTool.description = description;
1380
+ pi.registerTool(agentTool);
1381
+ }
1382
+ if (guidance) return { systemPrompt: event.systemPrompt + "\n\n" + guidance };
1383
+ });
1384
+
1367
1385
  /** Build the full type list text dynamically from available agents only. */
1368
1386
  const buildTypeListText = () => {
1369
1387
  const available = getAvailableTypes();
@@ -1578,10 +1596,11 @@ Terse command-style prompts produce shallow, generic work.
1578
1596
  // Held rather than registered inline: the mention clone reuses this exact
1579
1597
  // definition, so the agent it starts is an ordinary top-level spawn instead
1580
1598
  // of a second implementation that has to be kept in step with this one.
1599
+ const initialRoutingGuidance = routingGuidance(loadRoutingPolicy(process.cwd()));
1581
1600
  const agentTool = defineTool({
1582
1601
  name: SUBAGENT_TOOL_NAMES.AGENT,
1583
1602
  label: "Agent",
1584
- description: agentToolDescription,
1603
+ description: agentToolDescription + (initialRoutingGuidance ? "\n\n" + initialRoutingGuidance : ""),
1585
1604
  promptSnippet: "Launch autonomous sub-agents for complex multi-step tasks",
1586
1605
  promptGuidelines: [
1587
1606
  "Use Agent with specialized agents when the task matches an agent type's description. Subagents are valuable for parallelizing independent queries or for protecting the main context window from excessive results, but should not be used excessively when not needed. Importantly, avoid duplicating work that subagents are already doing — if you delegate research to a subagent, do not also perform the same searches yourself.",
@@ -1769,7 +1788,8 @@ Terse command-style prompts produce shallow, generic work.
1769
1788
  widget.setUICtx(ctx.ui as UICtx);
1770
1789
 
1771
1790
  // Reload custom agents so new project/global .md files are picked up without restart
1772
- reloadCustomAgents();
1791
+ const userAgents = reloadCustomAgents();
1792
+ const routingPolicy = loadRoutingPolicy(ctx.cwd, userAgents);
1773
1793
 
1774
1794
  const rawType = params.subagent_type as SubagentType;
1775
1795
  // Single decision point for dispatch (#183): unknown, disabled and
@@ -1916,6 +1936,10 @@ Terse command-style prompts produce shallow, generic work.
1916
1936
  const { modelName: recModelName, tags } = buildInvocationTags(rec.invocation);
1917
1937
  const recModeLabel = getPromptModeLabel(type);
1918
1938
  const recTags = recModeLabel ? [recModeLabel, ...tags] : tags;
1939
+ if (rec.routing?.source === "jev" && (rec.routingUsage || rec.routing.unpriced !== undefined)) {
1940
+ recTags.push(rec.routing.mode === "shadow" ? "Jev shadow" : "Jev");
1941
+ if (rec.routing.unpriced) recTags.push("Jev price unavailable");
1942
+ }
1919
1943
  return {
1920
1944
  displayName: getDisplayName(type),
1921
1945
  description: rec.description,
@@ -1952,7 +1976,7 @@ Terse command-style prompts produce shallow, generic work.
1952
1976
  subagent_type: requestedType,
1953
1977
  prompt: params.prompt as string,
1954
1978
  model: params.model as string | undefined,
1955
- thinking: thinking,
1979
+ thinking: params.thinking as typeof thinking,
1956
1980
  max_turns: effectiveMaxTurns,
1957
1981
  isolated: isolated,
1958
1982
  isolation: isolation,
@@ -2057,6 +2081,8 @@ Terse command-style prompts produce shallow, generic work.
2057
2081
  description: params.description,
2058
2082
  name: params.name as string | undefined,
2059
2083
  model,
2084
+ agentConfig: customConfig,
2085
+ routing: { policy: routingPolicy, modelExplicit: !!resolvedConfig.modelInput, thinkingExplicit: thinking !== undefined, entrypoint: "agent" },
2060
2086
  maxTurns: effectiveMaxTurns,
2061
2087
  isolated,
2062
2088
  inheritContext,
@@ -2211,6 +2237,8 @@ Terse command-style prompts produce shallow, generic work.
2211
2237
  description: params.description,
2212
2238
  name: params.name as string | undefined,
2213
2239
  model,
2240
+ agentConfig: customConfig,
2241
+ routing: { policy: routingPolicy, modelExplicit: !!resolvedConfig.modelInput, thinkingExplicit: thinking !== undefined, entrypoint: "agent" },
2214
2242
  maxTurns: effectiveMaxTurns,
2215
2243
  isolated,
2216
2244
  inheritContext,
@@ -2933,6 +2961,7 @@ Terse command-style prompts produce shallow, generic work.
2933
2961
  // Actions
2934
2962
  options.push("Create new agent");
2935
2963
  options.push("Settings");
2964
+ options.push("Model routing");
2936
2965
 
2937
2966
  const noAgentsMsg = allNames.length === 0 && agents.length === 0
2938
2967
  ? "No agents found. Create specialized subagents that can be delegated to.\n\n" +
@@ -2964,6 +2993,14 @@ Terse command-style prompts produce shallow, generic work.
2964
2993
  } else if (choice === "Settings") {
2965
2994
  await showSettings(ctx);
2966
2995
  await showAgentsMenu(ctx);
2996
+ } else if (choice === "Model routing") {
2997
+ const patch = await showRoutingMenu(ctx);
2998
+ if (patch) {
2999
+ const toast = saveAndEmitChanged({ ...snapshotSettings(), ...patch }, "Model routing settings updated", (event, payload) => pi.events.emit(event, payload), ctx.cwd);
3000
+ ctx.ui.notify(toast.message, toast.level);
3001
+ reloadCustomAgents();
3002
+ }
3003
+ await showAgentsMenu(ctx);
2967
3004
  }
2968
3005
  }
2969
3006
 
@@ -3331,6 +3368,7 @@ Write the file using the write tool. Only write the file, nothing else.`;
3331
3368
 
3332
3369
  const { record } = await manager.spawnAndWait(pi, ctx, "general-purpose", generatePrompt, {
3333
3370
  description: `Generate ${name} agent`,
3371
+ routing: { modelExplicit: false, thinkingExplicit: false, entrypoint: "internal" },
3334
3372
  maxTurns: 5,
3335
3373
  // Exempt from maxConcurrentForeground. This runs from a modal wizard, not
3336
3374
  // a tool call: it passes no signal, and Esc in `ctx.ui` never reaches the
@@ -3440,6 +3478,7 @@ Write the file using the write tool. Only write the file, nothing else.`;
3440
3478
  */
3441
3479
  function snapshotSettings() {
3442
3480
  return {
3481
+ ...projectRoutingSettings(process.cwd()),
3443
3482
  maxConcurrent: manager.getMaxConcurrent(),
3444
3483
  // 0 = unlimited, and the default — see SubagentsSettings.
3445
3484
  maxConcurrentForeground: manager.getMaxConcurrentForeground(),