@diousk/pi-subagents-fast 0.23.0 → 0.25.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -7,6 +7,22 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.25.1] - 2026-10-05
11
+
12
+ ### Fixed
13
+ - **GitHub Copilot agents omit `service_tier`.** Copilot rejects the field even on its Responses API. Configured tiers are ignored for Copilot requests and hidden from agent UI, including after model routing.
14
+
15
+ ## [0.25.0] - 2026-10-02
16
+
17
+ ### Added
18
+ - **Optional Jev model instructions.** Configure `instruction` per model in JSON or `/agents`; successful selection appends it to the child system prompt. Shadow and fallback preserve the original instructions.
19
+ - **Jev selects custom agent profiles.** In `jev` and `shadow` modes, enabled agent files and `jev.models` compete in one comparison; `jev` waits for the decision before starting the selected agent with its prompt, tools and model settings. Agent-only routing needs no duplicated model list, and runtime status shows selected and suggested agents. `auto` priority, low-confidence fallback and nested agent permissions are preserved.
20
+
21
+ ## [0.24.0] - 2026-10-02
22
+
23
+ ### Added
24
+ - **Routing runtime status in `/agents`.** Open Model routing → Runtime status to inspect effective configuration, credential presence, eligible profiles and the latest retained routing decisions, including fallback reasons and actual model/thinking. The refreshable, read-only view hides API keys and makes no classifier request.
25
+
10
26
  ## [0.23.0] - 2026-10-02
11
27
 
12
28
  ### Changed
package/README.md CHANGED
@@ -28,7 +28,7 @@ https://github.com/user-attachments/assets/8685261b-9338-4fea-8dfe-1c590d5df543
28
28
  - **Graceful turn limits** — agents get a "wrap up" warning before hard abort, producing clean partial results instead of cut-off output
29
29
  - **Case-insensitive agent types** — `"explore"`, `"Explore"`, `"EXPLORE"` all work. A type that doesn't resolve to exactly one *enabled* agent — unknown, disabled, or ambiguous between two agents differing only by case — falls back to general-purpose with a note, or is refused outright under [`fallbackSubagent: none`](#persistent-settings)
30
30
  - **Fuzzy model selection** — specify models by name (`"haiku"`, `"sonnet"`) instead of full IDs, with automatic filtering to only available/configured models
31
- - **Model routing** — custom agents first, then a user-supplied Markdown guideline, then optional Jev model selection. Configure it through `/agents → Model routing` or two ordinary JSON fields; see [Model routing](#model-routing)
31
+ - **Model routing** — custom agents first, then a user-supplied Markdown guideline, then optional Jev model selection. In `jev` mode, Jev compares custom agents and model profiles together before starting the child. Configure it through `/agents → Model routing`; see [Model routing](#model-routing)
32
32
  - **Context inheritance** — optionally fork the parent conversation into a sub-agent so it knows what's been discussed
33
33
  - **Persistent agent memory** — three scopes (project, local, user) with automatic read-only fallback for agents without write tools
34
34
  - **Git worktree isolation** — run agents in isolated repo copies; changes auto-committed to branches on completion
@@ -340,7 +340,7 @@ All fields are optional — sensible defaults for everything.
340
340
  | `disallowed_tools` | — | Comma-separated tools to deny even if extensions provide them |
341
341
  | `isolation` | — | Set to `worktree` to run in an isolated git worktree, or `off` to refuse one even when the caller passes `isolation: "worktree"` (frontmatter is authoritative). `none`, `no`, and `false` are accepted spellings of `off` |
342
342
  | `model` | inherit parent | Model — `provider/modelId` or fuzzy name (`"haiku"`, `"sonnet"`). Resolved tolerantly (`.`/`-` and a trailing date stamp are interchangeable) and falls back to the same model under another provider if the named one doesn't have it |
343
- | `service_tier` | — | OpenAI Responses/Codex processing tier: `auto`, `default`, `flex`, `fast`, `priority`, or `scale`. Applied only to `openai-responses` and `openai-codex-responses`; omitted preserves the provider default, and other APIs ignore it without displaying it as active |
343
+ | `service_tier` | — | OpenAI Responses/Codex processing tier: `auto`, `default`, `flex`, `fast`, `priority`, or `scale`. Applied only to `openai-responses` and `openai-codex-responses`, excluding GitHub Copilot; omitted preserves the provider default. GitHub Copilot and other APIs ignore it without displaying it as active |
344
344
  | `thinking` | inherit | off, minimal, low, medium, high, xhigh, max — actual availability depends on your pi version and model; pi maps or clamps unsupported levels |
345
345
  | `max_turns` | unlimited | Max agentic turns before graceful shutdown. `0` or omit for unlimited |
346
346
  | `persist_session` | `subagents.json` `rememberAgents` (default `true`) | Persist this subagent as a normal pi session instead of keeping the session in memory only; overrides the `rememberAgents` project default in both directions. It records its spawning session as parent, so it nests under it in `/resume`. The subagent's `.output` transcript is still written either way unless `output_transcript: false` |
@@ -353,7 +353,7 @@ All fields are optional — sensible defaults for everything.
353
353
  | `isolated` | `false` | Hermetic specialist mode: forces `extensions: false` + `skills: false` + drops `ext:` selectors. Only built-in tools. Distinct from `isolation: worktree` (filesystem) |
354
354
  | `enabled` | `true` | Set to `false` to disable an agent (useful for hiding a default agent per-project) |
355
355
 
356
- For an OpenAI Responses or Codex agent, set `service_tier: fast` to request fast processing; `priority` remains a supported alias. Availability depends on the provider and account. The UI shows the requested tier only when the effective model uses one of those APIs. See [OpenAI fast mode](https://developers.openai.com/api/docs/guides/fast-mode).
356
+ For an OpenAI Responses or Codex agent, set `service_tier: fast` to request fast processing; `priority` remains a supported alias. Availability depends on the provider and account. GitHub Copilot rejects `service_tier` even when its model uses `openai-responses`, so the extension skips the field and tier tag for Copilot. The UI shows the requested tier only when the effective model supports it. See [OpenAI fast mode](https://developers.openai.com/api/docs/guides/fast-mode).
357
357
 
358
358
  The extension forwards both tiers for `gpt-6.1-sol` and `gpt-6-luna` through Pi's OpenAI Responses and Codex transports. Pi 1.0.0's Codex adapter currently estimates a response marked `fast` at the standard rate; forwarding the tier works, but its displayed cost can be understated. This extension reports Pi's cost estimate without recalculating it. OpenAI documents Fast mode as unavailable for these models with EU data residency.
359
359
 
@@ -636,6 +636,8 @@ When background agents complete, they notify the main agent. The **join mode** c
636
636
 
637
637
  Open `/agents → Model routing` to choose a mode and configure a guideline path, models with descriptions and required thinking levels, or a masked TypeSafe API key. The menu shows the active mode and source and saves project settings. For machine-wide defaults, edit `~/.pi/agent/subagents.json`; project overrides go in `.pi/subagents.json`. `PI_CODING_AGENT_DIR` changes the global directory along with Pi's other configuration.
638
638
 
639
+ Use `/agents → Model routing → Runtime status` to inspect the effective mode and its project/global source, routing priority and fallback, guideline path, configured model/thinking profiles, current candidate eligibility, classifier availability and credential presence. No API key is displayed and opening the page does not classify a task. Credential presence does not confirm that the service will accept it. The page also shows the latest ten retained agent decisions, including selected versus actual model/thinking, shadow suggestions, confidence, fallback/error reasons and classifier usage. This is an in-memory view, not a durable log: evicted records and previous sessions are not included. Press `r` to refresh, Up/Down to scroll and Esc to return.
640
+
639
641
  Set one top-level field to control routing; omitting it means `auto`:
640
642
 
641
643
  ```json
@@ -645,12 +647,14 @@ Set one top-level field to control routing; omitting it means `auto`:
645
647
  | `routingMode` | Behavior |
646
648
  |---|---|
647
649
  | `auto` (default) | Use the priority table below |
648
- | `shadow` | Ask Jev and record its suggestion, but keep the model chosen from agent definitions, the guideline or the existing model. Jev requests can incur charges |
649
- | `jev` | Jev chooses first for every fresh task, even with custom agents, a guideline, explicit model parameters or agent-file model pins. Invalid configuration, unavailable credentials, low confidence or errors keep the default-priority choice |
650
+ | `shadow` | Compare custom agents and model profiles through Jev and record its suggestion, but keep the original agent, model and thinking. Jev requests can incur charges |
651
+ | `jev` | Wait for Jev to compare all eligible custom agents and `jev.models` together before starting each fresh task. Invalid configuration, unavailable credentials, low confidence or errors keep the default-priority choice |
650
652
  | `off` | No custom-agent routing selection guidance, guideline injection or Jev requests. Agents remain callable and existing model/thinking settings still apply |
651
653
 
652
654
  Mode changes apply to subsequent fresh launches. The main agent's routing guidance and tool description refresh before its next turn, so switching to `off` removes previously injected routing instructions.
653
655
 
656
+ In `jev` mode, a background `Agent` call also waits for routing before returning. If it queues behind the concurrency limit, it waits through dequeue and selection. Runtime status shows `pending` while Jev is deciding; no child session or worktree starts during that wait. After selection, normal foreground/background execution applies.
657
+
654
658
  The default priority is:
655
659
 
656
660
  | Priority | Configuration | Who chooses |
@@ -662,6 +666,22 @@ The default priority is:
662
666
 
663
667
  **Custom agents:** keep using `~/.pi/agent/agents/<name>.md`, `.pi/agents/<name>.md` or `.agents/agents/<name>.md`. No routing setting is needed. Built-in and disabled agents do not activate priority 1. Agent-file model/thinking pins supply the default choice over `Agent` parameters; a confident Jev choice in `jev` mode can replace both model and thinking level.
664
668
 
669
+ **Let Jev choose the agent:** keep your existing agent files and set `routingMode: "jev"`. The main agent submits the task and waits for the extension's routing decision; it does not choose the specialist itself. The required `subagent_type` supplies a fallback (`general-purpose` when available, an allowed type for nested calls). Jev compares each enabled custom agent's `description` with every `jev.models` description in one request. Agents using the same model remain separate candidates. Project definitions override global definitions with the same name; built-in defaults, disabled agents, unavailable models and agents outside the parent's nested allowlist are excluded.
670
+
671
+ For agent-only routing, this is enough when Pi/environment TypeSafe credentials are available:
672
+
673
+ ```json
674
+ { "routingMode": "jev" }
675
+ ```
676
+
677
+ Or supply a key without duplicating your agent profiles in JSON:
678
+
679
+ ```json
680
+ { "routingMode": "jev", "jev": { "TYPESAFE_API_KEY": "your-typesafe-key" } }
681
+ ```
682
+
683
+ A selected agent runs with its own prompt, tools, extensions, skills, memory, session persistence, service tier, model and thinking. An omitted agent model inherits the parent model; an unavailable model pin excludes that candidate. Its configured turn limit, context and isolation settings also apply; unspecified values retain the invocation's defaults. Foreground/background delivery, handles, workflow schema and ownership/depth remain properties of the original invocation. Tool/nested transcripts use the selected agent's `output_transcript`. A selected `jev.models` entry applies model/thinking and its optional `instruction`, retaining the submitted agent. Shadow applies neither kind of selection. Runtime status lists the combined candidates and the selected or suggested agent name.
684
+
665
685
  **Custom guideline:** write your routing rules in `~/.pi/agent/agents/custom-route.md`, then configure:
666
686
 
667
687
  ```json
@@ -686,7 +706,7 @@ For example, the Markdown can say “Use anthropic/claude-haiku-4-5 for simple e
686
706
  }
687
707
  ```
688
708
 
689
- `TYPESAFE_API_KEY` is optional when Pi already has TypeSafe credentials or the environment variable is set. A literal key must be a nonempty token without whitespace or control characters and applies only to that classifier request; it is saved in the settings file, masked in the menu and omitted from settings events, prompts and routing records. Without a literal key, Pi's native credential availability check runs before classification. Rejected credentials fall back without retrying. The extension does not change environment variables or register a routing provider. `models` accepts 1–254 unique entries with descriptions of 1–4000 characters. Every entry requires `thinkingLevel`: `minimal`, `low`, `medium`, `high`, `xhigh` or `max`. Existing profiles must add this field; a missing or invalid value disables the whole Jev block and uses the default-priority fallback. Jev entries use exact `provider/model-id` spelling; fuzzy names remain available for explicit `Agent` parameters.
709
+ `TYPESAFE_API_KEY` is optional when Pi already has TypeSafe credentials or the environment variable is set. A literal key must be a nonempty token without whitespace or control characters and applies only to that classifier request; it is saved in the settings file, masked in the menu and omitted from settings events, prompts and routing records. Without a literal key, Pi's native credential availability check runs before classification. Rejected credentials fall back without retrying. The extension does not change environment variables or register a routing provider. `models` is optional (defaults to `[]`) and accepts up to 254 unique entries with descriptions of 1–4000 characters. Every model entry requires `thinkingLevel`: `minimal`, `low`, `medium`, `high`, `xhigh` or `max`. Each model may also set `instruction` (up to 16000 characters; blank means omitted). It is appended to the selected child’s system prompt, preserving existing agent instructions. Only `description` is sent as selection criteria; `instruction` is applied only after a successful model-profile selection, never in shadow or fallback. Edit it through `/agents → Model routing → Jev models and descriptions`. Agent files keep their optional `thinking` field. A missing or invalid model-entry level disables the whole Jev block. The combined eligible agent/model pool must not exceed 254 profiles; an oversized pool falls back with a diagnostic instead of silently dropping profiles. Jev entries use exact `provider/model-id` spelling; fuzzy names remain available in agent files and explicit `Agent` parameters.
690
710
 
691
711
  Under `auto`, supplying **either** `model` or `thinking` explicitly skips Jev. An agent-file pin also skips it. In workflows, either `model` or `effort` skips it, and workflow options retain their precedence over agent-file defaults. Inherited models remain eligible for Jev. Under `jev`, these choices supply the fallback model, but do not skip classification. Under `shadow`, they remain the actual choice while Jev records a comparison. A selected Jev profile applies both its model and `thinkingLevel`. Pi clamps the requested thinking level to what that model supports. Shadow and fallback preserve the original model and thinking level.
692
712
 
@@ -731,7 +751,7 @@ Runtime tuning values set via `/agents` → Settings (max concurrency, max foreg
731
751
 
732
752
  **Precedence:** project overrides global on any field present in both. Missing fields fall back to the hardcoded defaults (max concurrency `10`, max foreground concurrency `0` = unlimited, default max turns unlimited, grace turns `5`, nested depth `2`, join mode `smart`, defaults enabled).
733
753
 
734
- Routing settings: `routingMode` (`auto | shadow | jev | off`, default `auto`), `customGuideline` (`string | false`, default unset) and `jev` (`{ TYPESAFE_API_KEY?: string, models: { model, description, thinkingLevel }[] } | false`, default unset). See [Model routing](#model-routing) for examples and the whole-block override rule.
754
+ Routing settings: `routingMode` (`auto | shadow | jev | off`, default `auto`), `customGuideline` (`string | false`, default unset) and `jev` (`{ TYPESAFE_API_KEY?: string, models?: { model, description, thinkingLevel, instruction? }[] } | false`, default unset). See [Model routing](#model-routing) for examples and the whole-block override rule.
735
755
 
736
756
  **Nested depth** (`maxSubagentDepth`, default `2`): the hard ceiling on [nested delegation](#nested-subagents), counted from the main session (main = 0, its subagents = 1). `0` or `1` disables nesting project-wide regardless of any agent's `allowed_subagents`. Read when a subagent session is built, so a change applies to agents started after it.
737
757
 
@@ -768,13 +788,13 @@ The `~` marks it as pi's estimate rather than a billed figure. **A cost is shown
768
788
 
769
789
  Independent of `reportUsage`: this one is what you read, that one is what your session counts. Toggle via `/agents → Settings → Show cost`; applied live.
770
790
 
771
- **Show model** (`showModel`, default `false`): whether the widget's running rows name the model, thinking level, and configured OpenAI service tier when the effective API supports it:
791
+ **Show model** (`showModel`, default `false`): whether the widget's running rows name the model, thinking level, and configured OpenAI service tier when the effective model supports it:
772
792
 
773
793
  ```text
774
794
  ├─ ⠹ Explore inspect code · gpt-5 · thinking: high · service tier: priority · ↻3 · 8.2k token · 4.1s
775
795
  ```
776
796
 
777
- Off by default because the row already carries the description, turns, tool uses, tokens and elapsed time, and every character it gains is one the description loses on a narrow terminal. The other surfaces show the model and thinking level either way: the `Agent` tool result names them beside its tags, and the conversation viewer's `↳` row spells out the canonical `provider/model-id`. They include the configured service tier only when the effective API supports it, so an unsupported provider is not presented as honoring the request.
797
+ Off by default because the row already carries the description, turns, tool uses, tokens and elapsed time, and every character it gains is one the description loses on a narrow terminal. The other surfaces show the model and thinking level either way: the `Agent` tool result names them beside its tags, and the conversation viewer's `↳` row spells out the canonical `provider/model-id`. They include the configured service tier only when the effective model supports it, so an unsupported provider is not presented as honoring the request.
778
798
 
779
799
  Both places report what the run *actually* used, read back from the child session once pi has resolved its defaults and clamped the level to what the model supports — not what the call asked for. Where those differ, the request is kept beside the effective value rather than dropped, whether pi clamped it or an agent file's frontmatter outranked it:
780
800
 
@@ -861,7 +881,7 @@ The four agent-lifecycle events — `subagents:started`, `:completed`, `:failed`
861
881
 
862
882
  `usage` answers the other question — what was billed — and so does include `cacheRead`, because the prefix really is re-read and re-charged on every call. It is a pi `Usage`, the same shape pi puts on `ToolResultEvent` and `AssistantMessage`, so `usage.cost.total` is where a listener already expects the money and anything pi adds to `Usage` arrives without a change here. Neither field derives from the other; `tokens` is a view model, `usage` is the data.
863
883
 
864
- Completed/failed payloads and persisted `subagents:record` entries also carry `routing` (`source`, `code`, `reason`, optional `model`, `thinkingLevel`, `suggestedModel`, `suggestedThinkingLevel`, supplied `description`, `confidence`, `unpriced`, `guidelinePath`, `guidelineHash`) and optional `routingUsage` (classifier-only Pi `Usage`). The guideline hash is SHA-256 of the original file contents. Coding `tokens` excludes classifier tokens; total cost includes the reported classifier cost once. Settings events omit `jev.TYPESAFE_API_KEY`.
884
+ Completed/failed payloads and persisted `subagents:record` entries also carry `routing` (`source`, `code`, `reason`, optional `agent`, `suggestedAgent`, `model`, `thinkingLevel`, `suggestedModel`, `suggestedThinkingLevel`, supplied `description`, `confidence`, `unpriced`, `guidelinePath`, `guidelineHash`) and optional `routingUsage` (classifier-only Pi `Usage`). The guideline hash is SHA-256 of the original file contents. Coding `tokens` excludes classifier tokens; total cost includes the reported classifier cost once. Settings events omit `jev.TYPESAFE_API_KEY`.
865
885
 
866
886
  ## Cross-Extension RPC
867
887
 
@@ -1117,6 +1137,7 @@ src/
1117
1137
  agent-mention.ts # `@` roster (running, resumable, and startable agents) + popup rows
1118
1138
  schedule-menu.ts # /agents → Scheduled jobs submenu
1119
1139
  model-routing-menu.ts # Guideline/model descriptions and masked API-key editor
1140
+ routing-status.ts # Read-only effective routing configuration and retained decisions
1120
1141
  select-item.ts # Collision-safe ctx.ui.select wrapper (numbered rows)
1121
1142
  workflow-card.ts # Inline workflow card (tool result and session entry)
1122
1143
  workflow-dialog.ts # /agents → Workflows two-pane inspector
@@ -172,7 +172,7 @@ interface SpawnOptions {
172
172
  /** Called on streaming text deltas from the assistant response. */
173
173
  onTextDelta?: (delta: string, fullText: string) => void;
174
174
  /** Called when the agent session is created (for accessing session stats). */
175
- onSessionCreated?: (session: AgentSession) => void;
175
+ onSessionCreated?: (session: AgentSession, agentConfig?: AgentConfig) => void;
176
176
  /** Called at the end of each agentic turn with the cumulative count. */
177
177
  onTurnEnd?: (turnCount: number) => void;
178
178
  /** Called once per assistant message_end with that message's usage delta. */
@@ -16,7 +16,7 @@
16
16
  import { randomUUID } from "node:crypto";
17
17
  import { statSync } from "node:fs";
18
18
  import { isAbsolute } from "node:path";
19
- import { resolveDefaultModel, resumeAgent, runAgent } from "./agent-runner.js";
19
+ import { resolveDefaultModel, resumeAgent, runAgent, supportsServiceTier } from "./agent-runner.js";
20
20
  import { getAgentConfig } from "./agent-types.js";
21
21
  import { assignHandle, handleBase } from "./mention.js";
22
22
  import { describeModel } from "./model-resolver.js";
@@ -472,6 +472,7 @@ export class AgentManager {
472
472
  this.runningBackground++;
473
473
  else if (pool === "foreground")
474
474
  this.runningForeground++;
475
+ let routingInstruction;
475
476
  const config = options.agentConfig;
476
477
  const provenance = options.routing;
477
478
  const explicit = provenance
@@ -487,6 +488,8 @@ export class AgentManager {
487
488
  const route = policy.mode === "jev" || policy.mode === "shadow" ||
488
489
  (policy.mode === "auto" && !explicit && !config?.model && !config?.thinking && policy.source === "jev");
489
490
  if (!options.resumeSessionFile && provenance?.entrypoint !== "internal" && route) {
491
+ record.routing.code = "pending";
492
+ record.routing.reason = "Waiting for Jev to compare agent and model profiles";
490
493
  const stop = () => this.abort(id);
491
494
  options.signal?.addEventListener("abort", stop, { once: true });
492
495
  if (options.signal?.aborted)
@@ -501,16 +504,31 @@ export class AgentManager {
501
504
  current.lifetimeUsage.cost = (current.lifetimeUsage.cost ?? 0) + usage.cost.total;
502
505
  current = current.parentAgentId ? this.agents.get(current.parentAgentId) : undefined;
503
506
  }
504
- });
507
+ }, provenance?.allowedAgentTypes);
505
508
  record.routing = { ...routed.decision, guidelinePath: policy.guidelinePath, guidelineHash: policy.guidelineHash };
506
509
  if (routed.model && policy.mode === "shadow") {
507
- record.routing = { ...record.routing, code: "shadow", model: undefined, thinkingLevel: undefined, suggestedModel: routed.decision.model,
510
+ record.routing = { ...record.routing, code: "shadow", model: undefined, agent: undefined, thinkingLevel: undefined, suggestedModel: routed.decision.model,
508
511
  fallbackSource: policy.source, reason: "Jev suggested a model; shadow mode kept the default-priority model" };
509
512
  }
510
513
  else if (routed.model) {
514
+ if (routed.agentConfig) {
515
+ const selected = routed.agentConfig;
516
+ type = selected.name;
517
+ record.type = type;
518
+ options.agentConfig = selected;
519
+ options.maxTurns = selected.maxTurns ?? options.maxTurns;
520
+ options.isolated = selected.isolated ?? options.isolated;
521
+ options.inheritContext = selected.inheritContext ?? options.inheritContext;
522
+ if (selected.isolation !== undefined)
523
+ options.isolation = selected.isolation === "worktree" ? "worktree" : undefined;
524
+ record.invocation = { ...record.invocation, maxTurns: options.maxTurns, isolated: options.isolated,
525
+ inheritContext: options.inheritContext, isolation: options.isolation };
526
+ }
527
+ routingInstruction = routed.instruction;
511
528
  options.model = routed.model;
512
529
  options.thinkingLevel = routed.thinkingLevel;
513
530
  record.invocation = { ...record.invocation, thinking: routed.thinkingLevel, requestedThinking: undefined };
531
+ record.invocation.serviceTier = supportsServiceTier(routed.model) ? options.agentConfig?.serviceTier : undefined;
514
532
  }
515
533
  }
516
534
  catch {
@@ -584,6 +602,7 @@ export class AgentManager {
584
602
  pi,
585
603
  agentId: id,
586
604
  agentConfig: options.agentConfig,
605
+ routingInstruction,
587
606
  model: options.model,
588
607
  maxTurns: options.maxTurns,
589
608
  isolated: options.isolated,
@@ -649,6 +668,9 @@ export class AgentManager {
649
668
  // AND, one line later, being replaced by the effective one.
650
669
  const requested = record.invocation.requestedThinking ?? record.invocation.thinking;
651
670
  Object.assign(record.invocation, describeModel(session.model));
671
+ if (options.agentConfig?.serviceTier) {
672
+ record.invocation.serviceTier = supportsServiceTier(session.model) ? options.agentConfig.serviceTier : undefined;
673
+ }
652
674
  // Guarded for the reason above: a session that reports no level keeps
653
675
  // the request rather than losing it. Overwriting unconditionally would
654
676
  // turn an older or stubbed session into a blank `thinking:` tag, which
@@ -667,7 +689,7 @@ export class AgentManager {
667
689
  }
668
690
  record.pendingSteers = undefined;
669
691
  }
670
- options.onSessionCreated?.(session);
692
+ options.onSessionCreated?.(session, options.agentConfig);
671
693
  },
672
694
  })
673
695
  .then(async ({ responseText, session, aborted, steered, failure, structuredJson, structuredRetried }) => {
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * agent-runner.ts — Core execution engine: creates sessions, runs agents, collects results.
3
3
  */
4
- import type { Model } from "@earendil-works/pi-ai";
4
+ import type { Api, Model } from "@earendil-works/pi-ai";
5
5
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
6
6
  import { type AgentSession, DefaultResourceLoader, type ExtensionAPI } from "@earendil-works/pi-coding-agent";
7
7
  import { type NestedAgentManager } from "./nested-tools.js";
@@ -20,8 +20,8 @@ export declare const SUBAGENT_TOOL_NAMES: {
20
20
  readonly GET_RESULT: "get_subagent_result";
21
21
  readonly STEER: "steer_subagent";
22
22
  };
23
- /** Whether an API accepts the OpenAI `service_tier` request field. */
24
- export declare function isServiceTierApi(api: string | undefined): boolean;
23
+ /** Copilot uses Responses but rejects OpenAI's `service_tier` request field. */
24
+ export declare function supportsServiceTier(model: Pick<Model<Api>, "api" | "provider"> | undefined): boolean;
25
25
  /**
26
26
  * Add a custom agent's service tier to compatible provider requests.
27
27
  *
@@ -157,6 +157,8 @@ export interface ToolActivity {
157
157
  toolName: string;
158
158
  }
159
159
  export interface RunOptions {
160
+ /** Additional instructions from an applied Jev model profile. */
161
+ routingInstruction?: string;
160
162
  /** Snapshot of the selected definition for this branch. */
161
163
  agentConfig?: AgentConfig;
162
164
  /** ExtensionAPI instance — used for pi.exec() instead of execSync. */
@@ -31,9 +31,9 @@ export const SUBAGENT_TOOL_NAMES = {
31
31
  const EXCLUDED_TOOL_NAMES = Object.values(SUBAGENT_TOOL_NAMES);
32
32
  /** APIs whose request payloads support OpenAI service tiers. */
33
33
  const SERVICE_TIER_APIS = new Set(["openai-codex-responses", "openai-responses"]);
34
- /** Whether an API accepts the OpenAI `service_tier` request field. */
35
- export function isServiceTierApi(api) {
36
- return api !== undefined && SERVICE_TIER_APIS.has(api);
34
+ /** Copilot uses Responses but rejects OpenAI's `service_tier` request field. */
35
+ export function supportsServiceTier(model) {
36
+ return model !== undefined && model.provider !== "github-copilot" && SERVICE_TIER_APIS.has(model.api);
37
37
  }
38
38
  function isObjectPayload(payload) {
39
39
  return typeof payload === "object" && payload !== null && !Array.isArray(payload);
@@ -54,7 +54,7 @@ export function installServiceTierPayload(session, serviceTier) {
54
54
  ? await priorOnPayload(payload, requestModel)
55
55
  : undefined;
56
56
  const effectivePayload = replacement === undefined ? payload : replacement;
57
- if (!isServiceTierApi(requestModel.api) || !isObjectPayload(effectivePayload)) {
57
+ if (!supportsServiceTier(requestModel) || !isObjectPayload(effectivePayload)) {
58
58
  return effectivePayload;
59
59
  }
60
60
  return { ...effectivePayload, service_tier: serviceTier };
@@ -515,6 +515,8 @@ export async function runAgent(ctx, type, prompt, options) {
515
515
  throw new Error(`No fallback config available for unknown type "${type}"`);
516
516
  systemPrompt = buildAgentPrompt({ ...fallback, name: type }, effectiveCwd, env, parentSystemPrompt, extras);
517
517
  }
518
+ if (options.routingInstruction)
519
+ systemPrompt += `\n\n<jev_model_instruction>\n${options.routingInstruction}\n</jev_model_instruction>`;
518
520
  // When skills is string[], we've already preloaded them into the prompt.
519
521
  // Still pass noSkills: true since we don't need the skill loader to load them again.
520
522
  const noSkills = skills === false || Array.isArray(skills);
package/dist/index.js CHANGED
@@ -18,7 +18,7 @@ import { abortable } from "./abortable.js";
18
18
  import { hasAgentBadge, renderAgentName } from "./agent-color.js";
19
19
  import { buildNewAgentFile, disableInContent, enableInContent, isEmptyStub, locateAgentFile, personalAgentsDir, projectAgentsDir, serializeAgentFile } from "./agent-file-toggle.js";
20
20
  import { AgentManager, isTopLevelAgent } from "./agent-manager.js";
21
- import { getAgentConversation, getDefaultMaxTurns, getGraceTurns, getRememberAgents, isServiceTierApi, normalizeMaxTurns, resolveEffectiveMaxTurns, SUBAGENT_TOOL_NAMES, setDefaultMaxTurns, setGraceTurns, setRememberAgents, steerAgent } from "./agent-runner.js";
21
+ import { getAgentConversation, getDefaultMaxTurns, getGraceTurns, getRememberAgents, normalizeMaxTurns, resolveEffectiveMaxTurns, SUBAGENT_TOOL_NAMES, setDefaultMaxTurns, setGraceTurns, setRememberAgents, steerAgent, supportsServiceTier } from "./agent-runner.js";
22
22
  import { BUILTIN_TOOL_NAMES, getAgentConfig, getAllTypes, getAvailableTypes, getConfig, getFallbackSubagent, isDefaultsDisabled, NO_FALLBACK, registerAgents, resolveSpawnType, resolveType, setDefaultsDisabled, setFallbackSubagent } from "./agent-types.js";
23
23
  import { inChildSessionContext } from "./child-context.js";
24
24
  import { registerRpcHandlers } from "./cross-extension-rpc.js";
@@ -1711,8 +1711,10 @@ Terse command-style prompts produce shallow, generic work.
1711
1711
  // downstream consumer keys off record.outputFile being set, so no spawn
1712
1712
  // path can re-enable the transcript by accident.
1713
1713
  const outputTranscript = customConfig?.outputTranscript ?? getOutputTranscriptDefault();
1714
- const attachTranscript = (rec, agentId) => {
1715
- if (!rec || !outputTranscript)
1714
+ const attachTranscript = (rec, agentId, config = customConfig) => {
1715
+ if (!rec || rec.outputFile || (!rec.session && (rec.status === "queued" || routingPolicy.mode === "jev" || rec.routing?.mode === "jev")))
1716
+ return;
1717
+ if (!(config?.outputTranscript ?? getOutputTranscriptDefault()))
1716
1718
  return;
1717
1719
  rec.outputFile = createOutputFilePath(ctx.cwd, agentId, ctx.sessionManager.getSessionId());
1718
1720
  writeInitialEntry(rec.outputFile, agentId, params.prompt, ctx.cwd);
@@ -1741,7 +1743,7 @@ Terse command-style prompts produce shallow, generic work.
1741
1743
  modelName,
1742
1744
  modelId,
1743
1745
  thinking,
1744
- serviceTier: model && isServiceTierApi(model.api) ? customConfig?.serviceTier : undefined,
1746
+ serviceTier: supportsServiceTier(model) ? customConfig?.serviceTier : undefined,
1745
1747
  // Only set where the agent file outranked the caller, so the surfaces can
1746
1748
  // disclose a parameter that was accepted but could not take effect (#182).
1747
1749
  requestedThinking: resolvedConfig.overridden?.thinking,
@@ -1899,9 +1901,11 @@ Terse command-style prompts produce shallow, generic work.
1899
1901
  // rather than closing over a value that doesn't exist yet.
1900
1902
  let id;
1901
1903
  const origBgOnSession = bgCallbacks.onSessionCreated;
1902
- bgCallbacks.onSessionCreated = (session) => {
1904
+ bgCallbacks.onSessionCreated = (session, config) => {
1903
1905
  origBgOnSession(session);
1904
1906
  const rec = manager.getRecord(id);
1907
+ attachTranscript(rec, id, config);
1908
+ bgState.maxTurns = normalizeMaxTurns(rec?.invocation?.maxTurns ?? getDefaultMaxTurns());
1905
1909
  if (rec?.outputFile) {
1906
1910
  rec.outputCleanup = streamToOutputFile(session, rec.outputFile, id, ctx.cwd);
1907
1911
  }
@@ -1938,6 +1942,8 @@ Terse command-style prompts produce shallow, generic work.
1938
1942
  // copy is an awaited git call. Wait for it here, after the synchronous
1939
1943
  // wiring above, so a strict-isolation failure still fails THIS tool
1940
1944
  // call instead of being reported as a subagent that ran (#179).
1945
+ if (routingPolicy.mode === "jev" && record?.startGate)
1946
+ await record.startGate;
1941
1947
  await manager.awaitStartup(id);
1942
1948
  if (joinMode == null || joinMode === 'async') {
1943
1949
  // Foreground/no join mode or explicit async — not part of any batch
@@ -1959,14 +1965,14 @@ Terse command-style prompts produce shallow, generic work.
1959
1965
  // Emit created event
1960
1966
  pi.events.emit("subagents:created", {
1961
1967
  id,
1962
- type: subagentType,
1968
+ type: record?.type ?? subagentType,
1963
1969
  description: params.description,
1964
1970
  isBackground: true,
1965
1971
  });
1966
1972
  const isQueued = record?.status === "queued";
1967
1973
  return textResult(`${fallbackNote}Agent ${isQueued ? "queued" : "started"} in background.\n` +
1968
1974
  `Agent ID: ${id}\n` +
1969
- `Type: ${displayName}\n` +
1975
+ `Type: ${record ? getDisplayName(record.type) : displayName}\n` +
1970
1976
  `Description: ${params.description}\n` +
1971
1977
  (record?.outputFile ? `Output file: ${record.outputFile}\n` : "") +
1972
1978
  (isQueued ? `Position: queued (max ${manager.getMaxConcurrent()} concurrent)\n` : "") +
@@ -2016,7 +2022,7 @@ Terse command-style prompts produce shallow, generic work.
2016
2022
  // The output file path is set synchronously after spawn (below),
2017
2023
  // before onSessionCreated fires — same pattern as background agents.
2018
2024
  const origOnSession = fgCallbacks.onSessionCreated;
2019
- fgCallbacks.onSessionCreated = (session) => {
2025
+ fgCallbacks.onSessionCreated = (session, config) => {
2020
2026
  origOnSession(session);
2021
2027
  // It really started — stop reporting it as queued, and repaint now
2022
2028
  // rather than leaving the stale line up for the next spinner tick.
@@ -2038,6 +2044,8 @@ Terse command-style prompts produce shallow, generic work.
2038
2044
  // Stream conversation to output file (foreground agent logging)
2039
2045
  if (fgId) {
2040
2046
  const rec = manager.getRecord(fgId);
2047
+ attachTranscript(rec, fgId, config);
2048
+ fgState.maxTurns = normalizeMaxTurns(rec?.invocation?.maxTurns ?? getDefaultMaxTurns());
2041
2049
  if (rec?.outputFile) {
2042
2050
  rec.outputCleanup = streamToOutputFile(session, rec.outputFile, fgId, ctx.cwd);
2043
2051
  }
@@ -2752,7 +2760,7 @@ Terse command-style prompts produce shallow, generic work.
2752
2760
  await showAgentsMenu(ctx);
2753
2761
  }
2754
2762
  else if (choice === "Model routing") {
2755
- const patch = await showRoutingMenu(ctx);
2763
+ const patch = await showRoutingMenu(ctx, () => manager.listAgents());
2756
2764
  if (patch) {
2757
2765
  const toast = saveAndEmitChanged({ ...snapshotSettings(), ...patch }, "Model routing settings updated", (event, payload) => pi.events.emit(event, payload), ctx.cwd);
2758
2766
  ctx.ui.notify(toast.message, toast.level);
@@ -3066,7 +3074,7 @@ description: <one-line description shown in UI>
3066
3074
  color: <optional agent name badge color: red, blue, green, yellow, purple, orange, pink, cyan, an Agency Agents alias, or quoted "#RRGGBB">
3067
3075
  tools: <comma-separated built-in tools: read, bash, edit, write, grep, find, ls. Use "none" for no tools. Omit for all tools>
3068
3076
  model: <optional model as "provider/modelId", e.g. "anthropic/claude-haiku-4-5". Omit to inherit parent model>
3069
- service_tier: <optional OpenAI Responses/Codex processing tier: auto, default, flex, priority, or scale. Use only with those APIs; omit for other APIs or the provider default>
3077
+ service_tier: <optional OpenAI Responses/Codex processing tier: auto, default, flex, priority, or scale. Omit for GitHub Copilot, other APIs, or the provider default>
3070
3078
  thinking: <optional thinking level: ${THINKING_LEVELS.join(", ")}. Omit to inherit>
3071
3079
  max_turns: <optional max agentic turns. 0 or omit for unlimited (default)>
3072
3080
  prompt_mode: <"replace" (body IS the full system prompt) or "append" (body is appended to default prompt). Default: replace>
@@ -3097,7 +3105,7 @@ Guidelines for choosing settings:
3097
3105
  - Use prompt_mode: replace for fully custom agents with their own personality/instructions
3098
3106
  - Set inherit_context: true if the agent needs to know what was discussed in the parent conversation
3099
3107
  - Set isolated: true if the agent should NOT have access to MCP servers or other extensions
3100
- - Set service_tier: priority only when the model uses an OpenAI Responses/Codex API; omit it for other providers
3108
+ - Set service_tier: priority only when the model uses an OpenAI Responses/Codex API; omit it for GitHub Copilot and other providers that do not support it
3101
3109
  - Set output_transcript: false to skip writing this agent's transcript; this alone doesn't keep the run off disk (persist_session, isolation: worktree commits, and memory still write) — set those too if that's the goal
3102
3110
  - Only include frontmatter fields that differ from defaults — omit fields where the default is fine
3103
3111
 
@@ -3397,7 +3405,7 @@ Write the file using the write tool. Only write the file, nothing else.`;
3397
3405
  {
3398
3406
  id: "showModel",
3399
3407
  label: "Show model",
3400
- description: "Name the model driving each agent, its thinking level, and configured OpenAI service tier when the effective API supports it, on the widget's running rows. The Agent tool result and the conversation viewer show these details when applicable — this adds them to the widget, where the row is already dense.",
3408
+ description: "Name the model driving each agent, its thinking level, and configured OpenAI service tier when the effective model supports it, on the widget's running rows. The Agent tool result and the conversation viewer show these details when applicable — this adds them to the widget, where the row is already dense.",
3401
3409
  currentValue: isShowModelEnabled() ? "on" : "off",
3402
3410
  values: ["on", "off"],
3403
3411
  },
@@ -11,6 +11,7 @@ export interface RoutingPolicy {
11
11
  name: string;
12
12
  description: string;
13
13
  }[];
14
+ agentProfiles?: AgentConfig[];
14
15
  guideline?: string;
15
16
  guidelinePath?: string;
16
17
  guidelineHash?: string;
@@ -21,6 +22,8 @@ export interface RoutingPolicy {
21
22
  export interface RoutingInput {
22
23
  /** Private launch snapshot; never accepted from external callers. */
23
24
  policy?: RoutingPolicy;
25
+ /** Parent permission boundary, supplied only by the nested tool. */
26
+ allowedAgentTypes?: string[];
24
27
  modelExplicit: boolean;
25
28
  thinkingExplicit: boolean;
26
29
  entrypoint: "agent" | "nested" | "workflow" | "schedule" | "internal";
@@ -30,9 +33,11 @@ export interface RoutingDecision {
30
33
  source: RoutingSource;
31
34
  fallbackSource?: RoutingSource;
32
35
  reason: string;
33
- code: "baseline" | "explicit" | "off" | "shadow" | "config_unavailable" | "credentials_unavailable" | "guideline_unavailable" | "no_candidates" | "classifier_unavailable" | "cancelled" | "timeout" | "invalid_answer" | "abstained" | "unavailable_choice" | "selected" | "classifier_error";
36
+ code: "baseline" | "pending" | "explicit" | "off" | "shadow" | "config_unavailable" | "credentials_unavailable" | "guideline_unavailable" | "no_candidates" | "classifier_unavailable" | "cancelled" | "timeout" | "invalid_answer" | "abstained" | "unavailable_choice" | "selected" | "classifier_error";
34
37
  model?: string;
35
38
  suggestedModel?: string;
39
+ agent?: string;
40
+ suggestedAgent?: string;
36
41
  thinkingLevel?: ThinkingLevel;
37
42
  suggestedThinkingLevel?: ThinkingLevel;
38
43
  description?: string;
@@ -44,14 +49,26 @@ export interface RoutingDecision {
44
49
  export declare function loadRoutingPolicy(cwd: string, loadedAgents?: Map<string, AgentConfig>): RoutingPolicy;
45
50
  /** Added in every description mode and refreshed before each main-agent turn. */
46
51
  export declare function routingGuidance(policy: RoutingPolicy): string;
52
+ export declare function eligibleModels(ctx: ExtensionContext): Map<string, Model<Api>>;
53
+ export interface RoutingCandidate {
54
+ instruction?: string;
55
+ model: string;
56
+ description: string;
57
+ thinkingLevel?: ThinkingLevel;
58
+ agentConfig?: AgentConfig;
59
+ }
60
+ /** Keep profiles distinct even when multiple specialists use the same model. */
61
+ export declare function routingCandidates(ctx: ExtensionContext, policy: RoutingPolicy, baseline: Model<Api> | undefined, allowedAgentTypes?: readonly string[]): RoutingCandidate[];
47
62
  /** Bounded classifier pool, independent of agent concurrency and nesting. */
48
63
  export declare class ModelRouter {
49
64
  private active;
50
65
  private waiters;
51
66
  private acquire;
52
- choose(ctx: ExtensionContext, policy: RoutingPolicy, prompt: string, description: string, baseline: Model<Api> | undefined, signal: AbortSignal, onUsage: (usage: Usage) => void): Promise<{
67
+ choose(ctx: ExtensionContext, policy: RoutingPolicy, prompt: string, description: string, baseline: Model<Api> | undefined, signal: AbortSignal, onUsage: (usage: Usage) => void, allowedAgentTypes?: readonly string[]): Promise<{
53
68
  model?: Model<Api>;
54
69
  thinkingLevel?: ThinkingLevel;
70
+ instruction?: string;
71
+ agentConfig?: AgentConfig;
55
72
  decision: RoutingDecision;
56
73
  }>;
57
74
  }