@signalridge/pi-subagents 1.10.2 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/ask-tools.ts CHANGED
@@ -4,7 +4,9 @@
4
4
  * `tools:` and `disallowed_tools:` are static: a tool is available for the whole
5
5
  * run or never. That forces a bad choice for the tools that are usually fine and
6
6
  * occasionally not — grant `bash` and hope, or withhold it and cripple the
7
- * agent. `ask_tools:` names the tools whose every call needs a person to agree.
7
+ * agent. `ask_tools:` names tools whose first call needs a person's approval
8
+ * for the current in-memory child, including resumed turns. A reopened child
9
+ * gets a new gate and must ask again.
8
10
  *
9
11
  * The approver is the HUMAN, deliberately. Upstream projects put an LLM in this
10
12
  * seat because their subagents run headless in another process and cannot reach
@@ -34,7 +36,7 @@ export interface AskGateContext {
34
36
  }
35
37
 
36
38
  /**
37
- * Build the per-call approval gate, or `undefined` when nothing needs asking.
39
+ * Build the session-scoped approval gate, or `undefined` when nothing needs asking.
38
40
  *
39
41
  * Returns a function that resolves to a block decision when the call must not
40
42
  * proceed, and `undefined` when it may.
@@ -45,8 +47,9 @@ export function createAskGate(
45
47
  const gated = new Set(context.askTools.map((name) => name.trim().toLowerCase()).filter(Boolean));
46
48
  if (gated.size === 0) return undefined;
47
49
 
48
- /** Tools the user has approved for the rest of this run. */
50
+ /** Tools approved for this in-memory child; never persisted on disk. */
49
51
  const approvedForRun = new Set<string>();
52
+ const pendingApprovals = new Map<string, Promise<{ approved: boolean; failed: boolean }>>();
50
53
 
51
54
  return async (toolName, input) => {
52
55
  const key = toolName.toLowerCase();
@@ -66,28 +69,35 @@ export function createAskGate(
66
69
  };
67
70
  }
68
71
 
69
- let approved: boolean;
70
- try {
71
- approved = await context.confirm(
72
- `${context.agentLabel} wants to use ${toolName}`,
73
- `${describeInput(input)}\n\nAllow this call?`,
74
- );
75
- } catch {
76
- // A prompt that cannot be shown is not an approval.
77
- return {
78
- block: true,
79
- reason: `Tool "${toolName}" requires approval (ask_tools) and the prompt could not be shown.`,
80
- };
72
+ // Parallel first calls share one decision for this tool. Otherwise the
73
+ // second dialog can be declined after the first one granted the same
74
+ // session-wide permission, making the two answers contradictory.
75
+ let pending = pendingApprovals.get(key);
76
+ if (!pending) {
77
+ const confirm = context.confirm;
78
+ pending = Promise.resolve()
79
+ .then(() => confirm(
80
+ `${context.agentLabel} wants to use ${toolName}`,
81
+ `Current call: ${describeInput(input)}\n\nAllow ${toolName} for this in-memory agent session, including resumed turns? Later calls in this session will not ask again. Reopening the child from disk or restarting Pi will ask again.`,
82
+ ))
83
+ .then((approved) => ({ approved, failed: false }), () => ({ approved: false, failed: true }));
84
+ pendingApprovals.set(key, pending);
85
+ const decision = pending;
86
+ void decision.then(() => {
87
+ if (pendingApprovals.get(key) === decision) pendingApprovals.delete(key);
88
+ });
81
89
  }
82
90
 
91
+ const { approved, failed } = await pending;
83
92
  if (!approved) {
84
93
  return {
85
94
  block: true,
86
- reason: `The user declined the "${toolName}" call. Do not retry it; continue without that tool or explain what you cannot do.`,
95
+ reason: failed
96
+ ? `Tool "${toolName}" requires approval (ask_tools) and the prompt could not be shown.`
97
+ : `The user declined the "${toolName}" call. Do not retry it; continue without that tool or explain what you cannot do.`,
87
98
  };
88
99
  }
89
- // Approved for the remainder of the run rather than for this call alone:
90
- // re-asking on every call of a tool the user just allowed trains them to
100
+ // Re-asking on every call of a tool the user just allowed trains them to
91
101
  // approve without reading, which is how an approval prompt stops working.
92
102
  approvedForRun.add(key);
93
103
  return undefined;
@@ -97,9 +107,9 @@ export function createAskGate(
97
107
  /**
98
108
  * One-line, bounded, inert rendering of a tool call's arguments.
99
109
  *
100
- * The user is being asked to approve a specific call, so the arguments are the
101
- * whole point — but they are model-authored and about to be drawn into a
102
- * terminal, so they are sanitized before truncation, never after.
110
+ * The first call's arguments show what prompted the session-wide request.
111
+ * They are model-authored and about to be drawn into a terminal, so they are
112
+ * sanitized before truncation, never after.
103
113
  */
104
114
  function describeInput(input: unknown): string {
105
115
  if (input === undefined || input === null) return "(no arguments)";
@@ -0,0 +1,54 @@
1
+ import type { AgentSession } from "@earendil-works/pi-coding-agent";
2
+
3
+ export const INHERIT_CONTEXT_UNAVAILABLE =
4
+ "Raw `inherit_context: true` is unavailable on this Pi host: the parent context hooks run after the session projection, " +
5
+ "and a child request cannot pin the same physical provider endpoint. Start a new Agent with `inherit_context: false` " +
6
+ "and put an explicitly sanitized summary in its task prompt, or persist a context_edit before starting it.";
7
+
8
+ export const INHERITED_SESSION_UNSAFE =
9
+ "This child session contains previously inherited parent history and cannot be resumed safely. " +
10
+ "Start a new Agent session with `inherit_context: false` and an explicitly sanitized summary in its task prompt.";
11
+
12
+ export function assertNoRawInheritance(inheritContext: boolean | undefined): void {
13
+ if (inheritContext === true) throw new Error(INHERIT_CONTEXT_UNAVAILABLE);
14
+ }
15
+
16
+ const PARENT_CONTEXT_MARKER = "# Parent Conversation Context\n";
17
+
18
+ function isRecord(value: unknown): value is Record<string, unknown> {
19
+ return typeof value === "object" && value !== null && !Array.isArray(value);
20
+ }
21
+
22
+ function hasInheritedPrompt(value: unknown): boolean {
23
+ if (!isRecord(value)) return false;
24
+ if (value.inheritContext === true || value.inherit_context === true) return true;
25
+ if (value.role === "user") {
26
+ const content = value.content;
27
+ if (typeof content === "string") return content.includes(PARENT_CONTEXT_MARKER);
28
+ if (Array.isArray(content)) return content.some((block) =>
29
+ isRecord(block) && block.type === "text" &&
30
+ typeof block.text === "string" && block.text.includes(PARENT_CONTEXT_MARKER));
31
+ }
32
+ return false;
33
+ }
34
+
35
+ /** Guard both the live view and the un-compacted journal before any new prompt. */
36
+ export function assertSafeChildSession(
37
+ session: Pick<AgentSession, "messages" | "sessionManager">,
38
+ inheritedMetadata = false,
39
+ ): void {
40
+ if (inheritedMetadata) throw new Error(INHERITED_SESSION_UNSAFE);
41
+ const manager = session.sessionManager;
42
+ const entries = manager?.getEntries?.() ?? manager?.getBranch?.() ?? [];
43
+ if (entries.some((entry: unknown) => isRecord(entry) &&
44
+ (hasInheritedPrompt(entry) || hasInheritedPrompt(entry.message) ||
45
+ hasInheritedPrompt(entry.data) ||
46
+ (isRecord(entry.data) && hasInheritedPrompt(entry.data.invocation)) ||
47
+ (isRecord(entry.message) && hasInheritedPrompt(entry.message.metadata))))) {
48
+ throw new Error(INHERITED_SESSION_UNSAFE);
49
+ }
50
+ if (session.messages?.some((message: unknown) => hasInheritedPrompt(message) ||
51
+ (isRecord(message) && hasInheritedPrompt(message.metadata)))) {
52
+ throw new Error(INHERITED_SESSION_UNSAFE);
53
+ }
54
+ }
package/src/context.ts CHANGED
@@ -4,6 +4,23 @@
4
4
 
5
5
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
6
6
 
7
+ type ProjectedMessage = ReturnType<ExtensionContext["sessionManager"]["buildSessionProjection"]>["messages"][number];
8
+
9
+ const HEADER = `# Parent Conversation Context
10
+ The following is the conversation history from the parent session that spawned you.
11
+ Use this context to understand what has been discussed and decided so far.
12
+
13
+ `;
14
+ const FOOTER = `
15
+
16
+ ---
17
+ # Your Task (below)
18
+ `;
19
+ const MAX_CONTEXT_CHARS = 32_000;
20
+ const MAX_MESSAGE_CHARS = 4_000;
21
+ const MAX_TOOL_CHARS = 2_000;
22
+ const CLIPPED = "...[truncated]...";
23
+
7
24
  /** Extract text from a message content block array. */
8
25
  export function extractText(content: unknown[]): string {
9
26
  return content
@@ -12,47 +29,115 @@ export function extractText(content: unknown[]): string {
12
29
  .join("\n");
13
30
  }
14
31
 
32
+ /** Clip without first joining or copying arbitrarily large tool outputs. */
33
+ function clipText(text: string, limit: number, preserveTail = false): string {
34
+ if (limit <= 0) return "";
35
+ if (text.length <= limit) return text;
36
+ if (limit <= CLIPPED.length) return text.slice(-limit);
37
+ const available = limit - CLIPPED.length;
38
+ if (preserveTail) {
39
+ const head = Math.min(300, Math.floor(available / 3));
40
+ return `${text.slice(0, head)}${CLIPPED}${text.slice(-(available - head))}`;
41
+ }
42
+ return `${text.slice(0, available)}${CLIPPED}`;
43
+ }
44
+
45
+ function boundedContent(content: string | unknown[], limit: number, preserveTail = false): string {
46
+ if (typeof content === "string") return clipText(content, limit, preserveTail).trim();
47
+ // Inspect each block only up to the remaining budget, never join a full tool
48
+ // result just to discard it. Reverse order retains failure/truncation tails.
49
+ const selected: string[] = [];
50
+ let remaining = limit;
51
+ for (let i = 0; i < content.length && remaining > 0; i++) {
52
+ const block = content[preserveTail ? content.length - 1 - i : i];
53
+ if (typeof block !== "object" || block === null || !("type" in block) || block.type !== "text" ||
54
+ !("text" in block) || typeof block.text !== "string") continue;
55
+ const text = clipText(block.text, remaining, preserveTail);
56
+ selected.push(text);
57
+ remaining -= text.length + 1;
58
+ }
59
+ const joined = (preserveTail ? selected.reverse() : selected).join("\n");
60
+ return clipText(joined, limit, preserveTail).trim();
61
+ }
62
+
63
+ /** Include shell commands shown to the model, with their outcome even on huge output. */
64
+ function boundedBash(msg: Extract<ProjectedMessage, { role: "bashExecution" }>): string {
65
+ const command = clipText(msg.command, 200);
66
+ const outcome = msg.cancelled ? "\n\n(command cancelled)"
67
+ : msg.exitCode !== null && msg.exitCode !== undefined && msg.exitCode !== 0
68
+ ? `\n\nCommand exited with code ${msg.exitCode}` : "";
69
+ const truncation = msg.truncated && msg.fullOutputPath
70
+ ? `\n\n[Output truncated. Full output: ${clipText(msg.fullOutputPath, 200)}]` : "";
71
+ const prefix = `Ran \`${command}\`\n`;
72
+ const suffix = `${outcome}${truncation}`;
73
+ const remaining = Math.max(0, MAX_TOOL_CHARS - prefix.length - suffix.length - 8);
74
+ const output = msg.output
75
+ ? `\`\`\`\n${clipText(msg.output, remaining, msg.cancelled || msg.truncated || msg.exitCode !== 0)}\n\`\`\``
76
+ : "(no output)";
77
+ return `${prefix}${output}${suffix}`;
78
+ }
79
+
15
80
  /**
16
- * Build a text representation of the parent conversation context.
17
- * Used when inherit_context is true to give the subagent visibility
18
- * into what has been discussed/done so far.
81
+ * Build a text representation of the parent conversation context. The child
82
+ * still needs room for its own prompt, tools and answer, so inherited context
83
+ * is at most one character per four context tokens (also capped for large
84
+ * models). This is conservative even for one-character-per-token content.
19
85
  */
20
- export function buildParentContext(ctx: ExtensionContext): string {
21
- const entries = ctx.sessionManager.getBranch();
22
- if (!entries || entries.length === 0) return "";
23
-
24
- const parts: string[] = [];
25
-
26
- for (const entry of entries) {
27
- if (entry.type === "message") {
28
- const msg = entry.message;
29
- if (msg.role === "user") {
30
- const text = typeof msg.content === "string"
31
- ? msg.content
32
- : extractText(msg.content);
33
- if (text.trim()) parts.push(`[User]: ${text.trim()}`);
34
- } else if (msg.role === "assistant") {
35
- const text = extractText(msg.content);
36
- if (text.trim()) parts.push(`[Assistant]: ${text.trim()}`);
86
+ export function buildParentContext(ctx: ExtensionContext, childContextWindow = ctx.model?.contextWindow): string {
87
+ const window = childContextWindow && Number.isFinite(childContextWindow) && childContextWindow > 0
88
+ ? childContextWindow : 128_000;
89
+ const budget = Math.max(0, Math.min(MAX_CONTEXT_CHARS, Math.floor(window / 4)) - HEADER.length - FOOTER.length);
90
+ if (!budget) return "";
91
+ const entries: { position: number; order: number; text: string }[] = [];
92
+ let used = 0;
93
+ const add = (position: number, order: number, label: string, content: string | unknown[], limit: number, tail = false): void => {
94
+ const room = Math.min(limit, budget - used - label.length - (entries.length ? 2 : 0));
95
+ if (room <= 0) return;
96
+ const text = boundedContent(content, room, tail);
97
+ if (!text) return;
98
+ entries.push({ position, order, text: `${label}${text}` });
99
+ used += label.length + text.length + (entries.length > 1 ? 2 : 0);
100
+ };
101
+
102
+ const sessionManager = ctx.sessionManager;
103
+ // Older Pi has no provenance-preserving projection, but buildSessionContext
104
+ // still applies compaction and branch selection. Its messages are precisely
105
+ // what the host sends to the model; getBranch() includes summarized raw turns.
106
+ const legacyManager = sessionManager as typeof sessionManager & {
107
+ buildSessionContext(): { messages: ProjectedMessage[] };
108
+ };
109
+ const projected: ProjectedMessage[][] = typeof sessionManager.buildSessionProjection === "function"
110
+ ? sessionManager.buildSessionProjection().entries.map((entry) => entry.messages)
111
+ : legacyManager.buildSessionContext().messages.map((message) => [message]);
112
+ // Reserve up to a quarter for recent summaries before spending the rest on
113
+ // recent turns. A long tool result must not evict the only compaction summary.
114
+ const summaryLimit = Math.floor(budget / 4);
115
+ for (let i = projected.length - 1; i >= 0 && used < summaryLimit; i--) {
116
+ for (let j = projected[i].length - 1; j >= 0 && used < summaryLimit; j--) {
117
+ const msg = projected[i][j];
118
+ if (msg.role === "compactionSummary" || msg.role === "branchSummary") {
119
+ add(i, j, "[Summary]: ", msg.summary, summaryLimit - used, true);
37
120
  }
38
- // Skip toolResult messages — too verbose for context
39
- } else if (entry.type === "compaction") {
40
- // Include compaction summaries — they're already condensed
41
- if (entry.summary) {
42
- parts.push(`[Summary]: ${entry.summary}`);
121
+ }
122
+ }
123
+ for (let i = projected.length - 1; i >= 0 && used < budget; i--) {
124
+ for (let j = projected[i].length - 1; j >= 0 && used < budget; j--) {
125
+ const msg = projected[i][j];
126
+ if (msg.role === "bashExecution") {
127
+ if (!msg.excludeFromContext) add(i, j, "[Bash Execution]: ", boundedBash(msg), MAX_TOOL_CHARS);
128
+ } else if (msg.role === "user" || msg.role === "assistant" || msg.role === "custom" || msg.role === "toolResult") {
129
+ // Provider transforms omit failed assistant attempts; a pure utility
130
+ // should not render them as successful model-visible context.
131
+ if (msg.role === "assistant" && (msg.stopReason === "error" || msg.stopReason === "aborted")) continue;
132
+ const label = msg.role === "toolResult" ? `[Tool Result (${msg.toolName})]: `
133
+ : msg.role === "custom" ? "[Context]: " : msg.role === "user" ? "[User]: " : "[Assistant]: ";
134
+ add(i, j, label, msg.content, msg.role === "toolResult" ? MAX_TOOL_CHARS : MAX_MESSAGE_CHARS, msg.role === "toolResult");
43
135
  }
136
+ // System messages are supplied by the child's own prompt policy.
44
137
  }
45
138
  }
46
139
 
47
- if (parts.length === 0) return "";
48
-
49
- return `# Parent Conversation Context
50
- The following is the conversation history from the parent session that spawned you.
51
- Use this context to understand what has been discussed and decided so far.
52
-
53
- ${parts.join("\n\n")}
54
-
55
- ---
56
- # Your Task (below)
57
- `;
140
+ if (entries.length === 0) return "";
141
+ entries.sort((a, b) => a.position - b.position || a.order - b.order);
142
+ return `${HEADER}${entries.map((entry) => entry.text).join("\n\n")}${FOOTER}`;
58
143
  }
@@ -225,6 +225,13 @@ function loadFromDir(
225
225
 
226
226
  const { builtinToolNames, extSelectors } = parseToolsField(fm.tools);
227
227
  warnLegacyModelFields(fm, path, warn);
228
+ if (fm.inherit_context === true) {
229
+ warn(
230
+ `Agent file ${path} sets inherit_context: true, which is unavailable on current Pi hosts; ` +
231
+ "spawns will be refused. Remove it and pass an explicitly sanitized summary in the Agent task.",
232
+ `inherit-context:${warningIdentity(path)}`,
233
+ );
234
+ }
228
235
 
229
236
  agents.set(name, {
230
237
  name,
@@ -9,26 +9,21 @@
9
9
  * project file has `enabledModels` set, it wholly replaces global's
10
10
  * (array fields are replaced, not concatenated).
11
11
  *
12
- * **Limited subset of upstream's resolveModelScope.** We support exact
13
- * `provider/modelId` matching only. Upstream (pi-coding-agent's
14
- * `core/model-resolver.ts`) additionally supports glob patterns
15
- * (`*sonnet*`, `anthropic/*`), bare model IDs without provider, and
16
- * thinking-level suffixes (`provider/*:high`). Those forms are silently
17
- * ignored here.
18
- *
19
- * In practice, pi's `/scoped-models` picker writes exact `provider/modelId`
20
- * entries, so the limitation is invisible for users who configure scope
21
- * through pi's UI. Hand-edited settings using globs or bare IDs will
22
- * produce an empty allowed set (scope check becomes a no-op).
12
+ * Resolve exact `provider/modelId` entries, unambiguous bare IDs, and Pi's
13
+ * glob patterns (`*sonnet*`, `anthropic/*`) with optional thinking suffixes.
14
+ * The suffix selects a thinking level in Pi; this guard checks only model
15
+ * membership. A configured pattern with no available match must not disable
16
+ * the opt-in scope check.
23
17
  *
24
18
  * Example:
25
19
  * enabledModels = ["anthropic/claude-sonnet-4-6", "anthropic/claude-opus-4-6"]
26
20
  * → resolves to { "anthropic/claude-sonnet-4-6", "anthropic/claude-opus-4-6" }
27
21
  */
28
22
 
29
- import { existsSync, readFileSync, statSync } from "node:fs";
23
+ import { existsSync, readFileSync } from "node:fs";
30
24
  import { join } from "node:path";
31
25
  import { getAgentDir } from "@earendil-works/pi-coding-agent";
26
+ import { minimatch } from "minimatch";
32
27
  import type { ModelEntry } from "./model-resolver.js";
33
28
 
34
29
  /** Minimal registry shape — only the methods resolveEnabledModels actually calls. */
@@ -50,7 +45,12 @@ function readField(path: string): string[] | undefined {
50
45
  if (!existsSync(path)) return undefined;
51
46
  try {
52
47
  const raw = JSON.parse(readFileSync(path, "utf-8"));
53
- if (Array.isArray(raw?.enabledModels)) return raw.enabledModels as string[];
48
+ if (Array.isArray(raw?.enabledModels)) {
49
+ // Keep the configured list present even if every item is malformed: an
50
+ // invalid project allowlist must not fall back to a broader global list
51
+ // or silently disable the opt-in scope check.
52
+ return raw.enabledModels.map((pattern: unknown) => typeof pattern === "string" ? pattern : "");
53
+ }
54
54
  } catch {
55
55
  /* corrupt file — silent */
56
56
  }
@@ -71,54 +71,25 @@ export function readEnabledModels(cwd: string): string[] | undefined {
71
71
  /**
72
72
  * Resolve enabledModels patterns → Set<"provider/modelId"> (lowercase keys).
73
73
  *
74
- * Only exact `provider/modelId` patterns are matched (case-insensitive).
75
- * Patterns without a slash, with glob characters, or with a `:thinking`
76
- * suffix are silently dropped. See module-level docstring for rationale.
74
+ * Matches exact references, unambiguous bare IDs, and case-insensitive glob
75
+ * patterns against the full `provider/modelId` or bare ID. A recognized
76
+ * `:thinking` suffix is ignored for this model-only policy.
77
77
  *
78
- * Cache: keyed on JSON.stringify(patterns) + mtime/size of *both*
79
- * project and global settings.json files. Re-resolves when either file
80
- * changes or the patterns argument differs.
78
+ * Resolves against the current registry on every call. Availability can
79
+ * change without a settings-file edit (or even a new registry instance).
80
+ * The optional cwd is retained for callers that pass it alongside patterns
81
+ * read from that project's settings.
81
82
  *
82
- * Returns undefined when no patterns are provided or no patterns match
83
- * (scope check becomes a no-op at the call site).
83
+ * Returns undefined only when there is no configured allowlist. A configured
84
+ * list with no available exact matches returns an empty set so caller-supplied
85
+ * models cannot bypass the scope check.
84
86
  */
85
-
86
- // Module-level cache — invalidated when either settings.json changes or patterns differ.
87
- let cachedAllowed: Set<string> | undefined;
88
- let cachedHash = "";
89
- let cachedPatternsKey = "";
90
-
91
- /** mtime+size hash of one file, or "missing" if absent. */
92
- function hashOf(path: string): string {
93
- try {
94
- const s = statSync(path);
95
- return `${s.mtimeMs}-${s.size}`;
96
- } catch {
97
- return "missing";
98
- }
99
- }
100
-
101
87
  export function resolveEnabledModels(
102
88
  patterns: string[] | undefined,
103
89
  registry: ModelRegistryRef,
104
- cwd: string = process.cwd(),
90
+ _cwd: string = process.cwd(),
105
91
  ): Set<string> | undefined {
106
- // Fast path: check cache (stat both project and global settings.json files)
107
- const patternsKey = JSON.stringify(patterns);
108
- const [project, global] = settingsPaths(cwd);
109
- const fileHash = `${hashOf(project)};${hashOf(global)}`;
110
-
111
- if (fileHash === cachedHash && patternsKey === cachedPatternsKey) {
112
- return cachedAllowed;
113
- }
114
-
115
- // Cache miss — resolve
116
- if (!patterns || patterns.length === 0) {
117
- cachedHash = fileHash;
118
- cachedPatternsKey = patternsKey;
119
- cachedAllowed = undefined;
120
- return undefined;
121
- }
92
+ if (!patterns || patterns.length === 0) return undefined;
122
93
 
123
94
  const available = (registry.getAvailable?.() ?? registry.getAll()) as ModelEntry[];
124
95
  const allowed = new Set<string>();
@@ -126,14 +97,10 @@ export function resolveEnabledModels(
126
97
  for (const pattern of patterns) {
127
98
  const trimmed = pattern.trim();
128
99
  if (!trimmed) continue; // skip empty/whitespace
129
- resolveExact(trimmed, available, allowed);
100
+ resolvePattern(trimmed, available, allowed);
130
101
  }
131
102
 
132
- const result = allowed.size > 0 ? allowed : undefined;
133
- cachedHash = fileHash;
134
- cachedPatternsKey = patternsKey;
135
- cachedAllowed = result;
136
- return result;
103
+ return allowed;
137
104
  }
138
105
 
139
106
 
@@ -155,26 +122,53 @@ function modelKey(model: { provider: string; id: string }): string {
155
122
  return `${model.provider}/${model.id}`.toLowerCase();
156
123
  }
157
124
 
158
- /**
159
- * Resolve exact model pattern. Example: "google/gemma-4-31b-it".
160
- */
161
- function resolveExact(
162
- pattern: string,
163
- available: ModelEntry[],
164
- allowed: Set<string>,
165
- ): void {
166
- // "provider/modelId" — exact (colon is part of id, not split)
167
- const slashIdx = pattern.indexOf("/");
168
- if (slashIdx === -1) return; // bare modelId not supported
169
-
170
- const provider = pattern.slice(0, slashIdx).toLowerCase();
171
- const modelId = pattern.slice(slashIdx + 1).toLowerCase();
172
- const exact = available.find(
173
- m => m.provider.toLowerCase() === provider && m.id.toLowerCase() === modelId,
174
- );
125
+ const THINKING_LEVELS = new Set(["off", "minimal", "low", "medium", "high", "xhigh", "max"]);
126
+
127
+ function resolvePattern(pattern: string, available: ModelEntry[], allowed: Set<string>): void {
128
+ const colon = pattern.lastIndexOf(":");
129
+ if (/[*?[]/.test(pattern)) {
130
+ // Pi strips a recognized thinking level from globs before matching. Trying
131
+ // the full glob first would select only colon-suffixed model IDs and miss
132
+ // their base models (e.g. custom/*:high).
133
+ const reference = colon >= 0 && THINKING_LEVELS.has(pattern.slice(colon + 1))
134
+ ? pattern.slice(0, colon) : pattern;
135
+ for (const model of available) {
136
+ if (minimatch(modelKey(model), reference, { nocase: true }) || minimatch(model.id, reference, { nocase: true })) {
137
+ allowed.add(modelKey(model));
138
+ }
139
+ }
140
+ return;
141
+ }
142
+
143
+ // Pi tries the complete reference before treating a colon as a thinking
144
+ // suffix: a model ID itself may contain ":high" or ":high-speed".
145
+ if (matchReference(pattern, available, allowed)) return;
146
+ // Invalid thinking suffixes on non-globs fall back to the prefix in scope
147
+ // mode, just as Pi does (though it reports a warning to its own UI).
148
+ if (colon >= 0) matchReference(pattern.slice(0, colon), available, allowed);
149
+ }
150
+
151
+ function matchReference(reference: string, available: ModelEntry[], allowed: Set<string>): boolean {
152
+ const exact = available.find((model) => modelKey(model) === reference.toLowerCase());
175
153
  if (exact) {
176
154
  allowed.add(modelKey(exact));
155
+ return true;
156
+ }
157
+ const bare = available.filter((model) => model.id.toLowerCase() === reference.toLowerCase());
158
+ if (bare.length === 1) {
159
+ allowed.add(modelKey(bare[0]));
160
+ return true;
177
161
  }
178
- }
179
162
 
163
+ // Pi's non-glob partial picker searches model IDs and names, not canonical
164
+ // provider/model keys. A provider-qualified partial is not an exact reference.
165
+ const query = reference.toLowerCase();
166
+ const matches = available.filter((model) =>
167
+ model.id.toLowerCase().includes(query) || model.name?.toLowerCase().includes(query)
168
+ );
169
+ const aliases = matches.filter((model) => model.id.endsWith("-latest") || !/-\d{8}$/.test(model.id));
170
+ const selected = (aliases.length > 0 ? aliases : matches).sort((a, b) => b.id.localeCompare(a.id))[0];
171
+ if (selected) allowed.add(modelKey(selected));
172
+ return selected !== undefined;
173
+ }
180
174