@phnx-labs/agents-cli 1.22.4 → 1.22.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/dist/bin/agents +0 -0
- package/dist/commands/events.js +8 -0
- package/dist/commands/exec.js +18 -6
- package/dist/commands/inspect.js +18 -5
- package/dist/commands/models.d.ts +1 -1
- package/dist/commands/models.js +68 -9
- package/dist/commands/monitors.js +1 -1
- package/dist/commands/view.d.ts +12 -0
- package/dist/commands/view.js +2 -0
- package/dist/commands/webhook.js +8 -4
- package/dist/index.js +2 -1
- package/dist/lib/browser/chrome.d.ts +12 -0
- package/dist/lib/browser/chrome.js +27 -11
- package/dist/lib/cloud/antigravity.js +8 -2
- package/dist/lib/crabbox/cli.js +72 -30
- package/dist/lib/event-stream.d.ts +3 -0
- package/dist/lib/event-stream.js +5 -0
- package/dist/lib/events.d.ts +2 -0
- package/dist/lib/events.js +6 -1
- package/dist/lib/feed.d.ts +1 -1
- package/dist/lib/feed.js +19 -0
- package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
- package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
- package/dist/lib/model-tier-overrides.d.ts +43 -0
- package/dist/lib/model-tier-overrides.js +97 -0
- package/dist/lib/model-tiers.d.ts +13 -7
- package/dist/lib/model-tiers.js +104 -35
- package/dist/lib/models.d.ts +16 -0
- package/dist/lib/models.js +17 -3
- package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
- package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
- package/dist/lib/secrets/index.d.ts +29 -0
- package/dist/lib/secrets/index.js +32 -0
- package/dist/lib/secrets/mcp.js +8 -4
- package/dist/lib/session/sync/config.d.ts +12 -11
- package/dist/lib/session/sync/config.js +40 -38
- package/dist/lib/share/config.js +4 -0
- package/dist/lib/types.d.ts +10 -0
- package/dist/lib/versions.js +35 -0
- package/package.json +1 -1
package/dist/lib/crabbox/cli.js
CHANGED
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
* from a secrets bundle when one is configured (see `crabboxEnv`).
|
|
13
13
|
*/
|
|
14
14
|
import { spawn, spawnSync } from 'child_process';
|
|
15
|
-
import { readAndResolveBundleEnv,
|
|
15
|
+
import { readAndResolveBundleEnv, listBundles, bundleExists } from '../secrets/bundles.js';
|
|
16
16
|
import { readMeta, writeMeta } from '../state.js';
|
|
17
17
|
/** Locate the crabbox binary, or throw an actionable error. */
|
|
18
18
|
export function findCrabbox() {
|
|
@@ -73,10 +73,14 @@ function resolveTailscaleBundleMemo() {
|
|
|
73
73
|
* and the single-key subset read (`keys: [ts.key]`) is rejected by
|
|
74
74
|
* `canCacheResolvedEnv` for broker auto-cache — so without this memo a
|
|
75
75
|
* non-broker-held tailscale bundle re-read the keychain on EVERY call (and,
|
|
76
|
-
* pre-guard, could pop a Touch ID sheet each time).
|
|
77
|
-
*
|
|
78
|
-
*
|
|
79
|
-
*
|
|
76
|
+
* pre-guard, could pop a Touch ID sheet each time). The read is always
|
|
77
|
+
* `agentOnly: true` (broker-only, SEC-13): tailscale is opt-in plumbing, a
|
|
78
|
+
* `--lease` run is headless by contract, and even an interactive `--lease`
|
|
79
|
+
* invocation must never pop an unwatched Touch ID sheet for this best-effort
|
|
80
|
+
* read — a locked bundle degrades silently to a public-network lease via the
|
|
81
|
+
* catch below. One read per process, success or failure (a failed read
|
|
82
|
+
* memoizes as undefined: the catch below already degrades to a public-network
|
|
83
|
+
* lease, retrying mid-process only repeats the same failure).
|
|
80
84
|
*/
|
|
81
85
|
let tailscaleValueMemo;
|
|
82
86
|
function resolveTailscaleKeyValueMemo(ts) {
|
|
@@ -86,7 +90,7 @@ function resolveTailscaleKeyValueMemo(ts) {
|
|
|
86
90
|
const { env } = readAndResolveBundleEnv(ts.name, {
|
|
87
91
|
caller: 'agents run --lease (crabbox tailscale)',
|
|
88
92
|
keys: [ts.key],
|
|
89
|
-
agentOnly:
|
|
93
|
+
agentOnly: true,
|
|
90
94
|
});
|
|
91
95
|
value = env[ts.key];
|
|
92
96
|
}
|
|
@@ -102,6 +106,7 @@ export function resetCrabboxSecretsMemosForTest() {
|
|
|
102
106
|
tailscaleBundleMemo = undefined;
|
|
103
107
|
tailscaleValueMemo = undefined;
|
|
104
108
|
leaseBundleMemo = undefined;
|
|
109
|
+
leaseEnvMemo = undefined;
|
|
105
110
|
}
|
|
106
111
|
/**
|
|
107
112
|
* The secrets bundle to feed crabbox, resolved in priority order:
|
|
@@ -150,6 +155,59 @@ function resolveLeaseBundleMemo() {
|
|
|
150
155
|
leaseBundleMemo = { value: resolveLeaseBundle() };
|
|
151
156
|
return leaseBundleMemo.value;
|
|
152
157
|
}
|
|
158
|
+
/**
|
|
159
|
+
* Process-lifetime memo for the RESOLVED provider-token env, resolved ONCE up
|
|
160
|
+
* front. `crabboxEnv` is called on every `crabboxWaitReady` poll iteration
|
|
161
|
+
* (crabboxWaitReady → crabboxFind → crabboxList → crabboxEnv), so without this
|
|
162
|
+
* memo the lease-token keychain read re-ran every ~5s of the ready wait — the
|
|
163
|
+
* per-poll storm this fix exists to kill. The read is now `agentOnly: true`
|
|
164
|
+
* (SEC-13: a `--lease` run is headless by contract and must never pop an
|
|
165
|
+
* unwatched Touch ID sheet). A locked `hold`/`always` bundle makes that read
|
|
166
|
+
* THROW the actionable "unlock <name>" message; we memoize the thrown error too
|
|
167
|
+
* and re-raise it on every subsequent call, so the failure surfaces loud ONCE
|
|
168
|
+
* up front (crabboxWarmup / the first crabboxList, before any poll loop) and the
|
|
169
|
+
* loop never re-issues the read. A `never`/no-ACL or broker-held bundle resolves
|
|
170
|
+
* silently. `undefined` when no lease bundle is configured (crabbox uses its own
|
|
171
|
+
* `crabbox login`). The memo (success OR failure) is cleared per test by
|
|
172
|
+
* resetCrabboxSecretsMemosForTest.
|
|
173
|
+
*/
|
|
174
|
+
let leaseEnvMemo;
|
|
175
|
+
function resolveLeaseEnvMemo(explicitBundle) {
|
|
176
|
+
if (!leaseEnvMemo) {
|
|
177
|
+
const resolved = explicitBundle
|
|
178
|
+
? { name: explicitBundle }
|
|
179
|
+
: resolveLeaseBundleMemo();
|
|
180
|
+
if (!resolved) {
|
|
181
|
+
leaseEnvMemo = {};
|
|
182
|
+
}
|
|
183
|
+
else {
|
|
184
|
+
try {
|
|
185
|
+
// Auto-detected bundle → inject ONLY the provider token key(s) (least
|
|
186
|
+
// privilege; an unrelated bundle can't leak its other secrets into crabbox).
|
|
187
|
+
// An explicitly-named bundle (env/config or `opts.secretsBundle`) injects
|
|
188
|
+
// whole — the user chose it. Same resolver `agents secrets exec` uses.
|
|
189
|
+
const { env } = readAndResolveBundleEnv(resolved.name, {
|
|
190
|
+
caller: 'agents run --lease (crabbox)',
|
|
191
|
+
keys: resolved.keys,
|
|
192
|
+
// --lease is headless by contract and a locked bundle must fail loud with
|
|
193
|
+
// an unlock hint, NEVER pop a Touch ID sheet — the read cannot be answered
|
|
194
|
+
// in a background lease (SEC-13).
|
|
195
|
+
agentOnly: true,
|
|
196
|
+
});
|
|
197
|
+
leaseEnvMemo = { env };
|
|
198
|
+
}
|
|
199
|
+
catch (e) {
|
|
200
|
+
leaseEnvMemo = {
|
|
201
|
+
error: new Error(`Could not load secrets bundle "${resolved.name}" for crabbox: ${e.message}. ` +
|
|
202
|
+
`Fix the bundle (agents secrets view ${resolved.name}) or unset lease.secretsBundle to use crabbox's own login.`),
|
|
203
|
+
};
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
if (leaseEnvMemo.error)
|
|
208
|
+
throw leaseEnvMemo.error;
|
|
209
|
+
return leaseEnvMemo.env;
|
|
210
|
+
}
|
|
153
211
|
/** Persist `lease.secretsBundle` in agents config so `--lease` needs no env var. */
|
|
154
212
|
export function setLeaseSecretsBundle(name) {
|
|
155
213
|
const meta = readMeta();
|
|
@@ -159,30 +217,14 @@ export function setLeaseSecretsBundle(name) {
|
|
|
159
217
|
/** Build the child env for crabbox, injecting a secrets bundle when configured. */
|
|
160
218
|
export function crabboxEnv(opts) {
|
|
161
219
|
const out = { ...process.env };
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
// whole — the user chose it. Same resolver `agents secrets exec` uses.
|
|
171
|
-
const { env } = readAndResolveBundleEnv(resolved.name, {
|
|
172
|
-
caller: 'agents run --lease (crabbox)',
|
|
173
|
-
keys: resolved.keys,
|
|
174
|
-
// --lease is headless by contract and crabboxEnv is called several times
|
|
175
|
-
// per run (list/wait/spawn/stop) — resolve broker-only so a keychain bundle
|
|
176
|
-
// can't pop repeated unwatched Touch ID sheets mid-lease.
|
|
177
|
-
agentOnly: isHeadlessSecretsContext(),
|
|
178
|
-
});
|
|
179
|
-
Object.assign(out, env);
|
|
180
|
-
}
|
|
181
|
-
catch (e) {
|
|
182
|
-
throw new Error(`Could not load secrets bundle "${resolved.name}" for crabbox: ${e.message}. ` +
|
|
183
|
-
`Fix the bundle (agents secrets view ${resolved.name}) or unset lease.secretsBundle to use crabbox's own login.`);
|
|
184
|
-
}
|
|
185
|
-
}
|
|
220
|
+
// The lease provider-token read is resolved ONCE up front and memoized (env or
|
|
221
|
+
// thrown error) for the process — crabboxEnv runs on every crabboxWaitReady
|
|
222
|
+
// poll, so re-reading here was the per-poll storm. A locked bundle re-raises
|
|
223
|
+
// the memoized "unlock <name>" error every call, so the failure surfaces loud
|
|
224
|
+
// on the first crabboxEnv (before any poll loop) and never re-issues the read.
|
|
225
|
+
const leaseEnv = resolveLeaseEnvMemo(opts.secretsBundle);
|
|
226
|
+
if (leaseEnv)
|
|
227
|
+
Object.assign(out, leaseEnv);
|
|
186
228
|
// Tailscale plumbing (F5): inject CRABBOX_TAILSCALE_AUTH_KEY from a bundle that
|
|
187
229
|
// declares a tailscale auth key, when the ambient env doesn't already set one.
|
|
188
230
|
// Best-effort and opt-in — a missing/unreadable tailscale bundle never fails a
|
|
@@ -20,6 +20,9 @@ export interface UnifiedQuery {
|
|
|
20
20
|
agent?: string;
|
|
21
21
|
/** Only events stamped with this session id (payload `sessionId`, the provenance floor). */
|
|
22
22
|
sessionId?: string;
|
|
23
|
+
/** Only events carrying this bundle name in their payload (e.g. secrets events).
|
|
24
|
+
* Combined with `sessionId`, answers "which session read this secrets bundle". */
|
|
25
|
+
bundle?: string;
|
|
23
26
|
caller?: string;
|
|
24
27
|
command?: string;
|
|
25
28
|
module?: string;
|
package/dist/lib/event-stream.js
CHANGED
|
@@ -31,6 +31,8 @@ function matches(r, q) {
|
|
|
31
31
|
return false;
|
|
32
32
|
if (q.sessionId && r.sessionId !== q.sessionId)
|
|
33
33
|
return false;
|
|
34
|
+
if (q.bundle && r.bundle !== q.bundle)
|
|
35
|
+
return false;
|
|
34
36
|
if (q.caller && r.caller !== q.caller)
|
|
35
37
|
return false;
|
|
36
38
|
if (q.command && r.command !== q.command &&
|
|
@@ -47,6 +49,8 @@ function matches(r, q) {
|
|
|
47
49
|
* result (each source is fetched up to `limit`, so the top-N is exact).
|
|
48
50
|
*/
|
|
49
51
|
export function readUnifiedEvents(q = {}) {
|
|
52
|
+
// `bundle` is filtered inside query()'s scan (before its limit cutoff) so a
|
|
53
|
+
// matching-bundle record older than the newest-`limit` window is not dropped.
|
|
50
54
|
const ops = query({
|
|
51
55
|
startDate: q.startDate,
|
|
52
56
|
endDate: q.endDate,
|
|
@@ -57,6 +61,7 @@ export function readUnifiedEvents(q = {}) {
|
|
|
57
61
|
caller: q.caller,
|
|
58
62
|
command: q.command,
|
|
59
63
|
module: q.module,
|
|
64
|
+
bundle: q.bundle,
|
|
60
65
|
limit: q.limit,
|
|
61
66
|
});
|
|
62
67
|
if (q.includeActivity === false)
|
package/dist/lib/events.d.ts
CHANGED
|
@@ -213,6 +213,8 @@ export declare function query(options: {
|
|
|
213
213
|
caller?: string;
|
|
214
214
|
command?: string;
|
|
215
215
|
module?: string;
|
|
216
|
+
/** Only events carrying this bundle name in their payload (e.g. secrets events). */
|
|
217
|
+
bundle?: string;
|
|
216
218
|
limit?: number;
|
|
217
219
|
}): EventRecord[];
|
|
218
220
|
/**
|
package/dist/lib/events.js
CHANGED
|
@@ -729,7 +729,7 @@ export function maybeRotate() {
|
|
|
729
729
|
* @returns Array of event records
|
|
730
730
|
*/
|
|
731
731
|
export function query(options) {
|
|
732
|
-
const { startDate, endDate = new Date(), eventTypes, level, agent, sessionId, caller, command, module, limit } = options;
|
|
732
|
+
const { startDate, endDate = new Date(), eventTypes, level, agent, sessionId, caller, command, module, bundle, limit } = options;
|
|
733
733
|
const results = [];
|
|
734
734
|
if (!fs.existsSync(eventsDir()))
|
|
735
735
|
return results;
|
|
@@ -783,6 +783,11 @@ export function query(options) {
|
|
|
783
783
|
continue;
|
|
784
784
|
if (module && record.module !== module)
|
|
785
785
|
continue;
|
|
786
|
+
// Filter bundle in the SAME scan, before the limit cutoff — a post-filter
|
|
787
|
+
// on the already-capped result silently drops matching-bundle records that
|
|
788
|
+
// fell outside the newest-`limit` window (a data-loss bug for an audit query).
|
|
789
|
+
if (bundle && record.bundle !== bundle)
|
|
790
|
+
continue;
|
|
786
791
|
results.push(record);
|
|
787
792
|
if (limit && results.length >= limit) {
|
|
788
793
|
return results;
|
package/dist/lib/feed.d.ts
CHANGED
|
@@ -218,7 +218,7 @@ export declare function removeBlock(blockId: string, root?: string): boolean;
|
|
|
218
218
|
* Embedded so it ships with the compiled CLI and can be installed to the
|
|
219
219
|
* CLI-writable user hooks dir without a separate file in the npm tarball.
|
|
220
220
|
*/
|
|
221
|
-
export declare const FEED_PUBLISH_HOOK_SCRIPT = "#!/usr/bin/env python3\n\"\"\"Publish and clear open-block records for `agents feed`.\n\nThe manifest invokes this script for top-level AskUserQuestion calls, waiting\nnotifications, question answers, and session lifecycle events. One atomic file\nper session means a new block replaces the previous block. Answer/resume/stop\nevents remove it so `agents feed` only lists decisions that are still open.\n\nSub-agent gate: when the PreToolUse payload carries `agent_type`, this is a\nTask/Agent subagent -- skip. Only the top-level agent publishes. Verified on\nClaude Code 2.1.170 (2026-07).\n\nFail-open: ANY error is swallowed so a feed hiccup never blocks a tool call.\n\"\"\"\nimport os\nimport sys\nimport json\nimport re\nimport socket\nimport tempfile\nfrom datetime import datetime, timezone\n\nWAITING_NOTIFICATION_TYPES = {\n \"permission_prompt\",\n \"idle_prompt\",\n \"elicitation_dialog\",\n}\nCLEAR_EVENTS = {\n \"PostToolUse\",\n \"Stop\",\n \"SessionEnd\",\n}\n# Codex emits a PermissionRequest event (not Claude's Notification) when it\n# blocks on an approval prompt. Claude never fires PermissionRequest, so the\n# same script handles both: PermissionRequest maps to an approval-class block\n# with a high cost-of-delay so 'agents feed --dispatch' pages it as urgent.\n\n\ndef read_json(path):\n try:\n with open(path) as f:\n return json.load(f)\n except Exception:\n return None\n\n\ndef write_json(path, value):\n dir_name = os.path.dirname(path)\n os.makedirs(dir_name, exist_ok=True)\n fd, tmp = tempfile.mkstemp(dir=dir_name, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(value, f, indent=2)\n os.replace(tmp, path)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\ndef main():\n raw = sys.stdin.read()\n try:\n payload = json.loads(raw) if raw.strip() else {}\n except Exception:\n return\n\n # Sub-agent gate.\n if payload.get(\"agent_type\"):\n return\n\n session_id = payload.get(\"session_id\", \"\")\n if not session_id:\n return\n\n safe_session_id = re.sub(r\"[^A-Za-z0-9._-]\", \"-\", session_id)\n block_id = f\"block-{safe_session_id}\"\n home = os.environ.get(\"HOME\") or os.path.expanduser(\"~\")\n feed_dir = os.path.join(home, \".agents\", \".history\", \"feed\")\n answered_dir = os.path.join(feed_dir, \"answered\")\n asks_dir = os.path.join(feed_dir, \"asks\")\n target = os.path.join(feed_dir, f\"{block_id}.json\")\n hook_event = payload.get(\"hook_event_name\", \"PreToolUse\")\n\n if hook_event in CLEAR_EVENTS:\n # A matcher-less PostToolUse clear (registered for Codex so an approved\n # tool clears its approval card) must NOT wipe an open AskUserQuestion\n # while an unrelated tool runs mid-question -- those are cleared only by\n # the AskUserQuestion-matched PostToolUse. So on PostToolUse, keep a\n # 'question' block; approval/notification blocks clear once the tool runs.\n if hook_event == \"PostToolUse\":\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\" and payload.get(\"tool_name\") != \"AskUserQuestion\":\n return\n except Exception:\n pass\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n # Also clear the answered marker so a future question for this session\n # is not permanently locked.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n # Terminal answers (human typed in the TUI) record an answered marker and\n # remove the block file so the feed stops showing it within one poll cycle.\n # The marker stays behind so a concurrent surface cannot double-answer.\n if hook_event == \"UserPromptSubmit\":\n os.makedirs(answered_dir, exist_ok=True)\n marker = os.path.join(answered_dir, f\"{block_id}.json\")\n try:\n fd = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)\n record = {\n \"answeredAt\": datetime.now(timezone.utc).isoformat(),\n \"answeredFrom\": \"terminal\",\n }\n with os.fdopen(fd, \"w\") as f:\n json.dump(record, f, indent=2)\n except FileExistsError:\n pass\n except Exception:\n pass\n # Remove the visible block so the feed drops the answered question.\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n notification_type = None\n codex_approval = False\n if hook_event == \"Notification\":\n notification_type = payload.get(\"notification_type\", \"\")\n if notification_type not in WAITING_NOTIFICATION_TYPES:\n return\n # Claude emits a generic permission notification after presenting an\n # AskUserQuestion. Keep the structured questions and options already\n # published for this session instead of replacing them with that less\n # useful notification text.\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\":\n return\n except Exception:\n pass\n message = payload.get(\"message\", \"\")\n if not message:\n return\n normalized_questions = [{\n \"text\": message,\n \"header\": payload.get(\"title\") or notification_type.replace(\"_\", \" \").title(),\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n elif hook_event == \"PermissionRequest\":\n # Codex approval prompt. The payload mirrors PreToolUse (tool_name,\n # tool_input) but carries no questions -- Codex is asking to run a tool,\n # not asking the operator a multiple-choice question. Publish it as a\n # notification-kind approval block naming the tool so the feed and the\n # phone notifier can surface it, and so the Factory extension can bridge\n # it to a VS Code notification.\n tool_name = payload.get(\"tool_name\") or \"a tool\"\n tool_input = payload.get(\"tool_input\", {})\n command = \"\"\n if isinstance(tool_input, dict):\n command = (\n tool_input.get(\"command\")\n or tool_input.get(\"cmd\")\n or tool_input.get(\"path\")\n or \"\"\n )\n if isinstance(command, list):\n command = \" \".join(str(c) for c in command)\n detail = f\": {command}\" if command else \"\"\n normalized_questions = [{\n \"text\": f\"Codex needs approval to run {tool_name}{detail}\",\n \"header\": \"Approval needed\",\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n notification_type = \"permission_prompt\"\n codex_approval = True\n else:\n tool_input = payload.get(\"tool_input\", {})\n questions = tool_input.get(\"questions\", [])\n if not questions:\n return\n normalized_questions = []\n for q in questions:\n if not isinstance(q, dict):\n continue\n question = {\n \"text\": q.get(\"question\", q.get(\"header\", \"\")),\n \"header\": q.get(\"header\"),\n \"multiSelect\": q.get(\"multiSelect\", False),\n }\n raw_opts = q.get(\"options\", [])\n if raw_opts:\n question[\"options\"] = [\n {\"label\": o.get(\"label\", \"\"), \"description\": o.get(\"description\")}\n for o in raw_opts\n if isinstance(o, dict)\n ]\n normalized_questions.append(question)\n if not normalized_questions:\n return\n kind = \"question\"\n\n # Identity from env (set by agents-cli at spawn).\n mailbox_id = os.path.basename(\n os.environ.get(\"AGENTS_MAILBOX_DIR\", \"\").rstrip(\"/\")\n ) or session_id\n\n now_iso = datetime.now(timezone.utc).isoformat()\n stats_path = os.path.join(asks_dir, f\"{safe_session_id}.json\")\n stats = read_json(stats_path) or {}\n recent = stats.get(\"recentAskTimestamps\") if isinstance(stats, dict) else []\n if not isinstance(recent, list):\n recent = []\n recent.append(now_iso)\n # Keep enough history for rolling one-hour needy detection without unbounded\n # per-session files. The TypeScript reader applies the exact time window.\n recent = recent[-200:]\n write_json(stats_path, {\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"firstAskAt\": stats.get(\"firstAskAt\") or now_iso,\n \"lastAskAt\": now_iso,\n \"totalAskCount\": int(stats.get(\"totalAskCount\") or 0) + 1,\n \"recentAskTimestamps\": recent,\n })\n\n hostname = os.environ.get(\"AGENTS_SYNC_MACHINE_ID\") or socket.gethostname()\n host = hostname.split(\".\")[0].strip().lower()\n host = re.sub(r\"[^a-z0-9_-]\", \"-\", host) or \"unknown\"\n\n runtime = os.environ.get(\"AGENTS_RUNTIME\", \"headless\")\n\n block = {\n \"blockId\": block_id,\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"host\": host,\n \"runtime\": runtime,\n \"ts\": now_iso,\n \"questions\": normalized_questions,\n \"kind\": kind,\n }\n if notification_type:\n block[\"notificationType\"] = notification_type\n\n # A Codex PermissionRequest is a real approval gate: mark it approval-class\n # with a high cost-of-delay so 'agents feed --dispatch' classifies it urgent\n # (isPhoneUrgent gates on costOfDelay >= phoneNotifyThreshold, default\n # 'medium') and pages the phone. A plain 'deny' is the safe default.\n if codex_approval:\n block[\"blockClass\"] = \"approval\"\n block[\"costOfDelay\"] = \"high\"\n block[\"safeDefault\"] = \"deny\"\n\n # Optional multi-operator control metadata passed by the agent in the\n # AskUserQuestion tool_input. Defaults keep the existing behavior. A Codex\n # PermissionRequest carries tool ARGS in tool_input (command/path), not\n # operator controls, so it is excluded here -- its class/cost is stamped\n # above from codex_approval.\n controls = payload.get(\"tool_input\", {}) if hook_event not in (\"Notification\", \"PermissionRequest\") else {}\n block_class = controls.get(\"blockClass\") if isinstance(controls, dict) else None\n if block_class in (\"approval\", \"decision\"):\n block[\"blockClass\"] = block_class\n consequence = controls.get(\"consequence\") if isinstance(controls, dict) else None\n if consequence:\n block[\"consequence\"] = consequence\n allowed = controls.get(\"allowedOperators\") if isinstance(controls, dict) else None\n if isinstance(allowed, list):\n block[\"allowedOperators\"] = [str(a) for a in allowed]\n timeout = controls.get(\"timeoutMinutes\") if isinstance(controls, dict) else None\n if isinstance(timeout, (int, float)) and timeout > 0:\n block[\"timeoutMinutes\"] = int(timeout)\n safe_default = controls.get(\"safeDefault\") if isinstance(controls, dict) else None\n if isinstance(safe_default, str):\n block[\"safeDefault\"] = safe_default\n cost = controls.get(\"costOfDelay\") if isinstance(controls, dict) else None\n if cost in (\"low\", \"medium\", \"high\"):\n block[\"costOfDelay\"] = cost\n\n # Publishing a new question clears any stale answered marker from the\n # previous question in this session.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n\n # Python's expanduser() ignores HOME on Windows, while agents-cli honors a\n # HOME override on every platform. Use the same anchor so hooks and the CLI\n # always read/write one feed store (including temp-home and sandbox runs).\n os.makedirs(feed_dir, exist_ok=True)\n\n fd, tmp = tempfile.mkstemp(dir=feed_dir, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(block, f, indent=2)\n os.replace(tmp, target)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\nif __name__ == \"__main__\":\n try:\n main()\n except Exception:\n pass # fail open\n";
|
|
221
|
+
export declare const FEED_PUBLISH_HOOK_SCRIPT = "#!/usr/bin/env python3\n\"\"\"Publish and clear open-block records for `agents feed`.\n\nThe manifest invokes this script for top-level AskUserQuestion calls, waiting\nnotifications, question answers, and session lifecycle events. One atomic file\nper session means a new block replaces the previous block. Answer/resume/stop\nevents remove it so `agents feed` only lists decisions that are still open.\n\nSub-agent gate: when the PreToolUse payload carries `agent_type`, this is a\nTask/Agent subagent -- skip. Only the top-level agent publishes. Verified on\nClaude Code 2.1.170 (2026-07).\n\nFail-open: ANY error is swallowed so a feed hiccup never blocks a tool call.\n\"\"\"\nimport os\nimport sys\nimport json\nimport re\nimport socket\nimport tempfile\nfrom datetime import datetime, timezone\n\nWAITING_NOTIFICATION_TYPES = {\n \"permission_prompt\",\n \"idle_prompt\",\n \"elicitation_dialog\",\n}\nCLEAR_EVENTS = {\n \"PostToolUse\",\n \"Stop\",\n \"SessionEnd\",\n}\n# Codex emits a PermissionRequest event (not Claude's Notification) when it\n# blocks on an approval prompt. Claude never fires PermissionRequest, so the\n# same script handles both: PermissionRequest maps to an approval-class block\n# with a high cost-of-delay so 'agents feed --dispatch' pages it as urgent.\n\n\ndef read_json(path):\n try:\n with open(path) as f:\n return json.load(f)\n except Exception:\n return None\n\n\ndef write_json(path, value):\n dir_name = os.path.dirname(path)\n os.makedirs(dir_name, exist_ok=True)\n fd, tmp = tempfile.mkstemp(dir=dir_name, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(value, f, indent=2)\n os.replace(tmp, path)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\ndef main():\n raw = sys.stdin.read()\n try:\n payload = json.loads(raw) if raw.strip() else {}\n except Exception:\n return\n\n # Sub-agent gate.\n if payload.get(\"agent_type\"):\n return\n\n session_id = payload.get(\"session_id\", \"\")\n if not session_id:\n return\n\n safe_session_id = re.sub(r\"[^A-Za-z0-9._-]\", \"-\", session_id)\n block_id = f\"block-{safe_session_id}\"\n home = os.environ.get(\"HOME\") or os.path.expanduser(\"~\")\n feed_dir = os.path.join(home, \".agents\", \".history\", \"feed\")\n answered_dir = os.path.join(feed_dir, \"answered\")\n asks_dir = os.path.join(feed_dir, \"asks\")\n target = os.path.join(feed_dir, f\"{block_id}.json\")\n hook_event = payload.get(\"hook_event_name\", \"PreToolUse\")\n\n if hook_event in CLEAR_EVENTS:\n # A declared block (`agents feed post --blocked`) is the agent explicitly\n # saying it is stuck. Unlike a question/notification/approval block -- which\n # tracks an in-flight harness prompt that a lifecycle event resolves -- a\n # declared block stays open until it is actually ANSWERED. So while it is\n # still UNANSWERED, Stop/SessionEnd/PostToolUse must never silently drop it:\n # otherwise the needs-you record vanishes the moment the agent parks the block\n # and its turn ends -- exactly when the owner still needs to see and answer it.\n # Once it IS answered (an answered marker exists), it clears like any other\n # block by falling through below -- which frees that marker too, so a later\n # `--blocked` in the same session is not falsely locked as already-answered\n # (recordAnswer creates the marker with O_EXCL).\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n answered = os.path.exists(os.path.join(answered_dir, f\"{block_id}.json\"))\n if existing.get(\"kind\") == \"declared\" and not answered:\n return\n except Exception:\n pass\n # A matcher-less PostToolUse clear (registered for Codex so an approved\n # tool clears its approval card) must NOT wipe an open AskUserQuestion\n # while an unrelated tool runs mid-question -- those are cleared only by\n # the AskUserQuestion-matched PostToolUse. So on PostToolUse, keep a\n # 'question' block; approval/notification blocks clear once the tool runs.\n if hook_event == \"PostToolUse\":\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\" and payload.get(\"tool_name\") != \"AskUserQuestion\":\n return\n except Exception:\n pass\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n # Also clear the answered marker so a future question for this session\n # is not permanently locked.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n # Terminal answers (human typed in the TUI) record an answered marker and\n # remove the block file so the feed stops showing it within one poll cycle.\n # The marker stays behind so a concurrent surface cannot double-answer.\n if hook_event == \"UserPromptSubmit\":\n os.makedirs(answered_dir, exist_ok=True)\n marker = os.path.join(answered_dir, f\"{block_id}.json\")\n try:\n fd = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)\n record = {\n \"answeredAt\": datetime.now(timezone.utc).isoformat(),\n \"answeredFrom\": \"terminal\",\n }\n with os.fdopen(fd, \"w\") as f:\n json.dump(record, f, indent=2)\n except FileExistsError:\n pass\n except Exception:\n pass\n # Remove the visible block so the feed drops the answered question.\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n notification_type = None\n codex_approval = False\n if hook_event == \"Notification\":\n notification_type = payload.get(\"notification_type\", \"\")\n if notification_type not in WAITING_NOTIFICATION_TYPES:\n return\n # Claude emits a generic permission notification after presenting an\n # AskUserQuestion. Keep the structured questions and options already\n # published for this session instead of replacing them with that less\n # useful notification text.\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\":\n return\n except Exception:\n pass\n message = payload.get(\"message\", \"\")\n if not message:\n return\n normalized_questions = [{\n \"text\": message,\n \"header\": payload.get(\"title\") or notification_type.replace(\"_\", \" \").title(),\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n elif hook_event == \"PermissionRequest\":\n # Codex approval prompt. The payload mirrors PreToolUse (tool_name,\n # tool_input) but carries no questions -- Codex is asking to run a tool,\n # not asking the operator a multiple-choice question. Publish it as a\n # notification-kind approval block naming the tool so the feed and the\n # phone notifier can surface it, and so the Factory extension can bridge\n # it to a VS Code notification.\n tool_name = payload.get(\"tool_name\") or \"a tool\"\n tool_input = payload.get(\"tool_input\", {})\n command = \"\"\n if isinstance(tool_input, dict):\n command = (\n tool_input.get(\"command\")\n or tool_input.get(\"cmd\")\n or tool_input.get(\"path\")\n or \"\"\n )\n if isinstance(command, list):\n command = \" \".join(str(c) for c in command)\n detail = f\": {command}\" if command else \"\"\n normalized_questions = [{\n \"text\": f\"Codex needs approval to run {tool_name}{detail}\",\n \"header\": \"Approval needed\",\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n notification_type = \"permission_prompt\"\n codex_approval = True\n else:\n tool_input = payload.get(\"tool_input\", {})\n questions = tool_input.get(\"questions\", [])\n if not questions:\n return\n normalized_questions = []\n for q in questions:\n if not isinstance(q, dict):\n continue\n question = {\n \"text\": q.get(\"question\", q.get(\"header\", \"\")),\n \"header\": q.get(\"header\"),\n \"multiSelect\": q.get(\"multiSelect\", False),\n }\n raw_opts = q.get(\"options\", [])\n if raw_opts:\n question[\"options\"] = [\n {\"label\": o.get(\"label\", \"\"), \"description\": o.get(\"description\")}\n for o in raw_opts\n if isinstance(o, dict)\n ]\n normalized_questions.append(question)\n if not normalized_questions:\n return\n kind = \"question\"\n\n # Identity from env (set by agents-cli at spawn).\n mailbox_id = os.path.basename(\n os.environ.get(\"AGENTS_MAILBOX_DIR\", \"\").rstrip(\"/\")\n ) or session_id\n\n now_iso = datetime.now(timezone.utc).isoformat()\n stats_path = os.path.join(asks_dir, f\"{safe_session_id}.json\")\n stats = read_json(stats_path) or {}\n recent = stats.get(\"recentAskTimestamps\") if isinstance(stats, dict) else []\n if not isinstance(recent, list):\n recent = []\n recent.append(now_iso)\n # Keep enough history for rolling one-hour needy detection without unbounded\n # per-session files. The TypeScript reader applies the exact time window.\n recent = recent[-200:]\n write_json(stats_path, {\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"firstAskAt\": stats.get(\"firstAskAt\") or now_iso,\n \"lastAskAt\": now_iso,\n \"totalAskCount\": int(stats.get(\"totalAskCount\") or 0) + 1,\n \"recentAskTimestamps\": recent,\n })\n\n hostname = os.environ.get(\"AGENTS_SYNC_MACHINE_ID\") or socket.gethostname()\n host = hostname.split(\".\")[0].strip().lower()\n host = re.sub(r\"[^a-z0-9_-]\", \"-\", host) or \"unknown\"\n\n runtime = os.environ.get(\"AGENTS_RUNTIME\", \"headless\")\n\n block = {\n \"blockId\": block_id,\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"host\": host,\n \"runtime\": runtime,\n \"ts\": now_iso,\n \"questions\": normalized_questions,\n \"kind\": kind,\n }\n if notification_type:\n block[\"notificationType\"] = notification_type\n\n # A Codex PermissionRequest is a real approval gate: mark it approval-class\n # with a high cost-of-delay so 'agents feed --dispatch' classifies it urgent\n # (isPhoneUrgent gates on costOfDelay >= phoneNotifyThreshold, default\n # 'medium') and pages the phone. A plain 'deny' is the safe default.\n if codex_approval:\n block[\"blockClass\"] = \"approval\"\n block[\"costOfDelay\"] = \"high\"\n block[\"safeDefault\"] = \"deny\"\n\n # Optional multi-operator control metadata passed by the agent in the\n # AskUserQuestion tool_input. Defaults keep the existing behavior. A Codex\n # PermissionRequest carries tool ARGS in tool_input (command/path), not\n # operator controls, so it is excluded here -- its class/cost is stamped\n # above from codex_approval.\n controls = payload.get(\"tool_input\", {}) if hook_event not in (\"Notification\", \"PermissionRequest\") else {}\n block_class = controls.get(\"blockClass\") if isinstance(controls, dict) else None\n if block_class in (\"approval\", \"decision\"):\n block[\"blockClass\"] = block_class\n consequence = controls.get(\"consequence\") if isinstance(controls, dict) else None\n if consequence:\n block[\"consequence\"] = consequence\n allowed = controls.get(\"allowedOperators\") if isinstance(controls, dict) else None\n if isinstance(allowed, list):\n block[\"allowedOperators\"] = [str(a) for a in allowed]\n timeout = controls.get(\"timeoutMinutes\") if isinstance(controls, dict) else None\n if isinstance(timeout, (int, float)) and timeout > 0:\n block[\"timeoutMinutes\"] = int(timeout)\n safe_default = controls.get(\"safeDefault\") if isinstance(controls, dict) else None\n if isinstance(safe_default, str):\n block[\"safeDefault\"] = safe_default\n cost = controls.get(\"costOfDelay\") if isinstance(controls, dict) else None\n if cost in (\"low\", \"medium\", \"high\"):\n block[\"costOfDelay\"] = cost\n\n # Publishing a new question clears any stale answered marker from the\n # previous question in this session.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n\n # Python's expanduser() ignores HOME on Windows, while agents-cli honors a\n # HOME override on every platform. Use the same anchor so hooks and the CLI\n # always read/write one feed store (including temp-home and sandbox runs).\n os.makedirs(feed_dir, exist_ok=True)\n\n fd, tmp = tempfile.mkstemp(dir=feed_dir, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(block, f, indent=2)\n os.replace(tmp, target)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\nif __name__ == \"__main__\":\n try:\n main()\n except Exception:\n pass # fail open\n";
|
|
222
222
|
/** Manifest entry for the feed-publish hook, matching the ManifestHook shape. */
|
|
223
223
|
export declare const FEED_PUBLISH_HOOK_MANIFEST: {
|
|
224
224
|
name: string;
|
package/dist/lib/feed.js
CHANGED
|
@@ -442,6 +442,25 @@ def main():
|
|
|
442
442
|
hook_event = payload.get("hook_event_name", "PreToolUse")
|
|
443
443
|
|
|
444
444
|
if hook_event in CLEAR_EVENTS:
|
|
445
|
+
# A declared block (\`agents feed post --blocked\`) is the agent explicitly
|
|
446
|
+
# saying it is stuck. Unlike a question/notification/approval block -- which
|
|
447
|
+
# tracks an in-flight harness prompt that a lifecycle event resolves -- a
|
|
448
|
+
# declared block stays open until it is actually ANSWERED. So while it is
|
|
449
|
+
# still UNANSWERED, Stop/SessionEnd/PostToolUse must never silently drop it:
|
|
450
|
+
# otherwise the needs-you record vanishes the moment the agent parks the block
|
|
451
|
+
# and its turn ends -- exactly when the owner still needs to see and answer it.
|
|
452
|
+
# Once it IS answered (an answered marker exists), it clears like any other
|
|
453
|
+
# block by falling through below -- which frees that marker too, so a later
|
|
454
|
+
# \`--blocked\` in the same session is not falsely locked as already-answered
|
|
455
|
+
# (recordAnswer creates the marker with O_EXCL).
|
|
456
|
+
try:
|
|
457
|
+
with open(target) as existing_file:
|
|
458
|
+
existing = json.load(existing_file)
|
|
459
|
+
answered = os.path.exists(os.path.join(answered_dir, f"{block_id}.json"))
|
|
460
|
+
if existing.get("kind") == "declared" and not answered:
|
|
461
|
+
return
|
|
462
|
+
except Exception:
|
|
463
|
+
pass
|
|
445
464
|
# A matcher-less PostToolUse clear (registered for Codex so an approved
|
|
446
465
|
# tool clears its approval card) must NOT wipe an open AskUserQuestion
|
|
447
466
|
# while an unrelated tool runs mid-question -- those are cleared only by
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Human overrides for the cost-tier -> model mapping.
|
|
3
|
+
*
|
|
4
|
+
* The auto-ranking (lib/model-tiers.ts) is a best guess; for subscription harnesses
|
|
5
|
+
* with no price signal (Kimi, Cursor) it can be wrong. A user pins the right model
|
|
6
|
+
* per tier with `agents models tier set <agent[@version]> <tier> <model>`, which
|
|
7
|
+
* writes here — the user never hand-edits the file.
|
|
8
|
+
*
|
|
9
|
+
* Stored under agents.yaml, same selector shape as run.defaults:
|
|
10
|
+
*
|
|
11
|
+
* model:
|
|
12
|
+
* tiers:
|
|
13
|
+
* "kimi:*": # applies to every installed kimi
|
|
14
|
+
* best: kimi-code/k3
|
|
15
|
+
* default: kimi-code/kimi-for-coding
|
|
16
|
+
* "kimi:0.19.2": # a specific version wins over the wildcard
|
|
17
|
+
* best: kimi-code/k3-256k
|
|
18
|
+
*
|
|
19
|
+
* Resolution (most-specific-first): `<agent>:<version>` beats `<agent>:*` beats the
|
|
20
|
+
* auto-ranking. resolveTierMap (model-tiers.ts) applies the result and falls back to
|
|
21
|
+
* auto for any tier whose overridden id isn't in that version's catalog.
|
|
22
|
+
*/
|
|
23
|
+
import type { AgentId } from './types.js';
|
|
24
|
+
import { type ModelTier } from './model-tiers.js';
|
|
25
|
+
export type TierOverrideMap = Partial<Record<ModelTier, string>>;
|
|
26
|
+
export interface TierOverrideEntry {
|
|
27
|
+
selector: string;
|
|
28
|
+
tiers: TierOverrideMap;
|
|
29
|
+
}
|
|
30
|
+
/** Validate a tier token, throwing a friendly error otherwise. */
|
|
31
|
+
export declare function parseTier(input: string): ModelTier;
|
|
32
|
+
/**
|
|
33
|
+
* The effective tier overrides for an (agent, version): the `<agent>:*` wildcard
|
|
34
|
+
* merged under the exact `<agent>:<version>` selector (exact wins per-tier).
|
|
35
|
+
*/
|
|
36
|
+
export declare function resolveTierOverrideFrom(all: Record<string, unknown>, agent: AgentId, version?: string | null): TierOverrideMap;
|
|
37
|
+
export declare function resolveTierOverride(agent: AgentId, version?: string | null): TierOverrideMap;
|
|
38
|
+
/** Every configured override entry, sorted by selector (for `agents models tier list`). */
|
|
39
|
+
export declare function listTierOverrides(): TierOverrideEntry[];
|
|
40
|
+
/** Pin `tier -> model` for a selector. Writes agents.yaml. */
|
|
41
|
+
export declare function setTierOverride(selectorInput: string, tierInput: string, model: string): TierOverrideEntry;
|
|
42
|
+
/** Clear one tier (or all tiers when `tierInput` is omitted) for a selector. Returns true if anything changed. */
|
|
43
|
+
export declare function clearTierOverride(selectorInput: string, tierInput?: string): boolean;
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
import { readMeta, updateMeta } from './state.js';
|
|
2
|
+
import { parseRunDefaultSelector } from './run-defaults.js';
|
|
3
|
+
import { MODEL_TIERS } from './model-tiers.js';
|
|
4
|
+
function isTier(value) {
|
|
5
|
+
return MODEL_TIERS.includes(value);
|
|
6
|
+
}
|
|
7
|
+
/** Validate a tier token, throwing a friendly error otherwise. */
|
|
8
|
+
export function parseTier(input) {
|
|
9
|
+
const t = input.trim().toLowerCase();
|
|
10
|
+
if (!isTier(t)) {
|
|
11
|
+
throw new Error(`Invalid tier '${input}'. Use one of: ${MODEL_TIERS.join(', ')}.`);
|
|
12
|
+
}
|
|
13
|
+
return t;
|
|
14
|
+
}
|
|
15
|
+
/** Normalize a stored selector's tier map, dropping unknown/empty entries. */
|
|
16
|
+
function normalize(raw) {
|
|
17
|
+
const out = {};
|
|
18
|
+
if (!raw || typeof raw !== 'object')
|
|
19
|
+
return out;
|
|
20
|
+
for (const [k, v] of Object.entries(raw)) {
|
|
21
|
+
if (isTier(k) && typeof v === 'string' && v.trim())
|
|
22
|
+
out[k] = v.trim();
|
|
23
|
+
}
|
|
24
|
+
return out;
|
|
25
|
+
}
|
|
26
|
+
function sortedSelectors(map) {
|
|
27
|
+
return Object.fromEntries(Object.entries(map).sort(([a], [b]) => a.localeCompare(b)));
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* The effective tier overrides for an (agent, version): the `<agent>:*` wildcard
|
|
31
|
+
* merged under the exact `<agent>:<version>` selector (exact wins per-tier).
|
|
32
|
+
*/
|
|
33
|
+
export function resolveTierOverrideFrom(all, agent, version) {
|
|
34
|
+
const merged = { ...normalize(all[`${agent}:*`]) };
|
|
35
|
+
if (version) {
|
|
36
|
+
for (const [tier, model] of Object.entries(normalize(all[`${agent}:${version}`]))) {
|
|
37
|
+
merged[tier] = model;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return merged;
|
|
41
|
+
}
|
|
42
|
+
export function resolveTierOverride(agent, version) {
|
|
43
|
+
return resolveTierOverrideFrom(readMeta().model?.tiers ?? {}, agent, version);
|
|
44
|
+
}
|
|
45
|
+
/** Every configured override entry, sorted by selector (for `agents models tier list`). */
|
|
46
|
+
export function listTierOverrides() {
|
|
47
|
+
const all = readMeta().model?.tiers ?? {};
|
|
48
|
+
return Object.entries(all)
|
|
49
|
+
.sort(([a], [b]) => a.localeCompare(b))
|
|
50
|
+
.map(([selector, tiers]) => ({ selector, tiers: normalize(tiers) }));
|
|
51
|
+
}
|
|
52
|
+
/** Pin `tier -> model` for a selector. Writes agents.yaml. */
|
|
53
|
+
export function setTierOverride(selectorInput, tierInput, model) {
|
|
54
|
+
const parsed = parseRunDefaultSelector(selectorInput);
|
|
55
|
+
const tier = parseTier(tierInput);
|
|
56
|
+
const id = model.trim();
|
|
57
|
+
if (!id)
|
|
58
|
+
throw new Error('A model id is required.');
|
|
59
|
+
updateMeta((meta) => {
|
|
60
|
+
const modelCfg = { ...(meta.model ?? {}) };
|
|
61
|
+
const tiers = { ...(modelCfg.tiers ?? {}) };
|
|
62
|
+
tiers[parsed.selector] = { ...(tiers[parsed.selector] ?? {}), [tier]: id };
|
|
63
|
+
modelCfg.tiers = sortedSelectors(tiers);
|
|
64
|
+
return { ...meta, model: modelCfg };
|
|
65
|
+
});
|
|
66
|
+
return { selector: parsed.selector, tiers: normalize(readMeta().model?.tiers?.[parsed.selector]) };
|
|
67
|
+
}
|
|
68
|
+
/** Clear one tier (or all tiers when `tierInput` is omitted) for a selector. Returns true if anything changed. */
|
|
69
|
+
export function clearTierOverride(selectorInput, tierInput) {
|
|
70
|
+
const parsed = parseRunDefaultSelector(selectorInput);
|
|
71
|
+
const tier = tierInput ? parseTier(tierInput) : null;
|
|
72
|
+
let changed = false;
|
|
73
|
+
updateMeta((meta) => {
|
|
74
|
+
if (!meta.model?.tiers?.[parsed.selector])
|
|
75
|
+
return meta;
|
|
76
|
+
const model = { ...meta.model };
|
|
77
|
+
const tiers = { ...(model.tiers ?? {}) };
|
|
78
|
+
if (tier) {
|
|
79
|
+
const entry = { ...tiers[parsed.selector] };
|
|
80
|
+
if (entry[tier] !== undefined) {
|
|
81
|
+
delete entry[tier];
|
|
82
|
+
changed = true;
|
|
83
|
+
}
|
|
84
|
+
if (Object.keys(entry).length === 0)
|
|
85
|
+
delete tiers[parsed.selector];
|
|
86
|
+
else
|
|
87
|
+
tiers[parsed.selector] = entry;
|
|
88
|
+
}
|
|
89
|
+
else {
|
|
90
|
+
delete tiers[parsed.selector];
|
|
91
|
+
changed = true;
|
|
92
|
+
}
|
|
93
|
+
model.tiers = tiers;
|
|
94
|
+
return { ...meta, model };
|
|
95
|
+
});
|
|
96
|
+
return changed;
|
|
97
|
+
}
|
|
@@ -36,18 +36,24 @@ export interface TierResolution {
|
|
|
36
36
|
clampedFrom?: ModelTier;
|
|
37
37
|
/** Human note (e.g. why it clamped, or that it is a curated/subscription mapping). */
|
|
38
38
|
note?: string;
|
|
39
|
+
/** Where the model came from: 'auto' ranking, a user 'override', or a 'curated' ladder. */
|
|
40
|
+
source?: 'auto' | 'override' | 'curated';
|
|
39
41
|
}
|
|
40
42
|
/**
|
|
41
|
-
* Resolve all four tiers for an (agent, version)
|
|
42
|
-
*
|
|
43
|
+
* Resolve all four tiers for an (agent, version) -- what `agents models` prints and
|
|
44
|
+
* `resolveTier` indexes. Precedence: user override -> curated ladder / auto-ranking.
|
|
43
45
|
*/
|
|
44
46
|
export declare function resolveTierMap(agent: AgentId, version: string): Record<ModelTier, TierResolution>;
|
|
45
47
|
/**
|
|
46
|
-
*
|
|
47
|
-
* so it is directly testable
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
|
|
48
|
+
* Apply user overrides on top of the auto/curated map. Pure (takes the resolved
|
|
49
|
+
* override map, no config lookup) so it is directly testable. An overridden id is
|
|
50
|
+
* used only when the version actually ships it (or when there is no catalog to
|
|
51
|
+
* check, e.g. Droid); otherwise the tier keeps its base value with a note.
|
|
52
|
+
*/
|
|
53
|
+
export declare function applyTierOverrides(overrides: Partial<Record<ModelTier, string>>, label: string, catalogIds: Set<string> | null, base: Record<ModelTier, TierResolution>): Record<ModelTier, TierResolution>;
|
|
54
|
+
/**
|
|
55
|
+
* Map a harness's catalog models onto the four tiers. Pure (no catalog lookup) so
|
|
56
|
+
* it is directly testable. A single-model harness maps the tiers to reasoning effort.
|
|
51
57
|
*/
|
|
52
58
|
export declare function tierizeModels(agent: AgentId, models: ModelInfo[]): Record<ModelTier, TierResolution>;
|
|
53
59
|
/** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
|