@phnx-labs/agents-cli 1.22.4 → 1.22.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/dist/bin/agents +0 -0
- package/dist/commands/events.js +8 -0
- package/dist/commands/exec.js +7 -5
- package/dist/commands/inspect.js +18 -5
- package/dist/commands/models.d.ts +1 -1
- package/dist/commands/models.js +68 -9
- package/dist/commands/view.d.ts +12 -0
- package/dist/commands/view.js +2 -0
- package/dist/commands/webhook.js +8 -4
- package/dist/lib/browser/chrome.d.ts +12 -0
- package/dist/lib/browser/chrome.js +27 -11
- package/dist/lib/cloud/antigravity.js +8 -2
- package/dist/lib/crabbox/cli.js +72 -30
- package/dist/lib/event-stream.d.ts +3 -0
- package/dist/lib/event-stream.js +5 -0
- package/dist/lib/events.d.ts +2 -0
- package/dist/lib/events.js +6 -1
- package/dist/lib/feed.d.ts +1 -1
- package/dist/lib/feed.js +19 -0
- package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
- package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
- package/dist/lib/model-tier-overrides.d.ts +43 -0
- package/dist/lib/model-tier-overrides.js +97 -0
- package/dist/lib/model-tiers.d.ts +13 -7
- package/dist/lib/model-tiers.js +104 -35
- package/dist/lib/models.d.ts +16 -0
- package/dist/lib/models.js +17 -3
- package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
- package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
- package/dist/lib/secrets/index.d.ts +29 -0
- package/dist/lib/secrets/index.js +32 -0
- package/dist/lib/secrets/mcp.js +8 -4
- package/dist/lib/session/sync/config.d.ts +12 -11
- package/dist/lib/session/sync/config.js +40 -38
- package/dist/lib/share/config.js +4 -0
- package/dist/lib/types.d.ts +10 -0
- package/dist/lib/versions.js +35 -0
- package/package.json +1 -1
package/dist/lib/event-stream.js
CHANGED
|
@@ -31,6 +31,8 @@ function matches(r, q) {
|
|
|
31
31
|
return false;
|
|
32
32
|
if (q.sessionId && r.sessionId !== q.sessionId)
|
|
33
33
|
return false;
|
|
34
|
+
if (q.bundle && r.bundle !== q.bundle)
|
|
35
|
+
return false;
|
|
34
36
|
if (q.caller && r.caller !== q.caller)
|
|
35
37
|
return false;
|
|
36
38
|
if (q.command && r.command !== q.command &&
|
|
@@ -47,6 +49,8 @@ function matches(r, q) {
|
|
|
47
49
|
* result (each source is fetched up to `limit`, so the top-N is exact).
|
|
48
50
|
*/
|
|
49
51
|
export function readUnifiedEvents(q = {}) {
|
|
52
|
+
// `bundle` is filtered inside query()'s scan (before its limit cutoff) so a
|
|
53
|
+
// matching-bundle record older than the newest-`limit` window is not dropped.
|
|
50
54
|
const ops = query({
|
|
51
55
|
startDate: q.startDate,
|
|
52
56
|
endDate: q.endDate,
|
|
@@ -57,6 +61,7 @@ export function readUnifiedEvents(q = {}) {
|
|
|
57
61
|
caller: q.caller,
|
|
58
62
|
command: q.command,
|
|
59
63
|
module: q.module,
|
|
64
|
+
bundle: q.bundle,
|
|
60
65
|
limit: q.limit,
|
|
61
66
|
});
|
|
62
67
|
if (q.includeActivity === false)
|
package/dist/lib/events.d.ts
CHANGED
|
@@ -213,6 +213,8 @@ export declare function query(options: {
|
|
|
213
213
|
caller?: string;
|
|
214
214
|
command?: string;
|
|
215
215
|
module?: string;
|
|
216
|
+
/** Only events carrying this bundle name in their payload (e.g. secrets events). */
|
|
217
|
+
bundle?: string;
|
|
216
218
|
limit?: number;
|
|
217
219
|
}): EventRecord[];
|
|
218
220
|
/**
|
package/dist/lib/events.js
CHANGED
|
@@ -729,7 +729,7 @@ export function maybeRotate() {
|
|
|
729
729
|
* @returns Array of event records
|
|
730
730
|
*/
|
|
731
731
|
export function query(options) {
|
|
732
|
-
const { startDate, endDate = new Date(), eventTypes, level, agent, sessionId, caller, command, module, limit } = options;
|
|
732
|
+
const { startDate, endDate = new Date(), eventTypes, level, agent, sessionId, caller, command, module, bundle, limit } = options;
|
|
733
733
|
const results = [];
|
|
734
734
|
if (!fs.existsSync(eventsDir()))
|
|
735
735
|
return results;
|
|
@@ -783,6 +783,11 @@ export function query(options) {
|
|
|
783
783
|
continue;
|
|
784
784
|
if (module && record.module !== module)
|
|
785
785
|
continue;
|
|
786
|
+
// Filter bundle in the SAME scan, before the limit cutoff — a post-filter
|
|
787
|
+
// on the already-capped result silently drops matching-bundle records that
|
|
788
|
+
// fell outside the newest-`limit` window (a data-loss bug for an audit query).
|
|
789
|
+
if (bundle && record.bundle !== bundle)
|
|
790
|
+
continue;
|
|
786
791
|
results.push(record);
|
|
787
792
|
if (limit && results.length >= limit) {
|
|
788
793
|
return results;
|
package/dist/lib/feed.d.ts
CHANGED
|
@@ -218,7 +218,7 @@ export declare function removeBlock(blockId: string, root?: string): boolean;
|
|
|
218
218
|
* Embedded so it ships with the compiled CLI and can be installed to the
|
|
219
219
|
* CLI-writable user hooks dir without a separate file in the npm tarball.
|
|
220
220
|
*/
|
|
221
|
-
export declare const FEED_PUBLISH_HOOK_SCRIPT = "#!/usr/bin/env python3\n\"\"\"Publish and clear open-block records for `agents feed`.\n\nThe manifest invokes this script for top-level AskUserQuestion calls, waiting\nnotifications, question answers, and session lifecycle events. One atomic file\nper session means a new block replaces the previous block. Answer/resume/stop\nevents remove it so `agents feed` only lists decisions that are still open.\n\nSub-agent gate: when the PreToolUse payload carries `agent_type`, this is a\nTask/Agent subagent -- skip. Only the top-level agent publishes. Verified on\nClaude Code 2.1.170 (2026-07).\n\nFail-open: ANY error is swallowed so a feed hiccup never blocks a tool call.\n\"\"\"\nimport os\nimport sys\nimport json\nimport re\nimport socket\nimport tempfile\nfrom datetime import datetime, timezone\n\nWAITING_NOTIFICATION_TYPES = {\n \"permission_prompt\",\n \"idle_prompt\",\n \"elicitation_dialog\",\n}\nCLEAR_EVENTS = {\n \"PostToolUse\",\n \"Stop\",\n \"SessionEnd\",\n}\n# Codex emits a PermissionRequest event (not Claude's Notification) when it\n# blocks on an approval prompt. Claude never fires PermissionRequest, so the\n# same script handles both: PermissionRequest maps to an approval-class block\n# with a high cost-of-delay so 'agents feed --dispatch' pages it as urgent.\n\n\ndef read_json(path):\n try:\n with open(path) as f:\n return json.load(f)\n except Exception:\n return None\n\n\ndef write_json(path, value):\n dir_name = os.path.dirname(path)\n os.makedirs(dir_name, exist_ok=True)\n fd, tmp = tempfile.mkstemp(dir=dir_name, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(value, f, indent=2)\n os.replace(tmp, path)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\ndef main():\n raw = sys.stdin.read()\n try:\n payload = json.loads(raw) if raw.strip() else {}\n except Exception:\n return\n\n # Sub-agent gate.\n if payload.get(\"agent_type\"):\n return\n\n session_id = payload.get(\"session_id\", \"\")\n if not session_id:\n return\n\n safe_session_id = re.sub(r\"[^A-Za-z0-9._-]\", \"-\", session_id)\n block_id = f\"block-{safe_session_id}\"\n home = os.environ.get(\"HOME\") or os.path.expanduser(\"~\")\n feed_dir = os.path.join(home, \".agents\", \".history\", \"feed\")\n answered_dir = os.path.join(feed_dir, \"answered\")\n asks_dir = os.path.join(feed_dir, \"asks\")\n target = os.path.join(feed_dir, f\"{block_id}.json\")\n hook_event = payload.get(\"hook_event_name\", \"PreToolUse\")\n\n if hook_event in CLEAR_EVENTS:\n # A matcher-less PostToolUse clear (registered for Codex so an approved\n # tool clears its approval card) must NOT wipe an open AskUserQuestion\n # while an unrelated tool runs mid-question -- those are cleared only by\n # the AskUserQuestion-matched PostToolUse. So on PostToolUse, keep a\n # 'question' block; approval/notification blocks clear once the tool runs.\n if hook_event == \"PostToolUse\":\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\" and payload.get(\"tool_name\") != \"AskUserQuestion\":\n return\n except Exception:\n pass\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n # Also clear the answered marker so a future question for this session\n # is not permanently locked.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n # Terminal answers (human typed in the TUI) record an answered marker and\n # remove the block file so the feed stops showing it within one poll cycle.\n # The marker stays behind so a concurrent surface cannot double-answer.\n if hook_event == \"UserPromptSubmit\":\n os.makedirs(answered_dir, exist_ok=True)\n marker = os.path.join(answered_dir, f\"{block_id}.json\")\n try:\n fd = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)\n record = {\n \"answeredAt\": datetime.now(timezone.utc).isoformat(),\n \"answeredFrom\": \"terminal\",\n }\n with os.fdopen(fd, \"w\") as f:\n json.dump(record, f, indent=2)\n except FileExistsError:\n pass\n except Exception:\n pass\n # Remove the visible block so the feed drops the answered question.\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n notification_type = None\n codex_approval = False\n if hook_event == \"Notification\":\n notification_type = payload.get(\"notification_type\", \"\")\n if notification_type not in WAITING_NOTIFICATION_TYPES:\n return\n # Claude emits a generic permission notification after presenting an\n # AskUserQuestion. Keep the structured questions and options already\n # published for this session instead of replacing them with that less\n # useful notification text.\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\":\n return\n except Exception:\n pass\n message = payload.get(\"message\", \"\")\n if not message:\n return\n normalized_questions = [{\n \"text\": message,\n \"header\": payload.get(\"title\") or notification_type.replace(\"_\", \" \").title(),\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n elif hook_event == \"PermissionRequest\":\n # Codex approval prompt. The payload mirrors PreToolUse (tool_name,\n # tool_input) but carries no questions -- Codex is asking to run a tool,\n # not asking the operator a multiple-choice question. Publish it as a\n # notification-kind approval block naming the tool so the feed and the\n # phone notifier can surface it, and so the Factory extension can bridge\n # it to a VS Code notification.\n tool_name = payload.get(\"tool_name\") or \"a tool\"\n tool_input = payload.get(\"tool_input\", {})\n command = \"\"\n if isinstance(tool_input, dict):\n command = (\n tool_input.get(\"command\")\n or tool_input.get(\"cmd\")\n or tool_input.get(\"path\")\n or \"\"\n )\n if isinstance(command, list):\n command = \" \".join(str(c) for c in command)\n detail = f\": {command}\" if command else \"\"\n normalized_questions = [{\n \"text\": f\"Codex needs approval to run {tool_name}{detail}\",\n \"header\": \"Approval needed\",\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n notification_type = \"permission_prompt\"\n codex_approval = True\n else:\n tool_input = payload.get(\"tool_input\", {})\n questions = tool_input.get(\"questions\", [])\n if not questions:\n return\n normalized_questions = []\n for q in questions:\n if not isinstance(q, dict):\n continue\n question = {\n \"text\": q.get(\"question\", q.get(\"header\", \"\")),\n \"header\": q.get(\"header\"),\n \"multiSelect\": q.get(\"multiSelect\", False),\n }\n raw_opts = q.get(\"options\", [])\n if raw_opts:\n question[\"options\"] = [\n {\"label\": o.get(\"label\", \"\"), \"description\": o.get(\"description\")}\n for o in raw_opts\n if isinstance(o, dict)\n ]\n normalized_questions.append(question)\n if not normalized_questions:\n return\n kind = \"question\"\n\n # Identity from env (set by agents-cli at spawn).\n mailbox_id = os.path.basename(\n os.environ.get(\"AGENTS_MAILBOX_DIR\", \"\").rstrip(\"/\")\n ) or session_id\n\n now_iso = datetime.now(timezone.utc).isoformat()\n stats_path = os.path.join(asks_dir, f\"{safe_session_id}.json\")\n stats = read_json(stats_path) or {}\n recent = stats.get(\"recentAskTimestamps\") if isinstance(stats, dict) else []\n if not isinstance(recent, list):\n recent = []\n recent.append(now_iso)\n # Keep enough history for rolling one-hour needy detection without unbounded\n # per-session files. The TypeScript reader applies the exact time window.\n recent = recent[-200:]\n write_json(stats_path, {\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"firstAskAt\": stats.get(\"firstAskAt\") or now_iso,\n \"lastAskAt\": now_iso,\n \"totalAskCount\": int(stats.get(\"totalAskCount\") or 0) + 1,\n \"recentAskTimestamps\": recent,\n })\n\n hostname = os.environ.get(\"AGENTS_SYNC_MACHINE_ID\") or socket.gethostname()\n host = hostname.split(\".\")[0].strip().lower()\n host = re.sub(r\"[^a-z0-9_-]\", \"-\", host) or \"unknown\"\n\n runtime = os.environ.get(\"AGENTS_RUNTIME\", \"headless\")\n\n block = {\n \"blockId\": block_id,\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"host\": host,\n \"runtime\": runtime,\n \"ts\": now_iso,\n \"questions\": normalized_questions,\n \"kind\": kind,\n }\n if notification_type:\n block[\"notificationType\"] = notification_type\n\n # A Codex PermissionRequest is a real approval gate: mark it approval-class\n # with a high cost-of-delay so 'agents feed --dispatch' classifies it urgent\n # (isPhoneUrgent gates on costOfDelay >= phoneNotifyThreshold, default\n # 'medium') and pages the phone. A plain 'deny' is the safe default.\n if codex_approval:\n block[\"blockClass\"] = \"approval\"\n block[\"costOfDelay\"] = \"high\"\n block[\"safeDefault\"] = \"deny\"\n\n # Optional multi-operator control metadata passed by the agent in the\n # AskUserQuestion tool_input. Defaults keep the existing behavior. A Codex\n # PermissionRequest carries tool ARGS in tool_input (command/path), not\n # operator controls, so it is excluded here -- its class/cost is stamped\n # above from codex_approval.\n controls = payload.get(\"tool_input\", {}) if hook_event not in (\"Notification\", \"PermissionRequest\") else {}\n block_class = controls.get(\"blockClass\") if isinstance(controls, dict) else None\n if block_class in (\"approval\", \"decision\"):\n block[\"blockClass\"] = block_class\n consequence = controls.get(\"consequence\") if isinstance(controls, dict) else None\n if consequence:\n block[\"consequence\"] = consequence\n allowed = controls.get(\"allowedOperators\") if isinstance(controls, dict) else None\n if isinstance(allowed, list):\n block[\"allowedOperators\"] = [str(a) for a in allowed]\n timeout = controls.get(\"timeoutMinutes\") if isinstance(controls, dict) else None\n if isinstance(timeout, (int, float)) and timeout > 0:\n block[\"timeoutMinutes\"] = int(timeout)\n safe_default = controls.get(\"safeDefault\") if isinstance(controls, dict) else None\n if isinstance(safe_default, str):\n block[\"safeDefault\"] = safe_default\n cost = controls.get(\"costOfDelay\") if isinstance(controls, dict) else None\n if cost in (\"low\", \"medium\", \"high\"):\n block[\"costOfDelay\"] = cost\n\n # Publishing a new question clears any stale answered marker from the\n # previous question in this session.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n\n # Python's expanduser() ignores HOME on Windows, while agents-cli honors a\n # HOME override on every platform. Use the same anchor so hooks and the CLI\n # always read/write one feed store (including temp-home and sandbox runs).\n os.makedirs(feed_dir, exist_ok=True)\n\n fd, tmp = tempfile.mkstemp(dir=feed_dir, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(block, f, indent=2)\n os.replace(tmp, target)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\nif __name__ == \"__main__\":\n try:\n main()\n except Exception:\n pass # fail open\n";
|
|
221
|
+
export declare const FEED_PUBLISH_HOOK_SCRIPT = "#!/usr/bin/env python3\n\"\"\"Publish and clear open-block records for `agents feed`.\n\nThe manifest invokes this script for top-level AskUserQuestion calls, waiting\nnotifications, question answers, and session lifecycle events. One atomic file\nper session means a new block replaces the previous block. Answer/resume/stop\nevents remove it so `agents feed` only lists decisions that are still open.\n\nSub-agent gate: when the PreToolUse payload carries `agent_type`, this is a\nTask/Agent subagent -- skip. Only the top-level agent publishes. Verified on\nClaude Code 2.1.170 (2026-07).\n\nFail-open: ANY error is swallowed so a feed hiccup never blocks a tool call.\n\"\"\"\nimport os\nimport sys\nimport json\nimport re\nimport socket\nimport tempfile\nfrom datetime import datetime, timezone\n\nWAITING_NOTIFICATION_TYPES = {\n \"permission_prompt\",\n \"idle_prompt\",\n \"elicitation_dialog\",\n}\nCLEAR_EVENTS = {\n \"PostToolUse\",\n \"Stop\",\n \"SessionEnd\",\n}\n# Codex emits a PermissionRequest event (not Claude's Notification) when it\n# blocks on an approval prompt. Claude never fires PermissionRequest, so the\n# same script handles both: PermissionRequest maps to an approval-class block\n# with a high cost-of-delay so 'agents feed --dispatch' pages it as urgent.\n\n\ndef read_json(path):\n try:\n with open(path) as f:\n return json.load(f)\n except Exception:\n return None\n\n\ndef write_json(path, value):\n dir_name = os.path.dirname(path)\n os.makedirs(dir_name, exist_ok=True)\n fd, tmp = tempfile.mkstemp(dir=dir_name, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(value, f, indent=2)\n os.replace(tmp, path)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\ndef main():\n raw = sys.stdin.read()\n try:\n payload = json.loads(raw) if raw.strip() else {}\n except Exception:\n return\n\n # Sub-agent gate.\n if payload.get(\"agent_type\"):\n return\n\n session_id = payload.get(\"session_id\", \"\")\n if not session_id:\n return\n\n safe_session_id = re.sub(r\"[^A-Za-z0-9._-]\", \"-\", session_id)\n block_id = f\"block-{safe_session_id}\"\n home = os.environ.get(\"HOME\") or os.path.expanduser(\"~\")\n feed_dir = os.path.join(home, \".agents\", \".history\", \"feed\")\n answered_dir = os.path.join(feed_dir, \"answered\")\n asks_dir = os.path.join(feed_dir, \"asks\")\n target = os.path.join(feed_dir, f\"{block_id}.json\")\n hook_event = payload.get(\"hook_event_name\", \"PreToolUse\")\n\n if hook_event in CLEAR_EVENTS:\n # A declared block (`agents feed post --blocked`) is the agent explicitly\n # saying it is stuck. Unlike a question/notification/approval block -- which\n # tracks an in-flight harness prompt that a lifecycle event resolves -- a\n # declared block stays open until it is actually ANSWERED. So while it is\n # still UNANSWERED, Stop/SessionEnd/PostToolUse must never silently drop it:\n # otherwise the needs-you record vanishes the moment the agent parks the block\n # and its turn ends -- exactly when the owner still needs to see and answer it.\n # Once it IS answered (an answered marker exists), it clears like any other\n # block by falling through below -- which frees that marker too, so a later\n # `--blocked` in the same session is not falsely locked as already-answered\n # (recordAnswer creates the marker with O_EXCL).\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n answered = os.path.exists(os.path.join(answered_dir, f\"{block_id}.json\"))\n if existing.get(\"kind\") == \"declared\" and not answered:\n return\n except Exception:\n pass\n # A matcher-less PostToolUse clear (registered for Codex so an approved\n # tool clears its approval card) must NOT wipe an open AskUserQuestion\n # while an unrelated tool runs mid-question -- those are cleared only by\n # the AskUserQuestion-matched PostToolUse. So on PostToolUse, keep a\n # 'question' block; approval/notification blocks clear once the tool runs.\n if hook_event == \"PostToolUse\":\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\" and payload.get(\"tool_name\") != \"AskUserQuestion\":\n return\n except Exception:\n pass\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n # Also clear the answered marker so a future question for this session\n # is not permanently locked.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n # Terminal answers (human typed in the TUI) record an answered marker and\n # remove the block file so the feed stops showing it within one poll cycle.\n # The marker stays behind so a concurrent surface cannot double-answer.\n if hook_event == \"UserPromptSubmit\":\n os.makedirs(answered_dir, exist_ok=True)\n marker = os.path.join(answered_dir, f\"{block_id}.json\")\n try:\n fd = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)\n record = {\n \"answeredAt\": datetime.now(timezone.utc).isoformat(),\n \"answeredFrom\": \"terminal\",\n }\n with os.fdopen(fd, \"w\") as f:\n json.dump(record, f, indent=2)\n except FileExistsError:\n pass\n except Exception:\n pass\n # Remove the visible block so the feed drops the answered question.\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n notification_type = None\n codex_approval = False\n if hook_event == \"Notification\":\n notification_type = payload.get(\"notification_type\", \"\")\n if notification_type not in WAITING_NOTIFICATION_TYPES:\n return\n # Claude emits a generic permission notification after presenting an\n # AskUserQuestion. Keep the structured questions and options already\n # published for this session instead of replacing them with that less\n # useful notification text.\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\":\n return\n except Exception:\n pass\n message = payload.get(\"message\", \"\")\n if not message:\n return\n normalized_questions = [{\n \"text\": message,\n \"header\": payload.get(\"title\") or notification_type.replace(\"_\", \" \").title(),\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n elif hook_event == \"PermissionRequest\":\n # Codex approval prompt. The payload mirrors PreToolUse (tool_name,\n # tool_input) but carries no questions -- Codex is asking to run a tool,\n # not asking the operator a multiple-choice question. Publish it as a\n # notification-kind approval block naming the tool so the feed and the\n # phone notifier can surface it, and so the Factory extension can bridge\n # it to a VS Code notification.\n tool_name = payload.get(\"tool_name\") or \"a tool\"\n tool_input = payload.get(\"tool_input\", {})\n command = \"\"\n if isinstance(tool_input, dict):\n command = (\n tool_input.get(\"command\")\n or tool_input.get(\"cmd\")\n or tool_input.get(\"path\")\n or \"\"\n )\n if isinstance(command, list):\n command = \" \".join(str(c) for c in command)\n detail = f\": {command}\" if command else \"\"\n normalized_questions = [{\n \"text\": f\"Codex needs approval to run {tool_name}{detail}\",\n \"header\": \"Approval needed\",\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n notification_type = \"permission_prompt\"\n codex_approval = True\n else:\n tool_input = payload.get(\"tool_input\", {})\n questions = tool_input.get(\"questions\", [])\n if not questions:\n return\n normalized_questions = []\n for q in questions:\n if not isinstance(q, dict):\n continue\n question = {\n \"text\": q.get(\"question\", q.get(\"header\", \"\")),\n \"header\": q.get(\"header\"),\n \"multiSelect\": q.get(\"multiSelect\", False),\n }\n raw_opts = q.get(\"options\", [])\n if raw_opts:\n question[\"options\"] = [\n {\"label\": o.get(\"label\", \"\"), \"description\": o.get(\"description\")}\n for o in raw_opts\n if isinstance(o, dict)\n ]\n normalized_questions.append(question)\n if not normalized_questions:\n return\n kind = \"question\"\n\n # Identity from env (set by agents-cli at spawn).\n mailbox_id = os.path.basename(\n os.environ.get(\"AGENTS_MAILBOX_DIR\", \"\").rstrip(\"/\")\n ) or session_id\n\n now_iso = datetime.now(timezone.utc).isoformat()\n stats_path = os.path.join(asks_dir, f\"{safe_session_id}.json\")\n stats = read_json(stats_path) or {}\n recent = stats.get(\"recentAskTimestamps\") if isinstance(stats, dict) else []\n if not isinstance(recent, list):\n recent = []\n recent.append(now_iso)\n # Keep enough history for rolling one-hour needy detection without unbounded\n # per-session files. The TypeScript reader applies the exact time window.\n recent = recent[-200:]\n write_json(stats_path, {\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"firstAskAt\": stats.get(\"firstAskAt\") or now_iso,\n \"lastAskAt\": now_iso,\n \"totalAskCount\": int(stats.get(\"totalAskCount\") or 0) + 1,\n \"recentAskTimestamps\": recent,\n })\n\n hostname = os.environ.get(\"AGENTS_SYNC_MACHINE_ID\") or socket.gethostname()\n host = hostname.split(\".\")[0].strip().lower()\n host = re.sub(r\"[^a-z0-9_-]\", \"-\", host) or \"unknown\"\n\n runtime = os.environ.get(\"AGENTS_RUNTIME\", \"headless\")\n\n block = {\n \"blockId\": block_id,\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"host\": host,\n \"runtime\": runtime,\n \"ts\": now_iso,\n \"questions\": normalized_questions,\n \"kind\": kind,\n }\n if notification_type:\n block[\"notificationType\"] = notification_type\n\n # A Codex PermissionRequest is a real approval gate: mark it approval-class\n # with a high cost-of-delay so 'agents feed --dispatch' classifies it urgent\n # (isPhoneUrgent gates on costOfDelay >= phoneNotifyThreshold, default\n # 'medium') and pages the phone. A plain 'deny' is the safe default.\n if codex_approval:\n block[\"blockClass\"] = \"approval\"\n block[\"costOfDelay\"] = \"high\"\n block[\"safeDefault\"] = \"deny\"\n\n # Optional multi-operator control metadata passed by the agent in the\n # AskUserQuestion tool_input. Defaults keep the existing behavior. A Codex\n # PermissionRequest carries tool ARGS in tool_input (command/path), not\n # operator controls, so it is excluded here -- its class/cost is stamped\n # above from codex_approval.\n controls = payload.get(\"tool_input\", {}) if hook_event not in (\"Notification\", \"PermissionRequest\") else {}\n block_class = controls.get(\"blockClass\") if isinstance(controls, dict) else None\n if block_class in (\"approval\", \"decision\"):\n block[\"blockClass\"] = block_class\n consequence = controls.get(\"consequence\") if isinstance(controls, dict) else None\n if consequence:\n block[\"consequence\"] = consequence\n allowed = controls.get(\"allowedOperators\") if isinstance(controls, dict) else None\n if isinstance(allowed, list):\n block[\"allowedOperators\"] = [str(a) for a in allowed]\n timeout = controls.get(\"timeoutMinutes\") if isinstance(controls, dict) else None\n if isinstance(timeout, (int, float)) and timeout > 0:\n block[\"timeoutMinutes\"] = int(timeout)\n safe_default = controls.get(\"safeDefault\") if isinstance(controls, dict) else None\n if isinstance(safe_default, str):\n block[\"safeDefault\"] = safe_default\n cost = controls.get(\"costOfDelay\") if isinstance(controls, dict) else None\n if cost in (\"low\", \"medium\", \"high\"):\n block[\"costOfDelay\"] = cost\n\n # Publishing a new question clears any stale answered marker from the\n # previous question in this session.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n\n # Python's expanduser() ignores HOME on Windows, while agents-cli honors a\n # HOME override on every platform. Use the same anchor so hooks and the CLI\n # always read/write one feed store (including temp-home and sandbox runs).\n os.makedirs(feed_dir, exist_ok=True)\n\n fd, tmp = tempfile.mkstemp(dir=feed_dir, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(block, f, indent=2)\n os.replace(tmp, target)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\nif __name__ == \"__main__\":\n try:\n main()\n except Exception:\n pass # fail open\n";
|
|
222
222
|
/** Manifest entry for the feed-publish hook, matching the ManifestHook shape. */
|
|
223
223
|
export declare const FEED_PUBLISH_HOOK_MANIFEST: {
|
|
224
224
|
name: string;
|
package/dist/lib/feed.js
CHANGED
|
@@ -442,6 +442,25 @@ def main():
|
|
|
442
442
|
hook_event = payload.get("hook_event_name", "PreToolUse")
|
|
443
443
|
|
|
444
444
|
if hook_event in CLEAR_EVENTS:
|
|
445
|
+
# A declared block (\`agents feed post --blocked\`) is the agent explicitly
|
|
446
|
+
# saying it is stuck. Unlike a question/notification/approval block -- which
|
|
447
|
+
# tracks an in-flight harness prompt that a lifecycle event resolves -- a
|
|
448
|
+
# declared block stays open until it is actually ANSWERED. So while it is
|
|
449
|
+
# still UNANSWERED, Stop/SessionEnd/PostToolUse must never silently drop it:
|
|
450
|
+
# otherwise the needs-you record vanishes the moment the agent parks the block
|
|
451
|
+
# and its turn ends -- exactly when the owner still needs to see and answer it.
|
|
452
|
+
# Once it IS answered (an answered marker exists), it clears like any other
|
|
453
|
+
# block by falling through below -- which frees that marker too, so a later
|
|
454
|
+
# \`--blocked\` in the same session is not falsely locked as already-answered
|
|
455
|
+
# (recordAnswer creates the marker with O_EXCL).
|
|
456
|
+
try:
|
|
457
|
+
with open(target) as existing_file:
|
|
458
|
+
existing = json.load(existing_file)
|
|
459
|
+
answered = os.path.exists(os.path.join(answered_dir, f"{block_id}.json"))
|
|
460
|
+
if existing.get("kind") == "declared" and not answered:
|
|
461
|
+
return
|
|
462
|
+
except Exception:
|
|
463
|
+
pass
|
|
445
464
|
# A matcher-less PostToolUse clear (registered for Codex so an approved
|
|
446
465
|
# tool clears its approval card) must NOT wipe an open AskUserQuestion
|
|
447
466
|
# while an unrelated tool runs mid-question -- those are cleared only by
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Human overrides for the cost-tier -> model mapping.
|
|
3
|
+
*
|
|
4
|
+
* The auto-ranking (lib/model-tiers.ts) is a best guess; for subscription harnesses
|
|
5
|
+
* with no price signal (Kimi, Cursor) it can be wrong. A user pins the right model
|
|
6
|
+
* per tier with `agents models tier set <agent[@version]> <tier> <model>`, which
|
|
7
|
+
* writes here — the user never hand-edits the file.
|
|
8
|
+
*
|
|
9
|
+
* Stored under agents.yaml, same selector shape as run.defaults:
|
|
10
|
+
*
|
|
11
|
+
* model:
|
|
12
|
+
* tiers:
|
|
13
|
+
* "kimi:*": # applies to every installed kimi
|
|
14
|
+
* best: kimi-code/k3
|
|
15
|
+
* default: kimi-code/kimi-for-coding
|
|
16
|
+
* "kimi:0.19.2": # a specific version wins over the wildcard
|
|
17
|
+
* best: kimi-code/k3-256k
|
|
18
|
+
*
|
|
19
|
+
* Resolution (most-specific-first): `<agent>:<version>` beats `<agent>:*` beats the
|
|
20
|
+
* auto-ranking. resolveTierMap (model-tiers.ts) applies the result and falls back to
|
|
21
|
+
* auto for any tier whose overridden id isn't in that version's catalog.
|
|
22
|
+
*/
|
|
23
|
+
import type { AgentId } from './types.js';
|
|
24
|
+
import { type ModelTier } from './model-tiers.js';
|
|
25
|
+
export type TierOverrideMap = Partial<Record<ModelTier, string>>;
|
|
26
|
+
export interface TierOverrideEntry {
|
|
27
|
+
selector: string;
|
|
28
|
+
tiers: TierOverrideMap;
|
|
29
|
+
}
|
|
30
|
+
/** Validate a tier token, throwing a friendly error otherwise. */
|
|
31
|
+
export declare function parseTier(input: string): ModelTier;
|
|
32
|
+
/**
|
|
33
|
+
* The effective tier overrides for an (agent, version): the `<agent>:*` wildcard
|
|
34
|
+
* merged under the exact `<agent>:<version>` selector (exact wins per-tier).
|
|
35
|
+
*/
|
|
36
|
+
export declare function resolveTierOverrideFrom(all: Record<string, unknown>, agent: AgentId, version?: string | null): TierOverrideMap;
|
|
37
|
+
export declare function resolveTierOverride(agent: AgentId, version?: string | null): TierOverrideMap;
|
|
38
|
+
/** Every configured override entry, sorted by selector (for `agents models tier list`). */
|
|
39
|
+
export declare function listTierOverrides(): TierOverrideEntry[];
|
|
40
|
+
/** Pin `tier -> model` for a selector. Writes agents.yaml. */
|
|
41
|
+
export declare function setTierOverride(selectorInput: string, tierInput: string, model: string): TierOverrideEntry;
|
|
42
|
+
/** Clear one tier (or all tiers when `tierInput` is omitted) for a selector. Returns true if anything changed. */
|
|
43
|
+
export declare function clearTierOverride(selectorInput: string, tierInput?: string): boolean;
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
import { readMeta, updateMeta } from './state.js';
|
|
2
|
+
import { parseRunDefaultSelector } from './run-defaults.js';
|
|
3
|
+
import { MODEL_TIERS } from './model-tiers.js';
|
|
4
|
+
function isTier(value) {
|
|
5
|
+
return MODEL_TIERS.includes(value);
|
|
6
|
+
}
|
|
7
|
+
/** Validate a tier token, throwing a friendly error otherwise. */
|
|
8
|
+
export function parseTier(input) {
|
|
9
|
+
const t = input.trim().toLowerCase();
|
|
10
|
+
if (!isTier(t)) {
|
|
11
|
+
throw new Error(`Invalid tier '${input}'. Use one of: ${MODEL_TIERS.join(', ')}.`);
|
|
12
|
+
}
|
|
13
|
+
return t;
|
|
14
|
+
}
|
|
15
|
+
/** Normalize a stored selector's tier map, dropping unknown/empty entries. */
|
|
16
|
+
function normalize(raw) {
|
|
17
|
+
const out = {};
|
|
18
|
+
if (!raw || typeof raw !== 'object')
|
|
19
|
+
return out;
|
|
20
|
+
for (const [k, v] of Object.entries(raw)) {
|
|
21
|
+
if (isTier(k) && typeof v === 'string' && v.trim())
|
|
22
|
+
out[k] = v.trim();
|
|
23
|
+
}
|
|
24
|
+
return out;
|
|
25
|
+
}
|
|
26
|
+
function sortedSelectors(map) {
|
|
27
|
+
return Object.fromEntries(Object.entries(map).sort(([a], [b]) => a.localeCompare(b)));
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* The effective tier overrides for an (agent, version): the `<agent>:*` wildcard
|
|
31
|
+
* merged under the exact `<agent>:<version>` selector (exact wins per-tier).
|
|
32
|
+
*/
|
|
33
|
+
export function resolveTierOverrideFrom(all, agent, version) {
|
|
34
|
+
const merged = { ...normalize(all[`${agent}:*`]) };
|
|
35
|
+
if (version) {
|
|
36
|
+
for (const [tier, model] of Object.entries(normalize(all[`${agent}:${version}`]))) {
|
|
37
|
+
merged[tier] = model;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return merged;
|
|
41
|
+
}
|
|
42
|
+
export function resolveTierOverride(agent, version) {
|
|
43
|
+
return resolveTierOverrideFrom(readMeta().model?.tiers ?? {}, agent, version);
|
|
44
|
+
}
|
|
45
|
+
/** Every configured override entry, sorted by selector (for `agents models tier list`). */
|
|
46
|
+
export function listTierOverrides() {
|
|
47
|
+
const all = readMeta().model?.tiers ?? {};
|
|
48
|
+
return Object.entries(all)
|
|
49
|
+
.sort(([a], [b]) => a.localeCompare(b))
|
|
50
|
+
.map(([selector, tiers]) => ({ selector, tiers: normalize(tiers) }));
|
|
51
|
+
}
|
|
52
|
+
/** Pin `tier -> model` for a selector. Writes agents.yaml. */
|
|
53
|
+
export function setTierOverride(selectorInput, tierInput, model) {
|
|
54
|
+
const parsed = parseRunDefaultSelector(selectorInput);
|
|
55
|
+
const tier = parseTier(tierInput);
|
|
56
|
+
const id = model.trim();
|
|
57
|
+
if (!id)
|
|
58
|
+
throw new Error('A model id is required.');
|
|
59
|
+
updateMeta((meta) => {
|
|
60
|
+
const modelCfg = { ...(meta.model ?? {}) };
|
|
61
|
+
const tiers = { ...(modelCfg.tiers ?? {}) };
|
|
62
|
+
tiers[parsed.selector] = { ...(tiers[parsed.selector] ?? {}), [tier]: id };
|
|
63
|
+
modelCfg.tiers = sortedSelectors(tiers);
|
|
64
|
+
return { ...meta, model: modelCfg };
|
|
65
|
+
});
|
|
66
|
+
return { selector: parsed.selector, tiers: normalize(readMeta().model?.tiers?.[parsed.selector]) };
|
|
67
|
+
}
|
|
68
|
+
/** Clear one tier (or all tiers when `tierInput` is omitted) for a selector. Returns true if anything changed. */
|
|
69
|
+
export function clearTierOverride(selectorInput, tierInput) {
|
|
70
|
+
const parsed = parseRunDefaultSelector(selectorInput);
|
|
71
|
+
const tier = tierInput ? parseTier(tierInput) : null;
|
|
72
|
+
let changed = false;
|
|
73
|
+
updateMeta((meta) => {
|
|
74
|
+
if (!meta.model?.tiers?.[parsed.selector])
|
|
75
|
+
return meta;
|
|
76
|
+
const model = { ...meta.model };
|
|
77
|
+
const tiers = { ...(model.tiers ?? {}) };
|
|
78
|
+
if (tier) {
|
|
79
|
+
const entry = { ...tiers[parsed.selector] };
|
|
80
|
+
if (entry[tier] !== undefined) {
|
|
81
|
+
delete entry[tier];
|
|
82
|
+
changed = true;
|
|
83
|
+
}
|
|
84
|
+
if (Object.keys(entry).length === 0)
|
|
85
|
+
delete tiers[parsed.selector];
|
|
86
|
+
else
|
|
87
|
+
tiers[parsed.selector] = entry;
|
|
88
|
+
}
|
|
89
|
+
else {
|
|
90
|
+
delete tiers[parsed.selector];
|
|
91
|
+
changed = true;
|
|
92
|
+
}
|
|
93
|
+
model.tiers = tiers;
|
|
94
|
+
return { ...meta, model };
|
|
95
|
+
});
|
|
96
|
+
return changed;
|
|
97
|
+
}
|
|
@@ -36,18 +36,24 @@ export interface TierResolution {
|
|
|
36
36
|
clampedFrom?: ModelTier;
|
|
37
37
|
/** Human note (e.g. why it clamped, or that it is a curated/subscription mapping). */
|
|
38
38
|
note?: string;
|
|
39
|
+
/** Where the model came from: 'auto' ranking, a user 'override', or a 'curated' ladder. */
|
|
40
|
+
source?: 'auto' | 'override' | 'curated';
|
|
39
41
|
}
|
|
40
42
|
/**
|
|
41
|
-
* Resolve all four tiers for an (agent, version)
|
|
42
|
-
*
|
|
43
|
+
* Resolve all four tiers for an (agent, version) -- what `agents models` prints and
|
|
44
|
+
* `resolveTier` indexes. Precedence: user override -> curated ladder / auto-ranking.
|
|
43
45
|
*/
|
|
44
46
|
export declare function resolveTierMap(agent: AgentId, version: string): Record<ModelTier, TierResolution>;
|
|
45
47
|
/**
|
|
46
|
-
*
|
|
47
|
-
* so it is directly testable
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
|
|
48
|
+
* Apply user overrides on top of the auto/curated map. Pure (takes the resolved
|
|
49
|
+
* override map, no config lookup) so it is directly testable. An overridden id is
|
|
50
|
+
* used only when the version actually ships it (or when there is no catalog to
|
|
51
|
+
* check, e.g. Droid); otherwise the tier keeps its base value with a note.
|
|
52
|
+
*/
|
|
53
|
+
export declare function applyTierOverrides(overrides: Partial<Record<ModelTier, string>>, label: string, catalogIds: Set<string> | null, base: Record<ModelTier, TierResolution>): Record<ModelTier, TierResolution>;
|
|
54
|
+
/**
|
|
55
|
+
* Map a harness's catalog models onto the four tiers. Pure (no catalog lookup) so
|
|
56
|
+
* it is directly testable. A single-model harness maps the tiers to reasoning effort.
|
|
51
57
|
*/
|
|
52
58
|
export declare function tierizeModels(agent: AgentId, models: ModelInfo[]): Record<ModelTier, TierResolution>;
|
|
53
59
|
/** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
|
package/dist/lib/model-tiers.js
CHANGED
|
@@ -1,11 +1,27 @@
|
|
|
1
1
|
import { getModelCatalog } from './models.js';
|
|
2
2
|
import { getModelPricing } from './pricing/index.js';
|
|
3
|
+
import { resolveTierOverride } from './model-tier-overrides.js';
|
|
3
4
|
/** The four cross-harness cost tiers, cheapest -> most capable. */
|
|
4
5
|
export const MODEL_TIERS = ['cheap', 'default', 'best', 'ultra'];
|
|
5
6
|
/** True if `s` is one of the four tier tokens (not a concrete model id). */
|
|
6
7
|
export function isTierToken(s) {
|
|
7
8
|
return !!s && MODEL_TIERS.includes(s);
|
|
8
9
|
}
|
|
10
|
+
/**
|
|
11
|
+
* Curated tier ladders for harnesses the auto-ranker can't order from names/price
|
|
12
|
+
* (subscription harnesses with no price signal). Each rung is `[tier, matcher]` in
|
|
13
|
+
* cheap -> best order; the newest catalog id matching each rung fills that tier,
|
|
14
|
+
* missing tiers clamp. Extend this table rather than adding per-harness branches.
|
|
15
|
+
*/
|
|
16
|
+
const CURATED_LADDERS = {
|
|
17
|
+
// Kimi: K2.7 Highspeed < K2.7 Coding < K3 (the 1M-context default; k3-256k folds
|
|
18
|
+
// into K3). No ultra. The name heuristic can't tell K3 > K2.7, so curate it.
|
|
19
|
+
kimi: [
|
|
20
|
+
{ tier: 'cheap', match: /highspeed/i },
|
|
21
|
+
{ tier: 'default', match: /for-coding(?!.*highspeed)/i },
|
|
22
|
+
{ tier: 'best', match: /(^|[-/])k3\b/i }, // K3 family incl. k3-256k; the plain id represents it
|
|
23
|
+
],
|
|
24
|
+
};
|
|
9
25
|
// --- single-model harnesses: the tier is reasoning effort, not a model ---------
|
|
10
26
|
const TIER_EFFORT = {
|
|
11
27
|
cheap: 'low',
|
|
@@ -169,29 +185,100 @@ function rankCatalog(agent, models) {
|
|
|
169
185
|
function rungIndexFor(tierIndex, n) {
|
|
170
186
|
return n >= 4 ? Math.round((tierIndex / 3) * (n - 1)) : Math.min(tierIndex, n - 1);
|
|
171
187
|
}
|
|
188
|
+
/** Bucket an ordered (cheap -> dear) rung list onto the four tiers, clamping when < 4. */
|
|
189
|
+
function bucketRungs(rungs) {
|
|
190
|
+
const n = rungs.length;
|
|
191
|
+
const map = {};
|
|
192
|
+
if (n === 0) {
|
|
193
|
+
// Fail-safe: no catalog -> every tier null, caller drops the --model flag.
|
|
194
|
+
for (const t of MODEL_TIERS)
|
|
195
|
+
map[t] = { tier: t, model: null };
|
|
196
|
+
return map;
|
|
197
|
+
}
|
|
198
|
+
// A tier that shares the rung of the tier below has no distinct rung -> mark clamped.
|
|
199
|
+
for (let i = 0; i < MODEL_TIERS.length; i++) {
|
|
200
|
+
const t = MODEL_TIERS[i];
|
|
201
|
+
const idx = rungIndexFor(i, n);
|
|
202
|
+
const shared = i > 0 && rungIndexFor(i - 1, n) === idx;
|
|
203
|
+
map[t] = shared
|
|
204
|
+
? { tier: t, model: rungs[idx].id, clampedFrom: MODEL_TIERS[i - 1], note: `no distinct ${t} rung; using ${MODEL_TIERS[i - 1]}`, source: 'auto' }
|
|
205
|
+
: { tier: t, model: rungs[idx].id, source: 'auto' };
|
|
206
|
+
}
|
|
207
|
+
return map;
|
|
208
|
+
}
|
|
209
|
+
/** Build a tier map from a curated ladder against a catalog (newest match per rung). */
|
|
210
|
+
function tierizeFromLadder(ladder, models) {
|
|
211
|
+
const usable = models.filter((m) => !PSEUDO.test(m.id));
|
|
212
|
+
const rungs = [];
|
|
213
|
+
for (const rung of ladder) {
|
|
214
|
+
const matches = usable.filter((m) => rung.match.test(m.id));
|
|
215
|
+
if (matches.length === 0)
|
|
216
|
+
continue;
|
|
217
|
+
// Prefer a plain id over a context-size variant (k3 over k3-256k) -- the
|
|
218
|
+
// variant folds into the rung but the plain model represents it -- then newest.
|
|
219
|
+
const plain = matches.filter((m) => !/-\d+[km]\b/i.test(m.id));
|
|
220
|
+
const pool = plain.length ? plain : matches;
|
|
221
|
+
rungs.push({ id: pool.reduce((a, b) => (newer(b.id, a.id) > 0 ? b : a)).id });
|
|
222
|
+
}
|
|
223
|
+
const map = bucketRungs(rungs);
|
|
224
|
+
for (const t of MODEL_TIERS)
|
|
225
|
+
if (map[t].model)
|
|
226
|
+
map[t].source = 'curated';
|
|
227
|
+
return map;
|
|
228
|
+
}
|
|
172
229
|
/**
|
|
173
|
-
* Resolve all four tiers for an (agent, version)
|
|
174
|
-
*
|
|
230
|
+
* Resolve all four tiers for an (agent, version) -- what `agents models` prints and
|
|
231
|
+
* `resolveTier` indexes. Precedence: user override -> curated ladder / auto-ranking.
|
|
175
232
|
*/
|
|
176
233
|
export function resolveTierMap(agent, version) {
|
|
177
|
-
|
|
234
|
+
let base;
|
|
235
|
+
let catalogIds;
|
|
178
236
|
if (agent === 'droid') {
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
237
|
+
// Droid: curated credit-multiplier map (no live catalog to validate against).
|
|
238
|
+
base = {
|
|
239
|
+
cheap: { tier: 'cheap', model: DROID_TIERS.cheap, note: 'Droid Core 0.55x', source: 'curated' },
|
|
240
|
+
default: { tier: 'default', model: DROID_TIERS.default, note: 'Droid Core 0.6x', source: 'curated' },
|
|
241
|
+
best: { tier: 'best', model: DROID_TIERS.best, note: '2x', source: 'curated' },
|
|
242
|
+
ultra: { tier: 'ultra', model: DROID_TIERS.ultra, clampedFrom: 'best', note: 'capped at 2x (4x excluded)', source: 'curated' },
|
|
184
243
|
};
|
|
244
|
+
catalogIds = null;
|
|
185
245
|
}
|
|
186
|
-
|
|
187
|
-
|
|
246
|
+
else {
|
|
247
|
+
const catalog = getModelCatalog(agent, version);
|
|
248
|
+
const models = catalog?.models ?? [];
|
|
249
|
+
const ladder = CURATED_LADDERS[agent];
|
|
250
|
+
base = ladder ? tierizeFromLadder(ladder, models) : tierizeModels(agent, models);
|
|
251
|
+
catalogIds = catalog ? new Set(models.map((m) => m.id)) : null;
|
|
252
|
+
}
|
|
253
|
+
const overrides = resolveTierOverride(agent, version);
|
|
254
|
+
return applyTierOverrides(overrides, `${agent}@${version}`, catalogIds, base);
|
|
255
|
+
}
|
|
256
|
+
/**
|
|
257
|
+
* Apply user overrides on top of the auto/curated map. Pure (takes the resolved
|
|
258
|
+
* override map, no config lookup) so it is directly testable. An overridden id is
|
|
259
|
+
* used only when the version actually ships it (or when there is no catalog to
|
|
260
|
+
* check, e.g. Droid); otherwise the tier keeps its base value with a note.
|
|
261
|
+
*/
|
|
262
|
+
export function applyTierOverrides(overrides, label, catalogIds, base) {
|
|
263
|
+
if (Object.keys(overrides).length === 0)
|
|
264
|
+
return base;
|
|
265
|
+
const out = { ...base };
|
|
266
|
+
for (const t of MODEL_TIERS) {
|
|
267
|
+
const id = overrides[t];
|
|
268
|
+
if (!id)
|
|
269
|
+
continue;
|
|
270
|
+
if (!catalogIds || catalogIds.has(id)) {
|
|
271
|
+
out[t] = { tier: t, model: id, source: 'override' };
|
|
272
|
+
}
|
|
273
|
+
else {
|
|
274
|
+
out[t] = { ...base[t], note: `override "${id}" not shipped by ${label}; kept the ${base[t].source ?? 'auto'} pick`, source: base[t].source ?? 'auto' };
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
return out;
|
|
188
278
|
}
|
|
189
279
|
/**
|
|
190
|
-
* Map a harness's catalog models onto the four tiers. Pure (no catalog lookup)
|
|
191
|
-
*
|
|
192
|
-
* variants, buckets onto cheap/default/best/ultra, and clamps absent tiers down
|
|
193
|
-
* to the nearest lower one. A single-model harness maps the tiers to reasoning
|
|
194
|
-
* effort instead of models.
|
|
280
|
+
* Map a harness's catalog models onto the four tiers. Pure (no catalog lookup) so
|
|
281
|
+
* it is directly testable. A single-model harness maps the tiers to reasoning effort.
|
|
195
282
|
*/
|
|
196
283
|
export function tierizeModels(agent, models) {
|
|
197
284
|
const rungs = rankCatalog(agent, models);
|
|
@@ -200,28 +287,10 @@ export function tierizeModels(agent, models) {
|
|
|
200
287
|
const only = rungs[0].id;
|
|
201
288
|
const map = {};
|
|
202
289
|
for (const t of MODEL_TIERS)
|
|
203
|
-
map[t] = { tier: t, model: only, effort: TIER_EFFORT[t], note: 'single model — tier maps to reasoning effort' };
|
|
204
|
-
return map;
|
|
205
|
-
}
|
|
206
|
-
const n = rungs.length;
|
|
207
|
-
const map = {};
|
|
208
|
-
if (n === 0) {
|
|
209
|
-
// Fail-safe: no catalog -> every tier null, caller drops the --model flag.
|
|
210
|
-
for (const t of MODEL_TIERS)
|
|
211
|
-
map[t] = { tier: t, model: null };
|
|
290
|
+
map[t] = { tier: t, model: only, effort: TIER_EFFORT[t], note: 'single model — tier maps to reasoning effort', source: 'auto' };
|
|
212
291
|
return map;
|
|
213
292
|
}
|
|
214
|
-
|
|
215
|
-
// has no distinct rung of its own, so mark it clamped for an honest display.
|
|
216
|
-
for (let i = 0; i < MODEL_TIERS.length; i++) {
|
|
217
|
-
const t = MODEL_TIERS[i];
|
|
218
|
-
const idx = rungIndexFor(i, n);
|
|
219
|
-
const shared = i > 0 && rungIndexFor(i - 1, n) === idx;
|
|
220
|
-
map[t] = shared
|
|
221
|
-
? { tier: t, model: rungs[idx].id, clampedFrom: MODEL_TIERS[i - 1], note: `no distinct ${t} rung; using ${MODEL_TIERS[i - 1]}` }
|
|
222
|
-
: { tier: t, model: rungs[idx].id };
|
|
223
|
-
}
|
|
224
|
-
return map;
|
|
293
|
+
return bucketRungs(rungs);
|
|
225
294
|
}
|
|
226
295
|
/** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
|
|
227
296
|
export function resolveTier(agent, version, tier) {
|
package/dist/lib/models.d.ts
CHANGED
|
@@ -60,6 +60,22 @@ export interface ModelSource {
|
|
|
60
60
|
* Returns null if nothing usable is found.
|
|
61
61
|
*/
|
|
62
62
|
export declare function locateModelSource(agent: AgentId, version: string): ModelSource | null;
|
|
63
|
+
/**
|
|
64
|
+
* Extract Claude's model catalog from its bundle/binary.
|
|
65
|
+
*
|
|
66
|
+
* Bundle/binary contains:
|
|
67
|
+
* - alias map: {opus:"claude-opus-4-7",sonnet:"claude-sonnet-4-6",haiku:"..."}
|
|
68
|
+
* - per-cloud maps: {firstParty:"claude-opus-4-5-...",bedrock:"...",vertex:"...",...}
|
|
69
|
+
* - constants: {OPUS_ID:"...",OPUS_NAME:"...",SONNET_ID:"...",...}
|
|
70
|
+
*/
|
|
71
|
+
/**
|
|
72
|
+
* Drop a bare `claude-<family>-<major>` (e.g. `claude-opus-4`) when a more specific
|
|
73
|
+
* sibling (`claude-opus-4-8`) is present. The bare form is only ever an internal
|
|
74
|
+
* `.includes("claude-opus-4")` prefix-check string in the binary, not a submittable
|
|
75
|
+
* id (issue #1892); a bare id with no sibling (e.g. `claude-sonnet-5`) is a real
|
|
76
|
+
* current model and is kept.
|
|
77
|
+
*/
|
|
78
|
+
export declare function dropBareLegacyIds(ids: string[]): string[];
|
|
63
79
|
/**
|
|
64
80
|
* Parse `grok models` stdout into a catalog. Exported for unit tests.
|
|
65
81
|
*
|
package/dist/lib/models.js
CHANGED
|
@@ -21,7 +21,7 @@ const CACHE_PATH = getModelsCachePath();
|
|
|
21
21
|
* Bump when the extractor logic changes shape in an incompatible way so cached
|
|
22
22
|
* catalogs from older agents-cli builds are re-extracted.
|
|
23
23
|
*/
|
|
24
|
-
const CACHE_SCHEMA_VERSION =
|
|
24
|
+
const CACHE_SCHEMA_VERSION = 4;
|
|
25
25
|
/**
|
|
26
26
|
* How long a cached 0-model extraction is trusted before we retry it. Bounds
|
|
27
27
|
* the self-healing window for a transient failure (mid-install, a broken
|
|
@@ -310,6 +310,19 @@ function extractStrings(filePath, minLen = 6) {
|
|
|
310
310
|
* - per-cloud maps: {firstParty:"claude-opus-4-5-...",bedrock:"...",vertex:"...",...}
|
|
311
311
|
* - constants: {OPUS_ID:"...",OPUS_NAME:"...",SONNET_ID:"...",...}
|
|
312
312
|
*/
|
|
313
|
+
/**
|
|
314
|
+
* Drop a bare `claude-<family>-<major>` (e.g. `claude-opus-4`) when a more specific
|
|
315
|
+
* sibling (`claude-opus-4-8`) is present. The bare form is only ever an internal
|
|
316
|
+
* `.includes("claude-opus-4")` prefix-check string in the binary, not a submittable
|
|
317
|
+
* id (issue #1892); a bare id with no sibling (e.g. `claude-sonnet-5`) is a real
|
|
318
|
+
* current model and is kept.
|
|
319
|
+
*/
|
|
320
|
+
export function dropBareLegacyIds(ids) {
|
|
321
|
+
return ids.filter((id) => {
|
|
322
|
+
const bareMajor = /^claude-[a-z]+-\d+$/.test(id);
|
|
323
|
+
return !(bareMajor && ids.some((o) => o !== id && o.startsWith(`${id}-`)));
|
|
324
|
+
});
|
|
325
|
+
}
|
|
313
326
|
function extractClaudeCatalog(text) {
|
|
314
327
|
const aliases = {};
|
|
315
328
|
const aliasMapMatch = text.match(/\{opus:"(claude-[^"]+)",sonnet:"(claude-[^"]+)",haiku:"(claude-[^"]+)"\}/);
|
|
@@ -380,8 +393,9 @@ function extractClaudeCatalog(text) {
|
|
|
380
393
|
let sm;
|
|
381
394
|
while ((sm = idRe.exec(text)) !== null)
|
|
382
395
|
scanned.add(sm[0]);
|
|
383
|
-
|
|
384
|
-
|
|
396
|
+
const filtered = dropBareLegacyIds([...scanned]);
|
|
397
|
+
if (filtered.length >= 2)
|
|
398
|
+
models = build(filtered);
|
|
385
399
|
}
|
|
386
400
|
return { models, aliases };
|
|
387
401
|
}
|
|
Binary file
|
|
Binary file
|
|
@@ -91,6 +91,21 @@ export declare function setKeychainBackendForTest(b: KeychainBackend | null): Ke
|
|
|
91
91
|
* fast-path must not engage. Always false in production (`backend` is null). */
|
|
92
92
|
export declare function isKeychainBackendOverridden(): boolean;
|
|
93
93
|
export declare const HMAC_KEY_ITEM = "agents-cli.hmackey";
|
|
94
|
+
interface HmacKeyRecord {
|
|
95
|
+
v: number;
|
|
96
|
+
/** 64 hex chars — the raw HMAC-SHA256 key. */
|
|
97
|
+
k: string;
|
|
98
|
+
/** True once the one-time re-key has moved every cleartext-named item. */
|
|
99
|
+
migrated: boolean;
|
|
100
|
+
/** Old cleartext services whose hashed copies are verified but whose
|
|
101
|
+
* originals are not yet deleted (crash-resume list; deletes are silent). */
|
|
102
|
+
pendingDeletes?: string[];
|
|
103
|
+
/** True once this record has been re-stored no-ACL to heal a hmackey item that
|
|
104
|
+
* an OLD helper (pre the metadata/hmackey no-ACL migration fix) re-stamped with
|
|
105
|
+
* a biometry ACL. Set on the first read that heals it, so the heal runs exactly
|
|
106
|
+
* once per machine and never churns the keychain afterward. */
|
|
107
|
+
healedNoAcl?: boolean;
|
|
108
|
+
}
|
|
94
109
|
/** Force hashed service names on with a fixed key (test only). Pass null to
|
|
95
110
|
* restore lazy production resolution. Composes with setKeychainBackendForTest
|
|
96
111
|
* so unit tests exercise the exact transform production uses. */
|
|
@@ -106,6 +121,20 @@ export declare function withRawKeychainServiceNames<T>(fn: () => T): T;
|
|
|
106
121
|
* the re-key migration and tests; runtime callers go through the primitives,
|
|
107
122
|
* which apply this transparently. */
|
|
108
123
|
export declare function hashedServiceName(item: string, key: Buffer): string;
|
|
124
|
+
/**
|
|
125
|
+
* Heal a `hmackey` item that an OLD helper (pre the metadata/hmackey no-ACL
|
|
126
|
+
* migration fix) re-stamped with a biometry ACL. Such an item makes EVERY hashed
|
|
127
|
+
* keychain lookup pop the generic "Agents CLI needs to authenticate" sheet,
|
|
128
|
+
* because the HMAC key is read before every hashed name resolves. The migration
|
|
129
|
+
* fix stopped the re-stamping but never un-stamped an already-damaged item, and
|
|
130
|
+
* nothing else re-stores it once hashing is already active — so it prompts forever.
|
|
131
|
+
*
|
|
132
|
+
* This re-stores the record no-ACL exactly once per machine (guarded by
|
|
133
|
+
* `healedNoAcl`), turning every future read silent. The read that produced `rec`
|
|
134
|
+
* has already happened (and already prompted if it was ACL'd); this only writes.
|
|
135
|
+
* Returns true if it healed. Exported for tests. No-op when already healed.
|
|
136
|
+
*/
|
|
137
|
+
export declare function healHmacKeyNoAclOnce(rec: HmacKeyRecord): boolean;
|
|
109
138
|
/**
|
|
110
139
|
* The storage-layer service name for `item`: hashed when hashing is active,
|
|
111
140
|
* the item itself otherwise. For callers that mix helper-enumerated
|