@phnx-labs/agents-cli 1.22.4 → 1.22.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/dist/bin/agents +0 -0
  3. package/dist/commands/events.js +8 -0
  4. package/dist/commands/exec.js +7 -5
  5. package/dist/commands/inspect.js +18 -5
  6. package/dist/commands/models.d.ts +1 -1
  7. package/dist/commands/models.js +68 -9
  8. package/dist/commands/view.d.ts +12 -0
  9. package/dist/commands/view.js +2 -0
  10. package/dist/commands/webhook.js +8 -4
  11. package/dist/lib/browser/chrome.d.ts +12 -0
  12. package/dist/lib/browser/chrome.js +27 -11
  13. package/dist/lib/cloud/antigravity.js +8 -2
  14. package/dist/lib/crabbox/cli.js +72 -30
  15. package/dist/lib/event-stream.d.ts +3 -0
  16. package/dist/lib/event-stream.js +5 -0
  17. package/dist/lib/events.d.ts +2 -0
  18. package/dist/lib/events.js +6 -1
  19. package/dist/lib/feed.d.ts +1 -1
  20. package/dist/lib/feed.js +19 -0
  21. package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
  22. package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
  23. package/dist/lib/model-tier-overrides.d.ts +43 -0
  24. package/dist/lib/model-tier-overrides.js +97 -0
  25. package/dist/lib/model-tiers.d.ts +13 -7
  26. package/dist/lib/model-tiers.js +104 -35
  27. package/dist/lib/models.d.ts +16 -0
  28. package/dist/lib/models.js +17 -3
  29. package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
  30. package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
  31. package/dist/lib/secrets/index.d.ts +29 -0
  32. package/dist/lib/secrets/index.js +32 -0
  33. package/dist/lib/secrets/mcp.js +8 -4
  34. package/dist/lib/session/sync/config.d.ts +12 -11
  35. package/dist/lib/session/sync/config.js +40 -38
  36. package/dist/lib/share/config.js +4 -0
  37. package/dist/lib/types.d.ts +10 -0
  38. package/dist/lib/versions.js +35 -0
  39. package/package.json +1 -1
@@ -31,6 +31,8 @@ function matches(r, q) {
31
31
  return false;
32
32
  if (q.sessionId && r.sessionId !== q.sessionId)
33
33
  return false;
34
+ if (q.bundle && r.bundle !== q.bundle)
35
+ return false;
34
36
  if (q.caller && r.caller !== q.caller)
35
37
  return false;
36
38
  if (q.command && r.command !== q.command &&
@@ -47,6 +49,8 @@ function matches(r, q) {
47
49
  * result (each source is fetched up to `limit`, so the top-N is exact).
48
50
  */
49
51
  export function readUnifiedEvents(q = {}) {
52
+ // `bundle` is filtered inside query()'s scan (before its limit cutoff) so a
53
+ // matching-bundle record older than the newest-`limit` window is not dropped.
50
54
  const ops = query({
51
55
  startDate: q.startDate,
52
56
  endDate: q.endDate,
@@ -57,6 +61,7 @@ export function readUnifiedEvents(q = {}) {
57
61
  caller: q.caller,
58
62
  command: q.command,
59
63
  module: q.module,
64
+ bundle: q.bundle,
60
65
  limit: q.limit,
61
66
  });
62
67
  if (q.includeActivity === false)
@@ -213,6 +213,8 @@ export declare function query(options: {
213
213
  caller?: string;
214
214
  command?: string;
215
215
  module?: string;
216
+ /** Only events carrying this bundle name in their payload (e.g. secrets events). */
217
+ bundle?: string;
216
218
  limit?: number;
217
219
  }): EventRecord[];
218
220
  /**
@@ -729,7 +729,7 @@ export function maybeRotate() {
729
729
  * @returns Array of event records
730
730
  */
731
731
  export function query(options) {
732
- const { startDate, endDate = new Date(), eventTypes, level, agent, sessionId, caller, command, module, limit } = options;
732
+ const { startDate, endDate = new Date(), eventTypes, level, agent, sessionId, caller, command, module, bundle, limit } = options;
733
733
  const results = [];
734
734
  if (!fs.existsSync(eventsDir()))
735
735
  return results;
@@ -783,6 +783,11 @@ export function query(options) {
783
783
  continue;
784
784
  if (module && record.module !== module)
785
785
  continue;
786
+ // Filter bundle in the SAME scan, before the limit cutoff — a post-filter
787
+ // on the already-capped result silently drops matching-bundle records that
788
+ // fell outside the newest-`limit` window (a data-loss bug for an audit query).
789
+ if (bundle && record.bundle !== bundle)
790
+ continue;
786
791
  results.push(record);
787
792
  if (limit && results.length >= limit) {
788
793
  return results;
@@ -218,7 +218,7 @@ export declare function removeBlock(blockId: string, root?: string): boolean;
218
218
  * Embedded so it ships with the compiled CLI and can be installed to the
219
219
  * CLI-writable user hooks dir without a separate file in the npm tarball.
220
220
  */
221
- export declare const FEED_PUBLISH_HOOK_SCRIPT = "#!/usr/bin/env python3\n\"\"\"Publish and clear open-block records for `agents feed`.\n\nThe manifest invokes this script for top-level AskUserQuestion calls, waiting\nnotifications, question answers, and session lifecycle events. One atomic file\nper session means a new block replaces the previous block. Answer/resume/stop\nevents remove it so `agents feed` only lists decisions that are still open.\n\nSub-agent gate: when the PreToolUse payload carries `agent_type`, this is a\nTask/Agent subagent -- skip. Only the top-level agent publishes. Verified on\nClaude Code 2.1.170 (2026-07).\n\nFail-open: ANY error is swallowed so a feed hiccup never blocks a tool call.\n\"\"\"\nimport os\nimport sys\nimport json\nimport re\nimport socket\nimport tempfile\nfrom datetime import datetime, timezone\n\nWAITING_NOTIFICATION_TYPES = {\n \"permission_prompt\",\n \"idle_prompt\",\n \"elicitation_dialog\",\n}\nCLEAR_EVENTS = {\n \"PostToolUse\",\n \"Stop\",\n \"SessionEnd\",\n}\n# Codex emits a PermissionRequest event (not Claude's Notification) when it\n# blocks on an approval prompt. Claude never fires PermissionRequest, so the\n# same script handles both: PermissionRequest maps to an approval-class block\n# with a high cost-of-delay so 'agents feed --dispatch' pages it as urgent.\n\n\ndef read_json(path):\n try:\n with open(path) as f:\n return json.load(f)\n except Exception:\n return None\n\n\ndef write_json(path, value):\n dir_name = os.path.dirname(path)\n os.makedirs(dir_name, exist_ok=True)\n fd, tmp = tempfile.mkstemp(dir=dir_name, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(value, f, indent=2)\n os.replace(tmp, path)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\ndef main():\n raw = sys.stdin.read()\n try:\n payload = json.loads(raw) if raw.strip() else {}\n except Exception:\n return\n\n # Sub-agent gate.\n if payload.get(\"agent_type\"):\n return\n\n session_id = payload.get(\"session_id\", \"\")\n if not session_id:\n return\n\n safe_session_id = re.sub(r\"[^A-Za-z0-9._-]\", \"-\", session_id)\n block_id = f\"block-{safe_session_id}\"\n home = os.environ.get(\"HOME\") or os.path.expanduser(\"~\")\n feed_dir = os.path.join(home, \".agents\", \".history\", \"feed\")\n answered_dir = os.path.join(feed_dir, \"answered\")\n asks_dir = os.path.join(feed_dir, \"asks\")\n target = os.path.join(feed_dir, f\"{block_id}.json\")\n hook_event = payload.get(\"hook_event_name\", \"PreToolUse\")\n\n if hook_event in CLEAR_EVENTS:\n # A matcher-less PostToolUse clear (registered for Codex so an approved\n # tool clears its approval card) must NOT wipe an open AskUserQuestion\n # while an unrelated tool runs mid-question -- those are cleared only by\n # the AskUserQuestion-matched PostToolUse. So on PostToolUse, keep a\n # 'question' block; approval/notification blocks clear once the tool runs.\n if hook_event == \"PostToolUse\":\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\" and payload.get(\"tool_name\") != \"AskUserQuestion\":\n return\n except Exception:\n pass\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n # Also clear the answered marker so a future question for this session\n # is not permanently locked.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n # Terminal answers (human typed in the TUI) record an answered marker and\n # remove the block file so the feed stops showing it within one poll cycle.\n # The marker stays behind so a concurrent surface cannot double-answer.\n if hook_event == \"UserPromptSubmit\":\n os.makedirs(answered_dir, exist_ok=True)\n marker = os.path.join(answered_dir, f\"{block_id}.json\")\n try:\n fd = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)\n record = {\n \"answeredAt\": datetime.now(timezone.utc).isoformat(),\n \"answeredFrom\": \"terminal\",\n }\n with os.fdopen(fd, \"w\") as f:\n json.dump(record, f, indent=2)\n except FileExistsError:\n pass\n except Exception:\n pass\n # Remove the visible block so the feed drops the answered question.\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n notification_type = None\n codex_approval = False\n if hook_event == \"Notification\":\n notification_type = payload.get(\"notification_type\", \"\")\n if notification_type not in WAITING_NOTIFICATION_TYPES:\n return\n # Claude emits a generic permission notification after presenting an\n # AskUserQuestion. Keep the structured questions and options already\n # published for this session instead of replacing them with that less\n # useful notification text.\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\":\n return\n except Exception:\n pass\n message = payload.get(\"message\", \"\")\n if not message:\n return\n normalized_questions = [{\n \"text\": message,\n \"header\": payload.get(\"title\") or notification_type.replace(\"_\", \" \").title(),\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n elif hook_event == \"PermissionRequest\":\n # Codex approval prompt. The payload mirrors PreToolUse (tool_name,\n # tool_input) but carries no questions -- Codex is asking to run a tool,\n # not asking the operator a multiple-choice question. Publish it as a\n # notification-kind approval block naming the tool so the feed and the\n # phone notifier can surface it, and so the Factory extension can bridge\n # it to a VS Code notification.\n tool_name = payload.get(\"tool_name\") or \"a tool\"\n tool_input = payload.get(\"tool_input\", {})\n command = \"\"\n if isinstance(tool_input, dict):\n command = (\n tool_input.get(\"command\")\n or tool_input.get(\"cmd\")\n or tool_input.get(\"path\")\n or \"\"\n )\n if isinstance(command, list):\n command = \" \".join(str(c) for c in command)\n detail = f\": {command}\" if command else \"\"\n normalized_questions = [{\n \"text\": f\"Codex needs approval to run {tool_name}{detail}\",\n \"header\": \"Approval needed\",\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n notification_type = \"permission_prompt\"\n codex_approval = True\n else:\n tool_input = payload.get(\"tool_input\", {})\n questions = tool_input.get(\"questions\", [])\n if not questions:\n return\n normalized_questions = []\n for q in questions:\n if not isinstance(q, dict):\n continue\n question = {\n \"text\": q.get(\"question\", q.get(\"header\", \"\")),\n \"header\": q.get(\"header\"),\n \"multiSelect\": q.get(\"multiSelect\", False),\n }\n raw_opts = q.get(\"options\", [])\n if raw_opts:\n question[\"options\"] = [\n {\"label\": o.get(\"label\", \"\"), \"description\": o.get(\"description\")}\n for o in raw_opts\n if isinstance(o, dict)\n ]\n normalized_questions.append(question)\n if not normalized_questions:\n return\n kind = \"question\"\n\n # Identity from env (set by agents-cli at spawn).\n mailbox_id = os.path.basename(\n os.environ.get(\"AGENTS_MAILBOX_DIR\", \"\").rstrip(\"/\")\n ) or session_id\n\n now_iso = datetime.now(timezone.utc).isoformat()\n stats_path = os.path.join(asks_dir, f\"{safe_session_id}.json\")\n stats = read_json(stats_path) or {}\n recent = stats.get(\"recentAskTimestamps\") if isinstance(stats, dict) else []\n if not isinstance(recent, list):\n recent = []\n recent.append(now_iso)\n # Keep enough history for rolling one-hour needy detection without unbounded\n # per-session files. The TypeScript reader applies the exact time window.\n recent = recent[-200:]\n write_json(stats_path, {\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"firstAskAt\": stats.get(\"firstAskAt\") or now_iso,\n \"lastAskAt\": now_iso,\n \"totalAskCount\": int(stats.get(\"totalAskCount\") or 0) + 1,\n \"recentAskTimestamps\": recent,\n })\n\n hostname = os.environ.get(\"AGENTS_SYNC_MACHINE_ID\") or socket.gethostname()\n host = hostname.split(\".\")[0].strip().lower()\n host = re.sub(r\"[^a-z0-9_-]\", \"-\", host) or \"unknown\"\n\n runtime = os.environ.get(\"AGENTS_RUNTIME\", \"headless\")\n\n block = {\n \"blockId\": block_id,\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"host\": host,\n \"runtime\": runtime,\n \"ts\": now_iso,\n \"questions\": normalized_questions,\n \"kind\": kind,\n }\n if notification_type:\n block[\"notificationType\"] = notification_type\n\n # A Codex PermissionRequest is a real approval gate: mark it approval-class\n # with a high cost-of-delay so 'agents feed --dispatch' classifies it urgent\n # (isPhoneUrgent gates on costOfDelay >= phoneNotifyThreshold, default\n # 'medium') and pages the phone. A plain 'deny' is the safe default.\n if codex_approval:\n block[\"blockClass\"] = \"approval\"\n block[\"costOfDelay\"] = \"high\"\n block[\"safeDefault\"] = \"deny\"\n\n # Optional multi-operator control metadata passed by the agent in the\n # AskUserQuestion tool_input. Defaults keep the existing behavior. A Codex\n # PermissionRequest carries tool ARGS in tool_input (command/path), not\n # operator controls, so it is excluded here -- its class/cost is stamped\n # above from codex_approval.\n controls = payload.get(\"tool_input\", {}) if hook_event not in (\"Notification\", \"PermissionRequest\") else {}\n block_class = controls.get(\"blockClass\") if isinstance(controls, dict) else None\n if block_class in (\"approval\", \"decision\"):\n block[\"blockClass\"] = block_class\n consequence = controls.get(\"consequence\") if isinstance(controls, dict) else None\n if consequence:\n block[\"consequence\"] = consequence\n allowed = controls.get(\"allowedOperators\") if isinstance(controls, dict) else None\n if isinstance(allowed, list):\n block[\"allowedOperators\"] = [str(a) for a in allowed]\n timeout = controls.get(\"timeoutMinutes\") if isinstance(controls, dict) else None\n if isinstance(timeout, (int, float)) and timeout > 0:\n block[\"timeoutMinutes\"] = int(timeout)\n safe_default = controls.get(\"safeDefault\") if isinstance(controls, dict) else None\n if isinstance(safe_default, str):\n block[\"safeDefault\"] = safe_default\n cost = controls.get(\"costOfDelay\") if isinstance(controls, dict) else None\n if cost in (\"low\", \"medium\", \"high\"):\n block[\"costOfDelay\"] = cost\n\n # Publishing a new question clears any stale answered marker from the\n # previous question in this session.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n\n # Python's expanduser() ignores HOME on Windows, while agents-cli honors a\n # HOME override on every platform. Use the same anchor so hooks and the CLI\n # always read/write one feed store (including temp-home and sandbox runs).\n os.makedirs(feed_dir, exist_ok=True)\n\n fd, tmp = tempfile.mkstemp(dir=feed_dir, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(block, f, indent=2)\n os.replace(tmp, target)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\nif __name__ == \"__main__\":\n try:\n main()\n except Exception:\n pass # fail open\n";
221
+ export declare const FEED_PUBLISH_HOOK_SCRIPT = "#!/usr/bin/env python3\n\"\"\"Publish and clear open-block records for `agents feed`.\n\nThe manifest invokes this script for top-level AskUserQuestion calls, waiting\nnotifications, question answers, and session lifecycle events. One atomic file\nper session means a new block replaces the previous block. Answer/resume/stop\nevents remove it so `agents feed` only lists decisions that are still open.\n\nSub-agent gate: when the PreToolUse payload carries `agent_type`, this is a\nTask/Agent subagent -- skip. Only the top-level agent publishes. Verified on\nClaude Code 2.1.170 (2026-07).\n\nFail-open: ANY error is swallowed so a feed hiccup never blocks a tool call.\n\"\"\"\nimport os\nimport sys\nimport json\nimport re\nimport socket\nimport tempfile\nfrom datetime import datetime, timezone\n\nWAITING_NOTIFICATION_TYPES = {\n \"permission_prompt\",\n \"idle_prompt\",\n \"elicitation_dialog\",\n}\nCLEAR_EVENTS = {\n \"PostToolUse\",\n \"Stop\",\n \"SessionEnd\",\n}\n# Codex emits a PermissionRequest event (not Claude's Notification) when it\n# blocks on an approval prompt. Claude never fires PermissionRequest, so the\n# same script handles both: PermissionRequest maps to an approval-class block\n# with a high cost-of-delay so 'agents feed --dispatch' pages it as urgent.\n\n\ndef read_json(path):\n try:\n with open(path) as f:\n return json.load(f)\n except Exception:\n return None\n\n\ndef write_json(path, value):\n dir_name = os.path.dirname(path)\n os.makedirs(dir_name, exist_ok=True)\n fd, tmp = tempfile.mkstemp(dir=dir_name, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(value, f, indent=2)\n os.replace(tmp, path)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\ndef main():\n raw = sys.stdin.read()\n try:\n payload = json.loads(raw) if raw.strip() else {}\n except Exception:\n return\n\n # Sub-agent gate.\n if payload.get(\"agent_type\"):\n return\n\n session_id = payload.get(\"session_id\", \"\")\n if not session_id:\n return\n\n safe_session_id = re.sub(r\"[^A-Za-z0-9._-]\", \"-\", session_id)\n block_id = f\"block-{safe_session_id}\"\n home = os.environ.get(\"HOME\") or os.path.expanduser(\"~\")\n feed_dir = os.path.join(home, \".agents\", \".history\", \"feed\")\n answered_dir = os.path.join(feed_dir, \"answered\")\n asks_dir = os.path.join(feed_dir, \"asks\")\n target = os.path.join(feed_dir, f\"{block_id}.json\")\n hook_event = payload.get(\"hook_event_name\", \"PreToolUse\")\n\n if hook_event in CLEAR_EVENTS:\n # A declared block (`agents feed post --blocked`) is the agent explicitly\n # saying it is stuck. Unlike a question/notification/approval block -- which\n # tracks an in-flight harness prompt that a lifecycle event resolves -- a\n # declared block stays open until it is actually ANSWERED. So while it is\n # still UNANSWERED, Stop/SessionEnd/PostToolUse must never silently drop it:\n # otherwise the needs-you record vanishes the moment the agent parks the block\n # and its turn ends -- exactly when the owner still needs to see and answer it.\n # Once it IS answered (an answered marker exists), it clears like any other\n # block by falling through below -- which frees that marker too, so a later\n # `--blocked` in the same session is not falsely locked as already-answered\n # (recordAnswer creates the marker with O_EXCL).\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n answered = os.path.exists(os.path.join(answered_dir, f\"{block_id}.json\"))\n if existing.get(\"kind\") == \"declared\" and not answered:\n return\n except Exception:\n pass\n # A matcher-less PostToolUse clear (registered for Codex so an approved\n # tool clears its approval card) must NOT wipe an open AskUserQuestion\n # while an unrelated tool runs mid-question -- those are cleared only by\n # the AskUserQuestion-matched PostToolUse. So on PostToolUse, keep a\n # 'question' block; approval/notification blocks clear once the tool runs.\n if hook_event == \"PostToolUse\":\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\" and payload.get(\"tool_name\") != \"AskUserQuestion\":\n return\n except Exception:\n pass\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n # Also clear the answered marker so a future question for this session\n # is not permanently locked.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n # Terminal answers (human typed in the TUI) record an answered marker and\n # remove the block file so the feed stops showing it within one poll cycle.\n # The marker stays behind so a concurrent surface cannot double-answer.\n if hook_event == \"UserPromptSubmit\":\n os.makedirs(answered_dir, exist_ok=True)\n marker = os.path.join(answered_dir, f\"{block_id}.json\")\n try:\n fd = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)\n record = {\n \"answeredAt\": datetime.now(timezone.utc).isoformat(),\n \"answeredFrom\": \"terminal\",\n }\n with os.fdopen(fd, \"w\") as f:\n json.dump(record, f, indent=2)\n except FileExistsError:\n pass\n except Exception:\n pass\n # Remove the visible block so the feed drops the answered question.\n try:\n os.unlink(target)\n except FileNotFoundError:\n pass\n except Exception:\n pass\n return\n\n notification_type = None\n codex_approval = False\n if hook_event == \"Notification\":\n notification_type = payload.get(\"notification_type\", \"\")\n if notification_type not in WAITING_NOTIFICATION_TYPES:\n return\n # Claude emits a generic permission notification after presenting an\n # AskUserQuestion. Keep the structured questions and options already\n # published for this session instead of replacing them with that less\n # useful notification text.\n try:\n with open(target) as existing_file:\n existing = json.load(existing_file)\n if existing.get(\"kind\") == \"question\":\n return\n except Exception:\n pass\n message = payload.get(\"message\", \"\")\n if not message:\n return\n normalized_questions = [{\n \"text\": message,\n \"header\": payload.get(\"title\") or notification_type.replace(\"_\", \" \").title(),\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n elif hook_event == \"PermissionRequest\":\n # Codex approval prompt. The payload mirrors PreToolUse (tool_name,\n # tool_input) but carries no questions -- Codex is asking to run a tool,\n # not asking the operator a multiple-choice question. Publish it as a\n # notification-kind approval block naming the tool so the feed and the\n # phone notifier can surface it, and so the Factory extension can bridge\n # it to a VS Code notification.\n tool_name = payload.get(\"tool_name\") or \"a tool\"\n tool_input = payload.get(\"tool_input\", {})\n command = \"\"\n if isinstance(tool_input, dict):\n command = (\n tool_input.get(\"command\")\n or tool_input.get(\"cmd\")\n or tool_input.get(\"path\")\n or \"\"\n )\n if isinstance(command, list):\n command = \" \".join(str(c) for c in command)\n detail = f\": {command}\" if command else \"\"\n normalized_questions = [{\n \"text\": f\"Codex needs approval to run {tool_name}{detail}\",\n \"header\": \"Approval needed\",\n \"multiSelect\": False,\n }]\n kind = \"notification\"\n notification_type = \"permission_prompt\"\n codex_approval = True\n else:\n tool_input = payload.get(\"tool_input\", {})\n questions = tool_input.get(\"questions\", [])\n if not questions:\n return\n normalized_questions = []\n for q in questions:\n if not isinstance(q, dict):\n continue\n question = {\n \"text\": q.get(\"question\", q.get(\"header\", \"\")),\n \"header\": q.get(\"header\"),\n \"multiSelect\": q.get(\"multiSelect\", False),\n }\n raw_opts = q.get(\"options\", [])\n if raw_opts:\n question[\"options\"] = [\n {\"label\": o.get(\"label\", \"\"), \"description\": o.get(\"description\")}\n for o in raw_opts\n if isinstance(o, dict)\n ]\n normalized_questions.append(question)\n if not normalized_questions:\n return\n kind = \"question\"\n\n # Identity from env (set by agents-cli at spawn).\n mailbox_id = os.path.basename(\n os.environ.get(\"AGENTS_MAILBOX_DIR\", \"\").rstrip(\"/\")\n ) or session_id\n\n now_iso = datetime.now(timezone.utc).isoformat()\n stats_path = os.path.join(asks_dir, f\"{safe_session_id}.json\")\n stats = read_json(stats_path) or {}\n recent = stats.get(\"recentAskTimestamps\") if isinstance(stats, dict) else []\n if not isinstance(recent, list):\n recent = []\n recent.append(now_iso)\n # Keep enough history for rolling one-hour needy detection without unbounded\n # per-session files. The TypeScript reader applies the exact time window.\n recent = recent[-200:]\n write_json(stats_path, {\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"firstAskAt\": stats.get(\"firstAskAt\") or now_iso,\n \"lastAskAt\": now_iso,\n \"totalAskCount\": int(stats.get(\"totalAskCount\") or 0) + 1,\n \"recentAskTimestamps\": recent,\n })\n\n hostname = os.environ.get(\"AGENTS_SYNC_MACHINE_ID\") or socket.gethostname()\n host = hostname.split(\".\")[0].strip().lower()\n host = re.sub(r\"[^a-z0-9_-]\", \"-\", host) or \"unknown\"\n\n runtime = os.environ.get(\"AGENTS_RUNTIME\", \"headless\")\n\n block = {\n \"blockId\": block_id,\n \"sessionId\": session_id,\n \"mailboxId\": mailbox_id,\n \"host\": host,\n \"runtime\": runtime,\n \"ts\": now_iso,\n \"questions\": normalized_questions,\n \"kind\": kind,\n }\n if notification_type:\n block[\"notificationType\"] = notification_type\n\n # A Codex PermissionRequest is a real approval gate: mark it approval-class\n # with a high cost-of-delay so 'agents feed --dispatch' classifies it urgent\n # (isPhoneUrgent gates on costOfDelay >= phoneNotifyThreshold, default\n # 'medium') and pages the phone. A plain 'deny' is the safe default.\n if codex_approval:\n block[\"blockClass\"] = \"approval\"\n block[\"costOfDelay\"] = \"high\"\n block[\"safeDefault\"] = \"deny\"\n\n # Optional multi-operator control metadata passed by the agent in the\n # AskUserQuestion tool_input. Defaults keep the existing behavior. A Codex\n # PermissionRequest carries tool ARGS in tool_input (command/path), not\n # operator controls, so it is excluded here -- its class/cost is stamped\n # above from codex_approval.\n controls = payload.get(\"tool_input\", {}) if hook_event not in (\"Notification\", \"PermissionRequest\") else {}\n block_class = controls.get(\"blockClass\") if isinstance(controls, dict) else None\n if block_class in (\"approval\", \"decision\"):\n block[\"blockClass\"] = block_class\n consequence = controls.get(\"consequence\") if isinstance(controls, dict) else None\n if consequence:\n block[\"consequence\"] = consequence\n allowed = controls.get(\"allowedOperators\") if isinstance(controls, dict) else None\n if isinstance(allowed, list):\n block[\"allowedOperators\"] = [str(a) for a in allowed]\n timeout = controls.get(\"timeoutMinutes\") if isinstance(controls, dict) else None\n if isinstance(timeout, (int, float)) and timeout > 0:\n block[\"timeoutMinutes\"] = int(timeout)\n safe_default = controls.get(\"safeDefault\") if isinstance(controls, dict) else None\n if isinstance(safe_default, str):\n block[\"safeDefault\"] = safe_default\n cost = controls.get(\"costOfDelay\") if isinstance(controls, dict) else None\n if cost in (\"low\", \"medium\", \"high\"):\n block[\"costOfDelay\"] = cost\n\n # Publishing a new question clears any stale answered marker from the\n # previous question in this session.\n try:\n os.unlink(os.path.join(answered_dir, f\"{block_id}.json\"))\n except FileNotFoundError:\n pass\n except Exception:\n pass\n\n # Python's expanduser() ignores HOME on Windows, while agents-cli honors a\n # HOME override on every platform. Use the same anchor so hooks and the CLI\n # always read/write one feed store (including temp-home and sandbox runs).\n os.makedirs(feed_dir, exist_ok=True)\n\n fd, tmp = tempfile.mkstemp(dir=feed_dir, suffix=\".tmp\")\n try:\n with os.fdopen(fd, \"w\") as f:\n json.dump(block, f, indent=2)\n os.replace(tmp, target)\n except Exception:\n try:\n os.unlink(tmp)\n except Exception:\n pass\n\n\nif __name__ == \"__main__\":\n try:\n main()\n except Exception:\n pass # fail open\n";
222
222
  /** Manifest entry for the feed-publish hook, matching the ManifestHook shape. */
223
223
  export declare const FEED_PUBLISH_HOOK_MANIFEST: {
224
224
  name: string;
package/dist/lib/feed.js CHANGED
@@ -442,6 +442,25 @@ def main():
442
442
  hook_event = payload.get("hook_event_name", "PreToolUse")
443
443
 
444
444
  if hook_event in CLEAR_EVENTS:
445
+ # A declared block (\`agents feed post --blocked\`) is the agent explicitly
446
+ # saying it is stuck. Unlike a question/notification/approval block -- which
447
+ # tracks an in-flight harness prompt that a lifecycle event resolves -- a
448
+ # declared block stays open until it is actually ANSWERED. So while it is
449
+ # still UNANSWERED, Stop/SessionEnd/PostToolUse must never silently drop it:
450
+ # otherwise the needs-you record vanishes the moment the agent parks the block
451
+ # and its turn ends -- exactly when the owner still needs to see and answer it.
452
+ # Once it IS answered (an answered marker exists), it clears like any other
453
+ # block by falling through below -- which frees that marker too, so a later
454
+ # \`--blocked\` in the same session is not falsely locked as already-answered
455
+ # (recordAnswer creates the marker with O_EXCL).
456
+ try:
457
+ with open(target) as existing_file:
458
+ existing = json.load(existing_file)
459
+ answered = os.path.exists(os.path.join(answered_dir, f"{block_id}.json"))
460
+ if existing.get("kind") == "declared" and not answered:
461
+ return
462
+ except Exception:
463
+ pass
445
464
  # A matcher-less PostToolUse clear (registered for Codex so an approved
446
465
  # tool clears its approval card) must NOT wipe an open AskUserQuestion
447
466
  # while an unrelated tool runs mid-question -- those are cleared only by
@@ -0,0 +1,43 @@
1
+ /**
2
+ * Human overrides for the cost-tier -> model mapping.
3
+ *
4
+ * The auto-ranking (lib/model-tiers.ts) is a best guess; for subscription harnesses
5
+ * with no price signal (Kimi, Cursor) it can be wrong. A user pins the right model
6
+ * per tier with `agents models tier set <agent[@version]> <tier> <model>`, which
7
+ * writes here — the user never hand-edits the file.
8
+ *
9
+ * Stored under agents.yaml, same selector shape as run.defaults:
10
+ *
11
+ * model:
12
+ * tiers:
13
+ * "kimi:*": # applies to every installed kimi
14
+ * best: kimi-code/k3
15
+ * default: kimi-code/kimi-for-coding
16
+ * "kimi:0.19.2": # a specific version wins over the wildcard
17
+ * best: kimi-code/k3-256k
18
+ *
19
+ * Resolution (most-specific-first): `<agent>:<version>` beats `<agent>:*` beats the
20
+ * auto-ranking. resolveTierMap (model-tiers.ts) applies the result and falls back to
21
+ * auto for any tier whose overridden id isn't in that version's catalog.
22
+ */
23
+ import type { AgentId } from './types.js';
24
+ import { type ModelTier } from './model-tiers.js';
25
+ export type TierOverrideMap = Partial<Record<ModelTier, string>>;
26
+ export interface TierOverrideEntry {
27
+ selector: string;
28
+ tiers: TierOverrideMap;
29
+ }
30
+ /** Validate a tier token, throwing a friendly error otherwise. */
31
+ export declare function parseTier(input: string): ModelTier;
32
+ /**
33
+ * The effective tier overrides for an (agent, version): the `<agent>:*` wildcard
34
+ * merged under the exact `<agent>:<version>` selector (exact wins per-tier).
35
+ */
36
+ export declare function resolveTierOverrideFrom(all: Record<string, unknown>, agent: AgentId, version?: string | null): TierOverrideMap;
37
+ export declare function resolveTierOverride(agent: AgentId, version?: string | null): TierOverrideMap;
38
+ /** Every configured override entry, sorted by selector (for `agents models tier list`). */
39
+ export declare function listTierOverrides(): TierOverrideEntry[];
40
+ /** Pin `tier -> model` for a selector. Writes agents.yaml. */
41
+ export declare function setTierOverride(selectorInput: string, tierInput: string, model: string): TierOverrideEntry;
42
+ /** Clear one tier (or all tiers when `tierInput` is omitted) for a selector. Returns true if anything changed. */
43
+ export declare function clearTierOverride(selectorInput: string, tierInput?: string): boolean;
@@ -0,0 +1,97 @@
1
+ import { readMeta, updateMeta } from './state.js';
2
+ import { parseRunDefaultSelector } from './run-defaults.js';
3
+ import { MODEL_TIERS } from './model-tiers.js';
4
+ function isTier(value) {
5
+ return MODEL_TIERS.includes(value);
6
+ }
7
+ /** Validate a tier token, throwing a friendly error otherwise. */
8
+ export function parseTier(input) {
9
+ const t = input.trim().toLowerCase();
10
+ if (!isTier(t)) {
11
+ throw new Error(`Invalid tier '${input}'. Use one of: ${MODEL_TIERS.join(', ')}.`);
12
+ }
13
+ return t;
14
+ }
15
+ /** Normalize a stored selector's tier map, dropping unknown/empty entries. */
16
+ function normalize(raw) {
17
+ const out = {};
18
+ if (!raw || typeof raw !== 'object')
19
+ return out;
20
+ for (const [k, v] of Object.entries(raw)) {
21
+ if (isTier(k) && typeof v === 'string' && v.trim())
22
+ out[k] = v.trim();
23
+ }
24
+ return out;
25
+ }
26
+ function sortedSelectors(map) {
27
+ return Object.fromEntries(Object.entries(map).sort(([a], [b]) => a.localeCompare(b)));
28
+ }
29
+ /**
30
+ * The effective tier overrides for an (agent, version): the `<agent>:*` wildcard
31
+ * merged under the exact `<agent>:<version>` selector (exact wins per-tier).
32
+ */
33
+ export function resolveTierOverrideFrom(all, agent, version) {
34
+ const merged = { ...normalize(all[`${agent}:*`]) };
35
+ if (version) {
36
+ for (const [tier, model] of Object.entries(normalize(all[`${agent}:${version}`]))) {
37
+ merged[tier] = model;
38
+ }
39
+ }
40
+ return merged;
41
+ }
42
+ export function resolveTierOverride(agent, version) {
43
+ return resolveTierOverrideFrom(readMeta().model?.tiers ?? {}, agent, version);
44
+ }
45
+ /** Every configured override entry, sorted by selector (for `agents models tier list`). */
46
+ export function listTierOverrides() {
47
+ const all = readMeta().model?.tiers ?? {};
48
+ return Object.entries(all)
49
+ .sort(([a], [b]) => a.localeCompare(b))
50
+ .map(([selector, tiers]) => ({ selector, tiers: normalize(tiers) }));
51
+ }
52
+ /** Pin `tier -> model` for a selector. Writes agents.yaml. */
53
+ export function setTierOverride(selectorInput, tierInput, model) {
54
+ const parsed = parseRunDefaultSelector(selectorInput);
55
+ const tier = parseTier(tierInput);
56
+ const id = model.trim();
57
+ if (!id)
58
+ throw new Error('A model id is required.');
59
+ updateMeta((meta) => {
60
+ const modelCfg = { ...(meta.model ?? {}) };
61
+ const tiers = { ...(modelCfg.tiers ?? {}) };
62
+ tiers[parsed.selector] = { ...(tiers[parsed.selector] ?? {}), [tier]: id };
63
+ modelCfg.tiers = sortedSelectors(tiers);
64
+ return { ...meta, model: modelCfg };
65
+ });
66
+ return { selector: parsed.selector, tiers: normalize(readMeta().model?.tiers?.[parsed.selector]) };
67
+ }
68
+ /** Clear one tier (or all tiers when `tierInput` is omitted) for a selector. Returns true if anything changed. */
69
+ export function clearTierOverride(selectorInput, tierInput) {
70
+ const parsed = parseRunDefaultSelector(selectorInput);
71
+ const tier = tierInput ? parseTier(tierInput) : null;
72
+ let changed = false;
73
+ updateMeta((meta) => {
74
+ if (!meta.model?.tiers?.[parsed.selector])
75
+ return meta;
76
+ const model = { ...meta.model };
77
+ const tiers = { ...(model.tiers ?? {}) };
78
+ if (tier) {
79
+ const entry = { ...tiers[parsed.selector] };
80
+ if (entry[tier] !== undefined) {
81
+ delete entry[tier];
82
+ changed = true;
83
+ }
84
+ if (Object.keys(entry).length === 0)
85
+ delete tiers[parsed.selector];
86
+ else
87
+ tiers[parsed.selector] = entry;
88
+ }
89
+ else {
90
+ delete tiers[parsed.selector];
91
+ changed = true;
92
+ }
93
+ model.tiers = tiers;
94
+ return { ...meta, model };
95
+ });
96
+ return changed;
97
+ }
@@ -36,18 +36,24 @@ export interface TierResolution {
36
36
  clampedFrom?: ModelTier;
37
37
  /** Human note (e.g. why it clamped, or that it is a curated/subscription mapping). */
38
38
  note?: string;
39
+ /** Where the model came from: 'auto' ranking, a user 'override', or a 'curated' ladder. */
40
+ source?: 'auto' | 'override' | 'curated';
39
41
  }
40
42
  /**
41
- * Resolve all four tiers for an (agent, version). The map is what `agents models`
42
- * prints and what `resolveTier` indexes into.
43
+ * Resolve all four tiers for an (agent, version) -- what `agents models` prints and
44
+ * `resolveTier` indexes. Precedence: user override -> curated ladder / auto-ranking.
43
45
  */
44
46
  export declare function resolveTierMap(agent: AgentId, version: string): Record<ModelTier, TierResolution>;
45
47
  /**
46
- * Map a harness's catalog models onto the four tiers. Pure (no catalog lookup)
47
- * so it is directly testable with synthetic inputs. Ranks the models, collapses
48
- * variants, buckets onto cheap/default/best/ultra, and clamps absent tiers down
49
- * to the nearest lower one. A single-model harness maps the tiers to reasoning
50
- * effort instead of models.
48
+ * Apply user overrides on top of the auto/curated map. Pure (takes the resolved
49
+ * override map, no config lookup) so it is directly testable. An overridden id is
50
+ * used only when the version actually ships it (or when there is no catalog to
51
+ * check, e.g. Droid); otherwise the tier keeps its base value with a note.
52
+ */
53
+ export declare function applyTierOverrides(overrides: Partial<Record<ModelTier, string>>, label: string, catalogIds: Set<string> | null, base: Record<ModelTier, TierResolution>): Record<ModelTier, TierResolution>;
54
+ /**
55
+ * Map a harness's catalog models onto the four tiers. Pure (no catalog lookup) so
56
+ * it is directly testable. A single-model harness maps the tiers to reasoning effort.
51
57
  */
52
58
  export declare function tierizeModels(agent: AgentId, models: ModelInfo[]): Record<ModelTier, TierResolution>;
53
59
  /** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
@@ -1,11 +1,27 @@
1
1
  import { getModelCatalog } from './models.js';
2
2
  import { getModelPricing } from './pricing/index.js';
3
+ import { resolveTierOverride } from './model-tier-overrides.js';
3
4
  /** The four cross-harness cost tiers, cheapest -> most capable. */
4
5
  export const MODEL_TIERS = ['cheap', 'default', 'best', 'ultra'];
5
6
  /** True if `s` is one of the four tier tokens (not a concrete model id). */
6
7
  export function isTierToken(s) {
7
8
  return !!s && MODEL_TIERS.includes(s);
8
9
  }
10
+ /**
11
+ * Curated tier ladders for harnesses the auto-ranker can't order from names/price
12
+ * (subscription harnesses with no price signal). Each rung is `[tier, matcher]` in
13
+ * cheap -> best order; the newest catalog id matching each rung fills that tier,
14
+ * missing tiers clamp. Extend this table rather than adding per-harness branches.
15
+ */
16
+ const CURATED_LADDERS = {
17
+ // Kimi: K2.7 Highspeed < K2.7 Coding < K3 (the 1M-context default; k3-256k folds
18
+ // into K3). No ultra. The name heuristic can't tell K3 > K2.7, so curate it.
19
+ kimi: [
20
+ { tier: 'cheap', match: /highspeed/i },
21
+ { tier: 'default', match: /for-coding(?!.*highspeed)/i },
22
+ { tier: 'best', match: /(^|[-/])k3\b/i }, // K3 family incl. k3-256k; the plain id represents it
23
+ ],
24
+ };
9
25
  // --- single-model harnesses: the tier is reasoning effort, not a model ---------
10
26
  const TIER_EFFORT = {
11
27
  cheap: 'low',
@@ -169,29 +185,100 @@ function rankCatalog(agent, models) {
169
185
  function rungIndexFor(tierIndex, n) {
170
186
  return n >= 4 ? Math.round((tierIndex / 3) * (n - 1)) : Math.min(tierIndex, n - 1);
171
187
  }
188
+ /** Bucket an ordered (cheap -> dear) rung list onto the four tiers, clamping when < 4. */
189
+ function bucketRungs(rungs) {
190
+ const n = rungs.length;
191
+ const map = {};
192
+ if (n === 0) {
193
+ // Fail-safe: no catalog -> every tier null, caller drops the --model flag.
194
+ for (const t of MODEL_TIERS)
195
+ map[t] = { tier: t, model: null };
196
+ return map;
197
+ }
198
+ // A tier that shares the rung of the tier below has no distinct rung -> mark clamped.
199
+ for (let i = 0; i < MODEL_TIERS.length; i++) {
200
+ const t = MODEL_TIERS[i];
201
+ const idx = rungIndexFor(i, n);
202
+ const shared = i > 0 && rungIndexFor(i - 1, n) === idx;
203
+ map[t] = shared
204
+ ? { tier: t, model: rungs[idx].id, clampedFrom: MODEL_TIERS[i - 1], note: `no distinct ${t} rung; using ${MODEL_TIERS[i - 1]}`, source: 'auto' }
205
+ : { tier: t, model: rungs[idx].id, source: 'auto' };
206
+ }
207
+ return map;
208
+ }
209
+ /** Build a tier map from a curated ladder against a catalog (newest match per rung). */
210
+ function tierizeFromLadder(ladder, models) {
211
+ const usable = models.filter((m) => !PSEUDO.test(m.id));
212
+ const rungs = [];
213
+ for (const rung of ladder) {
214
+ const matches = usable.filter((m) => rung.match.test(m.id));
215
+ if (matches.length === 0)
216
+ continue;
217
+ // Prefer a plain id over a context-size variant (k3 over k3-256k) -- the
218
+ // variant folds into the rung but the plain model represents it -- then newest.
219
+ const plain = matches.filter((m) => !/-\d+[km]\b/i.test(m.id));
220
+ const pool = plain.length ? plain : matches;
221
+ rungs.push({ id: pool.reduce((a, b) => (newer(b.id, a.id) > 0 ? b : a)).id });
222
+ }
223
+ const map = bucketRungs(rungs);
224
+ for (const t of MODEL_TIERS)
225
+ if (map[t].model)
226
+ map[t].source = 'curated';
227
+ return map;
228
+ }
172
229
  /**
173
- * Resolve all four tiers for an (agent, version). The map is what `agents models`
174
- * prints and what `resolveTier` indexes into.
230
+ * Resolve all four tiers for an (agent, version) -- what `agents models` prints and
231
+ * `resolveTier` indexes. Precedence: user override -> curated ladder / auto-ranking.
175
232
  */
176
233
  export function resolveTierMap(agent, version) {
177
- // Droid: curated credit-multiplier map (no live catalog).
234
+ let base;
235
+ let catalogIds;
178
236
  if (agent === 'droid') {
179
- return {
180
- cheap: { tier: 'cheap', model: DROID_TIERS.cheap, note: 'Droid Core 0.55x' },
181
- default: { tier: 'default', model: DROID_TIERS.default, note: 'Droid Core 0.6x' },
182
- best: { tier: 'best', model: DROID_TIERS.best, note: '2x' },
183
- ultra: { tier: 'ultra', model: DROID_TIERS.ultra, clampedFrom: 'best', note: 'capped at 2x (4x models excluded)' },
237
+ // Droid: curated credit-multiplier map (no live catalog to validate against).
238
+ base = {
239
+ cheap: { tier: 'cheap', model: DROID_TIERS.cheap, note: 'Droid Core 0.55x', source: 'curated' },
240
+ default: { tier: 'default', model: DROID_TIERS.default, note: 'Droid Core 0.6x', source: 'curated' },
241
+ best: { tier: 'best', model: DROID_TIERS.best, note: '2x', source: 'curated' },
242
+ ultra: { tier: 'ultra', model: DROID_TIERS.ultra, clampedFrom: 'best', note: 'capped at 2x (4x excluded)', source: 'curated' },
184
243
  };
244
+ catalogIds = null;
185
245
  }
186
- const catalog = getModelCatalog(agent, version);
187
- return tierizeModels(agent, catalog?.models ?? []);
246
+ else {
247
+ const catalog = getModelCatalog(agent, version);
248
+ const models = catalog?.models ?? [];
249
+ const ladder = CURATED_LADDERS[agent];
250
+ base = ladder ? tierizeFromLadder(ladder, models) : tierizeModels(agent, models);
251
+ catalogIds = catalog ? new Set(models.map((m) => m.id)) : null;
252
+ }
253
+ const overrides = resolveTierOverride(agent, version);
254
+ return applyTierOverrides(overrides, `${agent}@${version}`, catalogIds, base);
255
+ }
256
+ /**
257
+ * Apply user overrides on top of the auto/curated map. Pure (takes the resolved
258
+ * override map, no config lookup) so it is directly testable. An overridden id is
259
+ * used only when the version actually ships it (or when there is no catalog to
260
+ * check, e.g. Droid); otherwise the tier keeps its base value with a note.
261
+ */
262
+ export function applyTierOverrides(overrides, label, catalogIds, base) {
263
+ if (Object.keys(overrides).length === 0)
264
+ return base;
265
+ const out = { ...base };
266
+ for (const t of MODEL_TIERS) {
267
+ const id = overrides[t];
268
+ if (!id)
269
+ continue;
270
+ if (!catalogIds || catalogIds.has(id)) {
271
+ out[t] = { tier: t, model: id, source: 'override' };
272
+ }
273
+ else {
274
+ out[t] = { ...base[t], note: `override "${id}" not shipped by ${label}; kept the ${base[t].source ?? 'auto'} pick`, source: base[t].source ?? 'auto' };
275
+ }
276
+ }
277
+ return out;
188
278
  }
189
279
  /**
190
- * Map a harness's catalog models onto the four tiers. Pure (no catalog lookup)
191
- * so it is directly testable with synthetic inputs. Ranks the models, collapses
192
- * variants, buckets onto cheap/default/best/ultra, and clamps absent tiers down
193
- * to the nearest lower one. A single-model harness maps the tiers to reasoning
194
- * effort instead of models.
280
+ * Map a harness's catalog models onto the four tiers. Pure (no catalog lookup) so
281
+ * it is directly testable. A single-model harness maps the tiers to reasoning effort.
195
282
  */
196
283
  export function tierizeModels(agent, models) {
197
284
  const rungs = rankCatalog(agent, models);
@@ -200,28 +287,10 @@ export function tierizeModels(agent, models) {
200
287
  const only = rungs[0].id;
201
288
  const map = {};
202
289
  for (const t of MODEL_TIERS)
203
- map[t] = { tier: t, model: only, effort: TIER_EFFORT[t], note: 'single model — tier maps to reasoning effort' };
204
- return map;
205
- }
206
- const n = rungs.length;
207
- const map = {};
208
- if (n === 0) {
209
- // Fail-safe: no catalog -> every tier null, caller drops the --model flag.
210
- for (const t of MODEL_TIERS)
211
- map[t] = { tier: t, model: null };
290
+ map[t] = { tier: t, model: only, effort: TIER_EFFORT[t], note: 'single model — tier maps to reasoning effort', source: 'auto' };
212
291
  return map;
213
292
  }
214
- // Map each tier onto a rung; a tier that shares the rung of the tier below it
215
- // has no distinct rung of its own, so mark it clamped for an honest display.
216
- for (let i = 0; i < MODEL_TIERS.length; i++) {
217
- const t = MODEL_TIERS[i];
218
- const idx = rungIndexFor(i, n);
219
- const shared = i > 0 && rungIndexFor(i - 1, n) === idx;
220
- map[t] = shared
221
- ? { tier: t, model: rungs[idx].id, clampedFrom: MODEL_TIERS[i - 1], note: `no distinct ${t} rung; using ${MODEL_TIERS[i - 1]}` }
222
- : { tier: t, model: rungs[idx].id };
223
- }
224
- return map;
293
+ return bucketRungs(rungs);
225
294
  }
226
295
  /** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
227
296
  export function resolveTier(agent, version, tier) {
@@ -60,6 +60,22 @@ export interface ModelSource {
60
60
  * Returns null if nothing usable is found.
61
61
  */
62
62
  export declare function locateModelSource(agent: AgentId, version: string): ModelSource | null;
63
+ /**
64
+ * Extract Claude's model catalog from its bundle/binary.
65
+ *
66
+ * Bundle/binary contains:
67
+ * - alias map: {opus:"claude-opus-4-7",sonnet:"claude-sonnet-4-6",haiku:"..."}
68
+ * - per-cloud maps: {firstParty:"claude-opus-4-5-...",bedrock:"...",vertex:"...",...}
69
+ * - constants: {OPUS_ID:"...",OPUS_NAME:"...",SONNET_ID:"...",...}
70
+ */
71
+ /**
72
+ * Drop a bare `claude-<family>-<major>` (e.g. `claude-opus-4`) when a more specific
73
+ * sibling (`claude-opus-4-8`) is present. The bare form is only ever an internal
74
+ * `.includes("claude-opus-4")` prefix-check string in the binary, not a submittable
75
+ * id (issue #1892); a bare id with no sibling (e.g. `claude-sonnet-5`) is a real
76
+ * current model and is kept.
77
+ */
78
+ export declare function dropBareLegacyIds(ids: string[]): string[];
63
79
  /**
64
80
  * Parse `grok models` stdout into a catalog. Exported for unit tests.
65
81
  *
@@ -21,7 +21,7 @@ const CACHE_PATH = getModelsCachePath();
21
21
  * Bump when the extractor logic changes shape in an incompatible way so cached
22
22
  * catalogs from older agents-cli builds are re-extracted.
23
23
  */
24
- const CACHE_SCHEMA_VERSION = 3;
24
+ const CACHE_SCHEMA_VERSION = 4;
25
25
  /**
26
26
  * How long a cached 0-model extraction is trusted before we retry it. Bounds
27
27
  * the self-healing window for a transient failure (mid-install, a broken
@@ -310,6 +310,19 @@ function extractStrings(filePath, minLen = 6) {
310
310
  * - per-cloud maps: {firstParty:"claude-opus-4-5-...",bedrock:"...",vertex:"...",...}
311
311
  * - constants: {OPUS_ID:"...",OPUS_NAME:"...",SONNET_ID:"...",...}
312
312
  */
313
+ /**
314
+ * Drop a bare `claude-<family>-<major>` (e.g. `claude-opus-4`) when a more specific
315
+ * sibling (`claude-opus-4-8`) is present. The bare form is only ever an internal
316
+ * `.includes("claude-opus-4")` prefix-check string in the binary, not a submittable
317
+ * id (issue #1892); a bare id with no sibling (e.g. `claude-sonnet-5`) is a real
318
+ * current model and is kept.
319
+ */
320
+ export function dropBareLegacyIds(ids) {
321
+ return ids.filter((id) => {
322
+ const bareMajor = /^claude-[a-z]+-\d+$/.test(id);
323
+ return !(bareMajor && ids.some((o) => o !== id && o.startsWith(`${id}-`)));
324
+ });
325
+ }
313
326
  function extractClaudeCatalog(text) {
314
327
  const aliases = {};
315
328
  const aliasMapMatch = text.match(/\{opus:"(claude-[^"]+)",sonnet:"(claude-[^"]+)",haiku:"(claude-[^"]+)"\}/);
@@ -380,8 +393,9 @@ function extractClaudeCatalog(text) {
380
393
  let sm;
381
394
  while ((sm = idRe.exec(text)) !== null)
382
395
  scanned.add(sm[0]);
383
- if (scanned.size >= 2)
384
- models = build(scanned);
396
+ const filtered = dropBareLegacyIds([...scanned]);
397
+ if (filtered.length >= 2)
398
+ models = build(filtered);
385
399
  }
386
400
  return { models, aliases };
387
401
  }
@@ -91,6 +91,21 @@ export declare function setKeychainBackendForTest(b: KeychainBackend | null): Ke
91
91
  * fast-path must not engage. Always false in production (`backend` is null). */
92
92
  export declare function isKeychainBackendOverridden(): boolean;
93
93
  export declare const HMAC_KEY_ITEM = "agents-cli.hmackey";
94
+ interface HmacKeyRecord {
95
+ v: number;
96
+ /** 64 hex chars — the raw HMAC-SHA256 key. */
97
+ k: string;
98
+ /** True once the one-time re-key has moved every cleartext-named item. */
99
+ migrated: boolean;
100
+ /** Old cleartext services whose hashed copies are verified but whose
101
+ * originals are not yet deleted (crash-resume list; deletes are silent). */
102
+ pendingDeletes?: string[];
103
+ /** True once this record has been re-stored no-ACL to heal a hmackey item that
104
+ * an OLD helper (pre the metadata/hmackey no-ACL migration fix) re-stamped with
105
+ * a biometry ACL. Set on the first read that heals it, so the heal runs exactly
106
+ * once per machine and never churns the keychain afterward. */
107
+ healedNoAcl?: boolean;
108
+ }
94
109
  /** Force hashed service names on with a fixed key (test only). Pass null to
95
110
  * restore lazy production resolution. Composes with setKeychainBackendForTest
96
111
  * so unit tests exercise the exact transform production uses. */
@@ -106,6 +121,20 @@ export declare function withRawKeychainServiceNames<T>(fn: () => T): T;
106
121
  * the re-key migration and tests; runtime callers go through the primitives,
107
122
  * which apply this transparently. */
108
123
  export declare function hashedServiceName(item: string, key: Buffer): string;
124
+ /**
125
+ * Heal a `hmackey` item that an OLD helper (pre the metadata/hmackey no-ACL
126
+ * migration fix) re-stamped with a biometry ACL. Such an item makes EVERY hashed
127
+ * keychain lookup pop the generic "Agents CLI needs to authenticate" sheet,
128
+ * because the HMAC key is read before every hashed name resolves. The migration
129
+ * fix stopped the re-stamping but never un-stamped an already-damaged item, and
130
+ * nothing else re-stores it once hashing is already active — so it prompts forever.
131
+ *
132
+ * This re-stores the record no-ACL exactly once per machine (guarded by
133
+ * `healedNoAcl`), turning every future read silent. The read that produced `rec`
134
+ * has already happened (and already prompted if it was ACL'd); this only writes.
135
+ * Returns true if it healed. Exported for tests. No-op when already healed.
136
+ */
137
+ export declare function healHmacKeyNoAclOnce(rec: HmacKeyRecord): boolean;
109
138
  /**
110
139
  * The storage-layer service name for `item`: hashed when hashing is active,
111
140
  * the item itself otherwise. For callers that mix helper-enumerated