@namzu/sdk 45.1.0 → 46.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. package/CHANGELOG.md +111 -0
  2. package/dist/authorization/gate.d.ts +5 -2
  3. package/dist/authorization/gate.d.ts.map +1 -1
  4. package/dist/authorization/gate.js +25 -4
  5. package/dist/authorization/gate.js.map +1 -1
  6. package/dist/authorization/rules.d.ts.map +1 -1
  7. package/dist/authorization/rules.js +19 -0
  8. package/dist/authorization/rules.js.map +1 -1
  9. package/dist/authorization/shell-lexer.d.ts +24 -0
  10. package/dist/authorization/shell-lexer.d.ts.map +1 -1
  11. package/dist/authorization/shell-lexer.js +69 -51
  12. package/dist/authorization/shell-lexer.js.map +1 -1
  13. package/dist/pricing/catalogue.generated.d.ts.map +1 -1
  14. package/dist/pricing/catalogue.generated.js +28 -4
  15. package/dist/pricing/catalogue.generated.js.map +1 -1
  16. package/dist/public-runtime.d.ts +3 -3
  17. package/dist/public-runtime.d.ts.map +1 -1
  18. package/dist/public-runtime.js +4 -2
  19. package/dist/public-runtime.js.map +1 -1
  20. package/dist/public-tools.d.ts +3 -2
  21. package/dist/public-tools.d.ts.map +1 -1
  22. package/dist/public-tools.js +5 -2
  23. package/dist/public-tools.js.map +1 -1
  24. package/dist/public-types.d.ts +3 -3
  25. package/dist/public-types.d.ts.map +1 -1
  26. package/dist/registry/tool/callable.d.ts +22 -0
  27. package/dist/registry/tool/callable.d.ts.map +1 -0
  28. package/dist/registry/tool/callable.js +29 -0
  29. package/dist/registry/tool/callable.js.map +1 -0
  30. package/dist/registry/tool/execute.d.ts.map +1 -1
  31. package/dist/registry/tool/execute.js +8 -1
  32. package/dist/registry/tool/execute.js.map +1 -1
  33. package/dist/runtime/query/executor/tool-call-admission.d.ts +37 -11
  34. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -1
  35. package/dist/runtime/query/executor/tool-call-admission.js +38 -12
  36. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -1
  37. package/dist/runtime/query/executor.d.ts +4 -2
  38. package/dist/runtime/query/executor.d.ts.map +1 -1
  39. package/dist/runtime/query/executor.js +18 -5
  40. package/dist/runtime/query/executor.js.map +1 -1
  41. package/dist/runtime/query/review-policy.d.ts +43 -0
  42. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  43. package/dist/runtime/query/review-policy.js +59 -16
  44. package/dist/runtime/query/review-policy.js.map +1 -1
  45. package/dist/skills/registry.d.ts +6 -0
  46. package/dist/skills/registry.d.ts.map +1 -1
  47. package/dist/skills/registry.js +1 -0
  48. package/dist/skills/registry.js.map +1 -1
  49. package/dist/tools/builtins/computer-use-coordinates.d.ts +65 -0
  50. package/dist/tools/builtins/computer-use-coordinates.d.ts.map +1 -0
  51. package/dist/tools/builtins/computer-use-coordinates.js +123 -0
  52. package/dist/tools/builtins/computer-use-coordinates.js.map +1 -0
  53. package/dist/tools/builtins/computer-use-image.d.ts +77 -0
  54. package/dist/tools/builtins/computer-use-image.d.ts.map +1 -0
  55. package/dist/tools/builtins/computer-use-image.js +223 -0
  56. package/dist/tools/builtins/computer-use-image.js.map +1 -0
  57. package/dist/tools/builtins/computer-use.d.ts +519 -14
  58. package/dist/tools/builtins/computer-use.d.ts.map +1 -1
  59. package/dist/tools/builtins/computer-use.js +1188 -183
  60. package/dist/tools/builtins/computer-use.js.map +1 -1
  61. package/dist/tools/builtins/index.d.ts +2 -1
  62. package/dist/tools/builtins/index.d.ts.map +1 -1
  63. package/dist/tools/builtins/index.js +1 -1
  64. package/dist/tools/builtins/index.js.map +1 -1
  65. package/dist/tools/builtins/skill.d.ts +74 -0
  66. package/dist/tools/builtins/skill.d.ts.map +1 -1
  67. package/dist/tools/builtins/skill.js +214 -174
  68. package/dist/tools/builtins/skill.js.map +1 -1
  69. package/dist/tools/defineTool.d.ts +2 -0
  70. package/dist/tools/defineTool.d.ts.map +1 -1
  71. package/dist/tools/defineTool.js +7 -0
  72. package/dist/tools/defineTool.js.map +1 -1
  73. package/dist/tools/schedules/present.d.ts.map +1 -1
  74. package/dist/tools/schedules/present.js +2 -0
  75. package/dist/tools/schedules/present.js.map +1 -1
  76. package/dist/tools/schedules/schedule-tool.d.ts +6 -5
  77. package/dist/tools/schedules/schedule-tool.d.ts.map +1 -1
  78. package/dist/tools/schedules/schedule-tool.js +121 -30
  79. package/dist/tools/schedules/schedule-tool.js.map +1 -1
  80. package/dist/tools/schedules/types.d.ts +69 -1
  81. package/dist/tools/schedules/types.d.ts.map +1 -1
  82. package/dist/types/authorization/index.d.ts +21 -0
  83. package/dist/types/authorization/index.d.ts.map +1 -1
  84. package/dist/types/authorization/index.js +5 -0
  85. package/dist/types/authorization/index.js.map +1 -1
  86. package/dist/types/computer-use/index.d.ts +176 -0
  87. package/dist/types/computer-use/index.d.ts.map +1 -1
  88. package/dist/types/computer-use/index.js.map +1 -1
  89. package/dist/types/tool/index.d.ts +19 -0
  90. package/dist/types/tool/index.d.ts.map +1 -1
  91. package/dist/types/tool/index.js.map +1 -1
  92. package/package.json +3 -1
  93. package/src/authorization/gate.ts +25 -4
  94. package/src/authorization/rules.ts +18 -0
  95. package/src/authorization/shell-lexer.ts +73 -45
  96. package/src/pricing/catalogue.generated.ts +28 -4
  97. package/src/pricing/rates.source.json +27 -6
  98. package/src/public-runtime.ts +16 -2
  99. package/src/public-tools.ts +19 -1
  100. package/src/public-types.ts +5 -0
  101. package/src/registry/tool/callable.ts +37 -0
  102. package/src/registry/tool/execute.ts +8 -1
  103. package/src/runtime/query/executor/tool-call-admission.ts +60 -16
  104. package/src/runtime/query/executor.ts +19 -4
  105. package/src/runtime/query/review-policy.ts +103 -15
  106. package/src/skills/registry.ts +8 -0
  107. package/src/tools/builtins/computer-use-coordinates.ts +144 -0
  108. package/src/tools/builtins/computer-use-image.ts +278 -0
  109. package/src/tools/builtins/computer-use.ts +1485 -191
  110. package/src/tools/builtins/index.ts +7 -1
  111. package/src/tools/builtins/skill.ts +304 -174
  112. package/src/tools/defineTool.ts +10 -0
  113. package/src/tools/schedules/present.ts +2 -0
  114. package/src/tools/schedules/schedule-tool.ts +139 -33
  115. package/src/tools/schedules/types.ts +69 -1
  116. package/src/types/authorization/index.ts +21 -0
  117. package/src/types/computer-use/index.ts +202 -0
  118. package/src/types/tool/index.ts +19 -0
@@ -56,11 +56,32 @@
56
56
  {
57
57
  "providerId": "anthropic",
58
58
  "unmetered": false,
59
- "verified": "2026-08-10",
59
+ "verified": "2026-09-24",
60
60
  "promptIncludesCacheReads": false,
61
61
  "reportsCacheWrites": true,
62
- "note": "The last two rows exist because the driver's own offline catalogue offers those models and this table did not price them — a lookup-key gap that reads as `cost unknown` and is invisible to the generator's own check, which only proves the module matches this file. `providers/anthropic` asserts the two lists agree. Vendor list prices. Cache read is the vendor's published 0.1x of the input rate; cache write is its published 1.25x for the five-minute entry, which is the default TTL and the one the driver's `cacheControl: { type: 'auto' }` asks for. A time-limited introductory rate applied to one model at the time of entry; this table has no time dimension and records the list rate, so a turn inside a promotional window is over-reported rather than mis-modelled.",
62
+ "note": "The last two rows exist because the driver's own offline catalogue offers those models and this table did not price them — a lookup-key gap that reads as `cost unknown` and is invisible to the generator's own check, which only proves the module matches this file. `providers/anthropic` asserts the two lists agree. Vendor list prices, every row checked against its pricing page on the `verified` date. Cache read is the vendor's published 0.1x of the input rate, except on the three models it prices differently: 0.05x on claude-opus-5-5 and 0.025x on claude-fable-5-1 and claude-mythos-5-1, entered at the published rate rather than derived. Cache write is its published 1.25x for the five-minute entry, which is the default TTL and the one the driver's `cacheControl: { type: 'auto' }` asks for. claude-sonnet-5 launched at an introductory $2/$10 with $3/$15 scheduled from 2026-09-01, and this table recorded the scheduled rate; the vendor has since made $2/$10 the standard price and cancelled the increase. The table has no time dimension, so a future promotional window is still recorded at the list rate and over-reported rather than mis-modelled.",
63
63
  "models": [
64
+ {
65
+ "id": "claude-fable-5-1",
66
+ "inputPer1M": 10,
67
+ "outputPer1M": 50,
68
+ "cacheReadPer1M": 0.25,
69
+ "cacheWritePer1M": 12.5
70
+ },
71
+ {
72
+ "id": "claude-mythos-5-1",
73
+ "inputPer1M": 10,
74
+ "outputPer1M": 50,
75
+ "cacheReadPer1M": 0.25,
76
+ "cacheWritePer1M": 12.5
77
+ },
78
+ {
79
+ "id": "claude-opus-5-5",
80
+ "inputPer1M": 4,
81
+ "outputPer1M": 20,
82
+ "cacheReadPer1M": 0.2,
83
+ "cacheWritePer1M": 5
84
+ },
64
85
  {
65
86
  "id": "claude-fable-5",
66
87
  "inputPer1M": 10,
@@ -105,10 +126,10 @@
105
126
  },
106
127
  {
107
128
  "id": "claude-sonnet-5",
108
- "inputPer1M": 3,
109
- "outputPer1M": 15,
110
- "cacheReadPer1M": 0.3,
111
- "cacheWritePer1M": 3.75
129
+ "inputPer1M": 2,
130
+ "outputPer1M": 10,
131
+ "cacheReadPer1M": 0.2,
132
+ "cacheWritePer1M": 2.5
112
133
  },
113
134
  {
114
135
  "id": "claude-sonnet-4-6",
@@ -923,7 +923,13 @@ export {
923
923
  // The one reader of a bash command line the gate itself uses. A host that
924
924
  // writes a `predicate` rule about what a line runs decides on this reading,
925
925
  // so its rule and the SDK's own never disagree about where a quote ends.
926
- export { lexShellCommandLine } from './authorization/shell-lexer.js'
926
+ // `nestedShellCommand` says which `-c` payloads that reading already
927
+ // includes, so a host treats every other text a program runs as unread.
928
+ export {
929
+ NESTED_SHELLS,
930
+ lexShellCommandLine,
931
+ nestedShellCommand,
932
+ } from './authorization/shell-lexer.js'
927
933
 
928
934
  // NZ-BOOT-03: the module-attributed invariant registry. `compaction.ts` and
929
935
  // `claim-disk.ts` register themselves against the shared `invariants`
@@ -1302,6 +1308,8 @@ export {
1302
1308
  REVIEW_EXEMPT_WRITES,
1303
1309
  REVIEW_MODES,
1304
1310
  SANDBOX_ESCAPE_UNATTENDED_REFUSAL,
1311
+ SCREEN_CONSENT_DECLINED_FEEDBACK,
1312
+ SCREEN_CONSENT_UNATTENDED_REFUSAL,
1305
1313
  STRICT_MODE_REFUSAL,
1306
1314
  batchNeedsReview,
1307
1315
  createReviewHandler,
@@ -1433,7 +1441,13 @@ export type {
1433
1441
  export type { BroadcastHandoffDeps } from './session/handoff/broadcast.js'
1434
1442
  export type { SingleHandoffDeps } from './session/handoff/single.js'
1435
1443
  export type { InterventionChainLoader } from './session/intervention/prev-artifact.js'
1436
- export type { ActionInput } from './tools/builtins/computer-use.js'
1444
+ export type {
1445
+ ActionInput,
1446
+ ComputerUseTool,
1447
+ ComputerUseToolOptions,
1448
+ ImageSize,
1449
+ ScreenshotLimits,
1450
+ } from './tools/builtins/computer-use.js'
1437
1451
  export type { Project } from './types/project/entity.js'
1438
1452
  export type { CreatedLogger } from './utils/log/create-logger.js'
1439
1453
  export type {
@@ -66,7 +66,21 @@ export { WaitForJobTool } from './tools/builtins/wait-for-job.js'
66
66
  // NOT in the default builtin set: a turn with no skills has nothing for it
67
67
  // to do, and offering a tool that can only refuse is worse than not
68
68
  // offering it. Hosts register it alongside a skills registry.
69
- export { SKILL_TOOL_NAME, SkillTool, parseAllowedTools } from './tools/builtins/skill.js'
69
+ // `createSkillTool` takes what the host knows about where its model's tools
70
+ // run: `resolveModelDirectory` names the directory the model can open for
71
+ // each skill, which a sandboxed host's own path is not.
72
+ export {
73
+ SKILL_TOOL_NAME,
74
+ SkillTool,
75
+ createSkillTool,
76
+ parseAllowedTools,
77
+ } from './tools/builtins/skill.js'
78
+ export type {
79
+ SkillDirectoryContext,
80
+ SkillDirectoryRequest,
81
+ SkillDirectoryResolver,
82
+ SkillToolOptions,
83
+ } from './tools/builtins/skill.js'
70
84
  // Both declare `category: 'network'`, which is what the authorization
71
85
  // presets branch on. NOT in the default builtin set: a turn with no web
72
86
  // provider has nothing for them to do, and only the `unattended` preset --
@@ -105,7 +119,11 @@ export {
105
119
  } from './tools/builtins/structuredOutput.js'
106
120
  export {
107
121
  COMPUTER_USE_TOOL_NAME,
122
+ HIGH_RES_SCREENSHOT_LIMITS,
123
+ STANDARD_SCREENSHOT_LIMITS,
124
+ computerUseUnavailableReason,
108
125
  createComputerUseTool,
126
+ screenshotTargetSize,
109
127
  } from './tools/builtins/computer-use.js'
110
128
  // The browser contract: the two tools over a host a separate package
111
129
  // provides, and the canonicalisers a host and a site-rule compiler must share
@@ -272,6 +272,7 @@ export type { PermissionPreset } from './authorization/index.js'
272
272
  export type { EvaluateRuleOptions } from './authorization/rules.js'
273
273
  export type {
274
274
  ShellCommand,
275
+ NestedShellCommand,
275
276
  ShellLexOptions,
276
277
  ShellLexResult,
277
278
  ShellRedirection,
@@ -403,6 +404,7 @@ export type {
403
404
  ReviewExemption,
404
405
  ReviewMode,
405
406
  ReviewPolicyOptions,
407
+ ScreenConsentRecord,
406
408
  ToolReviewAnswer,
407
409
  ToolReviewPrompt,
408
410
  ToolReviewRequest,
@@ -671,11 +673,14 @@ export type {
671
673
  ScheduleBrowserSiteLevel,
672
674
  ScheduleConfirmAnswer,
673
675
  ScheduleConfirmRequest,
676
+ ScheduleJobChanges,
674
677
  ScheduleJobDraft,
675
678
  ScheduleJobPreview,
676
679
  ScheduleJobSummary,
680
+ ScheduleJobUpdateProposal,
677
681
  ScheduleRuleEffect,
678
682
  ScheduleToolHost,
683
+ ScheduleUpdateRequest,
679
684
  SessionLoop,
680
685
  SessionLoopHost,
681
686
  } from './tools/schedules/index.js'
@@ -0,0 +1,37 @@
1
+ import type { ToolRegistryContract } from '../../types/tool/index.js'
2
+
3
+ /**
4
+ * The tools a call on this step can reach: registered now, active, and on
5
+ * the step's allow-list when it has one. In registry order.
6
+ *
7
+ * This is what every "Available: …" a model is shown must say, because a
8
+ * model takes it literally and calls the next name on it. Two answers used
9
+ * to build that list separately and disagreed. A refusal echoed the step's
10
+ * allow-list, which is a snapshot taken when the request was built and so
11
+ * can name a tool unregistered since — a connector that disconnected — or
12
+ * one that is deferred or suspended, which the executor refuses. An
13
+ * unknown-tool error listed the whole registry, which on a narrowed step is
14
+ * mostly tools the step refuses. Each sent the model to a name the other
15
+ * one answered, and a run went round between them until it was stopped.
16
+ *
17
+ * `allowed` absent means the step is not narrowed. An EMPTY list is a step
18
+ * that may call nothing, and the answer is then empty too.
19
+ */
20
+ export function callableToolNames(
21
+ tools: Pick<ToolRegistryContract, 'listNames' | 'getAvailability'>,
22
+ allowed: readonly string[] | undefined,
23
+ ): string[] {
24
+ const permitted = allowed === undefined ? undefined : new Set(allowed)
25
+ return tools
26
+ .listNames()
27
+ .filter(
28
+ (name) =>
29
+ (permitted === undefined || permitted.has(name)) &&
30
+ tools.getAvailability(name) === 'active',
31
+ )
32
+ }
33
+
34
+ /** The list as a model reads it: comma-separated, or `(none)`. */
35
+ export function formatToolNames(names: readonly string[]): string {
36
+ return names.length > 0 ? names.join(', ') : '(none)'
37
+ }
@@ -20,6 +20,7 @@ import type {
20
20
  import { toErrorMessage } from '../../utils/error.js'
21
21
  import { cloneJsonValue as clonePreparedInput } from '../../utils/json-snapshot.js'
22
22
  import { ManagedRegistry } from '../ManagedRegistry.js'
23
+ import { callableToolNames, formatToolNames } from './callable.js'
23
24
  import { renderToolSchema, toolWireSchema } from './schema.js'
24
25
  import { ToolResultHalted, screenToolResult } from './screen.js'
25
26
 
@@ -636,9 +637,15 @@ Executable tool names, descriptions, and JSON input schemas are attached through
636
637
  // that may call nothing, and treating it as "no restriction" is
637
638
  // the fail-open reading this codebase has already been bitten by
638
639
  // once, in the delegate roster.
640
+ //
641
+ // What it says is available is what would run, not the list
642
+ // itself: the list was taken when the request was built, and a
643
+ // name on it may since have been unregistered, or be deferred or
644
+ // suspended. Echoing it sent a model to a name that was then
645
+ // answered "unknown tool". See `callableToolNames`.
639
646
  const allowed = context.allowedTools
640
647
  if (allowed !== undefined && !allowed.includes(toolName)) {
641
- const msg = `Tool "${toolName}" is not available on this step. Available: ${allowed.length > 0 ? allowed.join(', ') : '(none)'}`
648
+ const msg = `Tool "${toolName}" is not available on this step. Available: ${formatToolNames(callableToolNames(this, allowed))}`
642
649
  this.log.warn('Blocked a tool outside the step allow-list', {
643
650
  [GENAI.TOOL_NAME]: toolName,
644
651
  'namzu.registry.allowed': allowed.length,
@@ -1,4 +1,5 @@
1
1
  import { GENAI, NAMZU } from '../../../constants/telemetry/index.js'
2
+ import { callableToolNames, formatToolNames } from '../../../registry/tool/callable.js'
2
3
  import { renderToolSchema } from '../../../registry/tool/schema.js'
3
4
  import type { ToolCall } from '../../../types/message/index.js'
4
5
  import type { PluginHookResult } from '../../../types/plugin/index.js'
@@ -24,16 +25,43 @@ import { skippedToolResultText } from '../plugin-hooks.js'
24
25
  * here dispatches a tool, and nothing here writes a result: that is the
25
26
  * executor's half.
26
27
  *
27
- * The family reads exactly three things off the executor it serves, and they
28
+ * The family reads exactly four things off the executor it serves, and they
28
29
  * arrive as one value rather than as a captured reference: the tool registry
29
- * config, the event sink and the logger. `config` in particular is read
30
- * per call and never held — `ToolExecutor.setSandbox` REPLACES it, so a host
31
- * captured once would hand the next admission a stale sandbox.
30
+ * config, the event sink, the logger and the current step's allow-list.
31
+ * `config` in particular is read per call and never held —
32
+ * `ToolExecutor.setSandbox` REPLACES it, so a host captured once would hand
33
+ * the next admission a stale sandbox.
32
34
  */
33
35
  export interface ToolAdmissionHost {
34
36
  readonly config: ToolExecutorConfig
35
37
  readonly emitEvent: EmitEvent
36
38
  readonly log: Logger
39
+ /**
40
+ * What the current step may call: the step's own list where
41
+ * `prepareStep` gave one, the turn's `allowedTools` otherwise, absent
42
+ * when nothing narrowed the turn. NOT `config.allowedTools`, which is
43
+ * only the turn's; the step's list lives on the executor.
44
+ *
45
+ * Required, with `undefined` as a value, so that a host has to say.
46
+ * Leaving it out would list the whole registry to a model on a narrowed
47
+ * step, which is the answer that sent a model round in a circle.
48
+ */
49
+ readonly allowedTools: readonly string[] | undefined
50
+ }
51
+
52
+ /**
53
+ * The answer to a call naming a tool the registry does not hold, without
54
+ * the `Error: ` a direct call's result carries.
55
+ *
56
+ * It lists what the current step can call — the same list a step refusal
57
+ * gives — rather than what the registry holds. The registry's own
58
+ * "Not found" lists every tool it has, and on a narrowed step that is
59
+ * mostly tools the next call would be refused.
60
+ */
61
+ export function unknownToolMessage(host: ToolAdmissionHost, toolName: string): string {
62
+ return `Unknown tool "${toolName}". Available: ${formatToolNames(
63
+ callableToolNames(host.config.tools, host.allowedTools),
64
+ )}`
37
65
  }
38
66
 
39
67
  export async function runPreToolHook(
@@ -163,7 +191,11 @@ export async function prepareDirectCall(
163
191
  try {
164
192
  preparation = prepare.call(host.config.tools, toolName, parsed)
165
193
  } catch (err) {
166
- const message = `Error: Unknown or unavailable tool "${toolName}": ${toErrorMessage(err)}`
194
+ // A name the registry does not hold gets the step's list, not the
195
+ // registry's "Not found", which lists everything it holds.
196
+ const message = isUnregistered(host, toolName)
197
+ ? `Error: ${unknownToolMessage(host, toolName)}`
198
+ : `Error: Unknown or unavailable tool "${toolName}": ${toErrorMessage(err)}`
167
199
  const repair =
168
200
  !repairUsed && host.config.repairToolCall
169
201
  ? await requestRepair(host, toolCall, toolName, {
@@ -295,13 +327,15 @@ function interpretPreToolResults(
295
327
  * broken will not do better on a second look, and an unbounded loop
296
328
  * here is a hang rather than a degradation.
297
329
  *
298
- * `invalid_json` is the ONLY failure that stops the call here, and it
299
- * stopped it before this function existed too. `unknown_tool` and
300
- * `schema_validation` merely OFFER the repair and otherwise fall
301
- * through to the registry, which reports both with better messages —
302
- * its schema error already ships a "Required: <field>: <type>" hint the
303
- * model can self-correct from. So with no repairer configured this is
304
- * behaviorally identical to the bare `JSON.parse` it replaced.
330
+ * `invalid_json` stops the call here, and it stopped it before this
331
+ * function existed too. So does an `unknown_tool` the registry confirms
332
+ * it does not hold: the registry's own answer to that is "Not found",
333
+ * listing every tool it has, where the model needs the ones this step can
334
+ * call — see `unknownToolMessage`. `schema_validation`, and an unknown
335
+ * tool on a registry that cannot say, merely OFFER the repair and
336
+ * otherwise fall through to the registry, whose schema error already
337
+ * ships a "Required: <field>: <type>" hint the model can self-correct
338
+ * from.
305
339
  */
306
340
  export async function resolveCall(
307
341
  host: ToolAdmissionHost,
@@ -322,7 +356,10 @@ export async function resolveCall(
322
356
  : null
323
357
 
324
358
  if (!repair) {
325
- if (failure.reason === 'invalid_json') {
359
+ if (
360
+ failure.reason === 'invalid_json' ||
361
+ (failure.reason === 'unknown_tool' && isUnregistered(host, toolName))
362
+ ) {
326
363
  return { ok: false, toolName, message: failure.message }
327
364
  }
328
365
  return { ok: true, toolName, input: parseArguments(raw) }
@@ -392,11 +429,13 @@ function inspectCall(
392
429
  const tool = host.config.tools.get?.(toolName)
393
430
  if (!tool) {
394
431
  // Either the model named a tool that does not exist, or this
395
- // registry does not implement `get`. Both are the registry's to
396
- // answer; a repairer still gets offered the `unknown_tool` case.
432
+ // registry does not implement `get`. A repairer gets offered both;
433
+ // only the first is answered here, and with the step's list.
397
434
  return {
398
435
  reason: 'unknown_tool',
399
- message: `Error: Unknown tool "${toolName}"`,
436
+ message: isUnregistered(host, toolName)
437
+ ? `Error: ${unknownToolMessage(host, toolName)}`
438
+ : `Error: Unknown tool "${toolName}"`,
400
439
  }
401
440
  }
402
441
 
@@ -416,6 +455,11 @@ function inspectCall(
416
455
  return null
417
456
  }
418
457
 
458
+ /** The registry says it does not hold this name — not merely that it cannot look it up. */
459
+ export function isUnregistered(host: ToolAdmissionHost, toolName: string): boolean {
460
+ return typeof host.config.tools.has === 'function' && !host.config.tools.has(toolName)
461
+ }
462
+
419
463
  async function requestRepair(
420
464
  host: ToolAdmissionHost,
421
465
  toolCall: ToolCall,
@@ -55,11 +55,13 @@ import { type BackgroundJobRegistry, type JobProcess, bindOwner } from '../jobs/
55
55
  import {
56
56
  type ToolAdmissionHost,
57
57
  formatFailedToolOutput,
58
+ isUnregistered,
58
59
  prepareDirectCall,
59
60
  repairTruncatedCall,
60
61
  resolveCall,
61
62
  runPreToolHook,
62
63
  truncatedToolInputMessage,
64
+ unknownToolMessage,
63
65
  } from './executor/tool-call-admission.js'
64
66
  import { describeVisibleFileEvidence } from './file-evidence-context.js'
65
67
  import { seedObservationLedger } from './file-evidence-seed.js'
@@ -2202,10 +2204,12 @@ export class ToolExecutor {
2202
2204
  }
2203
2205
 
2204
2206
  /**
2205
- * The three things the admission family reads off this executor.
2207
+ * The four things the admission family reads off this executor.
2206
2208
  *
2207
2209
  * Built per call rather than held: `setSandbox` REPLACES `config`, so a
2208
- * host captured once would hand the next admission a stale sandbox.
2210
+ * host captured once would hand the next admission a stale sandbox. The
2211
+ * allow-list is the step's, taken the same way and for a like reason:
2212
+ * `setStepAllowedTools` replaces it every step.
2209
2213
  *
2210
2214
  * The one way this differs from the inline code it replaced, which
2211
2215
  * re-read `this.config` at every use: an admission that spans a
@@ -2216,7 +2220,12 @@ export class ToolExecutor {
2216
2220
  * before the loop, so nothing in this tree can tell them apart.
2217
2221
  */
2218
2222
  private admissionHost(): ToolAdmissionHost {
2219
- return { config: this.config, emitEvent: this.emitEvent, log: this.log }
2223
+ return {
2224
+ config: this.config,
2225
+ emitEvent: this.emitEvent,
2226
+ log: this.log,
2227
+ allowedTools: this.effectiveAllowedTools(),
2228
+ }
2220
2229
  }
2221
2230
 
2222
2231
  private async prepareNestedCall(
@@ -2251,10 +2260,16 @@ export class ToolExecutor {
2251
2260
  try {
2252
2261
  preparation = prepare.call(this.config.tools, toolName, input)
2253
2262
  } catch (err) {
2263
+ // A name the registry does not hold is answered with what this step
2264
+ // can call, as a model's own call is; the registry's "Not found"
2265
+ // lists everything it holds.
2266
+ const host = this.admissionHost()
2254
2267
  return {
2255
2268
  kind: 'synthetic',
2256
2269
  input,
2257
- message: `Tool "${toolName}" could not be prepared: ${toErrorMessage(err)}`,
2270
+ message: isUnregistered(host, toolName)
2271
+ ? unknownToolMessage(host, toolName)
2272
+ : `Tool "${toolName}" could not be prepared: ${toErrorMessage(err)}`,
2258
2273
  isError: true,
2259
2274
  }
2260
2275
  }
@@ -197,6 +197,28 @@ export const SANDBOX_ESCAPE_UNATTENDED_REFUSAL =
197
197
  export const OUTSIDE_ROOTS_UNATTENDED_REFUSAL =
198
198
  "Refused: a call in this batch reaches a path outside the working directory and the added directories, which needs a person to approve it each time, and nobody can be asked in this session. Nothing in this batch ran. Stay inside the working directory, or tell the user which directory you need so they can add it to the session (the CLI's --add-dir)."
199
199
 
200
+ /**
201
+ * What the model is told when a batch would show it the operator's screen for
202
+ * the first time in a session and nobody can be asked.
203
+ */
204
+ export const SCREEN_CONSENT_UNATTENDED_REFUSAL =
205
+ "Refused: this call would send what is on the user's screen to the model provider, which a person agrees to once per session, and nobody can be asked in this session. Nothing in this batch ran. Tell the user computer use needs their consent in an interactive session."
206
+
207
+ /** What the model is told when the operator declines to share the screen. */
208
+ export const SCREEN_CONSENT_DECLINED_FEEDBACK =
209
+ 'The user declined to share their screen in this session. Nothing in this batch ran. Do not take screenshots or read windows again; ask the user how they want to proceed.'
210
+
211
+ /**
212
+ * The sessions whose operator agreed to let the model see the screen.
213
+ *
214
+ * A host keeps one for as long as its sessions live and hands the same box to
215
+ * every policy it builds, so a mode switch keeps the answer and a new session
216
+ * (a different id) is asked again. The policy only adds to it.
217
+ */
218
+ export interface ScreenConsentRecord {
219
+ readonly sessions: Set<string>
220
+ }
221
+
200
222
  /** The batch a person is asked about. */
201
223
  export interface ToolReviewRequest {
202
224
  /** Originating session, preserved by createReviewHandler for host attribution. */
@@ -204,6 +226,14 @@ export interface ToolReviewRequest {
204
226
  /** Originating turn, preserved by createReviewHandler for host attribution. */
205
227
  readonly turnId?: TurnId
206
228
  readonly toolCalls: readonly ToolCallSummary[]
229
+ /**
230
+ * This batch would send the screen to the model provider for the first
231
+ * time in the session (see {@link ReviewPolicyOptions.screenConsent}).
232
+ * The question is whether the model may see the screen for the rest of
233
+ * the session; a yes also approves this batch, and later screen captures
234
+ * in the session run without asking.
235
+ */
236
+ readonly screenConsent?: true
207
237
  }
208
238
 
209
239
  export type ToolReviewAnswer =
@@ -250,6 +280,24 @@ export interface ReviewPolicyOptions {
250
280
  * either way.
251
281
  */
252
282
  readonly skillGrants?: 'honour' | 'ignore'
283
+ /**
284
+ * Ask once per session before the model first sees the screen.
285
+ *
286
+ * With a record, a batch holding a call that captures the screen
287
+ * ({@link capturesScreen}) in a session not yet in `sessions` is put to
288
+ * a person first, as a {@link ToolReviewRequest.screenConsent} request,
289
+ * even when every call in it only reads: in `prompt`, `accept-edits` and
290
+ * `plan`. A yes adds the session; a no refuses the batch. `strict`
291
+ * refuses such a call unless a rule allowed it, `auto` never asks, and a
292
+ * policy without a `prompt` refuses. A call a rule allowed is never
293
+ * asked about. Omitted, the screen is treated like any other read.
294
+ */
295
+ readonly screenConsent?: ScreenConsentRecord
296
+ /**
297
+ * Which calls capture the screen. Default: the tool's own
298
+ * `capturesScreen` declaration, read from `registry`; nothing without one.
299
+ */
300
+ readonly capturesScreen?: (name: string, input: unknown) => boolean
253
301
  }
254
302
 
255
303
  /**
@@ -276,10 +324,61 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
276
324
  options.exempt ??
277
325
  (registry ? (name, input) => isReviewExempt(registry, name, input) : () => false)
278
326
  const remembered = options.remembered ?? { all: false }
327
+ const capturesScreen =
328
+ options.capturesScreen ??
329
+ (registry
330
+ ? (name: string, input: unknown) => {
331
+ const tool = registry.get(name) ?? registry.get(name.toLowerCase())
332
+ return tool?.capturesScreen?.(input) === true
333
+ }
334
+ : () => false)
279
335
  return async (request): Promise<HITLResumeDecision> => {
280
336
  if (request.type !== 'tool_review') {
281
337
  return request.type === 'plan_approval' ? { action: 'approve_plan' } : { action: 'continue' }
282
338
  }
339
+ // The one answer a person gives for this batch. The screen question
340
+ // below shows the whole batch, so its yes also answers any question
341
+ // the rest of this function would ask; nobody is asked twice.
342
+ let answered: ToolReviewAnswer | undefined
343
+ const ask = async (screenConsent?: true): Promise<ToolReviewAnswer> =>
344
+ answered ??
345
+ (prompt as ToolReviewPrompt)({
346
+ sessionId: request.sessionId,
347
+ turnId: request.turnId,
348
+ toolCalls: request.toolCalls,
349
+ ...(screenConsent ? { screenConsent } : {}),
350
+ })
351
+ // The first look at the screen in a session is the operator's to allow:
352
+ // what is on it goes to the model provider, and a screenshot reads as
353
+ // harmlessly as a file read to every rule below. Asked once, in the
354
+ // modes where a person decides; a rule that allowed the call already
355
+ // said yes, and a call a rule denied will not run.
356
+ const consent = options.screenConsent
357
+ const sessionKey = String(request.sessionId ?? '')
358
+ if (
359
+ consent &&
360
+ mode !== 'auto' &&
361
+ !consent.sessions.has(sessionKey) &&
362
+ request.toolCalls.some(
363
+ (tc) =>
364
+ tc.authorization?.decision !== 'allow' &&
365
+ tc.authorization?.decision !== 'deny' &&
366
+ capturesScreen(tc.name, tc.input),
367
+ )
368
+ ) {
369
+ if (mode === 'strict') return { action: 'reject_tools', feedback: STRICT_MODE_REFUSAL }
370
+ if (!prompt) return { action: 'reject_tools', feedback: SCREEN_CONSENT_UNATTENDED_REFUSAL }
371
+ const answer = await ask(true)
372
+ if (answer.kind === 'reject') {
373
+ return {
374
+ action: 'reject_tools',
375
+ feedback: answer.feedback ?? SCREEN_CONSENT_DECLINED_FEEDBACK,
376
+ }
377
+ }
378
+ consent.sessions.add(sessionKey)
379
+ if (answer.kind === 'approve-all') remembered.all = true
380
+ answered = answer
381
+ }
283
382
  if (!batchNeedsReview(request.toolCalls, exempt)) {
284
383
  return { action: 'approve_tools' }
285
384
  }
@@ -326,11 +425,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
326
425
  ? { action: 'approve_tools', confirmedEscalations: escapes }
327
426
  : { action: 'reject_tools', feedback: SANDBOX_ESCAPE_UNATTENDED_REFUSAL }
328
427
  }
329
- const answer = await prompt({
330
- sessionId: request.sessionId,
331
- turnId: request.turnId,
332
- toolCalls: request.toolCalls,
333
- })
428
+ const answer = await ask()
334
429
  if (answer.kind === 'reject') {
335
430
  return {
336
431
  action: 'reject_tools',
@@ -348,11 +443,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
348
443
  // With nobody to ask it is refused, not approved.
349
444
  if (request.toolCalls.some((tc) => (tc.escalation?.outsidePaths?.length ?? 0) > 0)) {
350
445
  if (!prompt) return { action: 'reject_tools', feedback: OUTSIDE_ROOTS_UNATTENDED_REFUSAL }
351
- const answer = await prompt({
352
- sessionId: request.sessionId,
353
- turnId: request.turnId,
354
- toolCalls: request.toolCalls,
355
- })
446
+ const answer = await ask()
356
447
  if (answer.kind === 'reject') {
357
448
  return {
358
449
  action: 'reject_tools',
@@ -383,6 +474,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
383
474
  ),
384
475
  )
385
476
  if (
477
+ answered === undefined &&
386
478
  options.skillGrants !== 'ignore' &&
387
479
  needsPerson.length > 0 &&
388
480
  needsPerson.every(isSkillGranted)
@@ -392,11 +484,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
392
484
  skillGranted: needsPerson.map((tc) => tc.id),
393
485
  }
394
486
  }
395
- const answer = await prompt({
396
- sessionId: request.sessionId,
397
- turnId: request.turnId,
398
- toolCalls: request.toolCalls,
399
- })
487
+ const answer = await ask()
400
488
  switch (answer.kind) {
401
489
  case 'approve':
402
490
  return { action: 'approve_tools' }
@@ -220,6 +220,12 @@ export class SkillRegistry {
220
220
  registeredName: string
221
221
  description: string
222
222
  location: string
223
+ /**
224
+ * The skill's directory (`Skill.dirPath`); `location` is its SKILL.md.
225
+ * Always set here. Declared optional so a subclass that overrides
226
+ * `catalog()` without it still compiles.
227
+ */
228
+ directory?: string
223
229
  allowedTools?: string
224
230
  invocation?: 'model' | 'operator' | 'both'
225
231
  }[]
@@ -228,6 +234,7 @@ export class SkillRegistry {
228
234
  registeredName: string
229
235
  description: string
230
236
  location: string
237
+ directory: string
231
238
  allowedTools?: string
232
239
  invocation?: 'model' | 'operator' | 'both'
233
240
  }> = []
@@ -243,6 +250,7 @@ export class SkillRegistry {
243
250
  registeredName,
244
251
  description: skill.metadata.description,
245
252
  location: join(skill.dirPath, SKILL_FILENAME),
253
+ directory: skill.dirPath,
246
254
  ...(skill.metadata.allowedTools === undefined
247
255
  ? {}
248
256
  : { allowedTools: skill.metadata.allowedTools }),