@namzu/sdk 45.1.0 → 46.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +111 -0
- package/dist/authorization/gate.d.ts +5 -2
- package/dist/authorization/gate.d.ts.map +1 -1
- package/dist/authorization/gate.js +25 -4
- package/dist/authorization/gate.js.map +1 -1
- package/dist/authorization/rules.d.ts.map +1 -1
- package/dist/authorization/rules.js +19 -0
- package/dist/authorization/rules.js.map +1 -1
- package/dist/authorization/shell-lexer.d.ts +24 -0
- package/dist/authorization/shell-lexer.d.ts.map +1 -1
- package/dist/authorization/shell-lexer.js +69 -51
- package/dist/authorization/shell-lexer.js.map +1 -1
- package/dist/pricing/catalogue.generated.d.ts.map +1 -1
- package/dist/pricing/catalogue.generated.js +28 -4
- package/dist/pricing/catalogue.generated.js.map +1 -1
- package/dist/public-runtime.d.ts +3 -3
- package/dist/public-runtime.d.ts.map +1 -1
- package/dist/public-runtime.js +4 -2
- package/dist/public-runtime.js.map +1 -1
- package/dist/public-tools.d.ts +3 -2
- package/dist/public-tools.d.ts.map +1 -1
- package/dist/public-tools.js +5 -2
- package/dist/public-tools.js.map +1 -1
- package/dist/public-types.d.ts +3 -3
- package/dist/public-types.d.ts.map +1 -1
- package/dist/registry/tool/callable.d.ts +22 -0
- package/dist/registry/tool/callable.d.ts.map +1 -0
- package/dist/registry/tool/callable.js +29 -0
- package/dist/registry/tool/callable.js.map +1 -0
- package/dist/registry/tool/execute.d.ts.map +1 -1
- package/dist/registry/tool/execute.js +8 -1
- package/dist/registry/tool/execute.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +37 -11
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.js +38 -12
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -1
- package/dist/runtime/query/executor.d.ts +4 -2
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +18 -5
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/review-policy.d.ts +43 -0
- package/dist/runtime/query/review-policy.d.ts.map +1 -1
- package/dist/runtime/query/review-policy.js +59 -16
- package/dist/runtime/query/review-policy.js.map +1 -1
- package/dist/skills/registry.d.ts +6 -0
- package/dist/skills/registry.d.ts.map +1 -1
- package/dist/skills/registry.js +1 -0
- package/dist/skills/registry.js.map +1 -1
- package/dist/tools/builtins/computer-use-coordinates.d.ts +65 -0
- package/dist/tools/builtins/computer-use-coordinates.d.ts.map +1 -0
- package/dist/tools/builtins/computer-use-coordinates.js +123 -0
- package/dist/tools/builtins/computer-use-coordinates.js.map +1 -0
- package/dist/tools/builtins/computer-use-image.d.ts +77 -0
- package/dist/tools/builtins/computer-use-image.d.ts.map +1 -0
- package/dist/tools/builtins/computer-use-image.js +223 -0
- package/dist/tools/builtins/computer-use-image.js.map +1 -0
- package/dist/tools/builtins/computer-use.d.ts +519 -14
- package/dist/tools/builtins/computer-use.d.ts.map +1 -1
- package/dist/tools/builtins/computer-use.js +1188 -183
- package/dist/tools/builtins/computer-use.js.map +1 -1
- package/dist/tools/builtins/index.d.ts +2 -1
- package/dist/tools/builtins/index.d.ts.map +1 -1
- package/dist/tools/builtins/index.js +1 -1
- package/dist/tools/builtins/index.js.map +1 -1
- package/dist/tools/builtins/skill.d.ts +74 -0
- package/dist/tools/builtins/skill.d.ts.map +1 -1
- package/dist/tools/builtins/skill.js +214 -174
- package/dist/tools/builtins/skill.js.map +1 -1
- package/dist/tools/defineTool.d.ts +2 -0
- package/dist/tools/defineTool.d.ts.map +1 -1
- package/dist/tools/defineTool.js +7 -0
- package/dist/tools/defineTool.js.map +1 -1
- package/dist/tools/schedules/present.d.ts.map +1 -1
- package/dist/tools/schedules/present.js +2 -0
- package/dist/tools/schedules/present.js.map +1 -1
- package/dist/tools/schedules/schedule-tool.d.ts +6 -5
- package/dist/tools/schedules/schedule-tool.d.ts.map +1 -1
- package/dist/tools/schedules/schedule-tool.js +121 -30
- package/dist/tools/schedules/schedule-tool.js.map +1 -1
- package/dist/tools/schedules/types.d.ts +69 -1
- package/dist/tools/schedules/types.d.ts.map +1 -1
- package/dist/types/authorization/index.d.ts +21 -0
- package/dist/types/authorization/index.d.ts.map +1 -1
- package/dist/types/authorization/index.js +5 -0
- package/dist/types/authorization/index.js.map +1 -1
- package/dist/types/computer-use/index.d.ts +176 -0
- package/dist/types/computer-use/index.d.ts.map +1 -1
- package/dist/types/computer-use/index.js.map +1 -1
- package/dist/types/tool/index.d.ts +19 -0
- package/dist/types/tool/index.d.ts.map +1 -1
- package/dist/types/tool/index.js.map +1 -1
- package/package.json +3 -1
- package/src/authorization/gate.ts +25 -4
- package/src/authorization/rules.ts +18 -0
- package/src/authorization/shell-lexer.ts +73 -45
- package/src/pricing/catalogue.generated.ts +28 -4
- package/src/pricing/rates.source.json +27 -6
- package/src/public-runtime.ts +16 -2
- package/src/public-tools.ts +19 -1
- package/src/public-types.ts +5 -0
- package/src/registry/tool/callable.ts +37 -0
- package/src/registry/tool/execute.ts +8 -1
- package/src/runtime/query/executor/tool-call-admission.ts +60 -16
- package/src/runtime/query/executor.ts +19 -4
- package/src/runtime/query/review-policy.ts +103 -15
- package/src/skills/registry.ts +8 -0
- package/src/tools/builtins/computer-use-coordinates.ts +144 -0
- package/src/tools/builtins/computer-use-image.ts +278 -0
- package/src/tools/builtins/computer-use.ts +1485 -191
- package/src/tools/builtins/index.ts +7 -1
- package/src/tools/builtins/skill.ts +304 -174
- package/src/tools/defineTool.ts +10 -0
- package/src/tools/schedules/present.ts +2 -0
- package/src/tools/schedules/schedule-tool.ts +139 -33
- package/src/tools/schedules/types.ts +69 -1
- package/src/types/authorization/index.ts +21 -0
- package/src/types/computer-use/index.ts +202 -0
- package/src/types/tool/index.ts +19 -0
|
@@ -56,11 +56,32 @@
|
|
|
56
56
|
{
|
|
57
57
|
"providerId": "anthropic",
|
|
58
58
|
"unmetered": false,
|
|
59
|
-
"verified": "2026-
|
|
59
|
+
"verified": "2026-09-24",
|
|
60
60
|
"promptIncludesCacheReads": false,
|
|
61
61
|
"reportsCacheWrites": true,
|
|
62
|
-
"note": "The last two rows exist because the driver's own offline catalogue offers those models and this table did not price them — a lookup-key gap that reads as `cost unknown` and is invisible to the generator's own check, which only proves the module matches this file. `providers/anthropic` asserts the two lists agree. Vendor list prices. Cache read is the vendor's published 0.1x of the input rate
|
|
62
|
+
"note": "The last two rows exist because the driver's own offline catalogue offers those models and this table did not price them — a lookup-key gap that reads as `cost unknown` and is invisible to the generator's own check, which only proves the module matches this file. `providers/anthropic` asserts the two lists agree. Vendor list prices, every row checked against its pricing page on the `verified` date. Cache read is the vendor's published 0.1x of the input rate, except on the three models it prices differently: 0.05x on claude-opus-5-5 and 0.025x on claude-fable-5-1 and claude-mythos-5-1, entered at the published rate rather than derived. Cache write is its published 1.25x for the five-minute entry, which is the default TTL and the one the driver's `cacheControl: { type: 'auto' }` asks for. claude-sonnet-5 launched at an introductory $2/$10 with $3/$15 scheduled from 2026-09-01, and this table recorded the scheduled rate; the vendor has since made $2/$10 the standard price and cancelled the increase. The table has no time dimension, so a future promotional window is still recorded at the list rate and over-reported rather than mis-modelled.",
|
|
63
63
|
"models": [
|
|
64
|
+
{
|
|
65
|
+
"id": "claude-fable-5-1",
|
|
66
|
+
"inputPer1M": 10,
|
|
67
|
+
"outputPer1M": 50,
|
|
68
|
+
"cacheReadPer1M": 0.25,
|
|
69
|
+
"cacheWritePer1M": 12.5
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
"id": "claude-mythos-5-1",
|
|
73
|
+
"inputPer1M": 10,
|
|
74
|
+
"outputPer1M": 50,
|
|
75
|
+
"cacheReadPer1M": 0.25,
|
|
76
|
+
"cacheWritePer1M": 12.5
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
"id": "claude-opus-5-5",
|
|
80
|
+
"inputPer1M": 4,
|
|
81
|
+
"outputPer1M": 20,
|
|
82
|
+
"cacheReadPer1M": 0.2,
|
|
83
|
+
"cacheWritePer1M": 5
|
|
84
|
+
},
|
|
64
85
|
{
|
|
65
86
|
"id": "claude-fable-5",
|
|
66
87
|
"inputPer1M": 10,
|
|
@@ -105,10 +126,10 @@
|
|
|
105
126
|
},
|
|
106
127
|
{
|
|
107
128
|
"id": "claude-sonnet-5",
|
|
108
|
-
"inputPer1M":
|
|
109
|
-
"outputPer1M":
|
|
110
|
-
"cacheReadPer1M": 0.
|
|
111
|
-
"cacheWritePer1M":
|
|
129
|
+
"inputPer1M": 2,
|
|
130
|
+
"outputPer1M": 10,
|
|
131
|
+
"cacheReadPer1M": 0.2,
|
|
132
|
+
"cacheWritePer1M": 2.5
|
|
112
133
|
},
|
|
113
134
|
{
|
|
114
135
|
"id": "claude-sonnet-4-6",
|
package/src/public-runtime.ts
CHANGED
|
@@ -923,7 +923,13 @@ export {
|
|
|
923
923
|
// The one reader of a bash command line the gate itself uses. A host that
|
|
924
924
|
// writes a `predicate` rule about what a line runs decides on this reading,
|
|
925
925
|
// so its rule and the SDK's own never disagree about where a quote ends.
|
|
926
|
-
|
|
926
|
+
// `nestedShellCommand` says which `-c` payloads that reading already
|
|
927
|
+
// includes, so a host treats every other text a program runs as unread.
|
|
928
|
+
export {
|
|
929
|
+
NESTED_SHELLS,
|
|
930
|
+
lexShellCommandLine,
|
|
931
|
+
nestedShellCommand,
|
|
932
|
+
} from './authorization/shell-lexer.js'
|
|
927
933
|
|
|
928
934
|
// NZ-BOOT-03: the module-attributed invariant registry. `compaction.ts` and
|
|
929
935
|
// `claim-disk.ts` register themselves against the shared `invariants`
|
|
@@ -1302,6 +1308,8 @@ export {
|
|
|
1302
1308
|
REVIEW_EXEMPT_WRITES,
|
|
1303
1309
|
REVIEW_MODES,
|
|
1304
1310
|
SANDBOX_ESCAPE_UNATTENDED_REFUSAL,
|
|
1311
|
+
SCREEN_CONSENT_DECLINED_FEEDBACK,
|
|
1312
|
+
SCREEN_CONSENT_UNATTENDED_REFUSAL,
|
|
1305
1313
|
STRICT_MODE_REFUSAL,
|
|
1306
1314
|
batchNeedsReview,
|
|
1307
1315
|
createReviewHandler,
|
|
@@ -1433,7 +1441,13 @@ export type {
|
|
|
1433
1441
|
export type { BroadcastHandoffDeps } from './session/handoff/broadcast.js'
|
|
1434
1442
|
export type { SingleHandoffDeps } from './session/handoff/single.js'
|
|
1435
1443
|
export type { InterventionChainLoader } from './session/intervention/prev-artifact.js'
|
|
1436
|
-
export type {
|
|
1444
|
+
export type {
|
|
1445
|
+
ActionInput,
|
|
1446
|
+
ComputerUseTool,
|
|
1447
|
+
ComputerUseToolOptions,
|
|
1448
|
+
ImageSize,
|
|
1449
|
+
ScreenshotLimits,
|
|
1450
|
+
} from './tools/builtins/computer-use.js'
|
|
1437
1451
|
export type { Project } from './types/project/entity.js'
|
|
1438
1452
|
export type { CreatedLogger } from './utils/log/create-logger.js'
|
|
1439
1453
|
export type {
|
package/src/public-tools.ts
CHANGED
|
@@ -66,7 +66,21 @@ export { WaitForJobTool } from './tools/builtins/wait-for-job.js'
|
|
|
66
66
|
// NOT in the default builtin set: a turn with no skills has nothing for it
|
|
67
67
|
// to do, and offering a tool that can only refuse is worse than not
|
|
68
68
|
// offering it. Hosts register it alongside a skills registry.
|
|
69
|
-
|
|
69
|
+
// `createSkillTool` takes what the host knows about where its model's tools
|
|
70
|
+
// run: `resolveModelDirectory` names the directory the model can open for
|
|
71
|
+
// each skill, which a sandboxed host's own path is not.
|
|
72
|
+
export {
|
|
73
|
+
SKILL_TOOL_NAME,
|
|
74
|
+
SkillTool,
|
|
75
|
+
createSkillTool,
|
|
76
|
+
parseAllowedTools,
|
|
77
|
+
} from './tools/builtins/skill.js'
|
|
78
|
+
export type {
|
|
79
|
+
SkillDirectoryContext,
|
|
80
|
+
SkillDirectoryRequest,
|
|
81
|
+
SkillDirectoryResolver,
|
|
82
|
+
SkillToolOptions,
|
|
83
|
+
} from './tools/builtins/skill.js'
|
|
70
84
|
// Both declare `category: 'network'`, which is what the authorization
|
|
71
85
|
// presets branch on. NOT in the default builtin set: a turn with no web
|
|
72
86
|
// provider has nothing for them to do, and only the `unattended` preset --
|
|
@@ -105,7 +119,11 @@ export {
|
|
|
105
119
|
} from './tools/builtins/structuredOutput.js'
|
|
106
120
|
export {
|
|
107
121
|
COMPUTER_USE_TOOL_NAME,
|
|
122
|
+
HIGH_RES_SCREENSHOT_LIMITS,
|
|
123
|
+
STANDARD_SCREENSHOT_LIMITS,
|
|
124
|
+
computerUseUnavailableReason,
|
|
108
125
|
createComputerUseTool,
|
|
126
|
+
screenshotTargetSize,
|
|
109
127
|
} from './tools/builtins/computer-use.js'
|
|
110
128
|
// The browser contract: the two tools over a host a separate package
|
|
111
129
|
// provides, and the canonicalisers a host and a site-rule compiler must share
|
package/src/public-types.ts
CHANGED
|
@@ -272,6 +272,7 @@ export type { PermissionPreset } from './authorization/index.js'
|
|
|
272
272
|
export type { EvaluateRuleOptions } from './authorization/rules.js'
|
|
273
273
|
export type {
|
|
274
274
|
ShellCommand,
|
|
275
|
+
NestedShellCommand,
|
|
275
276
|
ShellLexOptions,
|
|
276
277
|
ShellLexResult,
|
|
277
278
|
ShellRedirection,
|
|
@@ -403,6 +404,7 @@ export type {
|
|
|
403
404
|
ReviewExemption,
|
|
404
405
|
ReviewMode,
|
|
405
406
|
ReviewPolicyOptions,
|
|
407
|
+
ScreenConsentRecord,
|
|
406
408
|
ToolReviewAnswer,
|
|
407
409
|
ToolReviewPrompt,
|
|
408
410
|
ToolReviewRequest,
|
|
@@ -671,11 +673,14 @@ export type {
|
|
|
671
673
|
ScheduleBrowserSiteLevel,
|
|
672
674
|
ScheduleConfirmAnswer,
|
|
673
675
|
ScheduleConfirmRequest,
|
|
676
|
+
ScheduleJobChanges,
|
|
674
677
|
ScheduleJobDraft,
|
|
675
678
|
ScheduleJobPreview,
|
|
676
679
|
ScheduleJobSummary,
|
|
680
|
+
ScheduleJobUpdateProposal,
|
|
677
681
|
ScheduleRuleEffect,
|
|
678
682
|
ScheduleToolHost,
|
|
683
|
+
ScheduleUpdateRequest,
|
|
679
684
|
SessionLoop,
|
|
680
685
|
SessionLoopHost,
|
|
681
686
|
} from './tools/schedules/index.js'
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import type { ToolRegistryContract } from '../../types/tool/index.js'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The tools a call on this step can reach: registered now, active, and on
|
|
5
|
+
* the step's allow-list when it has one. In registry order.
|
|
6
|
+
*
|
|
7
|
+
* This is what every "Available: …" a model is shown must say, because a
|
|
8
|
+
* model takes it literally and calls the next name on it. Two answers used
|
|
9
|
+
* to build that list separately and disagreed. A refusal echoed the step's
|
|
10
|
+
* allow-list, which is a snapshot taken when the request was built and so
|
|
11
|
+
* can name a tool unregistered since — a connector that disconnected — or
|
|
12
|
+
* one that is deferred or suspended, which the executor refuses. An
|
|
13
|
+
* unknown-tool error listed the whole registry, which on a narrowed step is
|
|
14
|
+
* mostly tools the step refuses. Each sent the model to a name the other
|
|
15
|
+
* one answered, and a run went round between them until it was stopped.
|
|
16
|
+
*
|
|
17
|
+
* `allowed` absent means the step is not narrowed. An EMPTY list is a step
|
|
18
|
+
* that may call nothing, and the answer is then empty too.
|
|
19
|
+
*/
|
|
20
|
+
export function callableToolNames(
|
|
21
|
+
tools: Pick<ToolRegistryContract, 'listNames' | 'getAvailability'>,
|
|
22
|
+
allowed: readonly string[] | undefined,
|
|
23
|
+
): string[] {
|
|
24
|
+
const permitted = allowed === undefined ? undefined : new Set(allowed)
|
|
25
|
+
return tools
|
|
26
|
+
.listNames()
|
|
27
|
+
.filter(
|
|
28
|
+
(name) =>
|
|
29
|
+
(permitted === undefined || permitted.has(name)) &&
|
|
30
|
+
tools.getAvailability(name) === 'active',
|
|
31
|
+
)
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** The list as a model reads it: comma-separated, or `(none)`. */
|
|
35
|
+
export function formatToolNames(names: readonly string[]): string {
|
|
36
|
+
return names.length > 0 ? names.join(', ') : '(none)'
|
|
37
|
+
}
|
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
import { toErrorMessage } from '../../utils/error.js'
|
|
21
21
|
import { cloneJsonValue as clonePreparedInput } from '../../utils/json-snapshot.js'
|
|
22
22
|
import { ManagedRegistry } from '../ManagedRegistry.js'
|
|
23
|
+
import { callableToolNames, formatToolNames } from './callable.js'
|
|
23
24
|
import { renderToolSchema, toolWireSchema } from './schema.js'
|
|
24
25
|
import { ToolResultHalted, screenToolResult } from './screen.js'
|
|
25
26
|
|
|
@@ -636,9 +637,15 @@ Executable tool names, descriptions, and JSON input schemas are attached through
|
|
|
636
637
|
// that may call nothing, and treating it as "no restriction" is
|
|
637
638
|
// the fail-open reading this codebase has already been bitten by
|
|
638
639
|
// once, in the delegate roster.
|
|
640
|
+
//
|
|
641
|
+
// What it says is available is what would run, not the list
|
|
642
|
+
// itself: the list was taken when the request was built, and a
|
|
643
|
+
// name on it may since have been unregistered, or be deferred or
|
|
644
|
+
// suspended. Echoing it sent a model to a name that was then
|
|
645
|
+
// answered "unknown tool". See `callableToolNames`.
|
|
639
646
|
const allowed = context.allowedTools
|
|
640
647
|
if (allowed !== undefined && !allowed.includes(toolName)) {
|
|
641
|
-
const msg = `Tool "${toolName}" is not available on this step. Available: ${
|
|
648
|
+
const msg = `Tool "${toolName}" is not available on this step. Available: ${formatToolNames(callableToolNames(this, allowed))}`
|
|
642
649
|
this.log.warn('Blocked a tool outside the step allow-list', {
|
|
643
650
|
[GENAI.TOOL_NAME]: toolName,
|
|
644
651
|
'namzu.registry.allowed': allowed.length,
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { GENAI, NAMZU } from '../../../constants/telemetry/index.js'
|
|
2
|
+
import { callableToolNames, formatToolNames } from '../../../registry/tool/callable.js'
|
|
2
3
|
import { renderToolSchema } from '../../../registry/tool/schema.js'
|
|
3
4
|
import type { ToolCall } from '../../../types/message/index.js'
|
|
4
5
|
import type { PluginHookResult } from '../../../types/plugin/index.js'
|
|
@@ -24,16 +25,43 @@ import { skippedToolResultText } from '../plugin-hooks.js'
|
|
|
24
25
|
* here dispatches a tool, and nothing here writes a result: that is the
|
|
25
26
|
* executor's half.
|
|
26
27
|
*
|
|
27
|
-
* The family reads exactly
|
|
28
|
+
* The family reads exactly four things off the executor it serves, and they
|
|
28
29
|
* arrive as one value rather than as a captured reference: the tool registry
|
|
29
|
-
* config, the event sink
|
|
30
|
-
* per call and never held —
|
|
31
|
-
*
|
|
30
|
+
* config, the event sink, the logger and the current step's allow-list.
|
|
31
|
+
* `config` in particular is read per call and never held —
|
|
32
|
+
* `ToolExecutor.setSandbox` REPLACES it, so a host captured once would hand
|
|
33
|
+
* the next admission a stale sandbox.
|
|
32
34
|
*/
|
|
33
35
|
export interface ToolAdmissionHost {
|
|
34
36
|
readonly config: ToolExecutorConfig
|
|
35
37
|
readonly emitEvent: EmitEvent
|
|
36
38
|
readonly log: Logger
|
|
39
|
+
/**
|
|
40
|
+
* What the current step may call: the step's own list where
|
|
41
|
+
* `prepareStep` gave one, the turn's `allowedTools` otherwise, absent
|
|
42
|
+
* when nothing narrowed the turn. NOT `config.allowedTools`, which is
|
|
43
|
+
* only the turn's; the step's list lives on the executor.
|
|
44
|
+
*
|
|
45
|
+
* Required, with `undefined` as a value, so that a host has to say.
|
|
46
|
+
* Leaving it out would list the whole registry to a model on a narrowed
|
|
47
|
+
* step, which is the answer that sent a model round in a circle.
|
|
48
|
+
*/
|
|
49
|
+
readonly allowedTools: readonly string[] | undefined
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* The answer to a call naming a tool the registry does not hold, without
|
|
54
|
+
* the `Error: ` a direct call's result carries.
|
|
55
|
+
*
|
|
56
|
+
* It lists what the current step can call — the same list a step refusal
|
|
57
|
+
* gives — rather than what the registry holds. The registry's own
|
|
58
|
+
* "Not found" lists every tool it has, and on a narrowed step that is
|
|
59
|
+
* mostly tools the next call would be refused.
|
|
60
|
+
*/
|
|
61
|
+
export function unknownToolMessage(host: ToolAdmissionHost, toolName: string): string {
|
|
62
|
+
return `Unknown tool "${toolName}". Available: ${formatToolNames(
|
|
63
|
+
callableToolNames(host.config.tools, host.allowedTools),
|
|
64
|
+
)}`
|
|
37
65
|
}
|
|
38
66
|
|
|
39
67
|
export async function runPreToolHook(
|
|
@@ -163,7 +191,11 @@ export async function prepareDirectCall(
|
|
|
163
191
|
try {
|
|
164
192
|
preparation = prepare.call(host.config.tools, toolName, parsed)
|
|
165
193
|
} catch (err) {
|
|
166
|
-
|
|
194
|
+
// A name the registry does not hold gets the step's list, not the
|
|
195
|
+
// registry's "Not found", which lists everything it holds.
|
|
196
|
+
const message = isUnregistered(host, toolName)
|
|
197
|
+
? `Error: ${unknownToolMessage(host, toolName)}`
|
|
198
|
+
: `Error: Unknown or unavailable tool "${toolName}": ${toErrorMessage(err)}`
|
|
167
199
|
const repair =
|
|
168
200
|
!repairUsed && host.config.repairToolCall
|
|
169
201
|
? await requestRepair(host, toolCall, toolName, {
|
|
@@ -295,13 +327,15 @@ function interpretPreToolResults(
|
|
|
295
327
|
* broken will not do better on a second look, and an unbounded loop
|
|
296
328
|
* here is a hang rather than a degradation.
|
|
297
329
|
*
|
|
298
|
-
* `invalid_json`
|
|
299
|
-
*
|
|
300
|
-
*
|
|
301
|
-
*
|
|
302
|
-
*
|
|
303
|
-
*
|
|
304
|
-
*
|
|
330
|
+
* `invalid_json` stops the call here, and it stopped it before this
|
|
331
|
+
* function existed too. So does an `unknown_tool` the registry confirms
|
|
332
|
+
* it does not hold: the registry's own answer to that is "Not found",
|
|
333
|
+
* listing every tool it has, where the model needs the ones this step can
|
|
334
|
+
* call — see `unknownToolMessage`. `schema_validation`, and an unknown
|
|
335
|
+
* tool on a registry that cannot say, merely OFFER the repair and
|
|
336
|
+
* otherwise fall through to the registry, whose schema error already
|
|
337
|
+
* ships a "Required: <field>: <type>" hint the model can self-correct
|
|
338
|
+
* from.
|
|
305
339
|
*/
|
|
306
340
|
export async function resolveCall(
|
|
307
341
|
host: ToolAdmissionHost,
|
|
@@ -322,7 +356,10 @@ export async function resolveCall(
|
|
|
322
356
|
: null
|
|
323
357
|
|
|
324
358
|
if (!repair) {
|
|
325
|
-
if (
|
|
359
|
+
if (
|
|
360
|
+
failure.reason === 'invalid_json' ||
|
|
361
|
+
(failure.reason === 'unknown_tool' && isUnregistered(host, toolName))
|
|
362
|
+
) {
|
|
326
363
|
return { ok: false, toolName, message: failure.message }
|
|
327
364
|
}
|
|
328
365
|
return { ok: true, toolName, input: parseArguments(raw) }
|
|
@@ -392,11 +429,13 @@ function inspectCall(
|
|
|
392
429
|
const tool = host.config.tools.get?.(toolName)
|
|
393
430
|
if (!tool) {
|
|
394
431
|
// Either the model named a tool that does not exist, or this
|
|
395
|
-
// registry does not implement `get`.
|
|
396
|
-
//
|
|
432
|
+
// registry does not implement `get`. A repairer gets offered both;
|
|
433
|
+
// only the first is answered here, and with the step's list.
|
|
397
434
|
return {
|
|
398
435
|
reason: 'unknown_tool',
|
|
399
|
-
message:
|
|
436
|
+
message: isUnregistered(host, toolName)
|
|
437
|
+
? `Error: ${unknownToolMessage(host, toolName)}`
|
|
438
|
+
: `Error: Unknown tool "${toolName}"`,
|
|
400
439
|
}
|
|
401
440
|
}
|
|
402
441
|
|
|
@@ -416,6 +455,11 @@ function inspectCall(
|
|
|
416
455
|
return null
|
|
417
456
|
}
|
|
418
457
|
|
|
458
|
+
/** The registry says it does not hold this name — not merely that it cannot look it up. */
|
|
459
|
+
export function isUnregistered(host: ToolAdmissionHost, toolName: string): boolean {
|
|
460
|
+
return typeof host.config.tools.has === 'function' && !host.config.tools.has(toolName)
|
|
461
|
+
}
|
|
462
|
+
|
|
419
463
|
async function requestRepair(
|
|
420
464
|
host: ToolAdmissionHost,
|
|
421
465
|
toolCall: ToolCall,
|
|
@@ -55,11 +55,13 @@ import { type BackgroundJobRegistry, type JobProcess, bindOwner } from '../jobs/
|
|
|
55
55
|
import {
|
|
56
56
|
type ToolAdmissionHost,
|
|
57
57
|
formatFailedToolOutput,
|
|
58
|
+
isUnregistered,
|
|
58
59
|
prepareDirectCall,
|
|
59
60
|
repairTruncatedCall,
|
|
60
61
|
resolveCall,
|
|
61
62
|
runPreToolHook,
|
|
62
63
|
truncatedToolInputMessage,
|
|
64
|
+
unknownToolMessage,
|
|
63
65
|
} from './executor/tool-call-admission.js'
|
|
64
66
|
import { describeVisibleFileEvidence } from './file-evidence-context.js'
|
|
65
67
|
import { seedObservationLedger } from './file-evidence-seed.js'
|
|
@@ -2202,10 +2204,12 @@ export class ToolExecutor {
|
|
|
2202
2204
|
}
|
|
2203
2205
|
|
|
2204
2206
|
/**
|
|
2205
|
-
* The
|
|
2207
|
+
* The four things the admission family reads off this executor.
|
|
2206
2208
|
*
|
|
2207
2209
|
* Built per call rather than held: `setSandbox` REPLACES `config`, so a
|
|
2208
|
-
* host captured once would hand the next admission a stale sandbox.
|
|
2210
|
+
* host captured once would hand the next admission a stale sandbox. The
|
|
2211
|
+
* allow-list is the step's, taken the same way and for a like reason:
|
|
2212
|
+
* `setStepAllowedTools` replaces it every step.
|
|
2209
2213
|
*
|
|
2210
2214
|
* The one way this differs from the inline code it replaced, which
|
|
2211
2215
|
* re-read `this.config` at every use: an admission that spans a
|
|
@@ -2216,7 +2220,12 @@ export class ToolExecutor {
|
|
|
2216
2220
|
* before the loop, so nothing in this tree can tell them apart.
|
|
2217
2221
|
*/
|
|
2218
2222
|
private admissionHost(): ToolAdmissionHost {
|
|
2219
|
-
return {
|
|
2223
|
+
return {
|
|
2224
|
+
config: this.config,
|
|
2225
|
+
emitEvent: this.emitEvent,
|
|
2226
|
+
log: this.log,
|
|
2227
|
+
allowedTools: this.effectiveAllowedTools(),
|
|
2228
|
+
}
|
|
2220
2229
|
}
|
|
2221
2230
|
|
|
2222
2231
|
private async prepareNestedCall(
|
|
@@ -2251,10 +2260,16 @@ export class ToolExecutor {
|
|
|
2251
2260
|
try {
|
|
2252
2261
|
preparation = prepare.call(this.config.tools, toolName, input)
|
|
2253
2262
|
} catch (err) {
|
|
2263
|
+
// A name the registry does not hold is answered with what this step
|
|
2264
|
+
// can call, as a model's own call is; the registry's "Not found"
|
|
2265
|
+
// lists everything it holds.
|
|
2266
|
+
const host = this.admissionHost()
|
|
2254
2267
|
return {
|
|
2255
2268
|
kind: 'synthetic',
|
|
2256
2269
|
input,
|
|
2257
|
-
message:
|
|
2270
|
+
message: isUnregistered(host, toolName)
|
|
2271
|
+
? unknownToolMessage(host, toolName)
|
|
2272
|
+
: `Tool "${toolName}" could not be prepared: ${toErrorMessage(err)}`,
|
|
2258
2273
|
isError: true,
|
|
2259
2274
|
}
|
|
2260
2275
|
}
|
|
@@ -197,6 +197,28 @@ export const SANDBOX_ESCAPE_UNATTENDED_REFUSAL =
|
|
|
197
197
|
export const OUTSIDE_ROOTS_UNATTENDED_REFUSAL =
|
|
198
198
|
"Refused: a call in this batch reaches a path outside the working directory and the added directories, which needs a person to approve it each time, and nobody can be asked in this session. Nothing in this batch ran. Stay inside the working directory, or tell the user which directory you need so they can add it to the session (the CLI's --add-dir)."
|
|
199
199
|
|
|
200
|
+
/**
|
|
201
|
+
* What the model is told when a batch would show it the operator's screen for
|
|
202
|
+
* the first time in a session and nobody can be asked.
|
|
203
|
+
*/
|
|
204
|
+
export const SCREEN_CONSENT_UNATTENDED_REFUSAL =
|
|
205
|
+
"Refused: this call would send what is on the user's screen to the model provider, which a person agrees to once per session, and nobody can be asked in this session. Nothing in this batch ran. Tell the user computer use needs their consent in an interactive session."
|
|
206
|
+
|
|
207
|
+
/** What the model is told when the operator declines to share the screen. */
|
|
208
|
+
export const SCREEN_CONSENT_DECLINED_FEEDBACK =
|
|
209
|
+
'The user declined to share their screen in this session. Nothing in this batch ran. Do not take screenshots or read windows again; ask the user how they want to proceed.'
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* The sessions whose operator agreed to let the model see the screen.
|
|
213
|
+
*
|
|
214
|
+
* A host keeps one for as long as its sessions live and hands the same box to
|
|
215
|
+
* every policy it builds, so a mode switch keeps the answer and a new session
|
|
216
|
+
* (a different id) is asked again. The policy only adds to it.
|
|
217
|
+
*/
|
|
218
|
+
export interface ScreenConsentRecord {
|
|
219
|
+
readonly sessions: Set<string>
|
|
220
|
+
}
|
|
221
|
+
|
|
200
222
|
/** The batch a person is asked about. */
|
|
201
223
|
export interface ToolReviewRequest {
|
|
202
224
|
/** Originating session, preserved by createReviewHandler for host attribution. */
|
|
@@ -204,6 +226,14 @@ export interface ToolReviewRequest {
|
|
|
204
226
|
/** Originating turn, preserved by createReviewHandler for host attribution. */
|
|
205
227
|
readonly turnId?: TurnId
|
|
206
228
|
readonly toolCalls: readonly ToolCallSummary[]
|
|
229
|
+
/**
|
|
230
|
+
* This batch would send the screen to the model provider for the first
|
|
231
|
+
* time in the session (see {@link ReviewPolicyOptions.screenConsent}).
|
|
232
|
+
* The question is whether the model may see the screen for the rest of
|
|
233
|
+
* the session; a yes also approves this batch, and later screen captures
|
|
234
|
+
* in the session run without asking.
|
|
235
|
+
*/
|
|
236
|
+
readonly screenConsent?: true
|
|
207
237
|
}
|
|
208
238
|
|
|
209
239
|
export type ToolReviewAnswer =
|
|
@@ -250,6 +280,24 @@ export interface ReviewPolicyOptions {
|
|
|
250
280
|
* either way.
|
|
251
281
|
*/
|
|
252
282
|
readonly skillGrants?: 'honour' | 'ignore'
|
|
283
|
+
/**
|
|
284
|
+
* Ask once per session before the model first sees the screen.
|
|
285
|
+
*
|
|
286
|
+
* With a record, a batch holding a call that captures the screen
|
|
287
|
+
* ({@link capturesScreen}) in a session not yet in `sessions` is put to
|
|
288
|
+
* a person first, as a {@link ToolReviewRequest.screenConsent} request,
|
|
289
|
+
* even when every call in it only reads: in `prompt`, `accept-edits` and
|
|
290
|
+
* `plan`. A yes adds the session; a no refuses the batch. `strict`
|
|
291
|
+
* refuses such a call unless a rule allowed it, `auto` never asks, and a
|
|
292
|
+
* policy without a `prompt` refuses. A call a rule allowed is never
|
|
293
|
+
* asked about. Omitted, the screen is treated like any other read.
|
|
294
|
+
*/
|
|
295
|
+
readonly screenConsent?: ScreenConsentRecord
|
|
296
|
+
/**
|
|
297
|
+
* Which calls capture the screen. Default: the tool's own
|
|
298
|
+
* `capturesScreen` declaration, read from `registry`; nothing without one.
|
|
299
|
+
*/
|
|
300
|
+
readonly capturesScreen?: (name: string, input: unknown) => boolean
|
|
253
301
|
}
|
|
254
302
|
|
|
255
303
|
/**
|
|
@@ -276,10 +324,61 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
|
|
|
276
324
|
options.exempt ??
|
|
277
325
|
(registry ? (name, input) => isReviewExempt(registry, name, input) : () => false)
|
|
278
326
|
const remembered = options.remembered ?? { all: false }
|
|
327
|
+
const capturesScreen =
|
|
328
|
+
options.capturesScreen ??
|
|
329
|
+
(registry
|
|
330
|
+
? (name: string, input: unknown) => {
|
|
331
|
+
const tool = registry.get(name) ?? registry.get(name.toLowerCase())
|
|
332
|
+
return tool?.capturesScreen?.(input) === true
|
|
333
|
+
}
|
|
334
|
+
: () => false)
|
|
279
335
|
return async (request): Promise<HITLResumeDecision> => {
|
|
280
336
|
if (request.type !== 'tool_review') {
|
|
281
337
|
return request.type === 'plan_approval' ? { action: 'approve_plan' } : { action: 'continue' }
|
|
282
338
|
}
|
|
339
|
+
// The one answer a person gives for this batch. The screen question
|
|
340
|
+
// below shows the whole batch, so its yes also answers any question
|
|
341
|
+
// the rest of this function would ask; nobody is asked twice.
|
|
342
|
+
let answered: ToolReviewAnswer | undefined
|
|
343
|
+
const ask = async (screenConsent?: true): Promise<ToolReviewAnswer> =>
|
|
344
|
+
answered ??
|
|
345
|
+
(prompt as ToolReviewPrompt)({
|
|
346
|
+
sessionId: request.sessionId,
|
|
347
|
+
turnId: request.turnId,
|
|
348
|
+
toolCalls: request.toolCalls,
|
|
349
|
+
...(screenConsent ? { screenConsent } : {}),
|
|
350
|
+
})
|
|
351
|
+
// The first look at the screen in a session is the operator's to allow:
|
|
352
|
+
// what is on it goes to the model provider, and a screenshot reads as
|
|
353
|
+
// harmlessly as a file read to every rule below. Asked once, in the
|
|
354
|
+
// modes where a person decides; a rule that allowed the call already
|
|
355
|
+
// said yes, and a call a rule denied will not run.
|
|
356
|
+
const consent = options.screenConsent
|
|
357
|
+
const sessionKey = String(request.sessionId ?? '')
|
|
358
|
+
if (
|
|
359
|
+
consent &&
|
|
360
|
+
mode !== 'auto' &&
|
|
361
|
+
!consent.sessions.has(sessionKey) &&
|
|
362
|
+
request.toolCalls.some(
|
|
363
|
+
(tc) =>
|
|
364
|
+
tc.authorization?.decision !== 'allow' &&
|
|
365
|
+
tc.authorization?.decision !== 'deny' &&
|
|
366
|
+
capturesScreen(tc.name, tc.input),
|
|
367
|
+
)
|
|
368
|
+
) {
|
|
369
|
+
if (mode === 'strict') return { action: 'reject_tools', feedback: STRICT_MODE_REFUSAL }
|
|
370
|
+
if (!prompt) return { action: 'reject_tools', feedback: SCREEN_CONSENT_UNATTENDED_REFUSAL }
|
|
371
|
+
const answer = await ask(true)
|
|
372
|
+
if (answer.kind === 'reject') {
|
|
373
|
+
return {
|
|
374
|
+
action: 'reject_tools',
|
|
375
|
+
feedback: answer.feedback ?? SCREEN_CONSENT_DECLINED_FEEDBACK,
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
consent.sessions.add(sessionKey)
|
|
379
|
+
if (answer.kind === 'approve-all') remembered.all = true
|
|
380
|
+
answered = answer
|
|
381
|
+
}
|
|
283
382
|
if (!batchNeedsReview(request.toolCalls, exempt)) {
|
|
284
383
|
return { action: 'approve_tools' }
|
|
285
384
|
}
|
|
@@ -326,11 +425,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
|
|
|
326
425
|
? { action: 'approve_tools', confirmedEscalations: escapes }
|
|
327
426
|
: { action: 'reject_tools', feedback: SANDBOX_ESCAPE_UNATTENDED_REFUSAL }
|
|
328
427
|
}
|
|
329
|
-
const answer = await
|
|
330
|
-
sessionId: request.sessionId,
|
|
331
|
-
turnId: request.turnId,
|
|
332
|
-
toolCalls: request.toolCalls,
|
|
333
|
-
})
|
|
428
|
+
const answer = await ask()
|
|
334
429
|
if (answer.kind === 'reject') {
|
|
335
430
|
return {
|
|
336
431
|
action: 'reject_tools',
|
|
@@ -348,11 +443,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
|
|
|
348
443
|
// With nobody to ask it is refused, not approved.
|
|
349
444
|
if (request.toolCalls.some((tc) => (tc.escalation?.outsidePaths?.length ?? 0) > 0)) {
|
|
350
445
|
if (!prompt) return { action: 'reject_tools', feedback: OUTSIDE_ROOTS_UNATTENDED_REFUSAL }
|
|
351
|
-
const answer = await
|
|
352
|
-
sessionId: request.sessionId,
|
|
353
|
-
turnId: request.turnId,
|
|
354
|
-
toolCalls: request.toolCalls,
|
|
355
|
-
})
|
|
446
|
+
const answer = await ask()
|
|
356
447
|
if (answer.kind === 'reject') {
|
|
357
448
|
return {
|
|
358
449
|
action: 'reject_tools',
|
|
@@ -383,6 +474,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
|
|
|
383
474
|
),
|
|
384
475
|
)
|
|
385
476
|
if (
|
|
477
|
+
answered === undefined &&
|
|
386
478
|
options.skillGrants !== 'ignore' &&
|
|
387
479
|
needsPerson.length > 0 &&
|
|
388
480
|
needsPerson.every(isSkillGranted)
|
|
@@ -392,11 +484,7 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
|
|
|
392
484
|
skillGranted: needsPerson.map((tc) => tc.id),
|
|
393
485
|
}
|
|
394
486
|
}
|
|
395
|
-
const answer = await
|
|
396
|
-
sessionId: request.sessionId,
|
|
397
|
-
turnId: request.turnId,
|
|
398
|
-
toolCalls: request.toolCalls,
|
|
399
|
-
})
|
|
487
|
+
const answer = await ask()
|
|
400
488
|
switch (answer.kind) {
|
|
401
489
|
case 'approve':
|
|
402
490
|
return { action: 'approve_tools' }
|
package/src/skills/registry.ts
CHANGED
|
@@ -220,6 +220,12 @@ export class SkillRegistry {
|
|
|
220
220
|
registeredName: string
|
|
221
221
|
description: string
|
|
222
222
|
location: string
|
|
223
|
+
/**
|
|
224
|
+
* The skill's directory (`Skill.dirPath`); `location` is its SKILL.md.
|
|
225
|
+
* Always set here. Declared optional so a subclass that overrides
|
|
226
|
+
* `catalog()` without it still compiles.
|
|
227
|
+
*/
|
|
228
|
+
directory?: string
|
|
223
229
|
allowedTools?: string
|
|
224
230
|
invocation?: 'model' | 'operator' | 'both'
|
|
225
231
|
}[]
|
|
@@ -228,6 +234,7 @@ export class SkillRegistry {
|
|
|
228
234
|
registeredName: string
|
|
229
235
|
description: string
|
|
230
236
|
location: string
|
|
237
|
+
directory: string
|
|
231
238
|
allowedTools?: string
|
|
232
239
|
invocation?: 'model' | 'operator' | 'both'
|
|
233
240
|
}> = []
|
|
@@ -243,6 +250,7 @@ export class SkillRegistry {
|
|
|
243
250
|
registeredName,
|
|
244
251
|
description: skill.metadata.description,
|
|
245
252
|
location: join(skill.dirPath, SKILL_FILENAME),
|
|
253
|
+
directory: skill.dirPath,
|
|
246
254
|
...(skill.metadata.allowedTools === undefined
|
|
247
255
|
? {}
|
|
248
256
|
: { allowedTools: skill.metadata.allowedTools }),
|