@kylecheng3146/agent-ops 0.1.22 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/packages/cli/src/args.js +25 -6
  2. package/dist/packages/cli/src/bin.js +6 -1
  3. package/dist/packages/cli/src/cli.js +136 -4
  4. package/dist/packages/cli/src/commands/hook.js +12 -13
  5. package/dist/packages/cli/src/commands/init.js +23 -10
  6. package/dist/packages/cli/src/commands/review.js +15 -5
  7. package/dist/packages/cli/src/hook-process.js +23 -5
  8. package/dist/runtime/src/adapters/claude/config.js +57 -7
  9. package/dist/runtime/src/adapters/claude/events.js +8 -0
  10. package/dist/runtime/src/adapters/claude/input.js +21 -7
  11. package/dist/runtime/src/adapters/claude/output.js +24 -0
  12. package/dist/runtime/src/hooks/completion-gate.js +19 -55
  13. package/dist/runtime/src/hooks/dispatch.js +11 -3
  14. package/dist/runtime/src/install/doctor.js +47 -1
  15. package/dist/runtime/src/install/harness.js +1 -1
  16. package/dist/runtime/src/install/ownership.js +7 -0
  17. package/dist/runtime/src/install/plan.js +24 -9
  18. package/dist/runtime/src/install/probes.js +18 -1
  19. package/dist/runtime/src/install/uninstall.js +2 -0
  20. package/dist/runtime/src/review/attestation.js +12 -1
  21. package/dist/runtime/src/review/execute.js +148 -24
  22. package/dist/runtime/src/review/host-sandbox.js +49 -0
  23. package/dist/runtime/src/review/invocation.js +18 -3
  24. package/dist/runtime/src/review/render.js +14 -0
  25. package/dist/runtime/src/security/trust.js +1 -2
  26. package/dist/runtime/src/task/completion.js +62 -0
  27. package/dist/runtime/src/task/service.js +76 -7
  28. package/dist/runtime/src/verify/change-surface.js +11 -2
  29. package/dist/runtime/src/verify/evidence.js +9 -1
  30. package/dist/runtime/src/verify/service.js +9 -1
  31. package/dist/runtime/src/verify/spawn.js +4 -3
  32. package/docs/en/guides/configuration.md +13 -9
  33. package/docs/en/spec/acceptance-and-evidence.md +25 -0
  34. package/docs/zh-TW/guides/configuration.md +10 -6
  35. package/docs/zh-TW/spec/acceptance-and-evidence.md +26 -1
  36. package/package.json +1 -1
@@ -284,8 +284,27 @@ export function parseArgs(argv) {
284
284
  if (helpSeen && versionSeen) {
285
285
  throw new CliArgumentError("CLI_CONFLICTING_ACTION", "--help and --version cannot be combined.");
286
286
  }
287
- if ((helpSeen || versionSeen) && command !== undefined) {
288
- throw new CliArgumentError("CLI_CONFLICTING_ACTION", "Global help or version cannot be combined with a command.");
287
+ // `<command> --help` asks about that command, so it answers instead of
288
+ // failing: an agent that cannot read a command's own option shapes guesses
289
+ // at them one rejected call at a time. Other options are ignored rather than
290
+ // rejected, so `task create --criterion <wrong> --help` still explains the
291
+ // shape it got wrong.
292
+ if (helpSeen &&
293
+ command !== undefined &&
294
+ command !== "help" &&
295
+ command !== "version") {
296
+ return {
297
+ command: "help",
298
+ helpTopic: command,
299
+ ...(action === undefined ? {} : { helpAction: action }),
300
+ profiles: [],
301
+ dryRun: false,
302
+ json,
303
+ yes: false
304
+ };
305
+ }
306
+ if (versionSeen && command !== undefined) {
307
+ throw new CliArgumentError("CLI_CONFLICTING_ACTION", "Global version cannot be combined with a command.");
289
308
  }
290
309
  if (helpSeen || versionSeen) {
291
310
  if (scope !== undefined ||
@@ -336,8 +355,9 @@ export function parseArgs(argv) {
336
355
  if (command !== "update" && targetVersion !== undefined) {
337
356
  throw new CliArgumentError("CLI_OPTION_NOT_ALLOWED", "--target-version may be used only with update.");
338
357
  }
339
- if (base !== undefined && command !== "verify" && command !== "review") {
340
- throw new CliArgumentError("CLI_OPTION_NOT_ALLOWED", "--base may be used only with verify or review.");
358
+ if (base !== undefined && command !== "verify" && command !== "review" &&
359
+ !(command === "task" && action === "complete")) {
360
+ throw new CliArgumentError("CLI_OPTION_NOT_ALLOWED", "--base may be used only with verify, review or task complete.");
341
361
  }
342
362
  if (hookTargets.length > 0 &&
343
363
  command !== "init" &&
@@ -401,9 +421,8 @@ export function parseArgs(argv) {
401
421
  evidence.length > 0 ||
402
422
  dryRun ||
403
423
  yes ||
404
- base !== undefined ||
405
424
  (taskId !== undefined && sessionId !== undefined))) {
406
- throw new CliArgumentError("CLI_OPTION_NOT_ALLOWED", "Verify accepts only scope, task or session, and json options.");
425
+ throw new CliArgumentError("CLI_OPTION_NOT_ALLOWED", "Verify accepts only scope, task or session, base, and json options.");
407
426
  }
408
427
  if (command === "review" &&
409
428
  (title !== undefined || sessionId !== undefined)) {
@@ -353,7 +353,12 @@ else {
353
353
  : { targetVersion: updateArgs.targetVersion })
354
354
  });
355
355
  }
356
- const taskService = new TaskService(new FileTaskStore(join(root, ".agent-ops", "tasks", "state.json"), root));
356
+ const taskService = new TaskService(new FileTaskStore(join(root, ".agent-ops", "tasks", "state.json"), root), { completion: {
357
+ root,
358
+ gitRunner: gitRunner(root),
359
+ ...(args.base === undefined ? {} : { base: args.base }),
360
+ loadConfig: async () => (await loadEffectiveConfig(root, args.scope === "user" ? "user" : "project")).config
361
+ } });
357
362
  if (args.command === "allow-stop") {
358
363
  const config = (await loadEffectiveConfig(root, "project")).config;
359
364
  return await runAllowStopCommand({
@@ -27,7 +27,7 @@ Commands:
27
27
  Manage independent task acceptance state
28
28
  verify Run configured verification
29
29
  review Run an independent review
30
- allow-stop Grant one fingerprint-bound agy Stop permit (requires --session)
30
+ allow-stop Grant one fingerprint-bound completion-gate Stop permit (requires --session)
31
31
  agy-run Run headless agy with a process-exit completion recheck
32
32
 
33
33
  Options:
@@ -37,7 +37,7 @@ Options:
37
37
  --profile <core|advisory|guardrails|loop> Repeatable
38
38
  --review-target <codex|agy|claude> Repeatable init option; external review
39
39
  targets in fallback-chain order
40
- --completion-gate Init only: enable the agy project-loop gate
40
+ --completion-gate Init only: enable the project-loop completion gate
41
41
  --check-auth Doctor only: probe each review target's
42
42
  authentication with one real call
43
43
  --task <id>
@@ -48,13 +48,141 @@ Options:
48
48
  --criterion <json> Repeatable
49
49
  --evidence <criterion-id=reference> Repeatable
50
50
  --session <id>
51
- --base <git-ref> Verify/review a clean committed range
51
+ --base <git-ref> Verify/review/complete a clean committed range
52
52
  --dry-run
53
53
  --json
54
54
  --yes
55
55
  --help
56
56
  --version
57
57
  `;
58
+ /**
59
+ * What `<command> --help` answers. Each entry states the option shapes that
60
+ * command actually accepts, because the global list cannot say which options
61
+ * belong to which command — and a caller that guesses learns only by being
62
+ * rejected.
63
+ */
64
+ export const COMMAND_HELP_TEXT = {
65
+ init: `Usage: agent-ops init [options]
66
+
67
+ Plan or install agent-ops into this repository.
68
+
69
+ Options:
70
+ --scope <project|user>
71
+ --harness <all|both|agy|claude|codex|opencode|comma-separated>
72
+ --hook-target <harness=surface-id> Repeatable
73
+ --profile <core|advisory|guardrails|loop> Repeatable
74
+ --review-target <codex|agy|claude> Repeatable, in fallback-chain order
75
+ --completion-gate Enable the project-loop completion gate
76
+ --dry-run Print the plan without writing
77
+ --json
78
+ --yes
79
+ `,
80
+ config: `Usage: agent-ops config explain [options]
81
+
82
+ Show where each effective configuration value came from.
83
+
84
+ Options:
85
+ --scope <project|user>
86
+ --json
87
+ `,
88
+ trust: `Usage: agent-ops trust <status|grant|revoke> [options]
89
+
90
+ Manage the explicit trust record this repository's commands require.
91
+
92
+ Options:
93
+ --scope <project|user>
94
+ --json
95
+ --yes
96
+ `,
97
+ doctor: `Usage: agent-ops doctor [options]
98
+
99
+ Diagnose the installation and report remediation for each failed check.
100
+
101
+ Options:
102
+ --scope <project|user>
103
+ --harness <all|both|agy|claude|codex|opencode|comma-separated>
104
+ --check-auth Probe each review target's authentication with one real call
105
+ --json
106
+ `,
107
+ update: `Usage: agent-ops update [options]
108
+
109
+ Update managed artifacts to this toolkit version.
110
+
111
+ Options:
112
+ --scope <project|user>
113
+ --harness <all|both|agy|claude|codex|opencode|comma-separated>
114
+ --target-version <version> Offline-capable update target
115
+ --dry-run
116
+ --json
117
+ --yes
118
+ `,
119
+ uninstall: `Usage: agent-ops uninstall [options]
120
+
121
+ Remove managed artifacts, leaving foreign handlers in place.
122
+
123
+ Options:
124
+ --scope <project|user>
125
+ --harness <all|both|agy|claude|codex|opencode|comma-separated>
126
+ --dry-run
127
+ --json
128
+ --yes
129
+ `,
130
+ task: `Usage: agent-ops task <create|status|attach|complete|archive|export> [options]
131
+
132
+ Manage independent task acceptance state. Task commands accept none of
133
+ --harness, --profile, --dry-run or --yes.
134
+
135
+ Options:
136
+ --title <text> create
137
+ --criterion <json> create, repeatable, two to five total
138
+ --parent <task-id> create: record a subtask; status: list subtasks
139
+ --task <id> status, attach, complete, archive, export
140
+ --session <id> attach, status
141
+ --evidence <criterion-id=reference> complete, repeatable
142
+ --base <git-ref> complete: a clean committed range
143
+ --json
144
+
145
+ Each --criterion is one JSON object with exactly these keys:
146
+
147
+ {"id":"kebab-id","description":"what must hold","verifierIds":["node-test"]}
148
+
149
+ Every criterion needs at least one verifierIds entry naming a verification
150
+ command id configured in .agent-ops/config.json. No other key is accepted.
151
+ `,
152
+ verify: `Usage: agent-ops verify [options]
153
+
154
+ Run the configured verification commands and record their evidence.
155
+
156
+ Options:
157
+ --task <id>
158
+ --session <id>
159
+ --base <git-ref> Verify a clean committed range
160
+ --json
161
+ `,
162
+ review: `Usage: agent-ops review [options]
163
+
164
+ Run one independent read-only review against the configured target chain.
165
+
166
+ Options:
167
+ --task <id>
168
+ --session <id>
169
+ --criterion <id> Repeatable: review only these task criteria
170
+ --harness <target> One configured review target
171
+ --base <git-ref>
172
+ --json
173
+ --yes Authorize the review call
174
+ `,
175
+ "allow-stop": `Usage: agent-ops allow-stop --session <id> [options]
176
+
177
+ Grant one fingerprint-bound Stop permit for the completion gate, on agy or
178
+ Claude Code. Requires user approval: the PreToolUse hook asks the user, so an
179
+ agent cannot self-authorize it.
180
+
181
+ Options:
182
+ --session <id> Required
183
+ --json
184
+ `
185
+ };
58
186
  function wantsJson(argv) {
59
187
  return argv.includes("--json");
60
188
  }
@@ -80,7 +208,11 @@ export async function runCli(argv, io, services) {
80
208
  return writeAndReturn(io, errorEnvelope("CLI_INTERNAL_ERROR", "Unable to parse command arguments."), json, 1);
81
209
  }
82
210
  if (args.command === "help") {
83
- return writeAndReturn(io, okEnvelope("CLI_HELP", { text: HELP_TEXT }), args.json, 0);
211
+ const topic = args.helpTopic;
212
+ return writeAndReturn(io, okEnvelope("CLI_HELP", {
213
+ text: topic === undefined ? HELP_TEXT : COMMAND_HELP_TEXT[topic],
214
+ ...(topic === undefined ? {} : { topic })
215
+ }), args.json, 0);
84
216
  }
85
217
  if (args.command === "version") {
86
218
  return writeAndReturn(io, okEnvelope("CLI_VERSION", { version: services.version }), args.json, 0);
@@ -20,25 +20,20 @@ export function normalizeHookInput(harness, input) {
20
20
  }
21
21
  }
22
22
  /**
23
- * Hooks are advisory infrastructure: every failure path stays fail-open with
24
- * exit code 0 so a broken toolkit can never wedge the harness.
23
+ * Advisory failures stay fail-open; an enabled completion Stop fails closed
24
+ * through the host's native decision protocol (still exit code 0).
25
25
  */
26
26
  export async function runHookCommand(options) {
27
27
  try {
28
28
  const { capabilities } = options.config.profiles.length === 0
29
29
  ? { capabilities: [] }
30
30
  : resolveCapabilities(options.config);
31
- let input;
32
- try {
33
- input = JSON.parse(options.stdin);
34
- }
35
- catch {
36
- return { exitCode: 0, stdout: "", stderr: "" };
37
- }
31
+ const input = JSON.parse(options.stdin);
38
32
  const descriptor = harnessDescriptor(options.harness);
39
33
  const normalized = normalizeHookInput(options.harness, input);
40
- if (normalized === null) {
41
- return { exitCode: 0, stdout: "", stderr: "" };
34
+ if (normalized === null ||
35
+ (options.completionGate !== undefined && options.event === "Stop" && normalized.event !== "stop")) {
36
+ throw new Error("Invalid hook input.");
42
37
  }
43
38
  const stopRegistration = descriptor.control.registrations.find(({ capability }) => capability === "optional-stop-verify");
44
39
  const stopVerification = options.stopVerification !== undefined &&
@@ -57,8 +52,12 @@ export async function runHookCommand(options) {
57
52
  return descriptor.runtime.formatOutput(options.event, result);
58
53
  }
59
54
  catch {
60
- if (options.harness === "agy" && options.completionGate !== undefined) {
61
- return harnessDescriptor("agy").runtime.formatOutput(options.event, {
55
+ // Fail closed on every host that enforces the gate: an exception here is
56
+ // exactly the case where a silent empty output would wave the stop through.
57
+ if ((options.harness === "agy" || options.harness === "claude") &&
58
+ options.event === "Stop" &&
59
+ options.completionGate !== undefined) {
60
+ return harnessDescriptor(options.harness).runtime.formatOutput(options.event, {
62
61
  action: "block",
63
62
  status: "UNKNOWN",
64
63
  code: "COMPLETION_GATE_UNAVAILABLE",
@@ -108,17 +108,30 @@ export async function runInitCommand(options) {
108
108
  : { completionGateEnabled: args.completionGate })
109
109
  });
110
110
  const trust = await trustChange(options, plan);
111
- const warnings = plan.harness.includes("agy") && options.agyWarning !== undefined
112
- ? (() => {
113
- try {
114
- const warning = options.agyWarning();
115
- return warning === undefined ? [] : [warning];
116
- }
117
- catch {
118
- return ["agy could not be probed; run `agent-ops doctor` to verify it."];
119
- }
120
- })()
111
+ // An installation with no verifier looks finished and is not: every task
112
+ // completion needs current PASS evidence from a required verifier, so the
113
+ // loop can never close. Detection stays conservative on purpose — it will
114
+ // not guess a test command — which makes saying so out loud the whole fix.
115
+ const verificationWarnings = plan.config.verification.commands.length === 0
116
+ ? [
117
+ "No verification command is configured, so no task can be completed. " +
118
+ "Add verification.commands to .agent-ops/config.json." +
119
+ (plan.verificationBlockers.length === 0
120
+ ? ""
121
+ : ` Detection stopped because — ${plan.verificationBlockers.join("; ")}`)
122
+ ]
121
123
  : [];
124
+ const warnings = [...verificationWarnings, ...(plan.harness.includes("agy") && options.agyWarning !== undefined
125
+ ? (() => {
126
+ try {
127
+ const warning = options.agyWarning();
128
+ return warning === undefined ? [] : [warning];
129
+ }
130
+ catch {
131
+ return ["agy could not be probed; run `agent-ops doctor` to verify it."];
132
+ }
133
+ })()
134
+ : [])];
122
135
  if (args.dryRun) {
123
136
  return okEnvelope("INIT_PLAN_READY", {
124
137
  applied: false,
@@ -1,7 +1,7 @@
1
1
  import { buildReviewPacket } from "../../../../runtime/src/review/packet.js";
2
2
  import { runIndependentReview } from "../../../../runtime/src/review/runner.js";
3
3
  import { renderReviewResult } from "../../../../runtime/src/review/render.js";
4
- import { saveReviewAttestation } from "../../../../runtime/src/review/attestation.js";
4
+ import { invalidateReviewAttestation, saveReviewAttestation } from "../../../../runtime/src/review/attestation.js";
5
5
  import { resolveReviewRole } from "../../../../runtime/src/review/roles.js";
6
6
  import { AgentOpsError } from "../../../../runtime/src/fs/paths.js";
7
7
  import { assertSafeSupportingPaths, isReviewerPolicyPath, resolveReviewScope, reviewScopeSignature } from "../../../../runtime/src/review/scope.js";
@@ -75,7 +75,8 @@ async function taskContext(options) {
75
75
  policyConfigHash: record.policyConfigHash,
76
76
  evidence: record.evidence,
77
77
  failureFingerprint: record.failureFingerprint,
78
- criteria
78
+ criteria,
79
+ allCriteriaReviewed: criteria.length === record.task.criteria.length
79
80
  };
80
81
  }
81
82
  function newestEvidence(values) {
@@ -123,11 +124,16 @@ async function currentEvidence(options, context, criterionId, commandId, configH
123
124
  return { current, hasReference, unreadable, stale };
124
125
  }
125
126
  async function preflightReview(options, context, sourceFingerprint) {
127
+ // A recorded failure is not stale evidence, and saying so sends the caller
128
+ // to re-run the verifier that just failed. The tests failed; that is the
129
+ // report.
130
+ if (context.failureFingerprint !== null) {
131
+ return { ok: false, reason: "verification-not-passed" };
132
+ }
126
133
  if (options.config === undefined ||
127
134
  options.evidenceStore === undefined ||
128
135
  options.root === undefined ||
129
- options.gitRunner === undefined ||
130
- context.failureFingerprint !== null) {
136
+ options.gitRunner === undefined) {
131
137
  return { ok: false, reason: "stale-verification" };
132
138
  }
133
139
  const configHash = calculateConfigHash(options.config);
@@ -287,6 +293,9 @@ export async function runReviewCommand(options) {
287
293
  });
288
294
  }
289
295
  sourceFingerprint = await calculateSourceFingerprint(options.root, scope, options.gitRunner);
296
+ if (options.authorized) {
297
+ await invalidateReviewAttestation(options.root, sourceFingerprint);
298
+ }
290
299
  if (context !== undefined && options.policyConfigHash !== undefined) {
291
300
  if (context.policyConfigHash === null) {
292
301
  return notRunEnvelope({
@@ -422,7 +431,8 @@ export async function runReviewCommand(options) {
422
431
  // changes again.
423
432
  if (options.root !== undefined &&
424
433
  result.status === "PASS" &&
425
- sourceFingerprint !== undefined) {
434
+ sourceFingerprint !== undefined &&
435
+ (context === undefined || context.allCriteriaReviewed)) {
426
436
  await saveReviewAttestation(options.root, {
427
437
  schemaVersion: 1,
428
438
  ...(context === undefined ? {} : { taskId: context.taskId }),
@@ -214,7 +214,11 @@ function shouldBuildStopVerification(harness, event, config, rawInput) {
214
214
  */
215
215
  export async function runHookProcess(argv, io, cliVersion, dependencies = {}) {
216
216
  const [harness, event] = argv;
217
- const completionGateInstalled = harness === "agy" &&
217
+ // Both hosts whose Stop hook can actually refuse a stop. codex never fires
218
+ // Stop under `codex exec` and rejects `permissionDecision:ask`, so its
219
+ // escape hatch could not be user-approved; opencode can only deny a tool
220
+ // call, never a stop.
221
+ const completionGateInstalled = (harness === "agy" || harness === "claude") &&
218
222
  event === "Stop" &&
219
223
  argv.includes("--completion-gate");
220
224
  if (harness === undefined ||
@@ -230,7 +234,7 @@ export async function runHookProcess(argv, io, cliVersion, dependencies = {}) {
230
234
  const hookEvent = event;
231
235
  if (process.env.AGENT_OPS_DISABLE === "1") {
232
236
  if (completionGateInstalled) {
233
- writeHookOutput(io, harnessDescriptor("agy").runtime.formatOutput("Stop", {
237
+ writeHookOutput(io, harnessDescriptor(harness).runtime.formatOutput("Stop", {
234
238
  action: "block",
235
239
  status: "UNKNOWN",
236
240
  code: "COMPLETION_GATE_DISABLE_REJECTED",
@@ -246,7 +250,7 @@ export async function runHookProcess(argv, io, cliVersion, dependencies = {}) {
246
250
  const configOutcome = await hookConfigOutcome(root, dependencies.loadConfig);
247
251
  if (configOutcome.kind === "invalid") {
248
252
  if (completionGateInstalled) {
249
- writeHookOutput(io, harnessDescriptor("agy").runtime.formatOutput("Stop", {
253
+ writeHookOutput(io, harnessDescriptor(harness).runtime.formatOutput("Stop", {
250
254
  action: "block",
251
255
  status: "UNKNOWN",
252
256
  code: "COMPLETION_GATE_CONFIG_INVALID",
@@ -267,6 +271,15 @@ export async function runHookProcess(argv, io, cliVersion, dependencies = {}) {
267
271
  return 0;
268
272
  }
269
273
  const config = configOutcome.config;
274
+ if (completionGateInstalled && !config.features.completionGate.enabled) {
275
+ writeHookOutput(io, harnessDescriptor(harness).runtime.formatOutput("Stop", {
276
+ action: "block",
277
+ status: "UNKNOWN",
278
+ code: "COMPLETION_GATE_CONFIG_DISABLED",
279
+ remedy: "Restore completionGate.enabled or explicitly uninstall the completion gate."
280
+ }));
281
+ return 0;
282
+ }
270
283
  const trustStatus = dependencies.trust === undefined
271
284
  ? await repositoryTrust(root, config, cliVersion)
272
285
  : await dependencies.trust(root, config, cliVersion);
@@ -284,7 +297,12 @@ export async function runHookProcess(argv, io, cliVersion, dependencies = {}) {
284
297
  processRunner
285
298
  })
286
299
  : undefined;
287
- const completionGate = harnessId === "agy" && config.features.completionGate.enabled
300
+ // Not `completionGateInstalled`: that is the Stop handler's own marker,
301
+ // and the gate also answers PreToolUse — where it turns a self-issued
302
+ // `allow-stop` into a question for the user. Narrowing this to Stop takes
303
+ // the escape hatch's approval step away.
304
+ const completionGate = (harnessId === "agy" || harnessId === "claude") &&
305
+ config.features.completionGate.enabled
288
306
  ? dependencies.completionGate ?? {
289
307
  handle: async (normalized) => await new CompletionGateService({
290
308
  root,
@@ -311,7 +329,7 @@ export async function runHookProcess(argv, io, cliVersion, dependencies = {}) {
311
329
  }
312
330
  catch {
313
331
  if (completionGateInstalled) {
314
- writeHookOutput(io, harnessDescriptor("agy").runtime.formatOutput("Stop", {
332
+ writeHookOutput(io, harnessDescriptor(harness).runtime.formatOutput("Stop", {
315
333
  action: "block",
316
334
  status: "UNKNOWN",
317
335
  code: "COMPLETION_GATE_UNAVAILABLE",
@@ -24,7 +24,13 @@ export function claudeSettingsTarget(scope) {
24
24
  requiresWorkspaceTrust: false
25
25
  };
26
26
  }
27
- function commandHook(event, runtimePath) {
27
+ /**
28
+ * The gate flag trails the ownership marker rather than preceding it, unlike
29
+ * agy's flat command string. The marker's position is what identifies a
30
+ * managed handler here, and moving it would make every existing installation
31
+ * read as foreign.
32
+ */
33
+ function commandHook(event, runtimePath, completionGate = false) {
28
34
  return {
29
35
  type: "command",
30
36
  command: "node",
@@ -32,15 +38,16 @@ function commandHook(event, runtimePath) {
32
38
  runtimePath,
33
39
  "claude",
34
40
  event,
35
- CLAUDE_HOOK_MARKER
41
+ CLAUDE_HOOK_MARKER,
42
+ ...(completionGate ? ["--completion-gate"] : [])
36
43
  ],
37
44
  timeout: 30
38
45
  };
39
46
  }
40
- function matcherGroup(event, runtimePath) {
47
+ function matcherGroup(event, runtimePath, completionGate = false) {
41
48
  return {
42
49
  ...(event === "PreToolUse" ? { matcher: "Bash" } : {}),
43
- hooks: [commandHook(event, runtimePath)]
50
+ hooks: [commandHook(event, runtimePath, completionGate)]
44
51
  };
45
52
  }
46
53
  function powershellLoopCommand(event) {
@@ -82,6 +89,21 @@ export function buildClaudeHookSettings(capabilities, runtimePath, platform = pr
82
89
  for (const event of CLAUDE_LOOP_EVENTS) {
83
90
  hooks[event] = [loopMatcherGroup(event, platform)];
84
91
  }
92
+ // The loop launcher runs a different process, and the completion gate does
93
+ // not live there. Its PreToolUse role is narrow but load-bearing: it is
94
+ // what turns a self-issued `allow-stop` into a question for the user, so
95
+ // the gate needs a handler of its own beside the loop's.
96
+ if (capabilities.includes("completion-gate")) {
97
+ // SessionStart for the same reason: the gate records its per-session
98
+ // baseline there, and a gate that never sees a session start refuses
99
+ // every stop as uninitialized.
100
+ for (const event of ["SessionStart", "PreToolUse"]) {
101
+ hooks[event] = [
102
+ ...(hooks[event] ?? []),
103
+ matcherGroup(event, runtimePath)
104
+ ];
105
+ }
106
+ }
85
107
  }
86
108
  else {
87
109
  if (capabilities.includes("lifecycle-summary")) {
@@ -91,7 +113,12 @@ export function buildClaudeHookSettings(capabilities, runtimePath, platform = pr
91
113
  hooks.PreToolUse = [matcherGroup("PreToolUse", runtimePath)];
92
114
  }
93
115
  }
94
- if (capabilities.includes("optional-stop-verify")) {
116
+ // The gate supersedes report-only Stop verification: one managed Stop
117
+ // handler, and the gated one already reports everything the other would.
118
+ if (capabilities.includes("completion-gate")) {
119
+ hooks.Stop = [matcherGroup("Stop", runtimePath, true)];
120
+ }
121
+ else if (capabilities.includes("optional-stop-verify")) {
95
122
  hooks.Stop = [matcherGroup("Stop", runtimePath)];
96
123
  }
97
124
  return { hooks };
@@ -99,6 +126,28 @@ export function buildClaudeHookSettings(capabilities, runtimePath, platform = pr
99
126
  function isRecord(value) {
100
127
  return typeof value === "object" && value !== null && !Array.isArray(value);
101
128
  }
129
+ /**
130
+ * The argument vector `commandHook` produces, and nothing else. The ownership
131
+ * marker alone is not proof: it is a plain string anyone may write, and a
132
+ * handler mistaken for ours is a handler `update` rewrites and `uninstall`
133
+ * deletes.
134
+ */
135
+ function isManagedNodeArgs(args) {
136
+ if (!Array.isArray(args) || args.length < 4 || args.length > 5) {
137
+ return false;
138
+ }
139
+ const [runtimePath, harness, event, marker, gate] = args;
140
+ // The event is checked for shape, not membership: a handler an older
141
+ // agent-ops wrote for an event this version no longer knows is still ours to
142
+ // remove, and rejecting it would orphan it in the user's settings forever.
143
+ return (typeof runtimePath === "string" &&
144
+ runtimePath.length > 0 &&
145
+ harness === "claude" &&
146
+ typeof event === "string" &&
147
+ event.length > 0 &&
148
+ marker === CLAUDE_HOOK_MARKER &&
149
+ (gate === undefined || gate === "--completion-gate"));
150
+ }
102
151
  /**
103
152
  * Matches only the two command shapes agent-ops actually generates. This is
104
153
  * reused by installation inspection so a foreign hook cannot masquerade as
@@ -109,11 +158,12 @@ export function isClaudeManagedHandler(handler) {
109
158
  return false;
110
159
  }
111
160
  return ((handler.command === "node" &&
112
- Array.isArray(handler.args) &&
113
- handler.args[3] === CLAUDE_HOOK_MARKER) ||
161
+ isManagedNodeArgs(handler.args)) ||
114
162
  (handler.command === "bash" &&
115
163
  Array.isArray(handler.args) &&
164
+ handler.args.length === 3 &&
116
165
  handler.args[0] === CLAUDE_LOOP_LAUNCHER &&
166
+ CLAUDE_LOOP_EVENTS.includes(handler.args[1]) &&
117
167
  handler.args[2] === CLAUDE_HOOK_MARKER) ||
118
168
  (handler.shell === "powershell" &&
119
169
  handler.args === undefined &&
@@ -37,5 +37,13 @@ export const CLAUDE_CAPABILITY_REGISTRATIONS = [
37
37
  surfaceId: "claude-settings",
38
38
  support: "supported",
39
39
  runtimeFailure: "fail-open"
40
+ },
41
+ {
42
+ capability: "completion-gate",
43
+ normalizedEvent: "stop",
44
+ nativeEvent: "Stop",
45
+ surfaceId: "claude-settings",
46
+ support: "supported",
47
+ runtimeFailure: "fail-closed"
40
48
  }
41
49
  ];
@@ -13,17 +13,31 @@ export function normalizeClaudeHookInput(input) {
13
13
  return normalizeHookEvent(input);
14
14
  }
15
15
  const projectRoot = input.cwd;
16
+ // The completion gate keys its per-session baseline on this. Without it every
17
+ // Stop is refused for the wrong reason and no SessionStart ever records a
18
+ // baseline to refuse against.
19
+ const sessionId = typeof input.session_id === "string"
20
+ ? input.session_id
21
+ : undefined;
16
22
  if (input.hook_event_name === "SessionStart") {
17
- return normalizeHookEvent({
18
- event: "session-start",
19
- projectRoot
20
- });
23
+ return {
24
+ ...normalizeHookEvent({ event: "session-start", projectRoot }),
25
+ ...(sessionId === undefined ? {} : { sessionId })
26
+ };
21
27
  }
22
28
  if (input.hook_event_name === "Stop") {
23
- return normalizeHookEvent({
29
+ // Claude publishes no termination reason: its Stop hook fires when the
30
+ // assistant has finished, which is agy's `model_stop`. The one distinction
31
+ // it does publish is recursion — a Stop the hook itself caused — and that
32
+ // is exactly the not-yet-idle case the gate lets through.
33
+ const stop = normalizeHookEvent({ event: "stop", projectRoot });
34
+ return {
24
35
  event: "stop",
25
- projectRoot
26
- });
36
+ projectRoot: stop.projectRoot,
37
+ ...(sessionId === undefined ? {} : { sessionId }),
38
+ terminationReason: "model_stop",
39
+ fullyIdle: input.stop_hook_active !== true
40
+ };
27
41
  }
28
42
  if (input.hook_event_name === "PreToolUse" &&
29
43
  input.tool_name === "Bash" &&
@@ -7,6 +7,30 @@ function json(value) {
7
7
  }
8
8
  export function claudeHookOutput(event, result) {
9
9
  const denialReason = result.remedy === undefined ? result.code : `${result.code}: ${result.remedy}`;
10
+ // The completion gate carries no verification evidence of its own — it reads
11
+ // evidence rather than producing it — so it never reaches the branch below
12
+ // and needs its own refusal.
13
+ if (event === "Stop" &&
14
+ result.action === "block" &&
15
+ result.code.startsWith("COMPLETION_GATE_")) {
16
+ return json({
17
+ decision: "block",
18
+ reason: `agent-ops: ${denialReason}`
19
+ });
20
+ }
21
+ if (event === "PreToolUse" &&
22
+ result.code === "COMPLETION_GATE_PERMIT_CONFIRMATION") {
23
+ // Asked, never allowed: a one-time Stop permit is the user's to grant, and
24
+ // an agent that could answer this for itself would hold the key to its own
25
+ // gate.
26
+ return json({
27
+ hookSpecificOutput: {
28
+ hookEventName: "PreToolUse",
29
+ permissionDecision: "ask",
30
+ permissionDecisionReason: denialReason
31
+ }
32
+ });
33
+ }
10
34
  if (event === "Stop" && result.evidence !== undefined) {
11
35
  if (result.status === "FAIL") {
12
36
  const failed = result.evidence.commandResults