@integrity-labs/agt-cli 0.28.854 → 0.28.856

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. package/dist/bin/agt.js +5 -5
  2. package/dist/{chunk-2E2NKILK.js → chunk-6CLS6G7Y.js} +2 -2
  3. package/dist/{chunk-FHOLG6VL.js → chunk-6I5IMMWR.js} +356 -194
  4. package/dist/chunk-6I5IMMWR.js.map +1 -0
  5. package/dist/{chunk-4Z4GFQAD.js → chunk-M6E42Z4Z.js} +111 -1
  6. package/dist/chunk-M6E42Z4Z.js.map +1 -0
  7. package/dist/{claude-pair-runtime-63TA3LT5.js → claude-pair-runtime-QQSSCLXU.js} +2 -2
  8. package/dist/lib/manager-worker.js +37 -21
  9. package/dist/lib/manager-worker.js.map +1 -1
  10. package/dist/mcp/computer-use-proxy.js +98 -1
  11. package/dist/mcp/direct-chat-channel.js +110 -0
  12. package/dist/mcp/index.js +110 -0
  13. package/dist/mcp/origami.js +110 -0
  14. package/dist/mcp/slack-channel.js +110 -0
  15. package/dist/mcp/telegram-channel.js +110 -0
  16. package/dist/{persistent-session-N4DJ4VXZ.js → persistent-session-GZLPDUMN.js} +3 -3
  17. package/dist/{responsiveness-probe-GYE5MIVB.js → responsiveness-probe-GSCW4NBC.js} +3 -3
  18. package/dist/{session-auth-dead-K6B6D3WR.js → session-auth-dead-NWLDIYTW.js} +2 -2
  19. package/package.json +1 -1
  20. package/dist/chunk-4Z4GFQAD.js.map +0 -1
  21. package/dist/chunk-FHOLG6VL.js.map +0 -1
  22. /package/dist/{chunk-2E2NKILK.js.map → chunk-6CLS6G7Y.js.map} +0 -0
  23. /package/dist/{claude-pair-runtime-63TA3LT5.js.map → claude-pair-runtime-QQSSCLXU.js.map} +0 -0
  24. /package/dist/{persistent-session-N4DJ4VXZ.js.map → persistent-session-GZLPDUMN.js.map} +0 -0
  25. /package/dist/{responsiveness-probe-GYE5MIVB.js.map → responsiveness-probe-GSCW4NBC.js.map} +0 -0
  26. /package/dist/{session-auth-dead-K6B6D3WR.js.map → session-auth-dead-NWLDIYTW.js.map} +0 -0
@@ -6,6 +6,89 @@ globalThis.require ??= __augmentedCreateRequire(import.meta.url);
6
6
  import { createInterface } from "readline";
7
7
  import { spawn } from "child_process";
8
8
  import { fileURLToPath } from "url";
9
+
10
+ // src/computer-use-ssh-identity.ts
11
+ import { chmodSync, mkdtempSync, rmSync, writeFileSync } from "fs";
12
+ import { tmpdir } from "os";
13
+ import { join } from "path";
14
+ var SSH_KEY_ENV = "OPEN_COMPUTER_USE_API_KEY";
15
+ var PEM_PREFIX = "-----BEGIN";
16
+ function decodePrivateKey(raw) {
17
+ const trimmed = raw.trim();
18
+ if (trimmed.length === 0) return null;
19
+ if (trimmed.startsWith(PEM_PREFIX)) return ensureTrailingNewline(trimmed);
20
+ let decoded;
21
+ try {
22
+ decoded = Buffer.from(trimmed, "base64").toString("utf8");
23
+ } catch {
24
+ return null;
25
+ }
26
+ if (!decoded.trimStart().startsWith(PEM_PREFIX)) return null;
27
+ return ensureTrailingNewline(decoded.trim());
28
+ }
29
+ function ensureTrailingNewline(pem) {
30
+ return pem.endsWith("\n") ? pem : `${pem}
31
+ `;
32
+ }
33
+ function removeIdentityDir(dir) {
34
+ if (dir === void 0) return;
35
+ try {
36
+ rmSync(dir, { recursive: true, force: true });
37
+ } catch {
38
+ }
39
+ }
40
+ function materializeSshIdentity(upstream, opts = {}) {
41
+ const env = opts.env ?? process.env;
42
+ const noop = { upstream: [...upstream], cleanup: () => {
43
+ }, problem: null, installed: false };
44
+ const raw = env[SSH_KEY_ENV];
45
+ if (raw === void 0 || raw.trim().length === 0) {
46
+ return noop;
47
+ }
48
+ if (raw.trim() === `\${${SSH_KEY_ENV}}`) {
49
+ return {
50
+ ...noop,
51
+ problem: `${SSH_KEY_ENV} is still the literal placeholder, so the manager has not materialized the credential for this agent yet`
52
+ };
53
+ }
54
+ if (upstream[0] !== "ssh") return noop;
55
+ const pem = decodePrivateKey(raw);
56
+ if (pem === null) {
57
+ return {
58
+ ...noop,
59
+ problem: `${SSH_KEY_ENV} is set but is neither base64-encoded PEM nor raw PEM (no ${PEM_PREFIX} header after decoding)`
60
+ };
61
+ }
62
+ let dir;
63
+ let keyPath;
64
+ try {
65
+ dir = mkdtempSync(join(opts.baseDir ?? tmpdir(), "agt-cu-"));
66
+ keyPath = join(dir, "id");
67
+ writeFileSync(keyPath, pem, { mode: 384 });
68
+ chmodSync(keyPath, 384);
69
+ } catch (err) {
70
+ removeIdentityDir(dir);
71
+ return {
72
+ ...noop,
73
+ problem: `could not write the ssh identity file: ${err.message}`
74
+ };
75
+ }
76
+ const createdDir = dir;
77
+ let removed = false;
78
+ const cleanup = () => {
79
+ if (removed) return;
80
+ removed = true;
81
+ removeIdentityDir(createdDir);
82
+ };
83
+ return {
84
+ upstream: [upstream[0], "-i", keyPath, ...upstream.slice(1)],
85
+ cleanup,
86
+ problem: null,
87
+ installed: true
88
+ };
89
+ }
90
+
91
+ // src/computer-use-proxy.ts
9
92
  import { resolve as resolvePath } from "path";
10
93
 
11
94
  // src/computer-use-gate.ts
@@ -543,9 +626,23 @@ async function main(argv = process.argv) {
543
626
  logErr("no upstream command; usage: computer-use-proxy -- <cmd> [args...]");
544
627
  process.exit(1);
545
628
  }
629
+ const identity = materializeSshIdentity(upstream);
630
+ if (identity.problem !== null) {
631
+ logErr(`ssh identity not installed: ${identity.problem}`);
632
+ }
633
+ if (identity.installed) {
634
+ process.on("exit", identity.cleanup);
635
+ for (const sig of ["SIGINT", "SIGTERM"]) {
636
+ process.on(sig, () => {
637
+ identity.cleanup();
638
+ process.exit(sig === "SIGINT" ? 130 : 143);
639
+ });
640
+ }
641
+ }
642
+ const spawnArgv = identity.upstream;
546
643
  let child;
547
644
  try {
548
- child = spawn(upstream[0], upstream.slice(1), { stdio: ["pipe", "pipe", "inherit"] });
645
+ child = spawn(spawnArgv[0], spawnArgv.slice(1), { stdio: ["pipe", "pipe", "inherit"] });
549
646
  } catch (err) {
550
647
  logErr(`failed to spawn upstream: ${err.message}`);
551
648
  process.exit(1);
@@ -33918,6 +33918,71 @@ var INTEGRATION_REGISTRY = [
33918
33918
  }
33919
33919
  }
33920
33920
  },
33921
+ {
33922
+ // ENG-10133 (epic ENG-10127). Drives a HUMAN'S macOS desktop — clicks,
33923
+ // keystrokes, window state — through `open-computer-use` running on that
33924
+ // Mac, reached over an SSH reverse tunnel the Mac itself opens.
33925
+ //
33926
+ // READ THIS BEFORE CHANGING ANY FIELD BELOW. Three properties are what make
33927
+ // this safe to ship at all, and each is easy to undo by accident:
33928
+ //
33929
+ // 1. Every forwarded call is gated per-action against a human-granted
33930
+ // approval (ENG-10132, packages/mcp/src/computer-use-gate.ts). The gate
33931
+ // lives INSIDE the proxy binary rather than beside it, so there is no
33932
+ // build in which the tools exist and the gate does not.
33933
+ // 2. The `open-computer-use` FLAG_REGISTRY flag decides whether the row is
33934
+ // forwarded to the host at all, and it is `sensitive` — arming it costs
33935
+ // a confirmation. The flag governs whether the PATH EXISTS; the gate
33936
+ // governs whether a call may travel it. Neither substitutes for the other.
33937
+ // 3. `beta` keeps it off every customer's picker until the operator runbook
33938
+ // (ENG-10135) exists. Setting up the Mac side — launchd tunnel, the
33939
+ // `from="127.0.0.1,::1"` pin on authorized_keys, Accessibility
33940
+ // permissions — is not something a connect dialog can do, and offering
33941
+ // it without those steps produces an integration that installs cleanly
33942
+ // and does nothing.
33943
+ //
33944
+ // NOT a `nativeMcp` spec: the entry needs a host-absolute proxy path and
33945
+ // per-install ssh config (user, port, remote command), neither of which the
33946
+ // ENG-5815 templating vocabulary can express. Rendered by
33947
+ // provisioning/computer-use-mcp.ts instead — see its header.
33948
+ id: "open-computer-use",
33949
+ name: "Computer Use (macOS)",
33950
+ category: "infrastructure",
33951
+ description: "Drive a macOS desktop by accessibility tree \u2014 read window state, click, type and press keys in any app. Every action requires a human approval granted per session and bound to one application.",
33952
+ // The credential is an OpenSSH private key, supplied by the operator who
33953
+ // set up the Mac. `api_key` is this codebase's "customer pastes a secret"
33954
+ // flow; there is no ssh-specific auth type and inventing one would buy
33955
+ // nothing the credential pipeline does not already do.
33956
+ supported_auth_types: ["api_key"],
33957
+ beta: true,
33958
+ // DELIBERATELY NOT `installable`, i.e. never in the Add Integration picker.
33959
+ // It joins augmented-admin / augmented-support / augmented-help-kb as a
33960
+ // GRANTED integration, and for a stronger version of their reason. A picker
33961
+ // entry promises self-service, and there is none to offer: the Mac side
33962
+ // needs a launchd tunnel, an authorized_keys entry pinned to loopback, an
33963
+ // SSH keypair and macOS Accessibility permission, none of which a connect
33964
+ // dialog can do. A customer who "installed" it from a picker would get a
33965
+ // dialog, paste something, and own a broken integration.
33966
+ //
33967
+ // It would also contradict the flag. `open-computer-use` is `sensitive`, so
33968
+ // arming it costs a deliberate confirmation — offering one-click install of
33969
+ // the thing behind it is incoherent.
33970
+ capabilities: [
33971
+ {
33972
+ id: "open-computer-use:read-screen",
33973
+ name: "Read Screen State",
33974
+ description: "List running applications and read a window's accessibility tree (element indices, labels, values). Screenshots are stripped by default and returned only on explicit request.",
33975
+ access: "read"
33976
+ },
33977
+ {
33978
+ id: "open-computer-use:control",
33979
+ name: "Control the Desktop",
33980
+ description: "Open apps, click, scroll, type text, press keys and set field values. Every call is refused unless a human has granted an approval for the target application.",
33981
+ access: "write"
33982
+ }
33983
+ ],
33984
+ docs_url: "https://github.com/anthropics/open-computer-use"
33985
+ },
33921
33986
  {
33922
33987
  // ENG-6195: admin-only debugging surface for Integrity Labs STAFF agents.
33923
33988
  // Provisions the @integrity-labs/augmented-admin-mcp stdio broker, which
@@ -36216,6 +36281,35 @@ var FLAG_REGISTRY = [
36216
36281
  defaultValue: false,
36217
36282
  envVar: "AGT_MODEL_API_ERROR_REPORTING_ENABLED"
36218
36283
  },
36284
+ {
36285
+ key: "model-policy-failover-trigger",
36286
+ description: "Automatic model-policy failover (ENG-10156, ADR-0074). off = the trigger does not run. shadow = it evaluates every ingested model-API error against the policy's failover_trigger config and LOGS the decision with its inputs, acting on nothing. Enum gate; ships OFF.",
36287
+ flagType: "enum",
36288
+ // `armed` is DELIBERATELY not an allowed value yet.
36289
+ //
36290
+ // Nothing writes model_policy_failover_state with source='auto' — that is
36291
+ // the actuation half, and it is gated behind a recorded shadow soak
36292
+ // (ENG-10156 AC8). Offering `armed` now would be a setting an operator can
36293
+ // select, that reports success, and that does nothing: the failure mode is
36294
+ // worse than the missing feature, because it looks like a fleet running
36295
+ // with automatic failover enabled. Add it in the same change that makes it
36296
+ // act, not before.
36297
+ allowedValues: ["off", "shadow"],
36298
+ // OFF, not `shadow`, and the contrast with `channel-quarantine-mode` is the
36299
+ // reason: that flag defaults to `shadow` because shadow WAS the live
36300
+ // manager's compiled behaviour, so anything else would have changed the
36301
+ // fleet. Here there is no incumbent behaviour to preserve — nothing has ever
36302
+ // evaluated this — so `off` is the honest default and turning shadow on is a
36303
+ // deliberate act with a soak attached to it.
36304
+ defaultValue: "off",
36305
+ envVar: "AGT_MODEL_POLICY_FAILOVER_TRIGGER_MODE",
36306
+ // Marked sensitive ahead of `armed` existing: the value this flag will
36307
+ // eventually carry moves a host onto a secondary binding whose credential
36308
+ // may be ours (credential_owner: 'platform'), i.e. it spends money with no
36309
+ // human in the loop. Declaring that now costs nothing and means the audited
36310
+ // -flip behaviour is already in place on the day the value is added.
36311
+ sensitive: true
36312
+ },
36219
36313
  {
36220
36314
  key: "claude-md-skills-index",
36221
36315
  description: `Manager injects the "## Available Skills" bullet list (one line per installed skill, name + frontmatter description) into the agent's project CLAUDE.md. Claude Code already surfaces installed skills to the model natively from .claude/skills/*/SKILL.md, so on current models the list is duplicated context - and an expensive one: it measured 12,045 chars across 29 skills on a prod agent, pushing project/CLAUDE.md to 51k against Claude Code's 40,000-char ceiling, past which the tail of the agent's own system prompt is silently truncated. OFF suppresses ONLY the skill bullets; the "Updating Integrations" guidance in the same managed block is real instruction and is always kept. Defaults ON (today's behaviour) - flip OFF per org to reclaim the headroom.`,
@@ -37520,6 +37614,22 @@ var FLAG_REGISTRY = [
37520
37614
  // No `since` for the same reason: there is no host-side reader and no fleet
37521
37615
  // to converge, so `classifyHostReach` returning `honors` for every host is
37522
37616
  // the literal truth here rather than the trap it is one flag above.
37617
+ },
37618
+ {
37619
+ key: "open-computer-use",
37620
+ description: "ENG-10133 (epic ENG-10127) - gates whether an agent gets the `open-computer-use` stdio MCP server, i.e. whether it can drive a human's macOS desktop at all. OFF (default) = the `.mcp.json` entry is not written, so the tools do not exist for the agent and `--allowedTools` never carries the wildcard (the allowlist is DERIVED from the server keys - see apps/cli/src/lib/claude-tools.ts, which is why there is no second lever to flip). ON = the entry is materialized for any agent that also holds the integration row; the flag is necessary, never sufficient. THE FLAG IS NOT THE SAFETY CONTROL AND MUST NOT BE READ AS ONE. Every forwarded call is gated per-action against a human-granted approval by packages/mcp/src/computer-use-gate.ts (ENG-10132); this flag decides whether the PATH EXISTS, the gate decides whether a call may travel it. Turning this on for an agent whose host runs a proxy build predating that gate hands it an ungated route to someone's desktop - which is why the gate ships in the same proxy binary rather than beside it, so there is no build in which the path exists and the gate does not. Org/agent-scoped rather than global: a desktop belongs to one person, and the blast radius of a wrong answer is that person's open windows, mail client and password manager. Deliberately NO envVar - a host-level env override would let a compromised or misconfigured host arm itself, and this is precisely the gate that must be armed centrally (ENG-8303 pin, ENG-7972 4KB cap).",
37621
+ flagType: "boolean",
37622
+ // Dark. There is no "today's behaviour" to preserve - nothing on the fleet
37623
+ // has this integration - so the safe direction is unambiguous.
37624
+ defaultValue: false,
37625
+ // Enforcement gate (ADR-0022 section 4). Arming it is "relaxing a security
37626
+ // control" in the most literal sense available in this codebase: it is the
37627
+ // difference between an agent that can and cannot move a human's mouse.
37628
+ // One confirmation is a cheap price for that.
37629
+ sensitive: true
37630
+ // Deliberately NOT public. The browser has no decision to make here; the
37631
+ // consumers are the host payload that builds `input.integrations` and the
37632
+ // grant route that refuses to install the row while the flag is off.
37523
37633
  }
37524
37634
  ];
37525
37635
  var REGISTRY_BY_KEY = new Map(FLAG_REGISTRY.map((definition) => [definition.key, definition]));
package/dist/mcp/index.js CHANGED
@@ -26467,6 +26467,71 @@ var INTEGRATION_REGISTRY = [
26467
26467
  }
26468
26468
  }
26469
26469
  },
26470
+ {
26471
+ // ENG-10133 (epic ENG-10127). Drives a HUMAN'S macOS desktop — clicks,
26472
+ // keystrokes, window state — through `open-computer-use` running on that
26473
+ // Mac, reached over an SSH reverse tunnel the Mac itself opens.
26474
+ //
26475
+ // READ THIS BEFORE CHANGING ANY FIELD BELOW. Three properties are what make
26476
+ // this safe to ship at all, and each is easy to undo by accident:
26477
+ //
26478
+ // 1. Every forwarded call is gated per-action against a human-granted
26479
+ // approval (ENG-10132, packages/mcp/src/computer-use-gate.ts). The gate
26480
+ // lives INSIDE the proxy binary rather than beside it, so there is no
26481
+ // build in which the tools exist and the gate does not.
26482
+ // 2. The `open-computer-use` FLAG_REGISTRY flag decides whether the row is
26483
+ // forwarded to the host at all, and it is `sensitive` — arming it costs
26484
+ // a confirmation. The flag governs whether the PATH EXISTS; the gate
26485
+ // governs whether a call may travel it. Neither substitutes for the other.
26486
+ // 3. `beta` keeps it off every customer's picker until the operator runbook
26487
+ // (ENG-10135) exists. Setting up the Mac side — launchd tunnel, the
26488
+ // `from="127.0.0.1,::1"` pin on authorized_keys, Accessibility
26489
+ // permissions — is not something a connect dialog can do, and offering
26490
+ // it without those steps produces an integration that installs cleanly
26491
+ // and does nothing.
26492
+ //
26493
+ // NOT a `nativeMcp` spec: the entry needs a host-absolute proxy path and
26494
+ // per-install ssh config (user, port, remote command), neither of which the
26495
+ // ENG-5815 templating vocabulary can express. Rendered by
26496
+ // provisioning/computer-use-mcp.ts instead — see its header.
26497
+ id: "open-computer-use",
26498
+ name: "Computer Use (macOS)",
26499
+ category: "infrastructure",
26500
+ description: "Drive a macOS desktop by accessibility tree \u2014 read window state, click, type and press keys in any app. Every action requires a human approval granted per session and bound to one application.",
26501
+ // The credential is an OpenSSH private key, supplied by the operator who
26502
+ // set up the Mac. `api_key` is this codebase's "customer pastes a secret"
26503
+ // flow; there is no ssh-specific auth type and inventing one would buy
26504
+ // nothing the credential pipeline does not already do.
26505
+ supported_auth_types: ["api_key"],
26506
+ beta: true,
26507
+ // DELIBERATELY NOT `installable`, i.e. never in the Add Integration picker.
26508
+ // It joins augmented-admin / augmented-support / augmented-help-kb as a
26509
+ // GRANTED integration, and for a stronger version of their reason. A picker
26510
+ // entry promises self-service, and there is none to offer: the Mac side
26511
+ // needs a launchd tunnel, an authorized_keys entry pinned to loopback, an
26512
+ // SSH keypair and macOS Accessibility permission, none of which a connect
26513
+ // dialog can do. A customer who "installed" it from a picker would get a
26514
+ // dialog, paste something, and own a broken integration.
26515
+ //
26516
+ // It would also contradict the flag. `open-computer-use` is `sensitive`, so
26517
+ // arming it costs a deliberate confirmation — offering one-click install of
26518
+ // the thing behind it is incoherent.
26519
+ capabilities: [
26520
+ {
26521
+ id: "open-computer-use:read-screen",
26522
+ name: "Read Screen State",
26523
+ description: "List running applications and read a window's accessibility tree (element indices, labels, values). Screenshots are stripped by default and returned only on explicit request.",
26524
+ access: "read"
26525
+ },
26526
+ {
26527
+ id: "open-computer-use:control",
26528
+ name: "Control the Desktop",
26529
+ description: "Open apps, click, scroll, type text, press keys and set field values. Every call is refused unless a human has granted an approval for the target application.",
26530
+ access: "write"
26531
+ }
26532
+ ],
26533
+ docs_url: "https://github.com/anthropics/open-computer-use"
26534
+ },
26470
26535
  {
26471
26536
  // ENG-6195: admin-only debugging surface for Integrity Labs STAFF agents.
26472
26537
  // Provisions the @integrity-labs/augmented-admin-mcp stdio broker, which
@@ -28486,6 +28551,35 @@ var FLAG_REGISTRY = [
28486
28551
  defaultValue: false,
28487
28552
  envVar: "AGT_MODEL_API_ERROR_REPORTING_ENABLED"
28488
28553
  },
28554
+ {
28555
+ key: "model-policy-failover-trigger",
28556
+ description: "Automatic model-policy failover (ENG-10156, ADR-0074). off = the trigger does not run. shadow = it evaluates every ingested model-API error against the policy's failover_trigger config and LOGS the decision with its inputs, acting on nothing. Enum gate; ships OFF.",
28557
+ flagType: "enum",
28558
+ // `armed` is DELIBERATELY not an allowed value yet.
28559
+ //
28560
+ // Nothing writes model_policy_failover_state with source='auto' — that is
28561
+ // the actuation half, and it is gated behind a recorded shadow soak
28562
+ // (ENG-10156 AC8). Offering `armed` now would be a setting an operator can
28563
+ // select, that reports success, and that does nothing: the failure mode is
28564
+ // worse than the missing feature, because it looks like a fleet running
28565
+ // with automatic failover enabled. Add it in the same change that makes it
28566
+ // act, not before.
28567
+ allowedValues: ["off", "shadow"],
28568
+ // OFF, not `shadow`, and the contrast with `channel-quarantine-mode` is the
28569
+ // reason: that flag defaults to `shadow` because shadow WAS the live
28570
+ // manager's compiled behaviour, so anything else would have changed the
28571
+ // fleet. Here there is no incumbent behaviour to preserve — nothing has ever
28572
+ // evaluated this — so `off` is the honest default and turning shadow on is a
28573
+ // deliberate act with a soak attached to it.
28574
+ defaultValue: "off",
28575
+ envVar: "AGT_MODEL_POLICY_FAILOVER_TRIGGER_MODE",
28576
+ // Marked sensitive ahead of `armed` existing: the value this flag will
28577
+ // eventually carry moves a host onto a secondary binding whose credential
28578
+ // may be ours (credential_owner: 'platform'), i.e. it spends money with no
28579
+ // human in the loop. Declaring that now costs nothing and means the audited
28580
+ // -flip behaviour is already in place on the day the value is added.
28581
+ sensitive: true
28582
+ },
28489
28583
  {
28490
28584
  key: "claude-md-skills-index",
28491
28585
  description: `Manager injects the "## Available Skills" bullet list (one line per installed skill, name + frontmatter description) into the agent's project CLAUDE.md. Claude Code already surfaces installed skills to the model natively from .claude/skills/*/SKILL.md, so on current models the list is duplicated context - and an expensive one: it measured 12,045 chars across 29 skills on a prod agent, pushing project/CLAUDE.md to 51k against Claude Code's 40,000-char ceiling, past which the tail of the agent's own system prompt is silently truncated. OFF suppresses ONLY the skill bullets; the "Updating Integrations" guidance in the same managed block is real instruction and is always kept. Defaults ON (today's behaviour) - flip OFF per org to reclaim the headroom.`,
@@ -29790,6 +29884,22 @@ var FLAG_REGISTRY = [
29790
29884
  // No `since` for the same reason: there is no host-side reader and no fleet
29791
29885
  // to converge, so `classifyHostReach` returning `honors` for every host is
29792
29886
  // the literal truth here rather than the trap it is one flag above.
29887
+ },
29888
+ {
29889
+ key: "open-computer-use",
29890
+ description: "ENG-10133 (epic ENG-10127) - gates whether an agent gets the `open-computer-use` stdio MCP server, i.e. whether it can drive a human's macOS desktop at all. OFF (default) = the `.mcp.json` entry is not written, so the tools do not exist for the agent and `--allowedTools` never carries the wildcard (the allowlist is DERIVED from the server keys - see apps/cli/src/lib/claude-tools.ts, which is why there is no second lever to flip). ON = the entry is materialized for any agent that also holds the integration row; the flag is necessary, never sufficient. THE FLAG IS NOT THE SAFETY CONTROL AND MUST NOT BE READ AS ONE. Every forwarded call is gated per-action against a human-granted approval by packages/mcp/src/computer-use-gate.ts (ENG-10132); this flag decides whether the PATH EXISTS, the gate decides whether a call may travel it. Turning this on for an agent whose host runs a proxy build predating that gate hands it an ungated route to someone's desktop - which is why the gate ships in the same proxy binary rather than beside it, so there is no build in which the path exists and the gate does not. Org/agent-scoped rather than global: a desktop belongs to one person, and the blast radius of a wrong answer is that person's open windows, mail client and password manager. Deliberately NO envVar - a host-level env override would let a compromised or misconfigured host arm itself, and this is precisely the gate that must be armed centrally (ENG-8303 pin, ENG-7972 4KB cap).",
29891
+ flagType: "boolean",
29892
+ // Dark. There is no "today's behaviour" to preserve - nothing on the fleet
29893
+ // has this integration - so the safe direction is unambiguous.
29894
+ defaultValue: false,
29895
+ // Enforcement gate (ADR-0022 section 4). Arming it is "relaxing a security
29896
+ // control" in the most literal sense available in this codebase: it is the
29897
+ // difference between an agent that can and cannot move a human's mouse.
29898
+ // One confirmation is a cheap price for that.
29899
+ sensitive: true
29900
+ // Deliberately NOT public. The browser has no decision to make here; the
29901
+ // consumers are the host payload that builds `input.integrations` and the
29902
+ // grant route that refuses to install the row while the flag is off.
29793
29903
  }
29794
29904
  ];
29795
29905
  var REGISTRY_BY_KEY = new Map(FLAG_REGISTRY.map((definition) => [definition.key, definition]));
@@ -40623,6 +40623,71 @@ var INTEGRATION_REGISTRY = [
40623
40623
  }
40624
40624
  }
40625
40625
  },
40626
+ {
40627
+ // ENG-10133 (epic ENG-10127). Drives a HUMAN'S macOS desktop — clicks,
40628
+ // keystrokes, window state — through `open-computer-use` running on that
40629
+ // Mac, reached over an SSH reverse tunnel the Mac itself opens.
40630
+ //
40631
+ // READ THIS BEFORE CHANGING ANY FIELD BELOW. Three properties are what make
40632
+ // this safe to ship at all, and each is easy to undo by accident:
40633
+ //
40634
+ // 1. Every forwarded call is gated per-action against a human-granted
40635
+ // approval (ENG-10132, packages/mcp/src/computer-use-gate.ts). The gate
40636
+ // lives INSIDE the proxy binary rather than beside it, so there is no
40637
+ // build in which the tools exist and the gate does not.
40638
+ // 2. The `open-computer-use` FLAG_REGISTRY flag decides whether the row is
40639
+ // forwarded to the host at all, and it is `sensitive` — arming it costs
40640
+ // a confirmation. The flag governs whether the PATH EXISTS; the gate
40641
+ // governs whether a call may travel it. Neither substitutes for the other.
40642
+ // 3. `beta` keeps it off every customer's picker until the operator runbook
40643
+ // (ENG-10135) exists. Setting up the Mac side — launchd tunnel, the
40644
+ // `from="127.0.0.1,::1"` pin on authorized_keys, Accessibility
40645
+ // permissions — is not something a connect dialog can do, and offering
40646
+ // it without those steps produces an integration that installs cleanly
40647
+ // and does nothing.
40648
+ //
40649
+ // NOT a `nativeMcp` spec: the entry needs a host-absolute proxy path and
40650
+ // per-install ssh config (user, port, remote command), neither of which the
40651
+ // ENG-5815 templating vocabulary can express. Rendered by
40652
+ // provisioning/computer-use-mcp.ts instead — see its header.
40653
+ id: "open-computer-use",
40654
+ name: "Computer Use (macOS)",
40655
+ category: "infrastructure",
40656
+ description: "Drive a macOS desktop by accessibility tree \u2014 read window state, click, type and press keys in any app. Every action requires a human approval granted per session and bound to one application.",
40657
+ // The credential is an OpenSSH private key, supplied by the operator who
40658
+ // set up the Mac. `api_key` is this codebase's "customer pastes a secret"
40659
+ // flow; there is no ssh-specific auth type and inventing one would buy
40660
+ // nothing the credential pipeline does not already do.
40661
+ supported_auth_types: ["api_key"],
40662
+ beta: true,
40663
+ // DELIBERATELY NOT `installable`, i.e. never in the Add Integration picker.
40664
+ // It joins augmented-admin / augmented-support / augmented-help-kb as a
40665
+ // GRANTED integration, and for a stronger version of their reason. A picker
40666
+ // entry promises self-service, and there is none to offer: the Mac side
40667
+ // needs a launchd tunnel, an authorized_keys entry pinned to loopback, an
40668
+ // SSH keypair and macOS Accessibility permission, none of which a connect
40669
+ // dialog can do. A customer who "installed" it from a picker would get a
40670
+ // dialog, paste something, and own a broken integration.
40671
+ //
40672
+ // It would also contradict the flag. `open-computer-use` is `sensitive`, so
40673
+ // arming it costs a deliberate confirmation — offering one-click install of
40674
+ // the thing behind it is incoherent.
40675
+ capabilities: [
40676
+ {
40677
+ id: "open-computer-use:read-screen",
40678
+ name: "Read Screen State",
40679
+ description: "List running applications and read a window's accessibility tree (element indices, labels, values). Screenshots are stripped by default and returned only on explicit request.",
40680
+ access: "read"
40681
+ },
40682
+ {
40683
+ id: "open-computer-use:control",
40684
+ name: "Control the Desktop",
40685
+ description: "Open apps, click, scroll, type text, press keys and set field values. Every call is refused unless a human has granted an approval for the target application.",
40686
+ access: "write"
40687
+ }
40688
+ ],
40689
+ docs_url: "https://github.com/anthropics/open-computer-use"
40690
+ },
40626
40691
  {
40627
40692
  // ENG-6195: admin-only debugging surface for Integrity Labs STAFF agents.
40628
40693
  // Provisions the @integrity-labs/augmented-admin-mcp stdio broker, which
@@ -42439,6 +42504,35 @@ var FLAG_REGISTRY = [
42439
42504
  defaultValue: false,
42440
42505
  envVar: "AGT_MODEL_API_ERROR_REPORTING_ENABLED"
42441
42506
  },
42507
+ {
42508
+ key: "model-policy-failover-trigger",
42509
+ description: "Automatic model-policy failover (ENG-10156, ADR-0074). off = the trigger does not run. shadow = it evaluates every ingested model-API error against the policy's failover_trigger config and LOGS the decision with its inputs, acting on nothing. Enum gate; ships OFF.",
42510
+ flagType: "enum",
42511
+ // `armed` is DELIBERATELY not an allowed value yet.
42512
+ //
42513
+ // Nothing writes model_policy_failover_state with source='auto' — that is
42514
+ // the actuation half, and it is gated behind a recorded shadow soak
42515
+ // (ENG-10156 AC8). Offering `armed` now would be a setting an operator can
42516
+ // select, that reports success, and that does nothing: the failure mode is
42517
+ // worse than the missing feature, because it looks like a fleet running
42518
+ // with automatic failover enabled. Add it in the same change that makes it
42519
+ // act, not before.
42520
+ allowedValues: ["off", "shadow"],
42521
+ // OFF, not `shadow`, and the contrast with `channel-quarantine-mode` is the
42522
+ // reason: that flag defaults to `shadow` because shadow WAS the live
42523
+ // manager's compiled behaviour, so anything else would have changed the
42524
+ // fleet. Here there is no incumbent behaviour to preserve — nothing has ever
42525
+ // evaluated this — so `off` is the honest default and turning shadow on is a
42526
+ // deliberate act with a soak attached to it.
42527
+ defaultValue: "off",
42528
+ envVar: "AGT_MODEL_POLICY_FAILOVER_TRIGGER_MODE",
42529
+ // Marked sensitive ahead of `armed` existing: the value this flag will
42530
+ // eventually carry moves a host onto a secondary binding whose credential
42531
+ // may be ours (credential_owner: 'platform'), i.e. it spends money with no
42532
+ // human in the loop. Declaring that now costs nothing and means the audited
42533
+ // -flip behaviour is already in place on the day the value is added.
42534
+ sensitive: true
42535
+ },
42442
42536
  {
42443
42537
  key: "claude-md-skills-index",
42444
42538
  description: `Manager injects the "## Available Skills" bullet list (one line per installed skill, name + frontmatter description) into the agent's project CLAUDE.md. Claude Code already surfaces installed skills to the model natively from .claude/skills/*/SKILL.md, so on current models the list is duplicated context - and an expensive one: it measured 12,045 chars across 29 skills on a prod agent, pushing project/CLAUDE.md to 51k against Claude Code's 40,000-char ceiling, past which the tail of the agent's own system prompt is silently truncated. OFF suppresses ONLY the skill bullets; the "Updating Integrations" guidance in the same managed block is real instruction and is always kept. Defaults ON (today's behaviour) - flip OFF per org to reclaim the headroom.`,
@@ -43743,6 +43837,22 @@ var FLAG_REGISTRY = [
43743
43837
  // No `since` for the same reason: there is no host-side reader and no fleet
43744
43838
  // to converge, so `classifyHostReach` returning `honors` for every host is
43745
43839
  // the literal truth here rather than the trap it is one flag above.
43840
+ },
43841
+ {
43842
+ key: "open-computer-use",
43843
+ description: "ENG-10133 (epic ENG-10127) - gates whether an agent gets the `open-computer-use` stdio MCP server, i.e. whether it can drive a human's macOS desktop at all. OFF (default) = the `.mcp.json` entry is not written, so the tools do not exist for the agent and `--allowedTools` never carries the wildcard (the allowlist is DERIVED from the server keys - see apps/cli/src/lib/claude-tools.ts, which is why there is no second lever to flip). ON = the entry is materialized for any agent that also holds the integration row; the flag is necessary, never sufficient. THE FLAG IS NOT THE SAFETY CONTROL AND MUST NOT BE READ AS ONE. Every forwarded call is gated per-action against a human-granted approval by packages/mcp/src/computer-use-gate.ts (ENG-10132); this flag decides whether the PATH EXISTS, the gate decides whether a call may travel it. Turning this on for an agent whose host runs a proxy build predating that gate hands it an ungated route to someone's desktop - which is why the gate ships in the same proxy binary rather than beside it, so there is no build in which the path exists and the gate does not. Org/agent-scoped rather than global: a desktop belongs to one person, and the blast radius of a wrong answer is that person's open windows, mail client and password manager. Deliberately NO envVar - a host-level env override would let a compromised or misconfigured host arm itself, and this is precisely the gate that must be armed centrally (ENG-8303 pin, ENG-7972 4KB cap).",
43844
+ flagType: "boolean",
43845
+ // Dark. There is no "today's behaviour" to preserve - nothing on the fleet
43846
+ // has this integration - so the safe direction is unambiguous.
43847
+ defaultValue: false,
43848
+ // Enforcement gate (ADR-0022 section 4). Arming it is "relaxing a security
43849
+ // control" in the most literal sense available in this codebase: it is the
43850
+ // difference between an agent that can and cannot move a human's mouse.
43851
+ // One confirmation is a cheap price for that.
43852
+ sensitive: true
43853
+ // Deliberately NOT public. The browser has no decision to make here; the
43854
+ // consumers are the host payload that builds `input.integrations` and the
43855
+ // grant route that refuses to install the row while the flag is off.
43746
43856
  }
43747
43857
  ];
43748
43858
  var REGISTRY_BY_KEY = new Map(FLAG_REGISTRY.map((definition) => [definition.key, definition]));
@@ -34771,6 +34771,71 @@ var INTEGRATION_REGISTRY = [
34771
34771
  }
34772
34772
  }
34773
34773
  },
34774
+ {
34775
+ // ENG-10133 (epic ENG-10127). Drives a HUMAN'S macOS desktop — clicks,
34776
+ // keystrokes, window state — through `open-computer-use` running on that
34777
+ // Mac, reached over an SSH reverse tunnel the Mac itself opens.
34778
+ //
34779
+ // READ THIS BEFORE CHANGING ANY FIELD BELOW. Three properties are what make
34780
+ // this safe to ship at all, and each is easy to undo by accident:
34781
+ //
34782
+ // 1. Every forwarded call is gated per-action against a human-granted
34783
+ // approval (ENG-10132, packages/mcp/src/computer-use-gate.ts). The gate
34784
+ // lives INSIDE the proxy binary rather than beside it, so there is no
34785
+ // build in which the tools exist and the gate does not.
34786
+ // 2. The `open-computer-use` FLAG_REGISTRY flag decides whether the row is
34787
+ // forwarded to the host at all, and it is `sensitive` — arming it costs
34788
+ // a confirmation. The flag governs whether the PATH EXISTS; the gate
34789
+ // governs whether a call may travel it. Neither substitutes for the other.
34790
+ // 3. `beta` keeps it off every customer's picker until the operator runbook
34791
+ // (ENG-10135) exists. Setting up the Mac side — launchd tunnel, the
34792
+ // `from="127.0.0.1,::1"` pin on authorized_keys, Accessibility
34793
+ // permissions — is not something a connect dialog can do, and offering
34794
+ // it without those steps produces an integration that installs cleanly
34795
+ // and does nothing.
34796
+ //
34797
+ // NOT a `nativeMcp` spec: the entry needs a host-absolute proxy path and
34798
+ // per-install ssh config (user, port, remote command), neither of which the
34799
+ // ENG-5815 templating vocabulary can express. Rendered by
34800
+ // provisioning/computer-use-mcp.ts instead — see its header.
34801
+ id: "open-computer-use",
34802
+ name: "Computer Use (macOS)",
34803
+ category: "infrastructure",
34804
+ description: "Drive a macOS desktop by accessibility tree \u2014 read window state, click, type and press keys in any app. Every action requires a human approval granted per session and bound to one application.",
34805
+ // The credential is an OpenSSH private key, supplied by the operator who
34806
+ // set up the Mac. `api_key` is this codebase's "customer pastes a secret"
34807
+ // flow; there is no ssh-specific auth type and inventing one would buy
34808
+ // nothing the credential pipeline does not already do.
34809
+ supported_auth_types: ["api_key"],
34810
+ beta: true,
34811
+ // DELIBERATELY NOT `installable`, i.e. never in the Add Integration picker.
34812
+ // It joins augmented-admin / augmented-support / augmented-help-kb as a
34813
+ // GRANTED integration, and for a stronger version of their reason. A picker
34814
+ // entry promises self-service, and there is none to offer: the Mac side
34815
+ // needs a launchd tunnel, an authorized_keys entry pinned to loopback, an
34816
+ // SSH keypair and macOS Accessibility permission, none of which a connect
34817
+ // dialog can do. A customer who "installed" it from a picker would get a
34818
+ // dialog, paste something, and own a broken integration.
34819
+ //
34820
+ // It would also contradict the flag. `open-computer-use` is `sensitive`, so
34821
+ // arming it costs a deliberate confirmation — offering one-click install of
34822
+ // the thing behind it is incoherent.
34823
+ capabilities: [
34824
+ {
34825
+ id: "open-computer-use:read-screen",
34826
+ name: "Read Screen State",
34827
+ description: "List running applications and read a window's accessibility tree (element indices, labels, values). Screenshots are stripped by default and returned only on explicit request.",
34828
+ access: "read"
34829
+ },
34830
+ {
34831
+ id: "open-computer-use:control",
34832
+ name: "Control the Desktop",
34833
+ description: "Open apps, click, scroll, type text, press keys and set field values. Every call is refused unless a human has granted an approval for the target application.",
34834
+ access: "write"
34835
+ }
34836
+ ],
34837
+ docs_url: "https://github.com/anthropics/open-computer-use"
34838
+ },
34774
34839
  {
34775
34840
  // ENG-6195: admin-only debugging surface for Integrity Labs STAFF agents.
34776
34841
  // Provisions the @integrity-labs/augmented-admin-mcp stdio broker, which
@@ -36994,6 +37059,35 @@ var FLAG_REGISTRY = [
36994
37059
  defaultValue: false,
36995
37060
  envVar: "AGT_MODEL_API_ERROR_REPORTING_ENABLED"
36996
37061
  },
37062
+ {
37063
+ key: "model-policy-failover-trigger",
37064
+ description: "Automatic model-policy failover (ENG-10156, ADR-0074). off = the trigger does not run. shadow = it evaluates every ingested model-API error against the policy's failover_trigger config and LOGS the decision with its inputs, acting on nothing. Enum gate; ships OFF.",
37065
+ flagType: "enum",
37066
+ // `armed` is DELIBERATELY not an allowed value yet.
37067
+ //
37068
+ // Nothing writes model_policy_failover_state with source='auto' — that is
37069
+ // the actuation half, and it is gated behind a recorded shadow soak
37070
+ // (ENG-10156 AC8). Offering `armed` now would be a setting an operator can
37071
+ // select, that reports success, and that does nothing: the failure mode is
37072
+ // worse than the missing feature, because it looks like a fleet running
37073
+ // with automatic failover enabled. Add it in the same change that makes it
37074
+ // act, not before.
37075
+ allowedValues: ["off", "shadow"],
37076
+ // OFF, not `shadow`, and the contrast with `channel-quarantine-mode` is the
37077
+ // reason: that flag defaults to `shadow` because shadow WAS the live
37078
+ // manager's compiled behaviour, so anything else would have changed the
37079
+ // fleet. Here there is no incumbent behaviour to preserve — nothing has ever
37080
+ // evaluated this — so `off` is the honest default and turning shadow on is a
37081
+ // deliberate act with a soak attached to it.
37082
+ defaultValue: "off",
37083
+ envVar: "AGT_MODEL_POLICY_FAILOVER_TRIGGER_MODE",
37084
+ // Marked sensitive ahead of `armed` existing: the value this flag will
37085
+ // eventually carry moves a host onto a secondary binding whose credential
37086
+ // may be ours (credential_owner: 'platform'), i.e. it spends money with no
37087
+ // human in the loop. Declaring that now costs nothing and means the audited
37088
+ // -flip behaviour is already in place on the day the value is added.
37089
+ sensitive: true
37090
+ },
36997
37091
  {
36998
37092
  key: "claude-md-skills-index",
36999
37093
  description: `Manager injects the "## Available Skills" bullet list (one line per installed skill, name + frontmatter description) into the agent's project CLAUDE.md. Claude Code already surfaces installed skills to the model natively from .claude/skills/*/SKILL.md, so on current models the list is duplicated context - and an expensive one: it measured 12,045 chars across 29 skills on a prod agent, pushing project/CLAUDE.md to 51k against Claude Code's 40,000-char ceiling, past which the tail of the agent's own system prompt is silently truncated. OFF suppresses ONLY the skill bullets; the "Updating Integrations" guidance in the same managed block is real instruction and is always kept. Defaults ON (today's behaviour) - flip OFF per org to reclaim the headroom.`,
@@ -38298,6 +38392,22 @@ var FLAG_REGISTRY = [
38298
38392
  // No `since` for the same reason: there is no host-side reader and no fleet
38299
38393
  // to converge, so `classifyHostReach` returning `honors` for every host is
38300
38394
  // the literal truth here rather than the trap it is one flag above.
38395
+ },
38396
+ {
38397
+ key: "open-computer-use",
38398
+ description: "ENG-10133 (epic ENG-10127) - gates whether an agent gets the `open-computer-use` stdio MCP server, i.e. whether it can drive a human's macOS desktop at all. OFF (default) = the `.mcp.json` entry is not written, so the tools do not exist for the agent and `--allowedTools` never carries the wildcard (the allowlist is DERIVED from the server keys - see apps/cli/src/lib/claude-tools.ts, which is why there is no second lever to flip). ON = the entry is materialized for any agent that also holds the integration row; the flag is necessary, never sufficient. THE FLAG IS NOT THE SAFETY CONTROL AND MUST NOT BE READ AS ONE. Every forwarded call is gated per-action against a human-granted approval by packages/mcp/src/computer-use-gate.ts (ENG-10132); this flag decides whether the PATH EXISTS, the gate decides whether a call may travel it. Turning this on for an agent whose host runs a proxy build predating that gate hands it an ungated route to someone's desktop - which is why the gate ships in the same proxy binary rather than beside it, so there is no build in which the path exists and the gate does not. Org/agent-scoped rather than global: a desktop belongs to one person, and the blast radius of a wrong answer is that person's open windows, mail client and password manager. Deliberately NO envVar - a host-level env override would let a compromised or misconfigured host arm itself, and this is precisely the gate that must be armed centrally (ENG-8303 pin, ENG-7972 4KB cap).",
38399
+ flagType: "boolean",
38400
+ // Dark. There is no "today's behaviour" to preserve - nothing on the fleet
38401
+ // has this integration - so the safe direction is unambiguous.
38402
+ defaultValue: false,
38403
+ // Enforcement gate (ADR-0022 section 4). Arming it is "relaxing a security
38404
+ // control" in the most literal sense available in this codebase: it is the
38405
+ // difference between an agent that can and cannot move a human's mouse.
38406
+ // One confirmation is a cheap price for that.
38407
+ sensitive: true
38408
+ // Deliberately NOT public. The browser has no decision to make here; the
38409
+ // consumers are the host payload that builds `input.integrations` and the
38410
+ // grant route that refuses to install the row while the flag is off.
38301
38411
  }
38302
38412
  ];
38303
38413
  var REGISTRY_BY_KEY = new Map(FLAG_REGISTRY.map((definition) => [definition.key, definition]));