@cohortapp/agent-sdk 2.10.0 → 2.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/.claude/commands/init-maestro.md +16 -9
  2. package/docs/guides/mac-mini.md +11 -1
  3. package/docs/runbooks/cohort-cutover.md +16 -0
  4. package/lib/channels/inbox-item.mjs +12 -0
  5. package/lib/channels/inbox-item.test.mjs +33 -0
  6. package/lib/execution/disposition.mjs +13 -2
  7. package/lib/execution/disposition.test.mjs +19 -2
  8. package/lib/execution/pipeline.test.mjs +4 -1
  9. package/lib/mcp/server.test.mjs +16 -4
  10. package/lib/org/client.mjs +58 -1
  11. package/lib/org/messaging.mjs +5 -0
  12. package/lib/org/messaging.test.mjs +7 -0
  13. package/lib/org/protocol.checksum +1 -1
  14. package/lib/org/protocol.mjs +98 -0
  15. package/lib/org/protocol.test.mjs +19 -2
  16. package/lib/org/resource-tools.mjs +317 -0
  17. package/lib/org/resource-tools.test.mjs +361 -0
  18. package/lib/org/tool-access.mjs +176 -0
  19. package/lib/org/tool-access.test.mjs +144 -0
  20. package/lib/org/tool-surface.mjs +431 -5
  21. package/lib/org/tool-surface.test.mjs +385 -8
  22. package/lib/org/ui-parity.mjs +196 -3
  23. package/lib/org/ui-parity.test.mjs +126 -7
  24. package/lib/tool-definitions.js +23 -2
  25. package/package.json +2 -2
  26. package/plugins/maestro-skills/.claude-plugin/marketplace.json +1 -1
  27. package/plugins/maestro-skills/plugin.json +4 -0
  28. package/plugins/maestro-skills/skills/venture-deliverables.md +176 -0
  29. package/policies/information-barriers.yaml +34 -7
  30. package/scripts/ci/check-no-residual-identity.mjs +281 -9
  31. package/scripts/ci/check-no-residual-identity.test.mjs +115 -2
  32. package/scripts/cloud-relay/voice/relay-identity.test.mjs +96 -0
  33. package/scripts/cloud-relay/voice/server.mjs +42 -2
  34. package/scripts/cost/track-claude-usage-pricing.test.mjs +183 -0
  35. package/scripts/cost/track-claude-usage.mjs +113 -4
  36. package/scripts/daemon/agent-daemon.mjs +212 -5
  37. package/scripts/daemon/agent-daemon.test.mjs +307 -0
  38. package/scripts/daemon/assurance.mjs +38 -15
  39. package/scripts/daemon/assurance.test.mjs +39 -1
  40. package/scripts/daemon/cadence-handlers.mjs +48 -5
  41. package/scripts/daemon/cadence-handlers.test.mjs +57 -2
  42. package/scripts/daemon/classifier-identity.test.mjs +137 -0
  43. package/scripts/daemon/classifier.mjs +98 -17
  44. package/scripts/daemon/inbox-deferral.mjs +49 -24
  45. package/scripts/daemon/inbox-deferral.test.mjs +39 -1
  46. package/scripts/daemon/prompt-builder-preamble.test.mjs +210 -0
  47. package/scripts/daemon/prompt-builder.mjs +264 -41
  48. package/scripts/daemon/prompt-builder.test.mjs +5 -5
  49. package/scripts/daemon/responder.mjs +9 -0
  50. package/scripts/disclosure_boundaries.py +56 -5
  51. package/scripts/huddle/huddle-prompt.test.mjs +176 -0
  52. package/scripts/huddle/huddle-server.mjs +128 -13
  53. package/scripts/local-triggers/autoupdate.sh +83 -0
  54. package/scripts/local-triggers/generate-plists.sh +9 -0
  55. package/scripts/local-triggers/generate-plists.test.mjs +12 -10
  56. package/scripts/media-generation/brand-clause.test.mjs +135 -0
  57. package/scripts/media-generation/gemini-image-client.mjs +27 -9
  58. package/scripts/media-generation/generate-assets.mjs +102 -7
  59. package/scripts/poller/inbox-scan-poller.mjs +7 -0
  60. package/scripts/poller/utils.mjs +11 -0
  61. package/scripts/pre-draft-context.py +91 -15
  62. package/scripts/spawn-session.sh +36 -6
  63. package/scripts/test-employer-grounding.py +348 -0
  64. package/scripts/validate_outbound.py +190 -26
@@ -0,0 +1,144 @@
1
+ /**
2
+ * tool-access.test.mjs — the access-tier mechanism, and the two silent failures
3
+ * it exists to close.
4
+ *
5
+ * REGRESSION 1 — THE ADMIN TRAPDOOR. `access:"admin"` is the CEO-only tier
6
+ * (lib/tool-definitions.js withholds it from writeToolsLeadership AND from the
7
+ * default lookup set). A CAPABILITY parked there is shipped and dark for every
8
+ * seat below the operator's console — the exact state tool-surface.mjs's header
9
+ * describes when it says "for a leadership or default agent, 'the long tail is
10
+ * reachable via org_rpc' is false". The rule against it used to live in
11
+ * per-delta hand-listed name arrays, so it guarded the names someone
12
+ * remembered. ADMIN_TOOLS makes the tier a closed set instead.
13
+ *
14
+ * REGRESSION 2 — THE TYPO TRAPDOOR. The tiers were bucketed by exact string
15
+ * equality, so an entry whose `access` was missing or mis-cased matched no
16
+ * bucket and reached NO native tier at all, while the MCP plane kept publishing
17
+ * it. That is the tool-side twin of the caller-side footgun `normalizeAccessLevel`
18
+ * already fixes ("'CEO' yielded zero tools, a silent UNDER-privilege footgun").
19
+ * The test below drives the OLD comparison and the NEW one over the same bad
20
+ * row so the difference is on the record.
21
+ *
22
+ * Hermetic, no network, no env. Run: node --test lib/org/tool-access.test.mjs
23
+ */
24
+
25
+ "use strict";
26
+
27
+ import { test } from "node:test";
28
+ import assert from "node:assert/strict";
29
+
30
+ import { METHODS } from "./protocol.mjs";
31
+ import {
32
+ TOOL_ACCESS_LEVELS,
33
+ MOST_RESTRICTED_ACCESS,
34
+ ADMIN_TOOLS,
35
+ expectedAccessFor,
36
+ normalizeToolAccess,
37
+ partitionByAccess,
38
+ } from "./tool-access.mjs";
39
+
40
+ // ── the derivation ─────────────────────────────────────────────────────────
41
+
42
+ test("expectedAccessFor derives the tier from the protocol SCOPE, not from sideEffecting", () => {
43
+ assert.equal(expectedAccessFor("resource.list"), "read");
44
+ assert.equal(expectedAccessFor("resource.create"), "write");
45
+
46
+ // The five audited reads: side-effecting (they land on the chain or the AI
47
+ // meter) yet `.read`-scoped on purpose, because a VIEWER seat may run them —
48
+ // exactly the ability its human seat has. hq keeps the same list under
49
+ // AUDITED_READ_SCOPE_WRITES. A sideEffecting-derived rule would demote every
50
+ // one of them out of the lookup set that every tier gets.
51
+ for (const m of [
52
+ "knowledge.search",
53
+ "directory.ask",
54
+ "branding.renderTemplate",
55
+ "branding.rewriteInVoice",
56
+ "branding.exportKit",
57
+ ]) {
58
+ assert.equal(METHODS[m].sideEffecting, true, `${m} is side-effecting`);
59
+ assert.equal(expectedAccessFor(m), "read", `${m} is nonetheless a read-tier tool`);
60
+ }
61
+
62
+ // The mirror case: not side-effecting, but write-scoped.
63
+ assert.equal(METHODS["artifact.create"].sideEffecting, false);
64
+ assert.equal(expectedAccessFor("artifact.create"), "write");
65
+
66
+ // Scopes that are neither `.read` nor `.write` (board.work, approval.request,
67
+ // mandate.propose) are writes — they are not callable by a read-only seat.
68
+ assert.equal(expectedAccessFor("board.claim"), "write");
69
+ assert.equal(expectedAccessFor("approval.request"), "write");
70
+ assert.equal(expectedAccessFor("mandate.propose"), "write");
71
+
72
+ // An unknown method is null, not a guess — the caller decides how loud to be.
73
+ assert.equal(expectedAccessFor("nope.nothing"), null);
74
+ assert.equal(expectedAccessFor(""), null);
75
+ });
76
+
77
+ // ── the admin trapdoor ─────────────────────────────────────────────────────
78
+
79
+ test("ADMIN_TOOLS is the closed set of escape hatches — two names, both consoles", () => {
80
+ assert.deepEqual([...ADMIN_TOOLS].sort(), ["org_read", "org_rpc"]);
81
+ assert.equal(ADMIN_TOOLS.has("resource_create"), false);
82
+ // Frozen: a runtime widening would be a privilege change with no diff.
83
+ assert.throws(() => ADMIN_TOOLS.add("resource_create"));
84
+ });
85
+
86
+ // ── the typo trapdoor ──────────────────────────────────────────────────────
87
+
88
+ test("normalizeToolAccess forgives casing/whitespace — 'Read' must not mean 'nowhere'", () => {
89
+ for (const good of TOOL_ACCESS_LEVELS) assert.equal(normalizeToolAccess(good, "t", () => {}), good);
90
+ assert.equal(normalizeToolAccess("Read", "t", () => {}), "read");
91
+ assert.equal(normalizeToolAccess(" WRITE ", "t", () => {}), "write");
92
+ assert.equal(normalizeToolAccess("Admin", "t", () => {}), "admin");
93
+ });
94
+
95
+ test("an unrecognised tier fails CLOSED and NOISY, never invisible", () => {
96
+ for (const bad of [undefined, null, "", "readonly", 7, {}, []]) {
97
+ const warnings = [];
98
+ const got = normalizeToolAccess(bad, "broken_tool", (m) => warnings.push(m));
99
+ assert.equal(got, MOST_RESTRICTED_ACCESS, `${JSON.stringify(bad)} → most-restricted`);
100
+ assert.equal(warnings.length, 1, `${JSON.stringify(bad)} warns exactly once`);
101
+ assert.match(warnings[0], /broken_tool/, "the warning names the bad row so it is findable");
102
+ }
103
+ });
104
+
105
+ test("a throwing warn sink cannot take the tool table down", () => {
106
+ assert.equal(
107
+ normalizeToolAccess("nonsense", "t", () => {
108
+ throw new Error("logger exploded");
109
+ }),
110
+ MOST_RESTRICTED_ACCESS,
111
+ );
112
+ });
113
+
114
+ test("REGRESSION: the old equality bucketing dropped a bad row from EVERY tier", () => {
115
+ const table = [
116
+ { name: "good_read", access: "read" },
117
+ { name: "good_write", access: "write" },
118
+ { name: "mis_cased", access: "Read" }, // a plausible table typo
119
+ { name: "no_access" }, // an entry someone forgot to tier
120
+ ];
121
+
122
+ // What the code used to do, verbatim: three independent equality filters.
123
+ const oldBuckets = ["read", "write", "admin"].map((a) => table.filter((t) => t.access === a));
124
+ const oldReach = new Set(oldBuckets.flat().map((t) => t.name));
125
+ assert.equal(oldReach.has("mis_cased"), false, "the old filter lost a mis-cased row…");
126
+ assert.equal(oldReach.has("no_access"), false, "…and an untiered one");
127
+ assert.equal(oldReach.size, 2, "two of four tools reached no access level at all, silently");
128
+
129
+ // What it does now: an exhaustive partition. Nothing is dropped.
130
+ const warnings = [];
131
+ const now = partitionByAccess(table, (m) => warnings.push(m));
132
+ assert.equal(now.read.length + now.write.length + now.admin.length, table.length, "exhaustive");
133
+ assert.deepEqual(now.read.map((t) => t.name), ["good_read", "mis_cased"], "casing recovered into its real tier");
134
+ assert.deepEqual(now.admin.map((t) => t.name), ["no_access"], "the untiered row fails closed, not open");
135
+ assert.equal(warnings.length, 1, "and exactly the untiered row is reported");
136
+ assert.match(warnings[0], /no_access/);
137
+ });
138
+
139
+ test("partitionByAccess survives a null/empty table (fail-open, like the rest of the client)", () => {
140
+ const empty = partitionByAccess(null, () => {});
141
+ assert.deepEqual(empty, { read: [], write: [], admin: [] });
142
+ const nullish = partitionByAccess([null, undefined], () => {});
143
+ assert.equal(nullish.admin.length, 2, "a null row is tiered, not thrown on");
144
+ });
@@ -7,16 +7,30 @@
7
7
  * - the native function-call path (`lib/tool-definitions.js` category 7 +
8
8
  * `lib/action-executor.js` executor routing).
9
9
  *
10
- * 38 curated tools + 5 email tools + 5 artifact tools + 69 agent-desk tools
10
+ * 54 curated tools + 5 email tools + 5 artifact tools + 77 agent-desk tools
11
11
  * (mail app / files / calendar / crm / books / meetings.recapFile / directory /
12
- * design / call working sessions — the additive families register only when
13
- * the vendored protocol carries them — see emailFamilyAvailable /
14
- * artifactFamilyAvailable / deskFamilyAvailable) — NOT a one-to-one mirror of
15
- * the 460-method protocol table. The long tail
12
+ * design / call working sessions / venture deliverables — the additive families
13
+ * register only when the vendored protocol carries them — see
14
+ * emailFamilyAvailable / artifactFamilyAvailable / deskFamilyAvailable) — NOT a
15
+ * one-to-one mirror of the 483-method protocol table. The long tail
16
16
  * stays reachable through
17
17
  * the `org_rpc` / `org_read` escape hatches, which validate against the frozen
18
18
  * protocol table before any network I/O and are deliberately `access:"admin"`.
19
19
  *
20
+ * THE DESKS ARE HAND-WRITTEN; THE RESOURCE DESK IS NOT. Every block below down
21
+ * to "calls" transcribes hq's op list by hand, which is precisely why hq
22
+ * exposes 239 ops across 11 families and this table exposed 69 across 9 (files
23
+ * 29 vs 7, crm 44 vs 8). The `resource_*` block is GENERATED from the vendored
24
+ * protocol declaration by lib/org/resource-tools.mjs and parity-tested against
25
+ * it, so it cannot drift that way; it is the pattern the older blocks should
26
+ * migrate to.
27
+ *
28
+ * ACCESS IS DERIVED, NOT DECLARED. For an rpc-bound tool the tier is a pure
29
+ * function of the method's protocol scope (`expectedAccessFor`), and `admin` is
30
+ * a CLOSED set of two escape hatches (`ADMIN_TOOLS`) rather than whatever a
31
+ * table author typed. lib/org/tool-access.mjs records the two silent failures
32
+ * the old hand-written literals allowed.
33
+ *
20
34
  * WHAT "REACHABLE THROUGH THE ESCAPE HATCH" IS WORTH. `access:"admin"` puts
21
35
  * org_rpc/org_read in the CEO tier ONLY (lib/tool-definitions.js: they are
22
36
  * withheld from writeToolsLeadership and from the default lookup set). So for a
@@ -67,6 +81,14 @@ import {
67
81
  okFrame,
68
82
  errFrame,
69
83
  } from "./protocol.mjs";
84
+ // The access tier of an rpc-bound tool is DERIVED from its protocol scope, and
85
+ // the admin tier is a closed, named set — see lib/org/tool-access.mjs for the
86
+ // two silent failures that hand-written `access` literals used to allow.
87
+ import { ADMIN_TOOLS, expectedAccessFor, normalizeToolAccess } from "./tool-access.mjs";
88
+ // The resource desk is GENERATED from the protocol declaration, never
89
+ // hand-copied — the mechanism that produced hq's 239 ops against 69 curated
90
+ // tools (files 29 vs 7, crm 44 vs 8) is two tables written twice by hand.
91
+ import { RESOURCE_TOOLS } from "./resource-tools.mjs";
70
92
  import {
71
93
  loadOrgConfig,
72
94
  configFromAgent,
@@ -181,6 +203,10 @@ const DESK_PROBES = Object.freeze({
181
203
  // OLD calling method would light these tools up on a protocol that lacks
182
204
  // the app-share surface.
183
205
  calls: "calling.appShareStart",
206
+ // 2026-08 venture deliverables. Probes resource.list — the family landed
207
+ // whole (declaration first, handlers second), so any vendored protocol that
208
+ // carries one resource method carries all eight.
209
+ resource: "resource.list",
184
210
  });
185
211
 
186
212
  /**
@@ -335,6 +361,41 @@ export const ORG_TOOLS = Object.freeze([
335
361
  outbound: true,
336
362
  binding: { kind: "rpc", method: "messaging.send" },
337
363
  },
364
+ {
365
+ name: "messaging_send_voice_note",
366
+ title: "Send a voice note",
367
+ description:
368
+ "Speak. Records `text` as an audio VOICE NOTE in your OWN designed voice and posts it into a " +
369
+ "channel — a real voice note in the feed with a waveform, exactly like a human's. Use it whenever " +
370
+ "someone ASKS you to speak ('send me a voicenote with this update', 'record me a quick summary', " +
371
+ "'say that out loud'); you can do this even though the request itself arrived as text. Keep `text` " +
372
+ "short and plain — speech cannot carry code, tables, links or long lists, so leave those to `body` " +
373
+ "(the written message the recording rides on; defaults to `text`). REFUSES honestly when this " +
374
+ "deployment has no speech key, when your seat has no designed voice of its own, or when you have " +
375
+ "used the hour's recordings: relay that reason, never promise audio you cannot produce.",
376
+ input_schema: S(
377
+ {
378
+ channelId: str("Channel id (use messaging_open_dm for a member DM)."),
379
+ text: str("The words to SPEAK aloud. Plain prose, no markdown, no URLs."),
380
+ body: str("The written message body it rides on. Defaults to `text`."),
381
+ threadRootId: str("Reply under this thread root (optional)."),
382
+ },
383
+ ["channelId", "text"],
384
+ ),
385
+ access: "write",
386
+ // Outbound-screened like messaging_send: the spoken words and the written
387
+ // body are both content a human receives, and the audio is the harder of
388
+ // the two to retract once heard. The screen runs on BOTH strings before a
389
+ // single billable character is synthesised — see outboundScreenShape.
390
+ outbound: true,
391
+ // LOCAL, not a bare rpc binding: the capability is two hq methods
392
+ // (synthesize → send), split there because speech synthesis is a 20s
393
+ // provider round-trip that cannot ride hq's 5s transactional send lane. The
394
+ // model should not have to know that, so the seam is hidden here rather
395
+ // than in the prompt. The POST leg is plain messaging.send — same ACL, same
396
+ // audit row, same dedup as any typed message.
397
+ binding: { kind: "local" },
398
+ },
338
399
  {
339
400
  name: "messaging_open_dm",
340
401
  title: "Open a DM",
@@ -917,6 +978,67 @@ export const ORG_TOOLS = Object.freeze([
917
978
  access: "write",
918
979
  binding: { kind: "rpc", method: "memory.author" },
919
980
  },
981
+ // ── preferences: the durable facts hq already injects for free ───────────
982
+ // hq built the `preference.*` family FOR this plane and said so in the
983
+ // handler ("an SDK-driven agent could LEARN a durable fact about a colleague
984
+ // and had nowhere to put it — while the hq responder's `remember_preference`
985
+ // wrote it straight into PreferenceMemory"), then no tool was ever added
986
+ // here, so the reverse-parity fix never reached the plane it was built for —
987
+ // the same way `memory.recall` sat unreachable. The asymmetry it leaves is
988
+ // the one users actually feel: the workspace remembers your travel
989
+ // preference when hq answers and forgets it when your own seat does.
990
+ // BOTH halves ship together deliberately: hq injects known preferences into
991
+ // every responder prompt, an SDK seat gets no such injection and has to ask,
992
+ // so a write-only tool would let this seat store a fact it could never read
993
+ // back and re-derive it forever.
994
+ {
995
+ name: "preference_remember",
996
+ title: "Remember a preference",
997
+ description:
998
+ "Store one durable, member-scoped preference (preference.upsert) — the kind of standing " +
999
+ "fact a colleague should not have to repeat (\"always books aisle seats\", \"reviews on " +
1000
+ "Thursdays\", \"never call before 10\"). It converges on ONE row per (subject, domain, key), " +
1001
+ "so restating a preference updates it rather than accumulating contradictions. `why` is " +
1002
+ "REQUIRED and is the evidence rule, not paperwork: store only what was actually stated or " +
1003
+ "clearly demonstrated, and quote it. Never store a guess, a one-off request, or anything " +
1004
+ "sensitive that was not offered as a standing preference — this is injected into future " +
1005
+ "conversations, so a wrong row is wrong forever and in public.",
1006
+ input_schema: S(
1007
+ {
1008
+ domain: str("The area, e.g. \"travel\", \"meetings\", \"comms\"."),
1009
+ key: str("The specific preference, e.g. \"seat\", \"review-day\"."),
1010
+ // Deliberately a plain string, though hq's zod union also takes a
1011
+ // number, a boolean or a short array: every curated schema in this
1012
+ // table declares ONE type, the MCP plane republishes these verbatim,
1013
+ // and a preference value is read back as prose anyway. "aisle",
1014
+ // "Thursday", "never before 10:00" all survive the round trip.
1015
+ value: str("The preference value, as plain text — e.g. \"aisle\", \"Thursday\", \"never before 10:00\"."),
1016
+ why: str("The evidence: quote or paraphrase where this was stated or shown (5–300 chars)."),
1017
+ confidence: num("0–1, how sure you are (optional)."),
1018
+ messageId: str("The message the evidence came from (optional)."),
1019
+ subjectMemberId: str("Whose preference — omit for your own seat."),
1020
+ },
1021
+ ["domain", "key", "value", "why"],
1022
+ ),
1023
+ access: "write",
1024
+ binding: { kind: "rpc", method: "preference.upsert" },
1025
+ },
1026
+ {
1027
+ name: "preference_list",
1028
+ title: "Read remembered preferences",
1029
+ description:
1030
+ "The preferences remembered about a member (preference.list) — yours by default, or a " +
1031
+ "colleague's disclosed set, which is the same set their own profile page shows. Read this " +
1032
+ "before asking someone something they have already told the workspace: hq's chat lane gets " +
1033
+ "these injected into every prompt automatically, so a colleague reasonably expects you to " +
1034
+ "know them. Empty is a real answer — nothing has been remembered yet.",
1035
+ input_schema: S({
1036
+ memberId: str("Whose preferences — omit for your own seat."),
1037
+ domain: str("Narrow to one area, e.g. \"travel\"."),
1038
+ }),
1039
+ access: "read",
1040
+ binding: { kind: "rpc", method: "preference.list" },
1041
+ },
920
1042
  // ── members / escalation ─────────────────────────────────────────────────
921
1043
  {
922
1044
  name: "member_profile",
@@ -950,6 +1072,227 @@ export const ORG_TOOLS = Object.freeze([
950
1072
  access: "write",
951
1073
  binding: { kind: "rpc", method: "escalation.create" },
952
1074
  },
1075
+ // ── mandate: the seat's own accountability spine ─────────────────────────
1076
+ // THE ASYMMETRY THESE CLOSE, and it ran the wrong way. hq's chat-lane
1077
+ // colleague has carried the whole mandate belt since the spine landed
1078
+ // (read_mandate / read_agent_plan / propose_objective / record_kpi_sample —
1079
+ // src/server/llm-responder/tools.ts, under a docblock that names SDK parity
1080
+ // as the reason they exist) while the SDK plane they were written to match
1081
+ // had NOTHING: `mandate.*` was protocol-declared, wired into the daemon's own
1082
+ // internals (lib/mandate/*, lib/goals/loop.mjs) and reachable from a session
1083
+ // only through `org_rpc`, which is access:"admin" and therefore does not
1084
+ // exist for a leadership or default seat. So the DAEMON knew the seat's
1085
+ // objectives and the SESSION it spawned did not: asked "what are you working
1086
+ // toward", a session improvised. These five make the session read the same
1087
+ // spine its own daemon compiles from.
1088
+ //
1089
+ // The tier is the SCOPE's, not a new judgement: the three reads and
1090
+ // `propose` sit in DEFAULT_AGENT_SCOPES, `mandate.write` (adopt / retire)
1091
+ // pointedly does not — a seat may not make its own targets live — so there is
1092
+ // deliberately no adopt or retire tool in this table at all. Always-on rather
1093
+ // than family-gated: `mandate.*` ships in the same vendored protocol as this
1094
+ // file, so a probe could only ever answer yes.
1095
+ {
1096
+ name: "mandate_read",
1097
+ title: "Read your mandate",
1098
+ description:
1099
+ "Your ADOPTED mandate snapshot (mandate.get): {version, checksum, body} — the outcomes " +
1100
+ "you are actually held to, their metric, target, cadence and collaborators. Read this " +
1101
+ "before answering \"what are you working toward / what are you responsible for\" instead " +
1102
+ "of improvising from your prompt. `version: 0` with a null body is NOT an error and not " +
1103
+ "an outage: it means nothing has been adopted for this seat yet, and the honest answer " +
1104
+ "is to say exactly that. An objective is a commitment only once adopted; a proposed one " +
1105
+ "is a suggestion awaiting a human. `memberId` defaults to your own seat — naming another " +
1106
+ "requires mandate.write/admin and is refused in-domain, not silently widened.",
1107
+ input_schema: S({ memberId: str("Whose mandate — omit for your own.") }),
1108
+ access: "read",
1109
+ binding: { kind: "rpc", method: "mandate.get" },
1110
+ },
1111
+ {
1112
+ name: "mandate_objectives",
1113
+ title: "The org objective tree",
1114
+ description:
1115
+ "The org's objective tree (mandate.tree): a FLAT node list plus explicit `childIds` and " +
1116
+ "`roots` — pillars, objectives and goals, who owns each, its state and latest value. Use " +
1117
+ "it to see where your own objectives sit in the org's, or to answer \"who owns this " +
1118
+ "outcome\". All filters AND together; `includeRetired` defaults false because a retired " +
1119
+ "objective is noise in a fleet view. Flat-with-edges by design: the tree is " +
1120
+ "self-referencing and a bad parent edit would hang a nested serialiser.",
1121
+ input_schema: S({
1122
+ memberId: str("Only this seat's objectives."),
1123
+ kind: str("PILLAR | OBJECTIVE | GOAL."),
1124
+ state: str("proposed | active | at_risk | met | missed | retired."),
1125
+ workstreamId: str("Only objectives on this workstream."),
1126
+ includeRetired: bool("Include retired rows (default false)."),
1127
+ limit: num("Max nodes (server caps at 2000)."),
1128
+ }),
1129
+ access: "read",
1130
+ binding: { kind: "rpc", method: "mandate.tree" },
1131
+ },
1132
+ {
1133
+ name: "mandate_kpi_series",
1134
+ title: "An objective's measurements",
1135
+ description:
1136
+ "One objective's KPI samples plus the server-computed {value, target, gap, normalizedGap, " +
1137
+ "trend, staleness} (mandate.series). Quote `gap` rather than re-deriving it: a POSITIVE " +
1138
+ "gap always means \"short of target\" whichever way the metric points, and that convention " +
1139
+ "is what makes gaps comparable across unlike metrics. `staleness.stale` is computed " +
1140
+ "against 2x the objective's cadence — the same rule hq opens a stale-sensor drift on, so " +
1141
+ "you and hq cannot disagree about whether a number has gone cold.",
1142
+ input_schema: S(
1143
+ {
1144
+ objectiveId: str("The objective's id (from mandate_read or mandate_objectives)."),
1145
+ limit: num("Max samples, newest first (1–500; default 50)."),
1146
+ },
1147
+ ["objectiveId"],
1148
+ ),
1149
+ access: "read",
1150
+ binding: { kind: "rpc", method: "mandate.series" },
1151
+ },
1152
+ {
1153
+ name: "mandate_propose_objective",
1154
+ title: "Propose an objective",
1155
+ description:
1156
+ "PROPOSE a measurable outcome for yourself (mandate.propose). It is created in " +
1157
+ "`state:'proposed'` and is NOT live: a human — or a manager who is not the owner — adopts " +
1158
+ "it in hq, and you can never adopt your own. That is the whole anti-Goodhart law, not a " +
1159
+ "temporary limitation, so do not report a proposal as a commitment. Re-proposing the same " +
1160
+ "`key` UPDATES the proposal; an already-adopted objective is refused (CONFLICT) — retire " +
1161
+ "it and propose a successor. The server also refuses a structurally unadoptable proposal " +
1162
+ "at write time (no target, no charter edge, no sensor), which is a real answer about " +
1163
+ "measurement debt, not a validation nit: relay it.",
1164
+ input_schema: S(
1165
+ {
1166
+ key: str("Stable lowercase slug, e.g. \"pipeline-coverage\". The upsert key."),
1167
+ text: str("The outcome in one line, as a human would state it."),
1168
+ kind: str("PILLAR | OBJECTIVE | GOAL (default OBJECTIVE)."),
1169
+ metric: str("What is measured, e.g. \"pipeline_coverage_x\"."),
1170
+ unit: str("The unit the number is in."),
1171
+ direction: str("up | down | hold | band (default up)."),
1172
+ target: num("The number that counts as met."),
1173
+ baseline: num("Where it stands today."),
1174
+ tolerance: num("Allowed slack around target (>= 0)."),
1175
+ weight: num("Relative weight among your objectives (>= 0)."),
1176
+ cadence: str("daily | weekly | monthly | quarterly — how often it is measured."),
1177
+ dueAt: str("ISO-8601 timestamp this is due by."),
1178
+ parentId: str("The objective this one serves (id, same tenant)."),
1179
+ charterSectionId: str(
1180
+ "The charter PILLAR/CAPABILITY section this is accountable to. Read it off the seat's " +
1181
+ "charter first — an objective that traces to no clause is accountable to nothing.",
1182
+ ),
1183
+ workstreamId: str("The workstream this belongs to."),
1184
+ sensor: {
1185
+ type: "object",
1186
+ description:
1187
+ "How it will be MEASURED, e.g. {method: \"crm.listDeals\"}. A method sensor naming a " +
1188
+ "read-only protocol method is what later makes a sample count as measured rather " +
1189
+ "than asserted.",
1190
+ },
1191
+ memberId: str("Whose objective — omit for your own."),
1192
+ },
1193
+ ["key", "text"],
1194
+ ),
1195
+ access: "write",
1196
+ binding: { kind: "rpc", method: "mandate.propose" },
1197
+ },
1198
+ {
1199
+ name: "mandate_record_kpi_sample",
1200
+ title: "Report a measurement",
1201
+ description:
1202
+ "Record ONE measurement against an objective (mandate.sample). THE HONESTY GATE: the " +
1203
+ "sample's `source` is decided by the SERVER from your evidence and any value you send for " +
1204
+ "it is ignored outright. Cite a real read-only protocol method in `evidence.method` and " +
1205
+ "the reading counts as MEASURED; anything else is `llm` — advisory, and your own admission " +
1206
+ "gate refuses to create work from it. So do not dress up a number you inferred: an " +
1207
+ "honest advisory sample is useful, a fabricated `method` claim is caught, recorded on " +
1208
+ "the chain as downgraded, and teaches you nothing about your unwired sensor. Re-reporting " +
1209
+ "the same `window` UPDATES that row rather than inflating the series.",
1210
+ input_schema: S(
1211
+ {
1212
+ objectiveId: str("The objective's id (from mandate_read or mandate_objectives)."),
1213
+ value: num("The measured value."),
1214
+ window: str("The period: YYYY-MM-DD, YYYY-Www or YYYY-MM."),
1215
+ evidence: {
1216
+ type: "object",
1217
+ description:
1218
+ "Where the number came from, e.g. {method: \"books.reports\", note: \"quoted in #finance\"}. " +
1219
+ "`method` must be a real read-only protocol method for the sample to count as measured.",
1220
+ },
1221
+ },
1222
+ ["objectiveId", "value", "window"],
1223
+ ),
1224
+ access: "write",
1225
+ binding: { kind: "rpc", method: "mandate.sample" },
1226
+ },
1227
+ // ── sub-agent registry (layer 1) ─────────────────────────────────────────
1228
+ // The two halves of one feature that never met: maestro GENERATES sub-agents
1229
+ // locally and hq keeps the org-scoped registry, but a running session could
1230
+ // not read that registry through any curated tool — `subagent.*` was
1231
+ // protocol-declared and reachable only via the admin-tier escape hatch. The
1232
+ // wizard/CLI half already talks to hq through lib/subagents/client.mjs, which
1233
+ // owns the offline OUTBOX and the lock file; these three are READS ONLY for
1234
+ // exactly that reason. A curated write here would be a second writer that
1235
+ // silently loses the outbox queueing and never touches the lock — so
1236
+ // create/publishVersion/fork/pin/unpin/yank stay with `maestro subagents`,
1237
+ // where the local state they mutate actually lives.
1238
+ {
1239
+ name: "subagent_list",
1240
+ title: "List workspace sub-agents",
1241
+ description:
1242
+ "The workspace's sub-agent definitions (subagent.list) — ids, slugs, titles, summaries, " +
1243
+ "tags, functions and altitudes, NEVER bodies. Use it to find whether a specialist already " +
1244
+ "exists before proposing to write one. Yanked definitions are excluded unless asked for. " +
1245
+ "Org-scoped server-side: there is no global catalogue to leak from.",
1246
+ input_schema: S({
1247
+ query: str("Free-text match over slug/title/summary."),
1248
+ function: str("Filter by declared function."),
1249
+ altitude: str("Filter by declared altitude."),
1250
+ includeYanked: bool("Include yanked definitions (default false)."),
1251
+ limit: num("Max definitions (1–500)."),
1252
+ }),
1253
+ access: "read",
1254
+ binding: { kind: "rpc", method: "subagent.list" },
1255
+ },
1256
+ {
1257
+ name: "subagent_get",
1258
+ title: "Read one sub-agent",
1259
+ description:
1260
+ "One definition plus (by default) the BODY of its head version (subagent.get). Address it " +
1261
+ "by `definitionId` or `slug` — one is required. `version` addresses a version by NUMBER, " +
1262
+ "the same number a lock file carries. Bodies come back with {{agent.*}} tokens " +
1263
+ "UNRENDERED — that is deliberate: rendering happens on the agent at load, so treat a " +
1264
+ "token as a token and never paste one at a human as if it were a value. Set " +
1265
+ "includeBody:false when you only need metadata; the body is the expensive field.",
1266
+ input_schema: S({
1267
+ definitionId: str("The definition's id (or pass slug)."),
1268
+ slug: str("The definition's slug (or pass definitionId)."),
1269
+ version: num("A specific version NUMBER; omit for the head."),
1270
+ includeVersions: bool("Also return the bodyless version index."),
1271
+ includeBody: bool("Set false for metadata only (default true)."),
1272
+ }),
1273
+ access: "read",
1274
+ binding: { kind: "rpc", method: "subagent.get" },
1275
+ },
1276
+ {
1277
+ name: "subagent_resolve",
1278
+ title: "Resolve a seat's sub-agents",
1279
+ description:
1280
+ "Everything a seat can actually materialise (subagent.resolve): every ACTIVE pin with its " +
1281
+ "body, frontmatter and provenance, plus advisory rewire verdicts for `refs` you parsed out " +
1282
+ "of workflows. This is the EXPENSIVE read — it carries bodies — so reach for subagent_list " +
1283
+ "when you only need to know what exists. hq resolves ONE layer; precedence across SDK / " +
1284
+ "workspace / agent-local is decided locally on the agent, so this is an input to that " +
1285
+ "decision, not the final answer. `memberId` defaults to your own seat; naming another is " +
1286
+ "refused without admin rather than quietly widened.",
1287
+ input_schema: S({
1288
+ memberId: str("Whose sub-agents — omit for your own seat."),
1289
+ function: str("Filter by declared function."),
1290
+ altitude: str("Filter by declared altitude."),
1291
+ refs: arr({ type: "string" }, "Agent references parsed from your workflows, for rewire verdicts."),
1292
+ }),
1293
+ access: "read",
1294
+ binding: { kind: "rpc", method: "subagent.resolve" },
1295
+ },
953
1296
  // ── discoverability + escape hatches ─────────────────────────────────────
954
1297
  {
955
1298
  name: "org_describe",
@@ -2513,6 +2856,24 @@ export const ORG_TOOLS = Object.freeze([
2513
2856
  desk: "calls",
2514
2857
  binding: { kind: "rpc", method: "calling.getCallEvents" },
2515
2858
  },
2859
+ // ── resource desk (venture deliverables; GENERATED, not hand-written) ──
2860
+ //
2861
+ // The eight `resource_*` entries are produced by lib/org/resource-tools.mjs
2862
+ // from the vendored protocol declaration: the NAME, the ACCESS TIER, the
2863
+ // BINDING and the DESK TAG are all derived, and the only hand-written part is
2864
+ // presentation (title / description / input_schema). Every other desk block
2865
+ // above is a hand-copy of hq's op list, and that is exactly why hq exposes
2866
+ // 239 ops across 11 families where this table exposed 69 across 9 — files 29
2867
+ // vs 7, crm 44 vs 8. A ninth `resource.*` method in the protocol becomes a
2868
+ // ninth tool on the next import with no edit here, and
2869
+ // lib/org/resource-tools.test.mjs goes red until it is described.
2870
+ //
2871
+ // Reads land at access:"read" (org.read → every tier, including
2872
+ // least-privilege "default"); the six writes at access:"write" (org.write →
2873
+ // ceo + leadership). NONE at "admin": a venture's own colleagues are default-
2874
+ // and leadership-tier agents, so their deliverables at the CEO-only tier
2875
+ // would be as unreachable as the org_rpc route a curated tool replaces.
2876
+ ...RESOURCE_TOOLS,
2516
2877
  ]);
2517
2878
 
2518
2879
  // ---------------------------------------------------------------------------
@@ -2673,6 +3034,13 @@ async function screenOrgOutbound({ channel, recipient, text, agentRoot, screenIm
2673
3034
 
2674
3035
  /** Outbound screen inputs (recipient + text) for a screened tool/method. */
2675
3036
  function outboundScreenShape(toolName, method, input) {
3037
+ if (toolName === "messaging_send_voice_note") {
3038
+ // BOTH strings, because both reach a person: the spoken words and the
3039
+ // written body. Screened before synthesis, so a blocked note costs nothing
3040
+ // and — more to the point — is never recorded in a colleague's own voice.
3041
+ const text = [input?.text, input?.body].filter((v) => typeof v === "string" && v).join("\n");
3042
+ return { recipient: input?.channelId ?? "", text };
3043
+ }
2676
3044
  if (method === "messaging.send") {
2677
3045
  return { recipient: input?.channelId ?? "", text: input?.body ?? "" };
2678
3046
  }
@@ -2998,6 +3366,54 @@ async function executeOrgToolInner(name, input = {}, o = {}) {
2998
3366
  });
2999
3367
  }
3000
3368
 
3369
+ case "messaging_send_voice_note": {
3370
+ const channelId = input && input.channelId ? String(input.channelId) : "";
3371
+ const spokenText = input && input.text ? String(input.text) : "";
3372
+ if (!channelId || !spokenText) {
3373
+ return errFrame("BAD_REQUEST", "messaging_send_voice_note: channelId and text are required");
3374
+ }
3375
+ // 1. RECORD. A refusal is FINAL and is relayed as-is: no key, no
3376
+ // designed voice for this seat, or the hourly cap. The one thing
3377
+ // this tool must never do is come back ok with nothing recorded —
3378
+ // an agent that believes it spoke will not repeat itself in text.
3379
+ const rec = await call("messaging.synthesizeVoiceNote", { text: spokenText }, { ...callOpts });
3380
+ if (!rec.ok) return rec;
3381
+ const note = rec.result || {};
3382
+ if (!note.fileId) {
3383
+ return errFrame("INTERNAL", "messaging_send_voice_note: no recording came back");
3384
+ }
3385
+ // 2. POST. Plain messaging.send with the recording attached — the
3386
+ // server-side outbound gate, the channel ACL, the READY + own-upload
3387
+ // attachment rule, the dedup and the notifier all apply exactly as
3388
+ // they do to a typed message. There is no audio-only send path.
3389
+ const sent = await call(
3390
+ "messaging.send",
3391
+ {
3392
+ channelId,
3393
+ body: input.body ? String(input.body) : spokenText,
3394
+ ...(input.threadRootId ? { threadRootId: String(input.threadRootId) } : {}),
3395
+ attachments: [{ fileId: note.fileId }],
3396
+ idempotencyId: randomUUID(),
3397
+ },
3398
+ { ...callOpts, idempotencyKey: randomUUID() },
3399
+ );
3400
+ if (!sent.ok) {
3401
+ // Recorded but unposted. Said plainly rather than folded into a bare
3402
+ // transport error: the spend HAPPENED, and the caller's next move is
3403
+ // to retry the post with this fileId, not to record again.
3404
+ return errFrame(
3405
+ sent.error?.code || "INTERNAL",
3406
+ `voice note recorded (fileId ${note.fileId}) but the post failed: ${sent.error?.message || "unknown"}`,
3407
+ );
3408
+ }
3409
+ return okFrame({
3410
+ ...(sent.result || {}),
3411
+ fileId: note.fileId,
3412
+ durationMs: note.durationMs ?? null,
3413
+ spokenChars: spokenText.length,
3414
+ });
3415
+ }
3416
+
3001
3417
  case "org_events_tail": {
3002
3418
  const q = {};
3003
3419
  if (input && input.cursor != null) q.cursor = input.cursor;
@@ -3114,8 +3530,18 @@ async function executeOrgToolInner(name, input = {}, o = {}) {
3114
3530
  }
3115
3531
  }
3116
3532
 
3533
+ // Re-exported so both exposure planes and every parity test resolve the access
3534
+ // mechanism from ONE place — a second copy of "which tools may be admin" is how
3535
+ // the rule ended up enforced by hand-listed name arrays in the first place.
3536
+ export { ADMIN_TOOLS, expectedAccessFor, normalizeToolAccess };
3537
+ export { RESOURCE_TOOLS };
3538
+
3117
3539
  export default {
3118
3540
  ORG_TOOLS,
3541
+ RESOURCE_TOOLS,
3542
+ ADMIN_TOOLS,
3543
+ expectedAccessFor,
3544
+ normalizeToolAccess,
3119
3545
  getOrgTools,
3120
3546
  isOrgTool,
3121
3547
  orgToolDef,