@cohortapp/agent-sdk 2.9.1 → 2.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,7 +8,7 @@
8
8
  import { test } from "node:test";
9
9
  import assert from "node:assert/strict";
10
10
  import { promises as fsp } from "fs";
11
- import { mkdirSync, writeFileSync } from "node:fs";
11
+ import { mkdirSync, rmSync, writeFileSync } from "node:fs";
12
12
  import { tmpdir } from "os";
13
13
  import { join } from "path";
14
14
  import { fileURLToPath } from "node:url";
@@ -27,6 +27,9 @@ import {
27
27
  validateRoutingConfig,
28
28
  } from "./model-router.mjs";
29
29
  import { loadCatalog } from "./model-router/catalog.mjs";
30
+ // Read-only, to PROVE (not assume) that a provider the SDK does not bundle
31
+ // inherits credential pooling from the catalog rather than needing new code.
32
+ import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
30
33
 
31
34
  async function makeAgentRoot() {
32
35
  const path = join(
@@ -52,10 +55,15 @@ const ANTHROPIC_BACKEND = {
52
55
  base_url: "https://api.anthropic.com",
53
56
  capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
54
57
  pricing: { input_per_m: 3.0, output_per_m: 15.0 },
58
+ // A CONFIG-DECLARED model map. Deliberately NOT the same ids as
59
+ // ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
60
+ // resolveBackend serves, so pinning them to the built-in defaults would make
61
+ // the assertions vacuous (they would pass even if the config map were being
62
+ // ignored entirely). The built-in defaults get their own test below.
55
63
  models: {
56
64
  classifier: "claude-haiku-4-5-20251001",
57
- default: "claude-sonnet-4-6",
58
- premium: "claude-opus-4-7",
65
+ default: "config-declared-sonnet",
66
+ premium: "config-declared-opus",
59
67
  fast: "claude-haiku-4-5-20251001",
60
68
  },
61
69
  };
@@ -252,14 +260,59 @@ test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () =
252
260
  const opus = resolveBackend({ model_hint: "opus" }, { config });
253
261
  const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
254
262
  const haiku = resolveBackend({ model_hint: "haiku" }, { config });
255
- assert.equal(opus.model, "claude-opus-4-7");
256
- assert.equal(sonnet.model, "claude-sonnet-4-6");
263
+ assert.equal(opus.model, "config-declared-opus");
264
+ assert.equal(sonnet.model, "config-declared-sonnet");
257
265
  assert.equal(haiku.model, "claude-haiku-4-5-20251001");
258
266
  assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
259
267
  assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
260
268
  assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
261
269
  });
262
270
 
271
+ // ---------------------------------------------------------------------------
272
+ // Model freshness — the built-in Anthropic defaults (see the provenance block
273
+ // above ANTHROPIC_DEFAULT in model-router.mjs).
274
+ // ---------------------------------------------------------------------------
275
+
276
+ test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
277
+ // The floor an agent with no routing config lands on. A stale id here never
278
+ // reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
279
+ // DID reach resolved.model, which the cost ledger and telemetry store — so a
280
+ // stale id mis-prices and mis-attributes every unrouted session.
281
+ const models = defaultRoutingConfig().backends.anthropic.models;
282
+ assert.equal(models.premium, "claude-opus-5");
283
+ assert.equal(models.default, "claude-sonnet-5");
284
+ // The fast/classifier tier did NOT move in the 2026-08-13 refresh.
285
+ assert.equal(models.fast, "claude-haiku-4-5-20251001");
286
+ assert.equal(models.classifier, "claude-haiku-4-5-20251001");
287
+ // Guard the specific regression: no 4.x id survives anywhere in the map.
288
+ for (const [tier, id] of Object.entries(models)) {
289
+ assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
290
+ }
291
+ });
292
+
293
+ test("resolveBackend serves the current ids through the no-config default path", () => {
294
+ // End-to-end through the resolver, not just the constant: an agent with no
295
+ // config file (useDefault) resolves premium/default to the -5 pair.
296
+ const config = defaultRoutingConfig();
297
+ assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
298
+ assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
299
+ // …while the CLI flag stays version-agnostic, which is WHY the stale ids were
300
+ // survivable on the wire and only corrupted the ledger.
301
+ assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
302
+ });
303
+
304
+ test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
305
+ // Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
306
+ // agent adds backends.anthropic.models to config/model-routing.yaml and is
307
+ // current WITHOUT an SDK release. normaliseConfig only injects the built-in
308
+ // map when backends.anthropic is absent, so this must win outright.
309
+ const config = {
310
+ backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
311
+ routing_policy: [{ default: true, backend: "anthropic" }],
312
+ };
313
+ assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
314
+ });
315
+
263
316
  test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
264
317
  const config = {
265
318
  backends: { qwen: QWEN_BACKEND },
@@ -833,6 +886,253 @@ test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
833
886
  assert.equal(d.audit.fallback_reason, "no_v2_config");
834
887
  });
835
888
 
889
+ // ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
890
+ //
891
+ // The safety net fires on the three paths that have no chain to walk: the kill
892
+ // switch, "no v2 config", and "nothing survived the gates". It is the ONE place
893
+ // resolve.mjs still names Anthropic model ids, so it is the one place that can
894
+ // go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
895
+ // preference list carries the current -5 ids AND the 4-x succession fallback.
896
+
897
+ /** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
898
+ const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
899
+ agentConfig: {
900
+ catalog_overrides: [
901
+ {
902
+ ref: "anthropic/claude-opus-5",
903
+ status: "available",
904
+ context_tokens: 950000,
905
+ max_tokens: 128000,
906
+ tool_reliability: "A",
907
+ harness: { session: true, direct: true, batch: true },
908
+ cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
909
+ },
910
+ {
911
+ ref: "anthropic/claude-sonnet-5",
912
+ status: "available",
913
+ context_tokens: 950000,
914
+ max_tokens: 64000,
915
+ tool_reliability: "A",
916
+ harness: { session: true, direct: true, batch: true },
917
+ cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
918
+ },
919
+ ],
920
+ },
921
+ });
922
+
923
+ test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
924
+ const frontier = resolveChain(
925
+ { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
926
+ { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
927
+ );
928
+ assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
929
+
930
+ const workhorse = resolveChain(
931
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
932
+ { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
933
+ );
934
+ assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
935
+ });
936
+
937
+ test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
938
+ // This is why the 4-x refs stay in the preference list. Delete them and this
939
+ // does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
940
+ // i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
941
+ // ORDINARY work the frontier model and quietly quadruple its cost.
942
+ const d = resolveChain(
943
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
944
+ { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
945
+ );
946
+ assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
947
+ // Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
948
+ // the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
949
+ });
950
+
951
+ test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
952
+ // THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
953
+ // finished. The test above asserts `chosen.model`, which reads as bookkeeping.
954
+ // This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
955
+ // `--model` — against the catalog the SDK actually ships, with `config: null`,
956
+ // which is the DEFAULT state of every agent because this repo contains no
957
+ // config/model-routing.yaml at all.
958
+ //
959
+ // Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
960
+ // agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
961
+ // row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
962
+ // merely a mis-stamped ledger entry.
963
+ //
964
+ // WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
965
+ // the missing step. Update the expectations here, and DELETE the 4-x refs from
966
+ // pickAnthropicRow's preference lists in the same change, or the museum the
967
+ // provenance block warns about starts accumulating.
968
+ const cases = [
969
+ ["session.responder", {}, "claude-sonnet-4-6"],
970
+ ["decision", { needs_thinking: true }, "claude-opus-4-8"],
971
+ ];
972
+ for (const [task_class, extra, expected] of cases) {
973
+ const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
974
+ for (const [label, env] of [
975
+ ["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
976
+ ["no v2 config", {}],
977
+ ]) {
978
+ const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
979
+ assert.equal(
980
+ d.spawnArgs.modelFlag,
981
+ expected,
982
+ `${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
983
+ );
984
+ // Not the shorthand: nothing downstream re-resolves this to a current build.
985
+ assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
986
+ }
987
+ }
988
+ });
989
+
990
+ test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
991
+ // The last-ditch path: no catalog, so the id goes straight onto `claude
992
+ // --model`. It used to emit {model: "claude-opus-4-8", ref:
993
+ // "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
994
+ // feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
995
+ // both named a different model than the one that ran.
996
+ const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
997
+ mkdirSync(emptyDir, { recursive: true });
998
+ try {
999
+ const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
1000
+ assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
1001
+
1002
+ const d = resolveChain(
1003
+ { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
1004
+ { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1005
+ );
1006
+ assert.equal(d.chosen.model, "claude-opus-5");
1007
+ assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
1008
+ assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
1009
+
1010
+ const ordinary = resolveChain(
1011
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
1012
+ { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1013
+ );
1014
+ assert.equal(ordinary.chosen.model, "claude-sonnet-5");
1015
+ assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
1016
+ } finally {
1017
+ rmSync(emptyDir, { recursive: true, force: true });
1018
+ }
1019
+ });
1020
+
1021
+ // ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
1022
+ //
1023
+ // The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
1024
+ // is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
1025
+ // highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
1026
+ // introduce a whole provider, and resolveChain gates it exactly like a bundled
1027
+ // one. These tests prove the full path — row → alias → chain → credential gate
1028
+ // → chosen — so the remaining work is a bundled YAML file, not plumbing.
1029
+
1030
+ const XAI_PROVIDER_DOC = {
1031
+ provider: "xai",
1032
+ auth_env: "XAI_API_KEY",
1033
+ endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
1034
+ data_residency: "us",
1035
+ models: [
1036
+ {
1037
+ id: "grok-4-fast",
1038
+ status: "available",
1039
+ context_window: 2000000,
1040
+ context_tokens: 1900000,
1041
+ max_tokens: 30000,
1042
+ cost: { input: 0.2, output: 0.5 },
1043
+ cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
1044
+ compat: { thinking_format: "openai", cache_control: "implicit" },
1045
+ // Grade B like every other third-party row: good enough for the direct
1046
+ // lane, structurally barred from hosting a tool-using session (§6.6).
1047
+ tool_reliability: "B",
1048
+ harness: { session: true, direct: true, batch: false },
1049
+ },
1050
+ ],
1051
+ };
1052
+
1053
+ const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
1054
+ agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
1055
+ });
1056
+
1057
+ const XAI_CONFIG = Object.freeze({
1058
+ schema_version: 2,
1059
+ aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
1060
+ defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
1061
+ backends: {
1062
+ anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
1063
+ xai: { allowed_data_classes: ["public"] },
1064
+ },
1065
+ routing_policy: [
1066
+ { match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
1067
+ { default: true, chain: ["default"] },
1068
+ ],
1069
+ fallback_to_anthropic: true,
1070
+ });
1071
+
1072
+ test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
1073
+ const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
1074
+ assert.ok(row, "grok row present after the agent-config merge");
1075
+ assert.equal(row.provider, "xai");
1076
+ assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
1077
+ assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
1078
+ assert.ok(
1079
+ CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
1080
+ "introducing an unbundled provider warns (typo guard) — it does not fail"
1081
+ );
1082
+ });
1083
+
1084
+ test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
1085
+ const d = resolveChain(
1086
+ { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1087
+ { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1088
+ );
1089
+ assert.equal(d.chosen.provider, "xai");
1090
+ assert.equal(d.chosen.model, "grok-4-fast");
1091
+ assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
1092
+ });
1093
+
1094
+ test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
1095
+ const d = resolveChain(
1096
+ { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1097
+ { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
1098
+ );
1099
+ const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1100
+ assert.ok(skipped, "grok recorded in tried[]");
1101
+ assert.equal(skipped.reason, "missing_credential");
1102
+ assert.notEqual(d.chosen.provider, "xai");
1103
+ });
1104
+
1105
+ test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
1106
+ // auth-profiles derives provider → auth_env from the CATALOG, so a provider
1107
+ // the SDK has never heard of gets pooled multi-key rotation for free. This is
1108
+ // the "their own auth profiles" half of the cheap-tier requirement.
1109
+ const profiles = loadAuthProfiles(null, {
1110
+ configDoc: null, // bypass the FS
1111
+ env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
1112
+ authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
1113
+ });
1114
+ assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
1115
+ assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
1116
+ assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
1117
+ });
1118
+
1119
+ test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
1120
+ // Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
1121
+ // third-party row is graded A until the probe ledger earns it. Cheap-tier
1122
+ // reachability must NOT quietly become cheap-tier session hosting.
1123
+ const sessionCfg = {
1124
+ ...XAI_CONFIG,
1125
+ routing_policy: [{ default: true, chain: ["cheap", "default"] }],
1126
+ };
1127
+ const d = resolveChain(
1128
+ { task_class: "session.responder", data_class: "public", token_estimate: 2000 },
1129
+ { catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1130
+ );
1131
+ const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1132
+ assert.equal(skipped?.reason, "grade_too_low");
1133
+ assert.equal(d.chosen.provider, "anthropic");
1134
+ });
1135
+
836
1136
  // ── estCostUSD + ledger join fields ──────────────────────────────────────────
837
1137
 
838
1138
  test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
@@ -102,11 +102,15 @@ export function toMessageEvent(o = {}) {
102
102
  kind: def.kind,
103
103
  subject,
104
104
  // ONLY ever a real Cohort channel id, never an entity id standing in for
105
- // one. `scripts/daemon/responder.mjs sendCohortReply` routes its reply with
106
- // `messaging.send({channel: item.channel_id})`, so putting a task/decision
107
- // id here would post-to-a-non-channel and 400. A surface with no room leaves
108
- // this blank and is routed by `kind` + `raw_ref` instead (see the wiring
109
- // note in this directory's module docs).
105
+ // one. A reply to a room surface routes with
106
+ // `messaging.send({channelId: item.channel_id})`, so putting a task/decision
107
+ // id here would post-to-a-non-channel and come back NOT_FOUND. A surface
108
+ // with no room leaves this BLANK and is routed on its own surface instead —
109
+ // `scripts/daemon/deliver.mjs cohortReplyRoute` reads `raw_ref` for the
110
+ // surface and `scope_id`/`thread_id` for the entity, and sends a board
111
+ // comment to `board.addTaskComment`, a doc comment to `files.commentAdd`,
112
+ // and so on. That enforcement is new: for months the send path fell back to
113
+ // `channel_label` below, so a board reply was addressed to a task's TITLE.
110
114
  channel_id: channelId,
111
115
  channel_label: channelLabel,
112
116
  sender: s(from.name || from.id),
@@ -138,8 +138,13 @@ function agentIdOf(cfg) {
138
138
  *
139
139
  * @param {object} o - { cfg, agentRoot, action, channel }
140
140
  * @param {object} deps - { ledgerImpl?, catalog? } injectable for tests
141
+ *
142
+ * EXPORTED because room sends are no longer the only attributed outbound:
143
+ * `scripts/daemon/deliver.mjs` posts board/doc/decision comments straight onto
144
+ * their entity thread, bypassing `sendMessage`, and an outbound family missing
145
+ * from the ledger is an outbound family nobody can see.
141
146
  */
142
- async function attributeMessaging(o, deps = {}) {
147
+ export async function attributeMessaging(o, deps = {}) {
143
148
  // Opt-out: an explicit `ledgerImpl: null` (or false) disables attribution.
144
149
  if (Object.prototype.hasOwnProperty.call(deps, "ledgerImpl") && !deps.ledgerImpl) return;
145
150
  try {
@@ -1 +1 @@
1
- 358b95ed6a48faeffa78b7ca717a909905e8ff4e7b75278edf274a0acdaf5dd6
1
+ 9ecd8993a6a9aceac04da479c6626a3b3852378b81bfd2b7590d2eee2099c9a9
@@ -76,6 +76,25 @@ export const FAMILIES = Object.freeze([
76
76
  // seat can PERSIST one it learned rather than re-deriving it every session.
77
77
  // The chain rows carry subject + domain + key only — never the value.
78
78
  "preference",
79
+ // GENESIS (2026-08): the founder's pre-commit workspace draft — the ten-stage
80
+ // run, its editable sections, and the conversational planner over them. The
81
+ // family exists so an agent reaches the SAME draft the founder's own form
82
+ // does; every method is a thin shell over the server action behind the
83
+ // button, never a second implementation of its gates.
84
+ //
85
+ // Its chain rows land in the DRAFTING org — the tenant the founder is signed
86
+ // into while they draft — and say only that a mutation happened, by whom, to
87
+ // which section. The content-level provenance is a different spine: the
88
+ // action's own `GenesisFieldEdit` rows, which `bufferGenesisAudit` replays
89
+ // into the NEW workspace's chain at commit. Both are true and neither
90
+ // replaces the other.
91
+ "genesis",
92
+ // ARCHITECTURE (2026-08): the four tier review decisions and the one whole-
93
+ // architecture approval that gate a Genesis commit. Split from `genesis`
94
+ // because these are REVIEW acts over a draft rather than edits to it, and
95
+ // because the same surface outlives the run — the Org › Architecture view
96
+ // reads this state for a workspace that has long since committed.
97
+ "architecture",
79
98
  ]);
80
99
 
81
100
  /**
@@ -424,7 +443,12 @@ export const METHODS = Object.freeze({
424
443
  // --- file ---
425
444
  "file.pin": { family: "file", scope: "org.write", sideEffecting: true, idempotent: true },
426
445
  "file.unpin": { family: "file", scope: "org.write", sideEffecting: true },
427
- "file.addComment": { family: "file", scope: "org.write", sideEffecting: true },
446
+ // idempotent: a COMMENT CREATE keyed on the caller's message id. Without
447
+ // the flag hq ignores `x-idempotency-key` entirely (dispatch.ts replays only
448
+ // for `def.idempotent === true`), so a daemon retry after a timeout-post-
449
+ // commit posts the same comment two or three times on the thread. Same
450
+ // reasoning as messaging.send, which has carried the flag from the start.
451
+ "file.addComment": { family: "file", scope: "org.write", sideEffecting: true, idempotent: true },
428
452
  "file.listComments": { family: "file", scope: "org.read", sideEffecting: false },
429
453
  "file.listPinned": { family: "file", scope: "org.read", sideEffecting: false },
430
454
  // --- memory ---
@@ -458,7 +482,7 @@ export const METHODS = Object.freeze({
458
482
  "notification.updatePreferences": { family: "notification", scope: "org.write", sideEffecting: true },
459
483
  "notification.getPreferences": { family: "notification", scope: "org.read", sideEffecting: false },
460
484
  // --- decision ---
461
- "decision.comment": { family: "decision", scope: "decision.write", sideEffecting: true },
485
+ "decision.comment": { family: "decision", scope: "decision.write", sideEffecting: true, idempotent: true },
462
486
  "decision.sign": { family: "decision", scope: "decision.write", sideEffecting: true },
463
487
  "decision.reverse": { family: "decision", scope: "decision.write", sideEffecting: true },
464
488
  "decision.requestAdjustment": { family: "decision", scope: "decision.write", sideEffecting: true },
@@ -479,7 +503,7 @@ export const METHODS = Object.freeze({
479
503
  "board.updateTask": { family: "board", scope: "board.write", sideEffecting: true },
480
504
  "board.moveTask": { family: "board", scope: "board.write", sideEffecting: true },
481
505
  "board.assignTask": { family: "board", scope: "board.write", sideEffecting: true },
482
- "board.addTaskComment": { family: "board", scope: "board.write", sideEffecting: true },
506
+ "board.addTaskComment": { family: "board", scope: "board.write", sideEffecting: true, idempotent: true },
483
507
  "board.addTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true, idempotent: true },
484
508
  "board.updateTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true },
485
509
  "board.removeTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true },
@@ -933,6 +957,57 @@ export const METHODS = Object.freeze({
933
957
  // dispatcher cannot keep: the protocol advertises the method, callers get
934
958
  // through scope resolution, and the call 500s. Register it in the same change
935
959
  // that adds src/server/methods/escalation/answer.ts.
960
+ // --- GENESIS + ARCHITECTURE (2026-08): the pre-commit workspace draft and the
961
+ // review gates over it. 1:1 with the founder's own form surface — the ask
962
+ // was parity, so the bar is EQUAL power, never more.
963
+ //
964
+ // SCOPES REUSE `org.read` / `org.write` rather than minting a genesis.*
965
+ // pair. `org.write` is already defined as "EDITOR-equivalent: write
966
+ // org-structural content (members, teams, personas, ... reporting lines,
967
+ // chart layout)" and already sits in DEFAULT_AGENT_SCOPES under the
968
+ // comment "agent UI-parity". A Genesis draft IS that content, one commit
969
+ // earlier; a private scope would be a second name for a permission the
970
+ // org already grants, and every operator who had reasoned about who may
971
+ // edit their structure would have to reason about it twice.
972
+ //
973
+ // WHY REGISTERING THEM AT ALL IS THE FIX. `scopeForMethod()` resolves an
974
+ // UNKNOWN method to `admin`, and `admin` is in neither
975
+ // HUMAN_DEFAULT_SCOPES nor DEFAULT_AGENT_SCOPES — so an undeclared method
976
+ // is not merely undocumented, it is unreachable by every principal that
977
+ // exists. That default is correct and it was firing: the handlers shipped
978
+ // without descriptors and could not serve one request.
979
+ //
980
+ // EVERY WRITE IS `sideEffecting:true` AND APPENDS ONE CHAIN EVENT, which
981
+ // is not bookkeeping — dispatch REQUIRES exactly one append from a
982
+ // side-effecting handler and throws INTERNAL on zero. The tempting escape
983
+ // (declare a durable write `sideEffecting:false` so the requirement never
984
+ // applies) would buy a green test with a lie the whole protocol reads:
985
+ // no chain row, no idempotency replay, and `READS`-shaped callers free to
986
+ // retry a mutation they were told was safe.
987
+ //
988
+ // `idempotent:true` on the writes is opt-in at the caller (it engages only
989
+ // when an idempotencyKey is sent) and it is the honest answer for this
990
+ // family: every write already carries an `(attemptId, editSeq)` base and a
991
+ // replay of a committed op is refused CONFLICT by the stale-base gate, so
992
+ // without the cache a dropped response turns a SUCCEEDED edit into a
993
+ // confusing conflict the caller cannot distinguish from a real race.
994
+ //
995
+ // genesis.plan is a READ that spends tokens: it asks a model to turn an
996
+ // instruction into ops and mutates nothing. Whether it costs money is not
997
+ // what `sideEffecting` means. ---
998
+ "genesis.outline": { family: "genesis", scope: "org.read", sideEffecting: false },
999
+ "genesis.entity": { family: "genesis", scope: "org.read", sideEffecting: false },
1000
+ "genesis.grid": { family: "genesis", scope: "org.read", sideEffecting: false },
1001
+ "genesis.plan": { family: "genesis", scope: "org.read", sideEffecting: false },
1002
+ "genesis.set": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1003
+ "genesis.insert": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1004
+ "genesis.remove": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1005
+ "genesis.reorder": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1006
+ "genesis.revert": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1007
+ "genesis.applyPlan": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1008
+ "architecture.status": { family: "architecture", scope: "org.read", sideEffecting: false },
1009
+ "architecture.accept": { family: "architecture", scope: "org.write", sideEffecting: true, idempotent: true },
1010
+ "architecture.approve": { family: "architecture", scope: "org.write", sideEffecting: true, idempotent: true },
936
1011
  });
937
1012
 
938
1013
  /**
@@ -134,7 +134,12 @@ test("protocol: frozen family + method counts (additive evolution guard)", () =>
134
134
  // The 2026-08 protocol-convergence pass adds 4: subagent (the org-scoped
135
135
  // sub-agent registry), agent (the agent.wait control lane), mandate (the
136
136
  // MANDATE spine) and preference (member-scoped preferences) → 51.
137
- assert.equal(FAMILIES.length, 51, "family count");
137
+ // The 2026-08 Genesis agent-surface pass adds 2: genesis (the founder's
138
+ // pre-commit workspace draft — the ten-stage run and its editable sections)
139
+ // and architecture (the four tier review decisions and the one whole-
140
+ // architecture approval, which outlive the run as the Org › Architecture
141
+ // view) → 53.
142
+ assert.equal(FAMILIES.length, 53, "family count");
138
143
  // 357 = the mesh-protocol + agent UI-parity + calling + branding + email
139
144
  // methods, plus the Live Integrations surface: integration.toolsetVersion +
140
145
  // integration.listAgentTools (SP1) and integration.invokeTool (SP5, execute a
@@ -176,7 +181,14 @@ test("protocol: frozen family + method counts (additive evolution guard)", () =>
176
181
  // `server/work/ledger.ts` directly, the daemon reaches it over this wire, so
177
182
  // the threshold and the idempotency cannot fork → 461.
178
183
  // the descriptors; the checksum (read at runtime) is the primary drift guard.
179
- assert.equal(Object.keys(METHODS).length, 461, "method count");
184
+ // The 2026-08 Genesis agent-surface pass adds 13, the 1:1 programmatic
185
+ // parity for the founder's own Generate surface: genesis 10 (outline/entity/
186
+ // grid/plan reads + set/insert/remove/reorder/revert/applyPlan writes) and
187
+ // architecture 3 (status read + accept/approve review gates). The handlers
188
+ // had shipped WITHOUT descriptors, which made all 13 unreachable — dispatch
189
+ // rejects unknown methods and scopeForMethod resolves them to `admin`, which
190
+ // no principal holds → 474.
191
+ assert.equal(Object.keys(METHODS).length, 474, "method count");
180
192
  });
181
193
 
182
194
  test("protocol SP3: messaging + calling families/methods/scopes", async () => {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cohortapp/agent-sdk",
3
- "version": "2.9.1",
3
+ "version": "2.10.0",
4
4
  "description": "Cohort Agent SDK \u2014 autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -225,14 +225,25 @@ export function openObligations() {
225
225
  * object is gone. Content is truncated: this file is a debt record, not a
226
226
  * message archive.
227
227
  */
228
- function itemSnapshot(item) {
228
+ export function itemSnapshot(item) {
229
229
  return {
230
230
  id: item.id || null,
231
231
  raw_ref: item.raw_ref || null,
232
232
  message_id: item.message_id || null,
233
233
  service: item.service || null,
234
+ // `kind` IS AN ADDRESS TOO — `deliver.cohortSurfaceOf` falls back to it when
235
+ // `raw_ref` carries no surface name, and without it every such item
236
+ // re-routed from the sweep as a plain `dm` with no channel: undeliverable.
237
+ kind: item.kind || null,
234
238
  channel: item.channel || null,
235
239
  channel_id: item.channel_id || null,
240
+ // `scope_id` IS AN ADDRESS, not decoration. `deliver.cohortReplyRoute` reads
241
+ // it FIRST for every roomless Cohort surface (board, doc, decision). The
242
+ // sweep re-delivers from `rec.item`, so leaving it out meant the sweep
243
+ // routed a DIFFERENT item than the live path did — masked only by every
244
+ // hydrator happening to set `thread_id` too. Anything the routing table
245
+ // reads belongs in this snapshot.
246
+ scope_id: item.scope_id || null,
236
247
  thread_id: item.thread_id || null,
237
248
  sender: item.sender || null,
238
249
  sender_email: item.sender_email || null,