@cohortapp/agent-sdk 2.9.0 → 2.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,7 +8,7 @@
8
8
  import { test } from "node:test";
9
9
  import assert from "node:assert/strict";
10
10
  import { promises as fsp } from "fs";
11
- import { mkdirSync, writeFileSync } from "node:fs";
11
+ import { mkdirSync, rmSync, writeFileSync } from "node:fs";
12
12
  import { tmpdir } from "os";
13
13
  import { join } from "path";
14
14
  import { fileURLToPath } from "node:url";
@@ -27,6 +27,9 @@ import {
27
27
  validateRoutingConfig,
28
28
  } from "./model-router.mjs";
29
29
  import { loadCatalog } from "./model-router/catalog.mjs";
30
+ // Read-only, to PROVE (not assume) that a provider the SDK does not bundle
31
+ // inherits credential pooling from the catalog rather than needing new code.
32
+ import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
30
33
 
31
34
  async function makeAgentRoot() {
32
35
  const path = join(
@@ -52,10 +55,15 @@ const ANTHROPIC_BACKEND = {
52
55
  base_url: "https://api.anthropic.com",
53
56
  capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
54
57
  pricing: { input_per_m: 3.0, output_per_m: 15.0 },
58
+ // A CONFIG-DECLARED model map. Deliberately NOT the same ids as
59
+ // ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
60
+ // resolveBackend serves, so pinning them to the built-in defaults would make
61
+ // the assertions vacuous (they would pass even if the config map were being
62
+ // ignored entirely). The built-in defaults get their own test below.
55
63
  models: {
56
64
  classifier: "claude-haiku-4-5-20251001",
57
- default: "claude-sonnet-4-6",
58
- premium: "claude-opus-4-7",
65
+ default: "config-declared-sonnet",
66
+ premium: "config-declared-opus",
59
67
  fast: "claude-haiku-4-5-20251001",
60
68
  },
61
69
  };
@@ -252,14 +260,59 @@ test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () =
252
260
  const opus = resolveBackend({ model_hint: "opus" }, { config });
253
261
  const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
254
262
  const haiku = resolveBackend({ model_hint: "haiku" }, { config });
255
- assert.equal(opus.model, "claude-opus-4-7");
256
- assert.equal(sonnet.model, "claude-sonnet-4-6");
263
+ assert.equal(opus.model, "config-declared-opus");
264
+ assert.equal(sonnet.model, "config-declared-sonnet");
257
265
  assert.equal(haiku.model, "claude-haiku-4-5-20251001");
258
266
  assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
259
267
  assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
260
268
  assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
261
269
  });
262
270
 
271
+ // ---------------------------------------------------------------------------
272
+ // Model freshness — the built-in Anthropic defaults (see the provenance block
273
+ // above ANTHROPIC_DEFAULT in model-router.mjs).
274
+ // ---------------------------------------------------------------------------
275
+
276
+ test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
277
+ // The floor an agent with no routing config lands on. A stale id here never
278
+ // reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
279
+ // DID reach resolved.model, which the cost ledger and telemetry store — so a
280
+ // stale id mis-prices and mis-attributes every unrouted session.
281
+ const models = defaultRoutingConfig().backends.anthropic.models;
282
+ assert.equal(models.premium, "claude-opus-5");
283
+ assert.equal(models.default, "claude-sonnet-5");
284
+ // The fast/classifier tier did NOT move in the 2026-08-13 refresh.
285
+ assert.equal(models.fast, "claude-haiku-4-5-20251001");
286
+ assert.equal(models.classifier, "claude-haiku-4-5-20251001");
287
+ // Guard the specific regression: no 4.x id survives anywhere in the map.
288
+ for (const [tier, id] of Object.entries(models)) {
289
+ assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
290
+ }
291
+ });
292
+
293
+ test("resolveBackend serves the current ids through the no-config default path", () => {
294
+ // End-to-end through the resolver, not just the constant: an agent with no
295
+ // config file (useDefault) resolves premium/default to the -5 pair.
296
+ const config = defaultRoutingConfig();
297
+ assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
298
+ assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
299
+ // …while the CLI flag stays version-agnostic, which is WHY the stale ids were
300
+ // survivable on the wire and only corrupted the ledger.
301
+ assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
302
+ });
303
+
304
+ test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
305
+ // Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
306
+ // agent adds backends.anthropic.models to config/model-routing.yaml and is
307
+ // current WITHOUT an SDK release. normaliseConfig only injects the built-in
308
+ // map when backends.anthropic is absent, so this must win outright.
309
+ const config = {
310
+ backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
311
+ routing_policy: [{ default: true, backend: "anthropic" }],
312
+ };
313
+ assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
314
+ });
315
+
263
316
  test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
264
317
  const config = {
265
318
  backends: { qwen: QWEN_BACKEND },
@@ -833,6 +886,253 @@ test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
833
886
  assert.equal(d.audit.fallback_reason, "no_v2_config");
834
887
  });
835
888
 
889
+ // ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
890
+ //
891
+ // The safety net fires on the three paths that have no chain to walk: the kill
892
+ // switch, "no v2 config", and "nothing survived the gates". It is the ONE place
893
+ // resolve.mjs still names Anthropic model ids, so it is the one place that can
894
+ // go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
895
+ // preference list carries the current -5 ids AND the 4-x succession fallback.
896
+
897
+ /** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
898
+ const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
899
+ agentConfig: {
900
+ catalog_overrides: [
901
+ {
902
+ ref: "anthropic/claude-opus-5",
903
+ status: "available",
904
+ context_tokens: 950000,
905
+ max_tokens: 128000,
906
+ tool_reliability: "A",
907
+ harness: { session: true, direct: true, batch: true },
908
+ cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
909
+ },
910
+ {
911
+ ref: "anthropic/claude-sonnet-5",
912
+ status: "available",
913
+ context_tokens: 950000,
914
+ max_tokens: 64000,
915
+ tool_reliability: "A",
916
+ harness: { session: true, direct: true, batch: true },
917
+ cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
918
+ },
919
+ ],
920
+ },
921
+ });
922
+
923
+ test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
924
+ const frontier = resolveChain(
925
+ { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
926
+ { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
927
+ );
928
+ assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
929
+
930
+ const workhorse = resolveChain(
931
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
932
+ { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
933
+ );
934
+ assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
935
+ });
936
+
937
+ test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
938
+ // This is why the 4-x refs stay in the preference list. Delete them and this
939
+ // does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
940
+ // i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
941
+ // ORDINARY work the frontier model and quietly quadruple its cost.
942
+ const d = resolveChain(
943
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
944
+ { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
945
+ );
946
+ assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
947
+ // Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
948
+ // the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
949
+ });
950
+
951
+ test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
952
+ // THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
953
+ // finished. The test above asserts `chosen.model`, which reads as bookkeeping.
954
+ // This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
955
+ // `--model` — against the catalog the SDK actually ships, with `config: null`,
956
+ // which is the DEFAULT state of every agent because this repo contains no
957
+ // config/model-routing.yaml at all.
958
+ //
959
+ // Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
960
+ // agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
961
+ // row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
962
+ // merely a mis-stamped ledger entry.
963
+ //
964
+ // WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
965
+ // the missing step. Update the expectations here, and DELETE the 4-x refs from
966
+ // pickAnthropicRow's preference lists in the same change, or the museum the
967
+ // provenance block warns about starts accumulating.
968
+ const cases = [
969
+ ["session.responder", {}, "claude-sonnet-4-6"],
970
+ ["decision", { needs_thinking: true }, "claude-opus-4-8"],
971
+ ];
972
+ for (const [task_class, extra, expected] of cases) {
973
+ const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
974
+ for (const [label, env] of [
975
+ ["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
976
+ ["no v2 config", {}],
977
+ ]) {
978
+ const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
979
+ assert.equal(
980
+ d.spawnArgs.modelFlag,
981
+ expected,
982
+ `${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
983
+ );
984
+ // Not the shorthand: nothing downstream re-resolves this to a current build.
985
+ assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
986
+ }
987
+ }
988
+ });
989
+
990
+ test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
991
+ // The last-ditch path: no catalog, so the id goes straight onto `claude
992
+ // --model`. It used to emit {model: "claude-opus-4-8", ref:
993
+ // "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
994
+ // feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
995
+ // both named a different model than the one that ran.
996
+ const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
997
+ mkdirSync(emptyDir, { recursive: true });
998
+ try {
999
+ const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
1000
+ assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
1001
+
1002
+ const d = resolveChain(
1003
+ { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
1004
+ { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1005
+ );
1006
+ assert.equal(d.chosen.model, "claude-opus-5");
1007
+ assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
1008
+ assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
1009
+
1010
+ const ordinary = resolveChain(
1011
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
1012
+ { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1013
+ );
1014
+ assert.equal(ordinary.chosen.model, "claude-sonnet-5");
1015
+ assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
1016
+ } finally {
1017
+ rmSync(emptyDir, { recursive: true, force: true });
1018
+ }
1019
+ });
1020
+
1021
+ // ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
1022
+ //
1023
+ // The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
1024
+ // is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
1025
+ // highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
1026
+ // introduce a whole provider, and resolveChain gates it exactly like a bundled
1027
+ // one. These tests prove the full path — row → alias → chain → credential gate
1028
+ // → chosen — so the remaining work is a bundled YAML file, not plumbing.
1029
+
1030
+ const XAI_PROVIDER_DOC = {
1031
+ provider: "xai",
1032
+ auth_env: "XAI_API_KEY",
1033
+ endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
1034
+ data_residency: "us",
1035
+ models: [
1036
+ {
1037
+ id: "grok-4-fast",
1038
+ status: "available",
1039
+ context_window: 2000000,
1040
+ context_tokens: 1900000,
1041
+ max_tokens: 30000,
1042
+ cost: { input: 0.2, output: 0.5 },
1043
+ cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
1044
+ compat: { thinking_format: "openai", cache_control: "implicit" },
1045
+ // Grade B like every other third-party row: good enough for the direct
1046
+ // lane, structurally barred from hosting a tool-using session (§6.6).
1047
+ tool_reliability: "B",
1048
+ harness: { session: true, direct: true, batch: false },
1049
+ },
1050
+ ],
1051
+ };
1052
+
1053
+ const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
1054
+ agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
1055
+ });
1056
+
1057
+ const XAI_CONFIG = Object.freeze({
1058
+ schema_version: 2,
1059
+ aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
1060
+ defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
1061
+ backends: {
1062
+ anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
1063
+ xai: { allowed_data_classes: ["public"] },
1064
+ },
1065
+ routing_policy: [
1066
+ { match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
1067
+ { default: true, chain: ["default"] },
1068
+ ],
1069
+ fallback_to_anthropic: true,
1070
+ });
1071
+
1072
+ test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
1073
+ const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
1074
+ assert.ok(row, "grok row present after the agent-config merge");
1075
+ assert.equal(row.provider, "xai");
1076
+ assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
1077
+ assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
1078
+ assert.ok(
1079
+ CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
1080
+ "introducing an unbundled provider warns (typo guard) — it does not fail"
1081
+ );
1082
+ });
1083
+
1084
+ test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
1085
+ const d = resolveChain(
1086
+ { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1087
+ { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1088
+ );
1089
+ assert.equal(d.chosen.provider, "xai");
1090
+ assert.equal(d.chosen.model, "grok-4-fast");
1091
+ assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
1092
+ });
1093
+
1094
+ test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
1095
+ const d = resolveChain(
1096
+ { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1097
+ { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
1098
+ );
1099
+ const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1100
+ assert.ok(skipped, "grok recorded in tried[]");
1101
+ assert.equal(skipped.reason, "missing_credential");
1102
+ assert.notEqual(d.chosen.provider, "xai");
1103
+ });
1104
+
1105
+ test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
1106
+ // auth-profiles derives provider → auth_env from the CATALOG, so a provider
1107
+ // the SDK has never heard of gets pooled multi-key rotation for free. This is
1108
+ // the "their own auth profiles" half of the cheap-tier requirement.
1109
+ const profiles = loadAuthProfiles(null, {
1110
+ configDoc: null, // bypass the FS
1111
+ env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
1112
+ authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
1113
+ });
1114
+ assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
1115
+ assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
1116
+ assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
1117
+ });
1118
+
1119
+ test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
1120
+ // Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
1121
+ // third-party row is graded A until the probe ledger earns it. Cheap-tier
1122
+ // reachability must NOT quietly become cheap-tier session hosting.
1123
+ const sessionCfg = {
1124
+ ...XAI_CONFIG,
1125
+ routing_policy: [{ default: true, chain: ["cheap", "default"] }],
1126
+ };
1127
+ const d = resolveChain(
1128
+ { task_class: "session.responder", data_class: "public", token_estimate: 2000 },
1129
+ { catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1130
+ );
1131
+ const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1132
+ assert.equal(skipped?.reason, "grade_too_low");
1133
+ assert.equal(d.chosen.provider, "anthropic");
1134
+ });
1135
+
836
1136
  // ── estCostUSD + ledger join fields ──────────────────────────────────────────
837
1137
 
838
1138
  test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
@@ -102,11 +102,15 @@ export function toMessageEvent(o = {}) {
102
102
  kind: def.kind,
103
103
  subject,
104
104
  // ONLY ever a real Cohort channel id, never an entity id standing in for
105
- // one. `scripts/daemon/responder.mjs sendCohortReply` routes its reply with
106
- // `messaging.send({channel: item.channel_id})`, so putting a task/decision
107
- // id here would post-to-a-non-channel and 400. A surface with no room leaves
108
- // this blank and is routed by `kind` + `raw_ref` instead (see the wiring
109
- // note in this directory's module docs).
105
+ // one. A reply to a room surface routes with
106
+ // `messaging.send({channelId: item.channel_id})`, so putting a task/decision
107
+ // id here would post-to-a-non-channel and come back NOT_FOUND. A surface
108
+ // with no room leaves this BLANK and is routed on its own surface instead —
109
+ // `scripts/daemon/deliver.mjs cohortReplyRoute` reads `raw_ref` for the
110
+ // surface and `scope_id`/`thread_id` for the entity, and sends a board
111
+ // comment to `board.addTaskComment`, a doc comment to `files.commentAdd`,
112
+ // and so on. That enforcement is new: for months the send path fell back to
113
+ // `channel_label` below, so a board reply was addressed to a task's TITLE.
110
114
  channel_id: channelId,
111
115
  channel_label: channelLabel,
112
116
  sender: s(from.name || from.id),
@@ -138,8 +138,13 @@ function agentIdOf(cfg) {
138
138
  *
139
139
  * @param {object} o - { cfg, agentRoot, action, channel }
140
140
  * @param {object} deps - { ledgerImpl?, catalog? } injectable for tests
141
+ *
142
+ * EXPORTED because room sends are no longer the only attributed outbound:
143
+ * `scripts/daemon/deliver.mjs` posts board/doc/decision comments straight onto
144
+ * their entity thread, bypassing `sendMessage`, and an outbound family missing
145
+ * from the ledger is an outbound family nobody can see.
141
146
  */
142
- async function attributeMessaging(o, deps = {}) {
147
+ export async function attributeMessaging(o, deps = {}) {
143
148
  // Opt-out: an explicit `ledgerImpl: null` (or false) disables attribution.
144
149
  if (Object.prototype.hasOwnProperty.call(deps, "ledgerImpl") && !deps.ledgerImpl) return;
145
150
  try {
@@ -346,6 +346,30 @@ export const PARAM_CONTRACT = {
346
346
  "calling.invite": { alias: { invitees: "memberIds", participantIds: "memberIds" }, required: ["callId", "memberIds"] },
347
347
 
348
348
  // ── board ────────────────────────────────────────────────────────────────
349
+ // `board.create` had NO entry, so `normalizeParams` passed its params through
350
+ // byte-for-byte and the board mirror's `{id, col, obligationKey}` reached a
351
+ // `.strict()` schema that accepts none of the three — every mirrored row was
352
+ // refused before a row was ever written, and the only trace was one fail-open
353
+ // warn line. (That mirror has since been retired in favour of the shared work
354
+ // ledger — `board.create`'s remaining caller is the self-directed goal path,
355
+ // lib/goals/collaborate.mjs. The contract stands either way: it is hq's
356
+ // schema, not one caller's habits.) The allow-list below IS hq's createSchema
357
+ // (src/server/methods/board/create.ts); hq's own board-lifecycle test pins the
358
+ // same six-plus-one names from the other side.
359
+ "board.create": {
360
+ alias: { id: "itemId", col: "status" },
361
+ dropAlias: true,
362
+ strict: ["title", "detail", "status", "priority", "workstreamId", "itemId", "channelId"],
363
+ // `status` is refined against the protocol BOARD_STATUS, which is the same
364
+ // ten words as `boardColEnum` — BOARD_COLS above is that list.
365
+ enums: { status: BOARD_COLS, priority: TASK_PRIORITIES },
366
+ required: ["title"],
367
+ note:
368
+ "createSchema is .strict(): `col`/`obligationKey`/a bare `id` 400 the whole create. " +
369
+ "`priority` is the P-scale (P0..P4) — the classifier's critical|high|normal|ignore is " +
370
+ "NOT accepted and must be mapped before the call (boardPriority, scripts/daemon/board-mirror.mjs). " +
371
+ "There is no `assigneeId`: board.claim is what puts an owner on a created row.",
372
+ },
349
373
  "board.complete": {
350
374
  strict: ["itemId", "proof"],
351
375
  transform: completeProof,
@@ -1 +1 @@
1
- 358b95ed6a48faeffa78b7ca717a909905e8ff4e7b75278edf274a0acdaf5dd6
1
+ 9ecd8993a6a9aceac04da479c6626a3b3852378b81bfd2b7590d2eee2099c9a9
@@ -76,6 +76,25 @@ export const FAMILIES = Object.freeze([
76
76
  // seat can PERSIST one it learned rather than re-deriving it every session.
77
77
  // The chain rows carry subject + domain + key only — never the value.
78
78
  "preference",
79
+ // GENESIS (2026-08): the founder's pre-commit workspace draft — the ten-stage
80
+ // run, its editable sections, and the conversational planner over them. The
81
+ // family exists so an agent reaches the SAME draft the founder's own form
82
+ // does; every method is a thin shell over the server action behind the
83
+ // button, never a second implementation of its gates.
84
+ //
85
+ // Its chain rows land in the DRAFTING org — the tenant the founder is signed
86
+ // into while they draft — and say only that a mutation happened, by whom, to
87
+ // which section. The content-level provenance is a different spine: the
88
+ // action's own `GenesisFieldEdit` rows, which `bufferGenesisAudit` replays
89
+ // into the NEW workspace's chain at commit. Both are true and neither
90
+ // replaces the other.
91
+ "genesis",
92
+ // ARCHITECTURE (2026-08): the four tier review decisions and the one whole-
93
+ // architecture approval that gate a Genesis commit. Split from `genesis`
94
+ // because these are REVIEW acts over a draft rather than edits to it, and
95
+ // because the same surface outlives the run — the Org › Architecture view
96
+ // reads this state for a workspace that has long since committed.
97
+ "architecture",
79
98
  ]);
80
99
 
81
100
  /**
@@ -424,7 +443,12 @@ export const METHODS = Object.freeze({
424
443
  // --- file ---
425
444
  "file.pin": { family: "file", scope: "org.write", sideEffecting: true, idempotent: true },
426
445
  "file.unpin": { family: "file", scope: "org.write", sideEffecting: true },
427
- "file.addComment": { family: "file", scope: "org.write", sideEffecting: true },
446
+ // idempotent: a COMMENT CREATE keyed on the caller's message id. Without
447
+ // the flag hq ignores `x-idempotency-key` entirely (dispatch.ts replays only
448
+ // for `def.idempotent === true`), so a daemon retry after a timeout-post-
449
+ // commit posts the same comment two or three times on the thread. Same
450
+ // reasoning as messaging.send, which has carried the flag from the start.
451
+ "file.addComment": { family: "file", scope: "org.write", sideEffecting: true, idempotent: true },
428
452
  "file.listComments": { family: "file", scope: "org.read", sideEffecting: false },
429
453
  "file.listPinned": { family: "file", scope: "org.read", sideEffecting: false },
430
454
  // --- memory ---
@@ -458,7 +482,7 @@ export const METHODS = Object.freeze({
458
482
  "notification.updatePreferences": { family: "notification", scope: "org.write", sideEffecting: true },
459
483
  "notification.getPreferences": { family: "notification", scope: "org.read", sideEffecting: false },
460
484
  // --- decision ---
461
- "decision.comment": { family: "decision", scope: "decision.write", sideEffecting: true },
485
+ "decision.comment": { family: "decision", scope: "decision.write", sideEffecting: true, idempotent: true },
462
486
  "decision.sign": { family: "decision", scope: "decision.write", sideEffecting: true },
463
487
  "decision.reverse": { family: "decision", scope: "decision.write", sideEffecting: true },
464
488
  "decision.requestAdjustment": { family: "decision", scope: "decision.write", sideEffecting: true },
@@ -479,7 +503,7 @@ export const METHODS = Object.freeze({
479
503
  "board.updateTask": { family: "board", scope: "board.write", sideEffecting: true },
480
504
  "board.moveTask": { family: "board", scope: "board.write", sideEffecting: true },
481
505
  "board.assignTask": { family: "board", scope: "board.write", sideEffecting: true },
482
- "board.addTaskComment": { family: "board", scope: "board.write", sideEffecting: true },
506
+ "board.addTaskComment": { family: "board", scope: "board.write", sideEffecting: true, idempotent: true },
483
507
  "board.addTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true, idempotent: true },
484
508
  "board.updateTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true },
485
509
  "board.removeTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true },
@@ -933,6 +957,57 @@ export const METHODS = Object.freeze({
933
957
  // dispatcher cannot keep: the protocol advertises the method, callers get
934
958
  // through scope resolution, and the call 500s. Register it in the same change
935
959
  // that adds src/server/methods/escalation/answer.ts.
960
+ // --- GENESIS + ARCHITECTURE (2026-08): the pre-commit workspace draft and the
961
+ // review gates over it. 1:1 with the founder's own form surface — the ask
962
+ // was parity, so the bar is EQUAL power, never more.
963
+ //
964
+ // SCOPES REUSE `org.read` / `org.write` rather than minting a genesis.*
965
+ // pair. `org.write` is already defined as "EDITOR-equivalent: write
966
+ // org-structural content (members, teams, personas, ... reporting lines,
967
+ // chart layout)" and already sits in DEFAULT_AGENT_SCOPES under the
968
+ // comment "agent UI-parity". A Genesis draft IS that content, one commit
969
+ // earlier; a private scope would be a second name for a permission the
970
+ // org already grants, and every operator who had reasoned about who may
971
+ // edit their structure would have to reason about it twice.
972
+ //
973
+ // WHY REGISTERING THEM AT ALL IS THE FIX. `scopeForMethod()` resolves an
974
+ // UNKNOWN method to `admin`, and `admin` is in neither
975
+ // HUMAN_DEFAULT_SCOPES nor DEFAULT_AGENT_SCOPES — so an undeclared method
976
+ // is not merely undocumented, it is unreachable by every principal that
977
+ // exists. That default is correct and it was firing: the handlers shipped
978
+ // without descriptors and could not serve one request.
979
+ //
980
+ // EVERY WRITE IS `sideEffecting:true` AND APPENDS ONE CHAIN EVENT, which
981
+ // is not bookkeeping — dispatch REQUIRES exactly one append from a
982
+ // side-effecting handler and throws INTERNAL on zero. The tempting escape
983
+ // (declare a durable write `sideEffecting:false` so the requirement never
984
+ // applies) would buy a green test with a lie the whole protocol reads:
985
+ // no chain row, no idempotency replay, and `READS`-shaped callers free to
986
+ // retry a mutation they were told was safe.
987
+ //
988
+ // `idempotent:true` on the writes is opt-in at the caller (it engages only
989
+ // when an idempotencyKey is sent) and it is the honest answer for this
990
+ // family: every write already carries an `(attemptId, editSeq)` base and a
991
+ // replay of a committed op is refused CONFLICT by the stale-base gate, so
992
+ // without the cache a dropped response turns a SUCCEEDED edit into a
993
+ // confusing conflict the caller cannot distinguish from a real race.
994
+ //
995
+ // genesis.plan is a READ that spends tokens: it asks a model to turn an
996
+ // instruction into ops and mutates nothing. Whether it costs money is not
997
+ // what `sideEffecting` means. ---
998
+ "genesis.outline": { family: "genesis", scope: "org.read", sideEffecting: false },
999
+ "genesis.entity": { family: "genesis", scope: "org.read", sideEffecting: false },
1000
+ "genesis.grid": { family: "genesis", scope: "org.read", sideEffecting: false },
1001
+ "genesis.plan": { family: "genesis", scope: "org.read", sideEffecting: false },
1002
+ "genesis.set": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1003
+ "genesis.insert": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1004
+ "genesis.remove": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1005
+ "genesis.reorder": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1006
+ "genesis.revert": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1007
+ "genesis.applyPlan": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
1008
+ "architecture.status": { family: "architecture", scope: "org.read", sideEffecting: false },
1009
+ "architecture.accept": { family: "architecture", scope: "org.write", sideEffecting: true, idempotent: true },
1010
+ "architecture.approve": { family: "architecture", scope: "org.write", sideEffecting: true, idempotent: true },
936
1011
  });
937
1012
 
938
1013
  /**
@@ -134,7 +134,12 @@ test("protocol: frozen family + method counts (additive evolution guard)", () =>
134
134
  // The 2026-08 protocol-convergence pass adds 4: subagent (the org-scoped
135
135
  // sub-agent registry), agent (the agent.wait control lane), mandate (the
136
136
  // MANDATE spine) and preference (member-scoped preferences) → 51.
137
- assert.equal(FAMILIES.length, 51, "family count");
137
+ // The 2026-08 Genesis agent-surface pass adds 2: genesis (the founder's
138
+ // pre-commit workspace draft — the ten-stage run and its editable sections)
139
+ // and architecture (the four tier review decisions and the one whole-
140
+ // architecture approval, which outlive the run as the Org › Architecture
141
+ // view) → 53.
142
+ assert.equal(FAMILIES.length, 53, "family count");
138
143
  // 357 = the mesh-protocol + agent UI-parity + calling + branding + email
139
144
  // methods, plus the Live Integrations surface: integration.toolsetVersion +
140
145
  // integration.listAgentTools (SP1) and integration.invokeTool (SP5, execute a
@@ -176,7 +181,14 @@ test("protocol: frozen family + method counts (additive evolution guard)", () =>
176
181
  // `server/work/ledger.ts` directly, the daemon reaches it over this wire, so
177
182
  // the threshold and the idempotency cannot fork → 461.
178
183
  // the descriptors; the checksum (read at runtime) is the primary drift guard.
179
- assert.equal(Object.keys(METHODS).length, 461, "method count");
184
+ // The 2026-08 Genesis agent-surface pass adds 13, the 1:1 programmatic
185
+ // parity for the founder's own Generate surface: genesis 10 (outline/entity/
186
+ // grid/plan reads + set/insert/remove/reorder/revert/applyPlan writes) and
187
+ // architecture 3 (status read + accept/approve review gates). The handlers
188
+ // had shipped WITHOUT descriptors, which made all 13 unreachable — dispatch
189
+ // rejects unknown methods and scopeForMethod resolves them to `admin`, which
190
+ // no principal holds → 474.
191
+ assert.equal(Object.keys(METHODS).length, 474, "method count");
180
192
  });
181
193
 
182
194
  test("protocol SP3: messaging + calling families/methods/scopes", async () => {
@@ -138,6 +138,7 @@ function clip(s, n) {
138
138
  * (true for the session path — the same signal that fires the holding
139
139
  * message; false for the quick-reply path). The server's gate reads it.
140
140
  * @param {string} [a.title] row title (defaults server-side from the ask)
141
+ * @param {string} [a.priority] P0..P4, applied when the row is created
141
142
  * @param {string} [a.detail] row detail, written once at open
142
143
  * @param {string} [a.note] a progress comment for this step
143
144
  * @param {string[]} [a.notify] extra member ids to @-tag on this step
@@ -184,6 +185,10 @@ export async function recordWorkStep(a = {}) {
184
185
  const requester = requesterMemberId(a.item);
185
186
  if (requester) params.requesterId = requester;
186
187
  if (a.title) params.title = clip(a.title, MAX_TITLE);
188
+ // The classifier's urgency, already mapped onto hq's P-scale by the caller
189
+ // (`board-mirror.mjs#boardPriority` — hq 400s the whole call on anything
190
+ // that is not P0..P4). Only ever read when the row is CREATED.
191
+ if (a.priority) params.priority = String(a.priority);
187
192
  if (a.detail) params.detail = clip(a.detail, MAX_BODY);
188
193
  if (a.note) params.note = clip(a.note, MAX_NOTE);
189
194
  if (Array.isArray(a.notify) && a.notify.length) params.notify = a.notify.slice(0, 8);
@@ -133,6 +133,42 @@ test("an accepted step sends the ask's identity and the deferred fact", async ()
133
133
  assert.equal(res.taskId, "t_1");
134
134
  });
135
135
 
136
+ test("a `working` step is what puts the row where the live-work surface looks", async () => {
137
+ // `accepted` lands the row in `triage`, and NO live-work surface reads
138
+ // `triage` — hq's "On now" banner reads `running` and nothing else. Sending
139
+ // only the first step is why every daemon row existed and was still
140
+ // invisible, and why a second, ungated board writer came to exist beside it.
141
+ const { calls, impl } = capture();
142
+ await recordWorkStep({
143
+ item: inboxItem(),
144
+ stage: "working",
145
+ deferred: true,
146
+ cfg: CFG,
147
+ trackImpl: impl,
148
+ });
149
+ assert.equal(calls[0].params.stage, "working");
150
+ assert.equal(calls[0].opts.idempotencyKey, "board.track:ch_dm_1:msg_abc123:working");
151
+ });
152
+
153
+ test("the classifier's urgency rides along, already on hq's P-scale", async () => {
154
+ // hq validates `priority` against z.enum(["P0".."P4"]) and 400s the whole
155
+ // call on anything else, so the mapping happens before the wire
156
+ // (scripts/daemon/board-mirror.mjs#boardPriority) and this only forwards it.
157
+ const { calls, impl } = capture();
158
+ await recordWorkStep({
159
+ item: inboxItem(),
160
+ stage: "accepted",
161
+ priority: "P0",
162
+ cfg: CFG,
163
+ trackImpl: impl,
164
+ });
165
+ assert.equal(calls[0].params.priority, "P0");
166
+
167
+ const plain = capture();
168
+ await recordWorkStep({ item: inboxItem(), stage: "accepted", cfg: CFG, trackImpl: plain.impl });
169
+ assert.ok(!("priority" in plain.calls[0].params), "no priority ⇒ hq's own default answers");
170
+ });
171
+
136
172
  test("directedness comes from the ingest layer's real verdict, never a guess", async () => {
137
173
  const { calls, impl } = capture();
138
174
  await recordWorkStep({
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cohortapp/agent-sdk",
3
- "version": "2.9.0",
3
+ "version": "2.10.0",
4
4
  "description": "Cohort Agent SDK \u2014 autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
5
5
  "type": "module",
6
6
  "bin": {