@cohortapp/agent-sdk 2.9.1 → 2.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/.claude/commands/init-maestro.md +16 -9
  2. package/docs/guides/mac-mini.md +11 -1
  3. package/docs/runbooks/cohort-cutover.md +16 -0
  4. package/lib/channels/inbox-item.mjs +4 -0
  5. package/lib/comms/send-gate.mjs +23 -1
  6. package/lib/comms/send-gate.test.mjs +24 -0
  7. package/lib/mcp/server.test.mjs +16 -4
  8. package/lib/model-router/economics.mjs +53 -1
  9. package/lib/model-router/economics.test.mjs +76 -0
  10. package/lib/model-router/resolve.mjs +57 -4
  11. package/lib/model-router.mjs +95 -8
  12. package/lib/model-router.test.mjs +305 -5
  13. package/lib/org/client.mjs +58 -1
  14. package/lib/org/inbound/project.mjs +9 -5
  15. package/lib/org/messaging.mjs +6 -1
  16. package/lib/org/protocol.checksum +1 -1
  17. package/lib/org/protocol.mjs +176 -3
  18. package/lib/org/protocol.test.mjs +31 -2
  19. package/lib/org/resource-tools.mjs +317 -0
  20. package/lib/org/resource-tools.test.mjs +361 -0
  21. package/lib/org/tool-access.mjs +176 -0
  22. package/lib/org/tool-access.test.mjs +144 -0
  23. package/lib/org/tool-surface.mjs +431 -5
  24. package/lib/org/tool-surface.test.mjs +385 -8
  25. package/lib/org/ui-parity.mjs +196 -3
  26. package/lib/org/ui-parity.test.mjs +126 -7
  27. package/lib/tool-definitions.js +23 -2
  28. package/package.json +2 -2
  29. package/plugins/maestro-skills/.claude-plugin/marketplace.json +1 -1
  30. package/plugins/maestro-skills/plugin.json +4 -0
  31. package/plugins/maestro-skills/skills/venture-deliverables.md +176 -0
  32. package/policies/information-barriers.yaml +34 -7
  33. package/scripts/ci/check-no-residual-identity.mjs +281 -9
  34. package/scripts/ci/check-no-residual-identity.test.mjs +115 -2
  35. package/scripts/cloud-relay/voice/relay-identity.test.mjs +96 -0
  36. package/scripts/cloud-relay/voice/server.mjs +42 -2
  37. package/scripts/cost/track-claude-usage-pricing.test.mjs +183 -0
  38. package/scripts/cost/track-claude-usage.mjs +113 -4
  39. package/scripts/daemon/agent-daemon.mjs +150 -3
  40. package/scripts/daemon/agent-daemon.test.mjs +190 -0
  41. package/scripts/daemon/assurance.mjs +50 -16
  42. package/scripts/daemon/assurance.test.mjs +39 -1
  43. package/scripts/daemon/classifier-identity.test.mjs +137 -0
  44. package/scripts/daemon/classifier.mjs +98 -17
  45. package/scripts/daemon/deliver.mjs +457 -33
  46. package/scripts/daemon/deliver.test.mjs +564 -0
  47. package/scripts/daemon/prompt-builder-preamble.test.mjs +210 -0
  48. package/scripts/daemon/prompt-builder.mjs +264 -41
  49. package/scripts/daemon/prompt-builder.test.mjs +5 -5
  50. package/scripts/daemon/responder-history.test.mjs +18 -2
  51. package/scripts/daemon/responder.mjs +7 -1
  52. package/scripts/disclosure_boundaries.py +56 -5
  53. package/scripts/huddle/huddle-prompt.test.mjs +176 -0
  54. package/scripts/huddle/huddle-server.mjs +128 -13
  55. package/scripts/local-triggers/autoupdate.sh +83 -0
  56. package/scripts/local-triggers/generate-plists.sh +9 -0
  57. package/scripts/local-triggers/generate-plists.test.mjs +12 -10
  58. package/scripts/media-generation/brand-clause.test.mjs +135 -0
  59. package/scripts/media-generation/gemini-image-client.mjs +27 -9
  60. package/scripts/media-generation/generate-assets.mjs +102 -7
  61. package/scripts/pre-draft-context.py +91 -15
  62. package/scripts/spawn-session.sh +36 -6
  63. package/scripts/test-employer-grounding.py +348 -0
  64. package/scripts/validate_outbound.py +190 -26
@@ -8,7 +8,7 @@
8
8
  import { test } from "node:test";
9
9
  import assert from "node:assert/strict";
10
10
  import { promises as fsp } from "fs";
11
- import { mkdirSync, writeFileSync } from "node:fs";
11
+ import { mkdirSync, rmSync, writeFileSync } from "node:fs";
12
12
  import { tmpdir } from "os";
13
13
  import { join } from "path";
14
14
  import { fileURLToPath } from "node:url";
@@ -27,6 +27,9 @@ import {
27
27
  validateRoutingConfig,
28
28
  } from "./model-router.mjs";
29
29
  import { loadCatalog } from "./model-router/catalog.mjs";
30
+ // Read-only, to PROVE (not assume) that a provider the SDK does not bundle
31
+ // inherits credential pooling from the catalog rather than needing new code.
32
+ import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
30
33
 
31
34
  async function makeAgentRoot() {
32
35
  const path = join(
@@ -52,10 +55,15 @@ const ANTHROPIC_BACKEND = {
52
55
  base_url: "https://api.anthropic.com",
53
56
  capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
54
57
  pricing: { input_per_m: 3.0, output_per_m: 15.0 },
58
+ // A CONFIG-DECLARED model map. Deliberately NOT the same ids as
59
+ // ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
60
+ // resolveBackend serves, so pinning them to the built-in defaults would make
61
+ // the assertions vacuous (they would pass even if the config map were being
62
+ // ignored entirely). The built-in defaults get their own test below.
55
63
  models: {
56
64
  classifier: "claude-haiku-4-5-20251001",
57
- default: "claude-sonnet-4-6",
58
- premium: "claude-opus-4-7",
65
+ default: "config-declared-sonnet",
66
+ premium: "config-declared-opus",
59
67
  fast: "claude-haiku-4-5-20251001",
60
68
  },
61
69
  };
@@ -252,14 +260,59 @@ test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () =
252
260
  const opus = resolveBackend({ model_hint: "opus" }, { config });
253
261
  const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
254
262
  const haiku = resolveBackend({ model_hint: "haiku" }, { config });
255
- assert.equal(opus.model, "claude-opus-4-7");
256
- assert.equal(sonnet.model, "claude-sonnet-4-6");
263
+ assert.equal(opus.model, "config-declared-opus");
264
+ assert.equal(sonnet.model, "config-declared-sonnet");
257
265
  assert.equal(haiku.model, "claude-haiku-4-5-20251001");
258
266
  assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
259
267
  assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
260
268
  assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
261
269
  });
262
270
 
271
+ // ---------------------------------------------------------------------------
272
+ // Model freshness — the built-in Anthropic defaults (see the provenance block
273
+ // above ANTHROPIC_DEFAULT in model-router.mjs).
274
+ // ---------------------------------------------------------------------------
275
+
276
+ test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
277
+ // The floor an agent with no routing config lands on. A stale id here never
278
+ // reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
279
+ // DID reach resolved.model, which the cost ledger and telemetry store — so a
280
+ // stale id mis-prices and mis-attributes every unrouted session.
281
+ const models = defaultRoutingConfig().backends.anthropic.models;
282
+ assert.equal(models.premium, "claude-opus-5");
283
+ assert.equal(models.default, "claude-sonnet-5");
284
+ // The fast/classifier tier did NOT move in the 2026-08-13 refresh.
285
+ assert.equal(models.fast, "claude-haiku-4-5-20251001");
286
+ assert.equal(models.classifier, "claude-haiku-4-5-20251001");
287
+ // Guard the specific regression: no 4.x id survives anywhere in the map.
288
+ for (const [tier, id] of Object.entries(models)) {
289
+ assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
290
+ }
291
+ });
292
+
293
+ test("resolveBackend serves the current ids through the no-config default path", () => {
294
+ // End-to-end through the resolver, not just the constant: an agent with no
295
+ // config file (useDefault) resolves premium/default to the -5 pair.
296
+ const config = defaultRoutingConfig();
297
+ assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
298
+ assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
299
+ // …while the CLI flag stays version-agnostic, which is WHY the stale ids were
300
+ // survivable on the wire and only corrupted the ledger.
301
+ assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
302
+ });
303
+
304
+ test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
305
+ // Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
306
+ // agent adds backends.anthropic.models to config/model-routing.yaml and is
307
+ // current WITHOUT an SDK release. normaliseConfig only injects the built-in
308
+ // map when backends.anthropic is absent, so this must win outright.
309
+ const config = {
310
+ backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
311
+ routing_policy: [{ default: true, backend: "anthropic" }],
312
+ };
313
+ assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
314
+ });
315
+
263
316
  test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
264
317
  const config = {
265
318
  backends: { qwen: QWEN_BACKEND },
@@ -833,6 +886,253 @@ test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
833
886
  assert.equal(d.audit.fallback_reason, "no_v2_config");
834
887
  });
835
888
 
889
+ // ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
890
+ //
891
+ // The safety net fires on the three paths that have no chain to walk: the kill
892
+ // switch, "no v2 config", and "nothing survived the gates". It is the ONE place
893
+ // resolve.mjs still names Anthropic model ids, so it is the one place that can
894
+ // go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
895
+ // preference list carries the current -5 ids AND the 4-x succession fallback.
896
+
897
+ /** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
898
+ const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
899
+ agentConfig: {
900
+ catalog_overrides: [
901
+ {
902
+ ref: "anthropic/claude-opus-5",
903
+ status: "available",
904
+ context_tokens: 950000,
905
+ max_tokens: 128000,
906
+ tool_reliability: "A",
907
+ harness: { session: true, direct: true, batch: true },
908
+ cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
909
+ },
910
+ {
911
+ ref: "anthropic/claude-sonnet-5",
912
+ status: "available",
913
+ context_tokens: 950000,
914
+ max_tokens: 64000,
915
+ tool_reliability: "A",
916
+ harness: { session: true, direct: true, batch: true },
917
+ cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
918
+ },
919
+ ],
920
+ },
921
+ });
922
+
923
+ test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
924
+ const frontier = resolveChain(
925
+ { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
926
+ { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
927
+ );
928
+ assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
929
+
930
+ const workhorse = resolveChain(
931
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
932
+ { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
933
+ );
934
+ assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
935
+ });
936
+
937
+ test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
938
+ // This is why the 4-x refs stay in the preference list. Delete them and this
939
+ // does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
940
+ // i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
941
+ // ORDINARY work the frontier model and quietly quadruple its cost.
942
+ const d = resolveChain(
943
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
944
+ { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
945
+ );
946
+ assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
947
+ // Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
948
+ // the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
949
+ });
950
+
951
+ test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
952
+ // THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
953
+ // finished. The test above asserts `chosen.model`, which reads as bookkeeping.
954
+ // This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
955
+ // `--model` — against the catalog the SDK actually ships, with `config: null`,
956
+ // which is the DEFAULT state of every agent because this repo contains no
957
+ // config/model-routing.yaml at all.
958
+ //
959
+ // Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
960
+ // agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
961
+ // row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
962
+ // merely a mis-stamped ledger entry.
963
+ //
964
+ // WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
965
+ // the missing step. Update the expectations here, and DELETE the 4-x refs from
966
+ // pickAnthropicRow's preference lists in the same change, or the museum the
967
+ // provenance block warns about starts accumulating.
968
+ const cases = [
969
+ ["session.responder", {}, "claude-sonnet-4-6"],
970
+ ["decision", { needs_thinking: true }, "claude-opus-4-8"],
971
+ ];
972
+ for (const [task_class, extra, expected] of cases) {
973
+ const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
974
+ for (const [label, env] of [
975
+ ["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
976
+ ["no v2 config", {}],
977
+ ]) {
978
+ const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
979
+ assert.equal(
980
+ d.spawnArgs.modelFlag,
981
+ expected,
982
+ `${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
983
+ );
984
+ // Not the shorthand: nothing downstream re-resolves this to a current build.
985
+ assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
986
+ }
987
+ }
988
+ });
989
+
990
+ test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
991
+ // The last-ditch path: no catalog, so the id goes straight onto `claude
992
+ // --model`. It used to emit {model: "claude-opus-4-8", ref:
993
+ // "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
994
+ // feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
995
+ // both named a different model than the one that ran.
996
+ const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
997
+ mkdirSync(emptyDir, { recursive: true });
998
+ try {
999
+ const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
1000
+ assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
1001
+
1002
+ const d = resolveChain(
1003
+ { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
1004
+ { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1005
+ );
1006
+ assert.equal(d.chosen.model, "claude-opus-5");
1007
+ assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
1008
+ assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
1009
+
1010
+ const ordinary = resolveChain(
1011
+ { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
1012
+ { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1013
+ );
1014
+ assert.equal(ordinary.chosen.model, "claude-sonnet-5");
1015
+ assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
1016
+ } finally {
1017
+ rmSync(emptyDir, { recursive: true, force: true });
1018
+ }
1019
+ });
1020
+
1021
+ // ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
1022
+ //
1023
+ // The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
1024
+ // is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
1025
+ // highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
1026
+ // introduce a whole provider, and resolveChain gates it exactly like a bundled
1027
+ // one. These tests prove the full path — row → alias → chain → credential gate
1028
+ // → chosen — so the remaining work is a bundled YAML file, not plumbing.
1029
+
1030
+ const XAI_PROVIDER_DOC = {
1031
+ provider: "xai",
1032
+ auth_env: "XAI_API_KEY",
1033
+ endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
1034
+ data_residency: "us",
1035
+ models: [
1036
+ {
1037
+ id: "grok-4-fast",
1038
+ status: "available",
1039
+ context_window: 2000000,
1040
+ context_tokens: 1900000,
1041
+ max_tokens: 30000,
1042
+ cost: { input: 0.2, output: 0.5 },
1043
+ cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
1044
+ compat: { thinking_format: "openai", cache_control: "implicit" },
1045
+ // Grade B like every other third-party row: good enough for the direct
1046
+ // lane, structurally barred from hosting a tool-using session (§6.6).
1047
+ tool_reliability: "B",
1048
+ harness: { session: true, direct: true, batch: false },
1049
+ },
1050
+ ],
1051
+ };
1052
+
1053
+ const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
1054
+ agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
1055
+ });
1056
+
1057
+ const XAI_CONFIG = Object.freeze({
1058
+ schema_version: 2,
1059
+ aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
1060
+ defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
1061
+ backends: {
1062
+ anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
1063
+ xai: { allowed_data_classes: ["public"] },
1064
+ },
1065
+ routing_policy: [
1066
+ { match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
1067
+ { default: true, chain: ["default"] },
1068
+ ],
1069
+ fallback_to_anthropic: true,
1070
+ });
1071
+
1072
+ test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
1073
+ const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
1074
+ assert.ok(row, "grok row present after the agent-config merge");
1075
+ assert.equal(row.provider, "xai");
1076
+ assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
1077
+ assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
1078
+ assert.ok(
1079
+ CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
1080
+ "introducing an unbundled provider warns (typo guard) — it does not fail"
1081
+ );
1082
+ });
1083
+
1084
+ test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
1085
+ const d = resolveChain(
1086
+ { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1087
+ { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1088
+ );
1089
+ assert.equal(d.chosen.provider, "xai");
1090
+ assert.equal(d.chosen.model, "grok-4-fast");
1091
+ assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
1092
+ });
1093
+
1094
+ test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
1095
+ const d = resolveChain(
1096
+ { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1097
+ { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
1098
+ );
1099
+ const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1100
+ assert.ok(skipped, "grok recorded in tried[]");
1101
+ assert.equal(skipped.reason, "missing_credential");
1102
+ assert.notEqual(d.chosen.provider, "xai");
1103
+ });
1104
+
1105
+ test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
1106
+ // auth-profiles derives provider → auth_env from the CATALOG, so a provider
1107
+ // the SDK has never heard of gets pooled multi-key rotation for free. This is
1108
+ // the "their own auth profiles" half of the cheap-tier requirement.
1109
+ const profiles = loadAuthProfiles(null, {
1110
+ configDoc: null, // bypass the FS
1111
+ env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
1112
+ authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
1113
+ });
1114
+ assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
1115
+ assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
1116
+ assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
1117
+ });
1118
+
1119
+ test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
1120
+ // Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
1121
+ // third-party row is graded A until the probe ledger earns it. Cheap-tier
1122
+ // reachability must NOT quietly become cheap-tier session hosting.
1123
+ const sessionCfg = {
1124
+ ...XAI_CONFIG,
1125
+ routing_policy: [{ default: true, chain: ["cheap", "default"] }],
1126
+ };
1127
+ const d = resolveChain(
1128
+ { task_class: "session.responder", data_class: "public", token_estimate: 2000 },
1129
+ { catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1130
+ );
1131
+ const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1132
+ assert.equal(skipped?.reason, "grade_too_low");
1133
+ assert.equal(d.chosen.provider, "anthropic");
1134
+ });
1135
+
836
1136
  // ── estCostUSD + ledger join fields ──────────────────────────────────────────
837
1137
 
838
1138
  test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
@@ -109,9 +109,18 @@ export function isEnabled(cfg) {
109
109
  * Distil the org client config from an agent config object. Resolves:
110
110
  * - base: the server origin for the /v1 binding (explicit `org.cohort.base`,
111
111
  * else derived from `apiUrl` [+ `orgId` path] for back-compat).
112
- * - orgId: the org slug — config `org.cohort.orgId` else COHORT_ORG_ID env
112
+ * - orgId: the org's ID — config `org.cohort.orgId` else COHORT_ORG_ID env
113
113
  * (hq's documented var). Used by the legacy directory-path derivation
114
114
  * and, when known, sent as the `x-org-id` pin header (defence-in-depth).
115
+ *
116
+ * NOT THE SLUG, though this said "the org slug" until 2026-08-21 and
117
+ * `docs/guides/mac-mini.md` told operators to export one. hq compares
118
+ * the pin with STRICT EQUALITY against the key's `Org.id`
119
+ * (`src/server/auth/org-api-key.ts:163`) — no slug fallback — so a
120
+ * slug here 401s every call. The two are not interchangeable: the
121
+ * ID has the form `org_default_<slug>` (e.g. slug `acme` → ID
122
+ * `org_default_acme`), never the bare slug. Note `COHORT_AGENT_ID`
123
+ * beside it genuinely IS a slug.
115
124
  * - token: bearer credential — the RAW OrgApiKey. Resolution order:
116
125
  * config `token` → COHORT_API_TOKEN → COHORT_TOKEN → COHORT_API_KEY.
117
126
  * hq's auth doc names the agent var COHORT_API_KEY; the older
@@ -1314,6 +1323,54 @@ export function messagingSend(params, o = {}) {
1314
1323
  return call("messaging.send", params || {}, o);
1315
1324
  }
1316
1325
 
1326
+ /**
1327
+ * Ask hq who — if anyone — should respond to a message in a channel
1328
+ * (messaging.electResponder). This is the SHOULD-I-RESPOND decision, moved off
1329
+ * the daemon and onto hq's `routeResponders`, which for an undirected group
1330
+ * message elects exactly ONE responder or NONE, and already fails safe (no model
1331
+ * → quiet). hq derives the roster + trigger server-side from just
1332
+ * `{messageId, channelId}`, org-scoped, computes the verdict ONCE per messageId
1333
+ * and caches it in a durable ledger event, so concurrent daemons read the same
1334
+ * election (the loser of a race reads the winner).
1335
+ *
1336
+ * Read-style: agent-tier `messaging.read` scope (any paired seat may ask),
1337
+ * `sideEffecting:false` — no client idempotency key, and the server owns the
1338
+ * write-once ledger append. Returns a res frame whose `result` is
1339
+ * `{ election, mode, responders:[{slug,dimension}], deferredToHuman }`.
1340
+ *
1341
+ * FAIL-OPEN AT THE TRANSPORT ONLY: an unreachable/disabled server yields an
1342
+ * error frame, never a throw. The CALLER owns the fail-SAFE inversion — for an
1343
+ * ambient channel, "no verdict / not me" means STAY SILENT, never respond-to-all.
1344
+ *
1345
+ * @param {{messageId:string, channelId:string}} params
1346
+ * @param {object} [o] { base, token, orgId?, fetchImpl? }
1347
+ */
1348
+ export function electResponder(params, o = {}) {
1349
+ return call("messaging.electResponder", params || {}, o);
1350
+ }
1351
+
1352
+ /**
1353
+ * Record a voice note in THIS seat's own designed voice
1354
+ * (messaging.synthesizeVoiceNote) and return the READY `{fileId, durationMs,
1355
+ * sizeBytes}` to attach.
1356
+ *
1357
+ * This is the RECORD half only. Posting is an ordinary `messagingSend` with
1358
+ * `attachments:[{fileId}]` — hq deliberately keeps the two apart because
1359
+ * synthesis is a provider round-trip that cannot ride its transactional send
1360
+ * lane, and because a spoken message must be an ordinary message with a file on
1361
+ * it rather than a second send path with its own governance.
1362
+ *
1363
+ * Refusals come back as error frames naming the cause (no synthesis key on the
1364
+ * deployment, this seat has no designed voice, the hourly recording cap). Relay
1365
+ * the cause — never claim a voice note was sent when the frame was not ok.
1366
+ *
1367
+ * `messaging_send_voice_note` (tool-surface) composes both halves into one verb.
1368
+ * @param {{text:string}} params
1369
+ */
1370
+ export function messagingSynthesizeVoiceNote(params, o = {}) {
1371
+ return call("messaging.synthesizeVoiceNote", params || {}, o);
1372
+ }
1373
+
1317
1374
  /** React to an org message (messaging.react). */
1318
1375
  export function messagingReact(params, o = {}) {
1319
1376
  return call("messaging.react", params || {}, o);
@@ -102,11 +102,15 @@ export function toMessageEvent(o = {}) {
102
102
  kind: def.kind,
103
103
  subject,
104
104
  // ONLY ever a real Cohort channel id, never an entity id standing in for
105
- // one. `scripts/daemon/responder.mjs sendCohortReply` routes its reply with
106
- // `messaging.send({channel: item.channel_id})`, so putting a task/decision
107
- // id here would post-to-a-non-channel and 400. A surface with no room leaves
108
- // this blank and is routed by `kind` + `raw_ref` instead (see the wiring
109
- // note in this directory's module docs).
105
+ // one. A reply to a room surface routes with
106
+ // `messaging.send({channelId: item.channel_id})`, so putting a task/decision
107
+ // id here would post-to-a-non-channel and come back NOT_FOUND. A surface
108
+ // with no room leaves this BLANK and is routed on its own surface instead —
109
+ // `scripts/daemon/deliver.mjs cohortReplyRoute` reads `raw_ref` for the
110
+ // surface and `scope_id`/`thread_id` for the entity, and sends a board
111
+ // comment to `board.addTaskComment`, a doc comment to `files.commentAdd`,
112
+ // and so on. That enforcement is new: for months the send path fell back to
113
+ // `channel_label` below, so a board reply was addressed to a task's TITLE.
110
114
  channel_id: channelId,
111
115
  channel_label: channelLabel,
112
116
  sender: s(from.name || from.id),
@@ -138,8 +138,13 @@ function agentIdOf(cfg) {
138
138
  *
139
139
  * @param {object} o - { cfg, agentRoot, action, channel }
140
140
  * @param {object} deps - { ledgerImpl?, catalog? } injectable for tests
141
+ *
142
+ * EXPORTED because room sends are no longer the only attributed outbound:
143
+ * `scripts/daemon/deliver.mjs` posts board/doc/decision comments straight onto
144
+ * their entity thread, bypassing `sendMessage`, and an outbound family missing
145
+ * from the ledger is an outbound family nobody can see.
141
146
  */
142
- async function attributeMessaging(o, deps = {}) {
147
+ export async function attributeMessaging(o, deps = {}) {
143
148
  // Opt-out: an explicit `ledgerImpl: null` (or false) disables attribution.
144
149
  if (Object.prototype.hasOwnProperty.call(deps, "ledgerImpl") && !deps.ledgerImpl) return;
145
150
  try {
@@ -1 +1 @@
1
- 358b95ed6a48faeffa78b7ca717a909905e8ff4e7b75278edf274a0acdaf5dd6
1
+ b2cbf124c99800a0ad1b19d7aea6a88c3c72f07b53838eb44ef6f82d888fda43