@cohortapp/agent-sdk 2.9.1 → 2.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/channels/inbox-item.mjs +4 -0
- package/lib/comms/send-gate.mjs +23 -1
- package/lib/comms/send-gate.test.mjs +24 -0
- package/lib/model-router/economics.mjs +53 -1
- package/lib/model-router/economics.test.mjs +76 -0
- package/lib/model-router/resolve.mjs +57 -4
- package/lib/model-router.mjs +95 -8
- package/lib/model-router.test.mjs +305 -5
- package/lib/org/inbound/project.mjs +9 -5
- package/lib/org/messaging.mjs +6 -1
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +78 -3
- package/lib/org/protocol.test.mjs +14 -2
- package/package.json +1 -1
- package/scripts/daemon/assurance.mjs +12 -1
- package/scripts/daemon/deliver.mjs +457 -33
- package/scripts/daemon/deliver.test.mjs +564 -0
- package/scripts/daemon/responder-history.test.mjs +18 -2
- package/scripts/daemon/responder.mjs +7 -1
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
import { test } from "node:test";
|
|
9
9
|
import assert from "node:assert/strict";
|
|
10
10
|
import { promises as fsp } from "fs";
|
|
11
|
-
import { mkdirSync, writeFileSync } from "node:fs";
|
|
11
|
+
import { mkdirSync, rmSync, writeFileSync } from "node:fs";
|
|
12
12
|
import { tmpdir } from "os";
|
|
13
13
|
import { join } from "path";
|
|
14
14
|
import { fileURLToPath } from "node:url";
|
|
@@ -27,6 +27,9 @@ import {
|
|
|
27
27
|
validateRoutingConfig,
|
|
28
28
|
} from "./model-router.mjs";
|
|
29
29
|
import { loadCatalog } from "./model-router/catalog.mjs";
|
|
30
|
+
// Read-only, to PROVE (not assume) that a provider the SDK does not bundle
|
|
31
|
+
// inherits credential pooling from the catalog rather than needing new code.
|
|
32
|
+
import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
|
|
30
33
|
|
|
31
34
|
async function makeAgentRoot() {
|
|
32
35
|
const path = join(
|
|
@@ -52,10 +55,15 @@ const ANTHROPIC_BACKEND = {
|
|
|
52
55
|
base_url: "https://api.anthropic.com",
|
|
53
56
|
capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
|
|
54
57
|
pricing: { input_per_m: 3.0, output_per_m: 15.0 },
|
|
58
|
+
// A CONFIG-DECLARED model map. Deliberately NOT the same ids as
|
|
59
|
+
// ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
|
|
60
|
+
// resolveBackend serves, so pinning them to the built-in defaults would make
|
|
61
|
+
// the assertions vacuous (they would pass even if the config map were being
|
|
62
|
+
// ignored entirely). The built-in defaults get their own test below.
|
|
55
63
|
models: {
|
|
56
64
|
classifier: "claude-haiku-4-5-20251001",
|
|
57
|
-
default: "
|
|
58
|
-
premium: "
|
|
65
|
+
default: "config-declared-sonnet",
|
|
66
|
+
premium: "config-declared-opus",
|
|
59
67
|
fast: "claude-haiku-4-5-20251001",
|
|
60
68
|
},
|
|
61
69
|
};
|
|
@@ -252,14 +260,59 @@ test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () =
|
|
|
252
260
|
const opus = resolveBackend({ model_hint: "opus" }, { config });
|
|
253
261
|
const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
|
|
254
262
|
const haiku = resolveBackend({ model_hint: "haiku" }, { config });
|
|
255
|
-
assert.equal(opus.model, "
|
|
256
|
-
assert.equal(sonnet.model, "
|
|
263
|
+
assert.equal(opus.model, "config-declared-opus");
|
|
264
|
+
assert.equal(sonnet.model, "config-declared-sonnet");
|
|
257
265
|
assert.equal(haiku.model, "claude-haiku-4-5-20251001");
|
|
258
266
|
assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
|
|
259
267
|
assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
|
|
260
268
|
assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
|
|
261
269
|
});
|
|
262
270
|
|
|
271
|
+
// ---------------------------------------------------------------------------
|
|
272
|
+
// Model freshness — the built-in Anthropic defaults (see the provenance block
|
|
273
|
+
// above ANTHROPIC_DEFAULT in model-router.mjs).
|
|
274
|
+
// ---------------------------------------------------------------------------
|
|
275
|
+
|
|
276
|
+
test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
|
|
277
|
+
// The floor an agent with no routing config lands on. A stale id here never
|
|
278
|
+
// reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
|
|
279
|
+
// DID reach resolved.model, which the cost ledger and telemetry store — so a
|
|
280
|
+
// stale id mis-prices and mis-attributes every unrouted session.
|
|
281
|
+
const models = defaultRoutingConfig().backends.anthropic.models;
|
|
282
|
+
assert.equal(models.premium, "claude-opus-5");
|
|
283
|
+
assert.equal(models.default, "claude-sonnet-5");
|
|
284
|
+
// The fast/classifier tier did NOT move in the 2026-08-13 refresh.
|
|
285
|
+
assert.equal(models.fast, "claude-haiku-4-5-20251001");
|
|
286
|
+
assert.equal(models.classifier, "claude-haiku-4-5-20251001");
|
|
287
|
+
// Guard the specific regression: no 4.x id survives anywhere in the map.
|
|
288
|
+
for (const [tier, id] of Object.entries(models)) {
|
|
289
|
+
assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
|
|
290
|
+
}
|
|
291
|
+
});
|
|
292
|
+
|
|
293
|
+
test("resolveBackend serves the current ids through the no-config default path", () => {
|
|
294
|
+
// End-to-end through the resolver, not just the constant: an agent with no
|
|
295
|
+
// config file (useDefault) resolves premium/default to the -5 pair.
|
|
296
|
+
const config = defaultRoutingConfig();
|
|
297
|
+
assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
|
|
298
|
+
assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
|
|
299
|
+
// …while the CLI flag stays version-agnostic, which is WHY the stale ids were
|
|
300
|
+
// survivable on the wire and only corrupted the ledger.
|
|
301
|
+
assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
|
|
302
|
+
});
|
|
303
|
+
|
|
304
|
+
test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
|
|
305
|
+
// Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
|
|
306
|
+
// agent adds backends.anthropic.models to config/model-routing.yaml and is
|
|
307
|
+
// current WITHOUT an SDK release. normaliseConfig only injects the built-in
|
|
308
|
+
// map when backends.anthropic is absent, so this must win outright.
|
|
309
|
+
const config = {
|
|
310
|
+
backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
|
|
311
|
+
routing_policy: [{ default: true, backend: "anthropic" }],
|
|
312
|
+
};
|
|
313
|
+
assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
|
|
314
|
+
});
|
|
315
|
+
|
|
263
316
|
test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
|
|
264
317
|
const config = {
|
|
265
318
|
backends: { qwen: QWEN_BACKEND },
|
|
@@ -833,6 +886,253 @@ test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
|
|
|
833
886
|
assert.equal(d.audit.fallback_reason, "no_v2_config");
|
|
834
887
|
});
|
|
835
888
|
|
|
889
|
+
// ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
|
|
890
|
+
//
|
|
891
|
+
// The safety net fires on the three paths that have no chain to walk: the kill
|
|
892
|
+
// switch, "no v2 config", and "nothing survived the gates". It is the ONE place
|
|
893
|
+
// resolve.mjs still names Anthropic model ids, so it is the one place that can
|
|
894
|
+
// go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
|
|
895
|
+
// preference list carries the current -5 ids AND the 4-x succession fallback.
|
|
896
|
+
|
|
897
|
+
/** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
|
|
898
|
+
const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
899
|
+
agentConfig: {
|
|
900
|
+
catalog_overrides: [
|
|
901
|
+
{
|
|
902
|
+
ref: "anthropic/claude-opus-5",
|
|
903
|
+
status: "available",
|
|
904
|
+
context_tokens: 950000,
|
|
905
|
+
max_tokens: 128000,
|
|
906
|
+
tool_reliability: "A",
|
|
907
|
+
harness: { session: true, direct: true, batch: true },
|
|
908
|
+
cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
|
|
909
|
+
},
|
|
910
|
+
{
|
|
911
|
+
ref: "anthropic/claude-sonnet-5",
|
|
912
|
+
status: "available",
|
|
913
|
+
context_tokens: 950000,
|
|
914
|
+
max_tokens: 64000,
|
|
915
|
+
tool_reliability: "A",
|
|
916
|
+
harness: { session: true, direct: true, batch: true },
|
|
917
|
+
cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
|
|
918
|
+
},
|
|
919
|
+
],
|
|
920
|
+
},
|
|
921
|
+
});
|
|
922
|
+
|
|
923
|
+
test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
|
|
924
|
+
const frontier = resolveChain(
|
|
925
|
+
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
926
|
+
{ catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
927
|
+
);
|
|
928
|
+
assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
|
|
929
|
+
|
|
930
|
+
const workhorse = resolveChain(
|
|
931
|
+
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
932
|
+
{ catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
933
|
+
);
|
|
934
|
+
assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
|
|
935
|
+
});
|
|
936
|
+
|
|
937
|
+
test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
|
|
938
|
+
// This is why the 4-x refs stay in the preference list. Delete them and this
|
|
939
|
+
// does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
|
|
940
|
+
// i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
|
|
941
|
+
// ORDINARY work the frontier model and quietly quadruple its cost.
|
|
942
|
+
const d = resolveChain(
|
|
943
|
+
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
944
|
+
{ catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
945
|
+
);
|
|
946
|
+
assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
|
|
947
|
+
// Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
|
|
948
|
+
// the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
|
|
949
|
+
});
|
|
950
|
+
|
|
951
|
+
test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
|
|
952
|
+
// THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
|
|
953
|
+
// finished. The test above asserts `chosen.model`, which reads as bookkeeping.
|
|
954
|
+
// This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
|
|
955
|
+
// `--model` — against the catalog the SDK actually ships, with `config: null`,
|
|
956
|
+
// which is the DEFAULT state of every agent because this repo contains no
|
|
957
|
+
// config/model-routing.yaml at all.
|
|
958
|
+
//
|
|
959
|
+
// Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
|
|
960
|
+
// agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
|
|
961
|
+
// row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
|
|
962
|
+
// merely a mis-stamped ledger entry.
|
|
963
|
+
//
|
|
964
|
+
// WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
|
|
965
|
+
// the missing step. Update the expectations here, and DELETE the 4-x refs from
|
|
966
|
+
// pickAnthropicRow's preference lists in the same change, or the museum the
|
|
967
|
+
// provenance block warns about starts accumulating.
|
|
968
|
+
const cases = [
|
|
969
|
+
["session.responder", {}, "claude-sonnet-4-6"],
|
|
970
|
+
["decision", { needs_thinking: true }, "claude-opus-4-8"],
|
|
971
|
+
];
|
|
972
|
+
for (const [task_class, extra, expected] of cases) {
|
|
973
|
+
const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
|
|
974
|
+
for (const [label, env] of [
|
|
975
|
+
["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
|
|
976
|
+
["no v2 config", {}],
|
|
977
|
+
]) {
|
|
978
|
+
const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
|
|
979
|
+
assert.equal(
|
|
980
|
+
d.spawnArgs.modelFlag,
|
|
981
|
+
expected,
|
|
982
|
+
`${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
|
|
983
|
+
);
|
|
984
|
+
// Not the shorthand: nothing downstream re-resolves this to a current build.
|
|
985
|
+
assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
|
|
986
|
+
}
|
|
987
|
+
}
|
|
988
|
+
});
|
|
989
|
+
|
|
990
|
+
test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
|
|
991
|
+
// The last-ditch path: no catalog, so the id goes straight onto `claude
|
|
992
|
+
// --model`. It used to emit {model: "claude-opus-4-8", ref:
|
|
993
|
+
// "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
|
|
994
|
+
// feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
|
|
995
|
+
// both named a different model than the one that ran.
|
|
996
|
+
const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
|
|
997
|
+
mkdirSync(emptyDir, { recursive: true });
|
|
998
|
+
try {
|
|
999
|
+
const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
|
|
1000
|
+
assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
|
|
1001
|
+
|
|
1002
|
+
const d = resolveChain(
|
|
1003
|
+
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
1004
|
+
{ catalog: EMPTY, config: null, env: {}, now: CLOCK }
|
|
1005
|
+
);
|
|
1006
|
+
assert.equal(d.chosen.model, "claude-opus-5");
|
|
1007
|
+
assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
|
|
1008
|
+
assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
|
|
1009
|
+
|
|
1010
|
+
const ordinary = resolveChain(
|
|
1011
|
+
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
1012
|
+
{ catalog: EMPTY, config: null, env: {}, now: CLOCK }
|
|
1013
|
+
);
|
|
1014
|
+
assert.equal(ordinary.chosen.model, "claude-sonnet-5");
|
|
1015
|
+
assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
|
|
1016
|
+
} finally {
|
|
1017
|
+
rmSync(emptyDir, { recursive: true, force: true });
|
|
1018
|
+
}
|
|
1019
|
+
});
|
|
1020
|
+
|
|
1021
|
+
// ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
|
|
1022
|
+
//
|
|
1023
|
+
// The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
|
|
1024
|
+
// is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
|
|
1025
|
+
// highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
|
|
1026
|
+
// introduce a whole provider, and resolveChain gates it exactly like a bundled
|
|
1027
|
+
// one. These tests prove the full path — row → alias → chain → credential gate
|
|
1028
|
+
// → chosen — so the remaining work is a bundled YAML file, not plumbing.
|
|
1029
|
+
|
|
1030
|
+
const XAI_PROVIDER_DOC = {
|
|
1031
|
+
provider: "xai",
|
|
1032
|
+
auth_env: "XAI_API_KEY",
|
|
1033
|
+
endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
|
|
1034
|
+
data_residency: "us",
|
|
1035
|
+
models: [
|
|
1036
|
+
{
|
|
1037
|
+
id: "grok-4-fast",
|
|
1038
|
+
status: "available",
|
|
1039
|
+
context_window: 2000000,
|
|
1040
|
+
context_tokens: 1900000,
|
|
1041
|
+
max_tokens: 30000,
|
|
1042
|
+
cost: { input: 0.2, output: 0.5 },
|
|
1043
|
+
cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
|
|
1044
|
+
compat: { thinking_format: "openai", cache_control: "implicit" },
|
|
1045
|
+
// Grade B like every other third-party row: good enough for the direct
|
|
1046
|
+
// lane, structurally barred from hosting a tool-using session (§6.6).
|
|
1047
|
+
tool_reliability: "B",
|
|
1048
|
+
harness: { session: true, direct: true, batch: false },
|
|
1049
|
+
},
|
|
1050
|
+
],
|
|
1051
|
+
};
|
|
1052
|
+
|
|
1053
|
+
const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
1054
|
+
agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
|
|
1055
|
+
});
|
|
1056
|
+
|
|
1057
|
+
const XAI_CONFIG = Object.freeze({
|
|
1058
|
+
schema_version: 2,
|
|
1059
|
+
aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
|
|
1060
|
+
defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
|
|
1061
|
+
backends: {
|
|
1062
|
+
anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
|
|
1063
|
+
xai: { allowed_data_classes: ["public"] },
|
|
1064
|
+
},
|
|
1065
|
+
routing_policy: [
|
|
1066
|
+
{ match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
|
|
1067
|
+
{ default: true, chain: ["default"] },
|
|
1068
|
+
],
|
|
1069
|
+
fallback_to_anthropic: true,
|
|
1070
|
+
});
|
|
1071
|
+
|
|
1072
|
+
test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
|
|
1073
|
+
const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
|
|
1074
|
+
assert.ok(row, "grok row present after the agent-config merge");
|
|
1075
|
+
assert.equal(row.provider, "xai");
|
|
1076
|
+
assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
|
|
1077
|
+
assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
|
|
1078
|
+
assert.ok(
|
|
1079
|
+
CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
|
|
1080
|
+
"introducing an unbundled provider warns (typo guard) — it does not fail"
|
|
1081
|
+
);
|
|
1082
|
+
});
|
|
1083
|
+
|
|
1084
|
+
test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
|
|
1085
|
+
const d = resolveChain(
|
|
1086
|
+
{ task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
|
|
1087
|
+
{ catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
|
|
1088
|
+
);
|
|
1089
|
+
assert.equal(d.chosen.provider, "xai");
|
|
1090
|
+
assert.equal(d.chosen.model, "grok-4-fast");
|
|
1091
|
+
assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
|
|
1092
|
+
});
|
|
1093
|
+
|
|
1094
|
+
test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
|
|
1095
|
+
const d = resolveChain(
|
|
1096
|
+
{ task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
|
|
1097
|
+
{ catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
|
|
1098
|
+
);
|
|
1099
|
+
const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
|
|
1100
|
+
assert.ok(skipped, "grok recorded in tried[]");
|
|
1101
|
+
assert.equal(skipped.reason, "missing_credential");
|
|
1102
|
+
assert.notEqual(d.chosen.provider, "xai");
|
|
1103
|
+
});
|
|
1104
|
+
|
|
1105
|
+
test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
|
|
1106
|
+
// auth-profiles derives provider → auth_env from the CATALOG, so a provider
|
|
1107
|
+
// the SDK has never heard of gets pooled multi-key rotation for free. This is
|
|
1108
|
+
// the "their own auth profiles" half of the cheap-tier requirement.
|
|
1109
|
+
const profiles = loadAuthProfiles(null, {
|
|
1110
|
+
configDoc: null, // bypass the FS
|
|
1111
|
+
env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
|
|
1112
|
+
authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
|
|
1113
|
+
});
|
|
1114
|
+
assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
|
|
1115
|
+
assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
|
|
1116
|
+
assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
|
|
1117
|
+
});
|
|
1118
|
+
|
|
1119
|
+
test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
|
|
1120
|
+
// Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
|
|
1121
|
+
// third-party row is graded A until the probe ledger earns it. Cheap-tier
|
|
1122
|
+
// reachability must NOT quietly become cheap-tier session hosting.
|
|
1123
|
+
const sessionCfg = {
|
|
1124
|
+
...XAI_CONFIG,
|
|
1125
|
+
routing_policy: [{ default: true, chain: ["cheap", "default"] }],
|
|
1126
|
+
};
|
|
1127
|
+
const d = resolveChain(
|
|
1128
|
+
{ task_class: "session.responder", data_class: "public", token_estimate: 2000 },
|
|
1129
|
+
{ catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
|
|
1130
|
+
);
|
|
1131
|
+
const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
|
|
1132
|
+
assert.equal(skipped?.reason, "grade_too_low");
|
|
1133
|
+
assert.equal(d.chosen.provider, "anthropic");
|
|
1134
|
+
});
|
|
1135
|
+
|
|
836
1136
|
// ── estCostUSD + ledger join fields ──────────────────────────────────────────
|
|
837
1137
|
|
|
838
1138
|
test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
|
|
@@ -102,11 +102,15 @@ export function toMessageEvent(o = {}) {
|
|
|
102
102
|
kind: def.kind,
|
|
103
103
|
subject,
|
|
104
104
|
// ONLY ever a real Cohort channel id, never an entity id standing in for
|
|
105
|
-
// one.
|
|
106
|
-
// `messaging.send({
|
|
107
|
-
// id here would post-to-a-non-channel and
|
|
108
|
-
// this
|
|
109
|
-
//
|
|
105
|
+
// one. A reply to a room surface routes with
|
|
106
|
+
// `messaging.send({channelId: item.channel_id})`, so putting a task/decision
|
|
107
|
+
// id here would post-to-a-non-channel and come back NOT_FOUND. A surface
|
|
108
|
+
// with no room leaves this BLANK and is routed on its own surface instead —
|
|
109
|
+
// `scripts/daemon/deliver.mjs cohortReplyRoute` reads `raw_ref` for the
|
|
110
|
+
// surface and `scope_id`/`thread_id` for the entity, and sends a board
|
|
111
|
+
// comment to `board.addTaskComment`, a doc comment to `files.commentAdd`,
|
|
112
|
+
// and so on. That enforcement is new: for months the send path fell back to
|
|
113
|
+
// `channel_label` below, so a board reply was addressed to a task's TITLE.
|
|
110
114
|
channel_id: channelId,
|
|
111
115
|
channel_label: channelLabel,
|
|
112
116
|
sender: s(from.name || from.id),
|
package/lib/org/messaging.mjs
CHANGED
|
@@ -138,8 +138,13 @@ function agentIdOf(cfg) {
|
|
|
138
138
|
*
|
|
139
139
|
* @param {object} o - { cfg, agentRoot, action, channel }
|
|
140
140
|
* @param {object} deps - { ledgerImpl?, catalog? } injectable for tests
|
|
141
|
+
*
|
|
142
|
+
* EXPORTED because room sends are no longer the only attributed outbound:
|
|
143
|
+
* `scripts/daemon/deliver.mjs` posts board/doc/decision comments straight onto
|
|
144
|
+
* their entity thread, bypassing `sendMessage`, and an outbound family missing
|
|
145
|
+
* from the ledger is an outbound family nobody can see.
|
|
141
146
|
*/
|
|
142
|
-
async function attributeMessaging(o, deps = {}) {
|
|
147
|
+
export async function attributeMessaging(o, deps = {}) {
|
|
143
148
|
// Opt-out: an explicit `ledgerImpl: null` (or false) disables attribution.
|
|
144
149
|
if (Object.prototype.hasOwnProperty.call(deps, "ledgerImpl") && !deps.ledgerImpl) return;
|
|
145
150
|
try {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
9ecd8993a6a9aceac04da479c6626a3b3852378b81bfd2b7590d2eee2099c9a9
|
package/lib/org/protocol.mjs
CHANGED
|
@@ -76,6 +76,25 @@ export const FAMILIES = Object.freeze([
|
|
|
76
76
|
// seat can PERSIST one it learned rather than re-deriving it every session.
|
|
77
77
|
// The chain rows carry subject + domain + key only — never the value.
|
|
78
78
|
"preference",
|
|
79
|
+
// GENESIS (2026-08): the founder's pre-commit workspace draft — the ten-stage
|
|
80
|
+
// run, its editable sections, and the conversational planner over them. The
|
|
81
|
+
// family exists so an agent reaches the SAME draft the founder's own form
|
|
82
|
+
// does; every method is a thin shell over the server action behind the
|
|
83
|
+
// button, never a second implementation of its gates.
|
|
84
|
+
//
|
|
85
|
+
// Its chain rows land in the DRAFTING org — the tenant the founder is signed
|
|
86
|
+
// into while they draft — and say only that a mutation happened, by whom, to
|
|
87
|
+
// which section. The content-level provenance is a different spine: the
|
|
88
|
+
// action's own `GenesisFieldEdit` rows, which `bufferGenesisAudit` replays
|
|
89
|
+
// into the NEW workspace's chain at commit. Both are true and neither
|
|
90
|
+
// replaces the other.
|
|
91
|
+
"genesis",
|
|
92
|
+
// ARCHITECTURE (2026-08): the four tier review decisions and the one whole-
|
|
93
|
+
// architecture approval that gate a Genesis commit. Split from `genesis`
|
|
94
|
+
// because these are REVIEW acts over a draft rather than edits to it, and
|
|
95
|
+
// because the same surface outlives the run — the Org › Architecture view
|
|
96
|
+
// reads this state for a workspace that has long since committed.
|
|
97
|
+
"architecture",
|
|
79
98
|
]);
|
|
80
99
|
|
|
81
100
|
/**
|
|
@@ -424,7 +443,12 @@ export const METHODS = Object.freeze({
|
|
|
424
443
|
// --- file ---
|
|
425
444
|
"file.pin": { family: "file", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
426
445
|
"file.unpin": { family: "file", scope: "org.write", sideEffecting: true },
|
|
427
|
-
|
|
446
|
+
// idempotent: a COMMENT CREATE keyed on the caller's message id. Without
|
|
447
|
+
// the flag hq ignores `x-idempotency-key` entirely (dispatch.ts replays only
|
|
448
|
+
// for `def.idempotent === true`), so a daemon retry after a timeout-post-
|
|
449
|
+
// commit posts the same comment two or three times on the thread. Same
|
|
450
|
+
// reasoning as messaging.send, which has carried the flag from the start.
|
|
451
|
+
"file.addComment": { family: "file", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
428
452
|
"file.listComments": { family: "file", scope: "org.read", sideEffecting: false },
|
|
429
453
|
"file.listPinned": { family: "file", scope: "org.read", sideEffecting: false },
|
|
430
454
|
// --- memory ---
|
|
@@ -458,7 +482,7 @@ export const METHODS = Object.freeze({
|
|
|
458
482
|
"notification.updatePreferences": { family: "notification", scope: "org.write", sideEffecting: true },
|
|
459
483
|
"notification.getPreferences": { family: "notification", scope: "org.read", sideEffecting: false },
|
|
460
484
|
// --- decision ---
|
|
461
|
-
"decision.comment": { family: "decision", scope: "decision.write", sideEffecting: true },
|
|
485
|
+
"decision.comment": { family: "decision", scope: "decision.write", sideEffecting: true, idempotent: true },
|
|
462
486
|
"decision.sign": { family: "decision", scope: "decision.write", sideEffecting: true },
|
|
463
487
|
"decision.reverse": { family: "decision", scope: "decision.write", sideEffecting: true },
|
|
464
488
|
"decision.requestAdjustment": { family: "decision", scope: "decision.write", sideEffecting: true },
|
|
@@ -479,7 +503,7 @@ export const METHODS = Object.freeze({
|
|
|
479
503
|
"board.updateTask": { family: "board", scope: "board.write", sideEffecting: true },
|
|
480
504
|
"board.moveTask": { family: "board", scope: "board.write", sideEffecting: true },
|
|
481
505
|
"board.assignTask": { family: "board", scope: "board.write", sideEffecting: true },
|
|
482
|
-
"board.addTaskComment": { family: "board", scope: "board.write", sideEffecting: true },
|
|
506
|
+
"board.addTaskComment": { family: "board", scope: "board.write", sideEffecting: true, idempotent: true },
|
|
483
507
|
"board.addTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true, idempotent: true },
|
|
484
508
|
"board.updateTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true },
|
|
485
509
|
"board.removeTaskAttachment": { family: "board", scope: "board.write", sideEffecting: true },
|
|
@@ -933,6 +957,57 @@ export const METHODS = Object.freeze({
|
|
|
933
957
|
// dispatcher cannot keep: the protocol advertises the method, callers get
|
|
934
958
|
// through scope resolution, and the call 500s. Register it in the same change
|
|
935
959
|
// that adds src/server/methods/escalation/answer.ts.
|
|
960
|
+
// --- GENESIS + ARCHITECTURE (2026-08): the pre-commit workspace draft and the
|
|
961
|
+
// review gates over it. 1:1 with the founder's own form surface — the ask
|
|
962
|
+
// was parity, so the bar is EQUAL power, never more.
|
|
963
|
+
//
|
|
964
|
+
// SCOPES REUSE `org.read` / `org.write` rather than minting a genesis.*
|
|
965
|
+
// pair. `org.write` is already defined as "EDITOR-equivalent: write
|
|
966
|
+
// org-structural content (members, teams, personas, ... reporting lines,
|
|
967
|
+
// chart layout)" and already sits in DEFAULT_AGENT_SCOPES under the
|
|
968
|
+
// comment "agent UI-parity". A Genesis draft IS that content, one commit
|
|
969
|
+
// earlier; a private scope would be a second name for a permission the
|
|
970
|
+
// org already grants, and every operator who had reasoned about who may
|
|
971
|
+
// edit their structure would have to reason about it twice.
|
|
972
|
+
//
|
|
973
|
+
// WHY REGISTERING THEM AT ALL IS THE FIX. `scopeForMethod()` resolves an
|
|
974
|
+
// UNKNOWN method to `admin`, and `admin` is in neither
|
|
975
|
+
// HUMAN_DEFAULT_SCOPES nor DEFAULT_AGENT_SCOPES — so an undeclared method
|
|
976
|
+
// is not merely undocumented, it is unreachable by every principal that
|
|
977
|
+
// exists. That default is correct and it was firing: the handlers shipped
|
|
978
|
+
// without descriptors and could not serve one request.
|
|
979
|
+
//
|
|
980
|
+
// EVERY WRITE IS `sideEffecting:true` AND APPENDS ONE CHAIN EVENT, which
|
|
981
|
+
// is not bookkeeping — dispatch REQUIRES exactly one append from a
|
|
982
|
+
// side-effecting handler and throws INTERNAL on zero. The tempting escape
|
|
983
|
+
// (declare a durable write `sideEffecting:false` so the requirement never
|
|
984
|
+
// applies) would buy a green test with a lie the whole protocol reads:
|
|
985
|
+
// no chain row, no idempotency replay, and `READS`-shaped callers free to
|
|
986
|
+
// retry a mutation they were told was safe.
|
|
987
|
+
//
|
|
988
|
+
// `idempotent:true` on the writes is opt-in at the caller (it engages only
|
|
989
|
+
// when an idempotencyKey is sent) and it is the honest answer for this
|
|
990
|
+
// family: every write already carries an `(attemptId, editSeq)` base and a
|
|
991
|
+
// replay of a committed op is refused CONFLICT by the stale-base gate, so
|
|
992
|
+
// without the cache a dropped response turns a SUCCEEDED edit into a
|
|
993
|
+
// confusing conflict the caller cannot distinguish from a real race.
|
|
994
|
+
//
|
|
995
|
+
// genesis.plan is a READ that spends tokens: it asks a model to turn an
|
|
996
|
+
// instruction into ops and mutates nothing. Whether it costs money is not
|
|
997
|
+
// what `sideEffecting` means. ---
|
|
998
|
+
"genesis.outline": { family: "genesis", scope: "org.read", sideEffecting: false },
|
|
999
|
+
"genesis.entity": { family: "genesis", scope: "org.read", sideEffecting: false },
|
|
1000
|
+
"genesis.grid": { family: "genesis", scope: "org.read", sideEffecting: false },
|
|
1001
|
+
"genesis.plan": { family: "genesis", scope: "org.read", sideEffecting: false },
|
|
1002
|
+
"genesis.set": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1003
|
+
"genesis.insert": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1004
|
+
"genesis.remove": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1005
|
+
"genesis.reorder": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1006
|
+
"genesis.revert": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1007
|
+
"genesis.applyPlan": { family: "genesis", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1008
|
+
"architecture.status": { family: "architecture", scope: "org.read", sideEffecting: false },
|
|
1009
|
+
"architecture.accept": { family: "architecture", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
1010
|
+
"architecture.approve": { family: "architecture", scope: "org.write", sideEffecting: true, idempotent: true },
|
|
936
1011
|
});
|
|
937
1012
|
|
|
938
1013
|
/**
|
|
@@ -134,7 +134,12 @@ test("protocol: frozen family + method counts (additive evolution guard)", () =>
|
|
|
134
134
|
// The 2026-08 protocol-convergence pass adds 4: subagent (the org-scoped
|
|
135
135
|
// sub-agent registry), agent (the agent.wait control lane), mandate (the
|
|
136
136
|
// MANDATE spine) and preference (member-scoped preferences) → 51.
|
|
137
|
-
|
|
137
|
+
// The 2026-08 Genesis agent-surface pass adds 2: genesis (the founder's
|
|
138
|
+
// pre-commit workspace draft — the ten-stage run and its editable sections)
|
|
139
|
+
// and architecture (the four tier review decisions and the one whole-
|
|
140
|
+
// architecture approval, which outlive the run as the Org › Architecture
|
|
141
|
+
// view) → 53.
|
|
142
|
+
assert.equal(FAMILIES.length, 53, "family count");
|
|
138
143
|
// 357 = the mesh-protocol + agent UI-parity + calling + branding + email
|
|
139
144
|
// methods, plus the Live Integrations surface: integration.toolsetVersion +
|
|
140
145
|
// integration.listAgentTools (SP1) and integration.invokeTool (SP5, execute a
|
|
@@ -176,7 +181,14 @@ test("protocol: frozen family + method counts (additive evolution guard)", () =>
|
|
|
176
181
|
// `server/work/ledger.ts` directly, the daemon reaches it over this wire, so
|
|
177
182
|
// the threshold and the idempotency cannot fork → 461.
|
|
178
183
|
// the descriptors; the checksum (read at runtime) is the primary drift guard.
|
|
179
|
-
|
|
184
|
+
// The 2026-08 Genesis agent-surface pass adds 13, the 1:1 programmatic
|
|
185
|
+
// parity for the founder's own Generate surface: genesis 10 (outline/entity/
|
|
186
|
+
// grid/plan reads + set/insert/remove/reorder/revert/applyPlan writes) and
|
|
187
|
+
// architecture 3 (status read + accept/approve review gates). The handlers
|
|
188
|
+
// had shipped WITHOUT descriptors, which made all 13 unreachable — dispatch
|
|
189
|
+
// rejects unknown methods and scopeForMethod resolves them to `admin`, which
|
|
190
|
+
// no principal holds → 474.
|
|
191
|
+
assert.equal(Object.keys(METHODS).length, 474, "method count");
|
|
180
192
|
});
|
|
181
193
|
|
|
182
194
|
test("protocol SP3: messaging + calling families/methods/scopes", async () => {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cohortapp/agent-sdk",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.10.0",
|
|
4
4
|
"description": "Cohort Agent SDK \u2014 autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -225,14 +225,25 @@ export function openObligations() {
|
|
|
225
225
|
* object is gone. Content is truncated: this file is a debt record, not a
|
|
226
226
|
* message archive.
|
|
227
227
|
*/
|
|
228
|
-
function itemSnapshot(item) {
|
|
228
|
+
export function itemSnapshot(item) {
|
|
229
229
|
return {
|
|
230
230
|
id: item.id || null,
|
|
231
231
|
raw_ref: item.raw_ref || null,
|
|
232
232
|
message_id: item.message_id || null,
|
|
233
233
|
service: item.service || null,
|
|
234
|
+
// `kind` IS AN ADDRESS TOO — `deliver.cohortSurfaceOf` falls back to it when
|
|
235
|
+
// `raw_ref` carries no surface name, and without it every such item
|
|
236
|
+
// re-routed from the sweep as a plain `dm` with no channel: undeliverable.
|
|
237
|
+
kind: item.kind || null,
|
|
234
238
|
channel: item.channel || null,
|
|
235
239
|
channel_id: item.channel_id || null,
|
|
240
|
+
// `scope_id` IS AN ADDRESS, not decoration. `deliver.cohortReplyRoute` reads
|
|
241
|
+
// it FIRST for every roomless Cohort surface (board, doc, decision). The
|
|
242
|
+
// sweep re-delivers from `rec.item`, so leaving it out meant the sweep
|
|
243
|
+
// routed a DIFFERENT item than the live path did — masked only by every
|
|
244
|
+
// hydrator happening to set `thread_id` too. Anything the routing table
|
|
245
|
+
// reads belongs in this snapshot.
|
|
246
|
+
scope_id: item.scope_id || null,
|
|
236
247
|
thread_id: item.thread_id || null,
|
|
237
248
|
sender: item.sender || null,
|
|
238
249
|
sender_email: item.sender_email || null,
|