@cohortapp/agent-sdk 2.9.1 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/init-maestro.md +16 -9
- package/docs/guides/mac-mini.md +11 -1
- package/docs/runbooks/cohort-cutover.md +16 -0
- package/lib/channels/inbox-item.mjs +4 -0
- package/lib/comms/send-gate.mjs +23 -1
- package/lib/comms/send-gate.test.mjs +24 -0
- package/lib/mcp/server.test.mjs +16 -4
- package/lib/model-router/economics.mjs +53 -1
- package/lib/model-router/economics.test.mjs +76 -0
- package/lib/model-router/resolve.mjs +57 -4
- package/lib/model-router.mjs +95 -8
- package/lib/model-router.test.mjs +305 -5
- package/lib/org/client.mjs +58 -1
- package/lib/org/inbound/project.mjs +9 -5
- package/lib/org/messaging.mjs +6 -1
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +176 -3
- package/lib/org/protocol.test.mjs +31 -2
- package/lib/org/resource-tools.mjs +317 -0
- package/lib/org/resource-tools.test.mjs +361 -0
- package/lib/org/tool-access.mjs +176 -0
- package/lib/org/tool-access.test.mjs +144 -0
- package/lib/org/tool-surface.mjs +431 -5
- package/lib/org/tool-surface.test.mjs +385 -8
- package/lib/org/ui-parity.mjs +196 -3
- package/lib/org/ui-parity.test.mjs +126 -7
- package/lib/tool-definitions.js +23 -2
- package/package.json +2 -2
- package/plugins/maestro-skills/.claude-plugin/marketplace.json +1 -1
- package/plugins/maestro-skills/plugin.json +4 -0
- package/plugins/maestro-skills/skills/venture-deliverables.md +176 -0
- package/policies/information-barriers.yaml +34 -7
- package/scripts/ci/check-no-residual-identity.mjs +281 -9
- package/scripts/ci/check-no-residual-identity.test.mjs +115 -2
- package/scripts/cloud-relay/voice/relay-identity.test.mjs +96 -0
- package/scripts/cloud-relay/voice/server.mjs +42 -2
- package/scripts/cost/track-claude-usage-pricing.test.mjs +183 -0
- package/scripts/cost/track-claude-usage.mjs +113 -4
- package/scripts/daemon/agent-daemon.mjs +150 -3
- package/scripts/daemon/agent-daemon.test.mjs +190 -0
- package/scripts/daemon/assurance.mjs +50 -16
- package/scripts/daemon/assurance.test.mjs +39 -1
- package/scripts/daemon/classifier-identity.test.mjs +137 -0
- package/scripts/daemon/classifier.mjs +98 -17
- package/scripts/daemon/deliver.mjs +457 -33
- package/scripts/daemon/deliver.test.mjs +564 -0
- package/scripts/daemon/prompt-builder-preamble.test.mjs +210 -0
- package/scripts/daemon/prompt-builder.mjs +264 -41
- package/scripts/daemon/prompt-builder.test.mjs +5 -5
- package/scripts/daemon/responder-history.test.mjs +18 -2
- package/scripts/daemon/responder.mjs +7 -1
- package/scripts/disclosure_boundaries.py +56 -5
- package/scripts/huddle/huddle-prompt.test.mjs +176 -0
- package/scripts/huddle/huddle-server.mjs +128 -13
- package/scripts/local-triggers/autoupdate.sh +83 -0
- package/scripts/local-triggers/generate-plists.sh +9 -0
- package/scripts/local-triggers/generate-plists.test.mjs +12 -10
- package/scripts/media-generation/brand-clause.test.mjs +135 -0
- package/scripts/media-generation/gemini-image-client.mjs +27 -9
- package/scripts/media-generation/generate-assets.mjs +102 -7
- package/scripts/pre-draft-context.py +91 -15
- package/scripts/spawn-session.sh +36 -6
- package/scripts/test-employer-grounding.py +348 -0
- package/scripts/validate_outbound.py +190 -26
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
import { test } from "node:test";
|
|
9
9
|
import assert from "node:assert/strict";
|
|
10
10
|
import { promises as fsp } from "fs";
|
|
11
|
-
import { mkdirSync, writeFileSync } from "node:fs";
|
|
11
|
+
import { mkdirSync, rmSync, writeFileSync } from "node:fs";
|
|
12
12
|
import { tmpdir } from "os";
|
|
13
13
|
import { join } from "path";
|
|
14
14
|
import { fileURLToPath } from "node:url";
|
|
@@ -27,6 +27,9 @@ import {
|
|
|
27
27
|
validateRoutingConfig,
|
|
28
28
|
} from "./model-router.mjs";
|
|
29
29
|
import { loadCatalog } from "./model-router/catalog.mjs";
|
|
30
|
+
// Read-only, to PROVE (not assume) that a provider the SDK does not bundle
|
|
31
|
+
// inherits credential pooling from the catalog rather than needing new code.
|
|
32
|
+
import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
|
|
30
33
|
|
|
31
34
|
async function makeAgentRoot() {
|
|
32
35
|
const path = join(
|
|
@@ -52,10 +55,15 @@ const ANTHROPIC_BACKEND = {
|
|
|
52
55
|
base_url: "https://api.anthropic.com",
|
|
53
56
|
capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
|
|
54
57
|
pricing: { input_per_m: 3.0, output_per_m: 15.0 },
|
|
58
|
+
// A CONFIG-DECLARED model map. Deliberately NOT the same ids as
|
|
59
|
+
// ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
|
|
60
|
+
// resolveBackend serves, so pinning them to the built-in defaults would make
|
|
61
|
+
// the assertions vacuous (they would pass even if the config map were being
|
|
62
|
+
// ignored entirely). The built-in defaults get their own test below.
|
|
55
63
|
models: {
|
|
56
64
|
classifier: "claude-haiku-4-5-20251001",
|
|
57
|
-
default: "
|
|
58
|
-
premium: "
|
|
65
|
+
default: "config-declared-sonnet",
|
|
66
|
+
premium: "config-declared-opus",
|
|
59
67
|
fast: "claude-haiku-4-5-20251001",
|
|
60
68
|
},
|
|
61
69
|
};
|
|
@@ -252,14 +260,59 @@ test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () =
|
|
|
252
260
|
const opus = resolveBackend({ model_hint: "opus" }, { config });
|
|
253
261
|
const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
|
|
254
262
|
const haiku = resolveBackend({ model_hint: "haiku" }, { config });
|
|
255
|
-
assert.equal(opus.model, "
|
|
256
|
-
assert.equal(sonnet.model, "
|
|
263
|
+
assert.equal(opus.model, "config-declared-opus");
|
|
264
|
+
assert.equal(sonnet.model, "config-declared-sonnet");
|
|
257
265
|
assert.equal(haiku.model, "claude-haiku-4-5-20251001");
|
|
258
266
|
assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
|
|
259
267
|
assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
|
|
260
268
|
assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
|
|
261
269
|
});
|
|
262
270
|
|
|
271
|
+
// ---------------------------------------------------------------------------
|
|
272
|
+
// Model freshness — the built-in Anthropic defaults (see the provenance block
|
|
273
|
+
// above ANTHROPIC_DEFAULT in model-router.mjs).
|
|
274
|
+
// ---------------------------------------------------------------------------
|
|
275
|
+
|
|
276
|
+
test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
|
|
277
|
+
// The floor an agent with no routing config lands on. A stale id here never
|
|
278
|
+
// reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
|
|
279
|
+
// DID reach resolved.model, which the cost ledger and telemetry store — so a
|
|
280
|
+
// stale id mis-prices and mis-attributes every unrouted session.
|
|
281
|
+
const models = defaultRoutingConfig().backends.anthropic.models;
|
|
282
|
+
assert.equal(models.premium, "claude-opus-5");
|
|
283
|
+
assert.equal(models.default, "claude-sonnet-5");
|
|
284
|
+
// The fast/classifier tier did NOT move in the 2026-08-13 refresh.
|
|
285
|
+
assert.equal(models.fast, "claude-haiku-4-5-20251001");
|
|
286
|
+
assert.equal(models.classifier, "claude-haiku-4-5-20251001");
|
|
287
|
+
// Guard the specific regression: no 4.x id survives anywhere in the map.
|
|
288
|
+
for (const [tier, id] of Object.entries(models)) {
|
|
289
|
+
assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
|
|
290
|
+
}
|
|
291
|
+
});
|
|
292
|
+
|
|
293
|
+
test("resolveBackend serves the current ids through the no-config default path", () => {
|
|
294
|
+
// End-to-end through the resolver, not just the constant: an agent with no
|
|
295
|
+
// config file (useDefault) resolves premium/default to the -5 pair.
|
|
296
|
+
const config = defaultRoutingConfig();
|
|
297
|
+
assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
|
|
298
|
+
assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
|
|
299
|
+
// …while the CLI flag stays version-agnostic, which is WHY the stale ids were
|
|
300
|
+
// survivable on the wire and only corrupted the ledger.
|
|
301
|
+
assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
|
|
302
|
+
});
|
|
303
|
+
|
|
304
|
+
test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
|
|
305
|
+
// Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
|
|
306
|
+
// agent adds backends.anthropic.models to config/model-routing.yaml and is
|
|
307
|
+
// current WITHOUT an SDK release. normaliseConfig only injects the built-in
|
|
308
|
+
// map when backends.anthropic is absent, so this must win outright.
|
|
309
|
+
const config = {
|
|
310
|
+
backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
|
|
311
|
+
routing_policy: [{ default: true, backend: "anthropic" }],
|
|
312
|
+
};
|
|
313
|
+
assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
|
|
314
|
+
});
|
|
315
|
+
|
|
263
316
|
test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
|
|
264
317
|
const config = {
|
|
265
318
|
backends: { qwen: QWEN_BACKEND },
|
|
@@ -833,6 +886,253 @@ test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
|
|
|
833
886
|
assert.equal(d.audit.fallback_reason, "no_v2_config");
|
|
834
887
|
});
|
|
835
888
|
|
|
889
|
+
// ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
|
|
890
|
+
//
|
|
891
|
+
// The safety net fires on the three paths that have no chain to walk: the kill
|
|
892
|
+
// switch, "no v2 config", and "nothing survived the gates". It is the ONE place
|
|
893
|
+
// resolve.mjs still names Anthropic model ids, so it is the one place that can
|
|
894
|
+
// go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
|
|
895
|
+
// preference list carries the current -5 ids AND the 4-x succession fallback.
|
|
896
|
+
|
|
897
|
+
/** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
|
|
898
|
+
const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
899
|
+
agentConfig: {
|
|
900
|
+
catalog_overrides: [
|
|
901
|
+
{
|
|
902
|
+
ref: "anthropic/claude-opus-5",
|
|
903
|
+
status: "available",
|
|
904
|
+
context_tokens: 950000,
|
|
905
|
+
max_tokens: 128000,
|
|
906
|
+
tool_reliability: "A",
|
|
907
|
+
harness: { session: true, direct: true, batch: true },
|
|
908
|
+
cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
|
|
909
|
+
},
|
|
910
|
+
{
|
|
911
|
+
ref: "anthropic/claude-sonnet-5",
|
|
912
|
+
status: "available",
|
|
913
|
+
context_tokens: 950000,
|
|
914
|
+
max_tokens: 64000,
|
|
915
|
+
tool_reliability: "A",
|
|
916
|
+
harness: { session: true, direct: true, batch: true },
|
|
917
|
+
cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
|
|
918
|
+
},
|
|
919
|
+
],
|
|
920
|
+
},
|
|
921
|
+
});
|
|
922
|
+
|
|
923
|
+
test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
|
|
924
|
+
const frontier = resolveChain(
|
|
925
|
+
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
926
|
+
{ catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
927
|
+
);
|
|
928
|
+
assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
|
|
929
|
+
|
|
930
|
+
const workhorse = resolveChain(
|
|
931
|
+
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
932
|
+
{ catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
933
|
+
);
|
|
934
|
+
assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
|
|
935
|
+
});
|
|
936
|
+
|
|
937
|
+
test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
|
|
938
|
+
// This is why the 4-x refs stay in the preference list. Delete them and this
|
|
939
|
+
// does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
|
|
940
|
+
// i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
|
|
941
|
+
// ORDINARY work the frontier model and quietly quadruple its cost.
|
|
942
|
+
const d = resolveChain(
|
|
943
|
+
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
944
|
+
{ catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
945
|
+
);
|
|
946
|
+
assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
|
|
947
|
+
// Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
|
|
948
|
+
// the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
|
|
949
|
+
});
|
|
950
|
+
|
|
951
|
+
test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
|
|
952
|
+
// THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
|
|
953
|
+
// finished. The test above asserts `chosen.model`, which reads as bookkeeping.
|
|
954
|
+
// This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
|
|
955
|
+
// `--model` — against the catalog the SDK actually ships, with `config: null`,
|
|
956
|
+
// which is the DEFAULT state of every agent because this repo contains no
|
|
957
|
+
// config/model-routing.yaml at all.
|
|
958
|
+
//
|
|
959
|
+
// Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
|
|
960
|
+
// agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
|
|
961
|
+
// row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
|
|
962
|
+
// merely a mis-stamped ledger entry.
|
|
963
|
+
//
|
|
964
|
+
// WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
|
|
965
|
+
// the missing step. Update the expectations here, and DELETE the 4-x refs from
|
|
966
|
+
// pickAnthropicRow's preference lists in the same change, or the museum the
|
|
967
|
+
// provenance block warns about starts accumulating.
|
|
968
|
+
const cases = [
|
|
969
|
+
["session.responder", {}, "claude-sonnet-4-6"],
|
|
970
|
+
["decision", { needs_thinking: true }, "claude-opus-4-8"],
|
|
971
|
+
];
|
|
972
|
+
for (const [task_class, extra, expected] of cases) {
|
|
973
|
+
const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
|
|
974
|
+
for (const [label, env] of [
|
|
975
|
+
["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
|
|
976
|
+
["no v2 config", {}],
|
|
977
|
+
]) {
|
|
978
|
+
const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
|
|
979
|
+
assert.equal(
|
|
980
|
+
d.spawnArgs.modelFlag,
|
|
981
|
+
expected,
|
|
982
|
+
`${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
|
|
983
|
+
);
|
|
984
|
+
// Not the shorthand: nothing downstream re-resolves this to a current build.
|
|
985
|
+
assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
|
|
986
|
+
}
|
|
987
|
+
}
|
|
988
|
+
});
|
|
989
|
+
|
|
990
|
+
test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
|
|
991
|
+
// The last-ditch path: no catalog, so the id goes straight onto `claude
|
|
992
|
+
// --model`. It used to emit {model: "claude-opus-4-8", ref:
|
|
993
|
+
// "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
|
|
994
|
+
// feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
|
|
995
|
+
// both named a different model than the one that ran.
|
|
996
|
+
const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
|
|
997
|
+
mkdirSync(emptyDir, { recursive: true });
|
|
998
|
+
try {
|
|
999
|
+
const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
|
|
1000
|
+
assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
|
|
1001
|
+
|
|
1002
|
+
const d = resolveChain(
|
|
1003
|
+
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
1004
|
+
{ catalog: EMPTY, config: null, env: {}, now: CLOCK }
|
|
1005
|
+
);
|
|
1006
|
+
assert.equal(d.chosen.model, "claude-opus-5");
|
|
1007
|
+
assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
|
|
1008
|
+
assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
|
|
1009
|
+
|
|
1010
|
+
const ordinary = resolveChain(
|
|
1011
|
+
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
1012
|
+
{ catalog: EMPTY, config: null, env: {}, now: CLOCK }
|
|
1013
|
+
);
|
|
1014
|
+
assert.equal(ordinary.chosen.model, "claude-sonnet-5");
|
|
1015
|
+
assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
|
|
1016
|
+
} finally {
|
|
1017
|
+
rmSync(emptyDir, { recursive: true, force: true });
|
|
1018
|
+
}
|
|
1019
|
+
});
|
|
1020
|
+
|
|
1021
|
+
// ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
|
|
1022
|
+
//
|
|
1023
|
+
// The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
|
|
1024
|
+
// is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
|
|
1025
|
+
// highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
|
|
1026
|
+
// introduce a whole provider, and resolveChain gates it exactly like a bundled
|
|
1027
|
+
// one. These tests prove the full path — row → alias → chain → credential gate
|
|
1028
|
+
// → chosen — so the remaining work is a bundled YAML file, not plumbing.
|
|
1029
|
+
|
|
1030
|
+
const XAI_PROVIDER_DOC = {
|
|
1031
|
+
provider: "xai",
|
|
1032
|
+
auth_env: "XAI_API_KEY",
|
|
1033
|
+
endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
|
|
1034
|
+
data_residency: "us",
|
|
1035
|
+
models: [
|
|
1036
|
+
{
|
|
1037
|
+
id: "grok-4-fast",
|
|
1038
|
+
status: "available",
|
|
1039
|
+
context_window: 2000000,
|
|
1040
|
+
context_tokens: 1900000,
|
|
1041
|
+
max_tokens: 30000,
|
|
1042
|
+
cost: { input: 0.2, output: 0.5 },
|
|
1043
|
+
cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
|
|
1044
|
+
compat: { thinking_format: "openai", cache_control: "implicit" },
|
|
1045
|
+
// Grade B like every other third-party row: good enough for the direct
|
|
1046
|
+
// lane, structurally barred from hosting a tool-using session (§6.6).
|
|
1047
|
+
tool_reliability: "B",
|
|
1048
|
+
harness: { session: true, direct: true, batch: false },
|
|
1049
|
+
},
|
|
1050
|
+
],
|
|
1051
|
+
};
|
|
1052
|
+
|
|
1053
|
+
const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
1054
|
+
agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
|
|
1055
|
+
});
|
|
1056
|
+
|
|
1057
|
+
const XAI_CONFIG = Object.freeze({
|
|
1058
|
+
schema_version: 2,
|
|
1059
|
+
aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
|
|
1060
|
+
defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
|
|
1061
|
+
backends: {
|
|
1062
|
+
anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
|
|
1063
|
+
xai: { allowed_data_classes: ["public"] },
|
|
1064
|
+
},
|
|
1065
|
+
routing_policy: [
|
|
1066
|
+
{ match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
|
|
1067
|
+
{ default: true, chain: ["default"] },
|
|
1068
|
+
],
|
|
1069
|
+
fallback_to_anthropic: true,
|
|
1070
|
+
});
|
|
1071
|
+
|
|
1072
|
+
test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
|
|
1073
|
+
const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
|
|
1074
|
+
assert.ok(row, "grok row present after the agent-config merge");
|
|
1075
|
+
assert.equal(row.provider, "xai");
|
|
1076
|
+
assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
|
|
1077
|
+
assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
|
|
1078
|
+
assert.ok(
|
|
1079
|
+
CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
|
|
1080
|
+
"introducing an unbundled provider warns (typo guard) — it does not fail"
|
|
1081
|
+
);
|
|
1082
|
+
});
|
|
1083
|
+
|
|
1084
|
+
test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
|
|
1085
|
+
const d = resolveChain(
|
|
1086
|
+
{ task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
|
|
1087
|
+
{ catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
|
|
1088
|
+
);
|
|
1089
|
+
assert.equal(d.chosen.provider, "xai");
|
|
1090
|
+
assert.equal(d.chosen.model, "grok-4-fast");
|
|
1091
|
+
assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
|
|
1092
|
+
});
|
|
1093
|
+
|
|
1094
|
+
test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
|
|
1095
|
+
const d = resolveChain(
|
|
1096
|
+
{ task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
|
|
1097
|
+
{ catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
|
|
1098
|
+
);
|
|
1099
|
+
const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
|
|
1100
|
+
assert.ok(skipped, "grok recorded in tried[]");
|
|
1101
|
+
assert.equal(skipped.reason, "missing_credential");
|
|
1102
|
+
assert.notEqual(d.chosen.provider, "xai");
|
|
1103
|
+
});
|
|
1104
|
+
|
|
1105
|
+
test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
|
|
1106
|
+
// auth-profiles derives provider → auth_env from the CATALOG, so a provider
|
|
1107
|
+
// the SDK has never heard of gets pooled multi-key rotation for free. This is
|
|
1108
|
+
// the "their own auth profiles" half of the cheap-tier requirement.
|
|
1109
|
+
const profiles = loadAuthProfiles(null, {
|
|
1110
|
+
configDoc: null, // bypass the FS
|
|
1111
|
+
env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
|
|
1112
|
+
authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
|
|
1113
|
+
});
|
|
1114
|
+
assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
|
|
1115
|
+
assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
|
|
1116
|
+
assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
|
|
1117
|
+
});
|
|
1118
|
+
|
|
1119
|
+
test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
|
|
1120
|
+
// Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
|
|
1121
|
+
// third-party row is graded A until the probe ledger earns it. Cheap-tier
|
|
1122
|
+
// reachability must NOT quietly become cheap-tier session hosting.
|
|
1123
|
+
const sessionCfg = {
|
|
1124
|
+
...XAI_CONFIG,
|
|
1125
|
+
routing_policy: [{ default: true, chain: ["cheap", "default"] }],
|
|
1126
|
+
};
|
|
1127
|
+
const d = resolveChain(
|
|
1128
|
+
{ task_class: "session.responder", data_class: "public", token_estimate: 2000 },
|
|
1129
|
+
{ catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
|
|
1130
|
+
);
|
|
1131
|
+
const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
|
|
1132
|
+
assert.equal(skipped?.reason, "grade_too_low");
|
|
1133
|
+
assert.equal(d.chosen.provider, "anthropic");
|
|
1134
|
+
});
|
|
1135
|
+
|
|
836
1136
|
// ── estCostUSD + ledger join fields ──────────────────────────────────────────
|
|
837
1137
|
|
|
838
1138
|
test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
|
package/lib/org/client.mjs
CHANGED
|
@@ -109,9 +109,18 @@ export function isEnabled(cfg) {
|
|
|
109
109
|
* Distil the org client config from an agent config object. Resolves:
|
|
110
110
|
* - base: the server origin for the /v1 binding (explicit `org.cohort.base`,
|
|
111
111
|
* else derived from `apiUrl` [+ `orgId` path] for back-compat).
|
|
112
|
-
* - orgId: the org
|
|
112
|
+
* - orgId: the org's ID — config `org.cohort.orgId` else COHORT_ORG_ID env
|
|
113
113
|
* (hq's documented var). Used by the legacy directory-path derivation
|
|
114
114
|
* and, when known, sent as the `x-org-id` pin header (defence-in-depth).
|
|
115
|
+
*
|
|
116
|
+
* NOT THE SLUG, though this said "the org slug" until 2026-08-21 and
|
|
117
|
+
* `docs/guides/mac-mini.md` told operators to export one. hq compares
|
|
118
|
+
* the pin with STRICT EQUALITY against the key's `Org.id`
|
|
119
|
+
* (`src/server/auth/org-api-key.ts:163`) — no slug fallback — so a
|
|
120
|
+
* slug here 401s every call. The two are not interchangeable: the
|
|
121
|
+
* ID has the form `org_default_<slug>` (e.g. slug `acme` → ID
|
|
122
|
+
* `org_default_acme`), never the bare slug. Note `COHORT_AGENT_ID`
|
|
123
|
+
* beside it genuinely IS a slug.
|
|
115
124
|
* - token: bearer credential — the RAW OrgApiKey. Resolution order:
|
|
116
125
|
* config `token` → COHORT_API_TOKEN → COHORT_TOKEN → COHORT_API_KEY.
|
|
117
126
|
* hq's auth doc names the agent var COHORT_API_KEY; the older
|
|
@@ -1314,6 +1323,54 @@ export function messagingSend(params, o = {}) {
|
|
|
1314
1323
|
return call("messaging.send", params || {}, o);
|
|
1315
1324
|
}
|
|
1316
1325
|
|
|
1326
|
+
/**
|
|
1327
|
+
* Ask hq who — if anyone — should respond to a message in a channel
|
|
1328
|
+
* (messaging.electResponder). This is the SHOULD-I-RESPOND decision, moved off
|
|
1329
|
+
* the daemon and onto hq's `routeResponders`, which for an undirected group
|
|
1330
|
+
* message elects exactly ONE responder or NONE, and already fails safe (no model
|
|
1331
|
+
* → quiet). hq derives the roster + trigger server-side from just
|
|
1332
|
+
* `{messageId, channelId}`, org-scoped, computes the verdict ONCE per messageId
|
|
1333
|
+
* and caches it in a durable ledger event, so concurrent daemons read the same
|
|
1334
|
+
* election (the loser of a race reads the winner).
|
|
1335
|
+
*
|
|
1336
|
+
* Read-style: agent-tier `messaging.read` scope (any paired seat may ask),
|
|
1337
|
+
* `sideEffecting:false` — no client idempotency key, and the server owns the
|
|
1338
|
+
* write-once ledger append. Returns a res frame whose `result` is
|
|
1339
|
+
* `{ election, mode, responders:[{slug,dimension}], deferredToHuman }`.
|
|
1340
|
+
*
|
|
1341
|
+
* FAIL-OPEN AT THE TRANSPORT ONLY: an unreachable/disabled server yields an
|
|
1342
|
+
* error frame, never a throw. The CALLER owns the fail-SAFE inversion — for an
|
|
1343
|
+
* ambient channel, "no verdict / not me" means STAY SILENT, never respond-to-all.
|
|
1344
|
+
*
|
|
1345
|
+
* @param {{messageId:string, channelId:string}} params
|
|
1346
|
+
* @param {object} [o] { base, token, orgId?, fetchImpl? }
|
|
1347
|
+
*/
|
|
1348
|
+
export function electResponder(params, o = {}) {
|
|
1349
|
+
return call("messaging.electResponder", params || {}, o);
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
/**
|
|
1353
|
+
* Record a voice note in THIS seat's own designed voice
|
|
1354
|
+
* (messaging.synthesizeVoiceNote) and return the READY `{fileId, durationMs,
|
|
1355
|
+
* sizeBytes}` to attach.
|
|
1356
|
+
*
|
|
1357
|
+
* This is the RECORD half only. Posting is an ordinary `messagingSend` with
|
|
1358
|
+
* `attachments:[{fileId}]` — hq deliberately keeps the two apart because
|
|
1359
|
+
* synthesis is a provider round-trip that cannot ride its transactional send
|
|
1360
|
+
* lane, and because a spoken message must be an ordinary message with a file on
|
|
1361
|
+
* it rather than a second send path with its own governance.
|
|
1362
|
+
*
|
|
1363
|
+
* Refusals come back as error frames naming the cause (no synthesis key on the
|
|
1364
|
+
* deployment, this seat has no designed voice, the hourly recording cap). Relay
|
|
1365
|
+
* the cause — never claim a voice note was sent when the frame was not ok.
|
|
1366
|
+
*
|
|
1367
|
+
* `messaging_send_voice_note` (tool-surface) composes both halves into one verb.
|
|
1368
|
+
* @param {{text:string}} params
|
|
1369
|
+
*/
|
|
1370
|
+
export function messagingSynthesizeVoiceNote(params, o = {}) {
|
|
1371
|
+
return call("messaging.synthesizeVoiceNote", params || {}, o);
|
|
1372
|
+
}
|
|
1373
|
+
|
|
1317
1374
|
/** React to an org message (messaging.react). */
|
|
1318
1375
|
export function messagingReact(params, o = {}) {
|
|
1319
1376
|
return call("messaging.react", params || {}, o);
|
|
@@ -102,11 +102,15 @@ export function toMessageEvent(o = {}) {
|
|
|
102
102
|
kind: def.kind,
|
|
103
103
|
subject,
|
|
104
104
|
// ONLY ever a real Cohort channel id, never an entity id standing in for
|
|
105
|
-
// one.
|
|
106
|
-
// `messaging.send({
|
|
107
|
-
// id here would post-to-a-non-channel and
|
|
108
|
-
// this
|
|
109
|
-
//
|
|
105
|
+
// one. A reply to a room surface routes with
|
|
106
|
+
// `messaging.send({channelId: item.channel_id})`, so putting a task/decision
|
|
107
|
+
// id here would post-to-a-non-channel and come back NOT_FOUND. A surface
|
|
108
|
+
// with no room leaves this BLANK and is routed on its own surface instead —
|
|
109
|
+
// `scripts/daemon/deliver.mjs cohortReplyRoute` reads `raw_ref` for the
|
|
110
|
+
// surface and `scope_id`/`thread_id` for the entity, and sends a board
|
|
111
|
+
// comment to `board.addTaskComment`, a doc comment to `files.commentAdd`,
|
|
112
|
+
// and so on. That enforcement is new: for months the send path fell back to
|
|
113
|
+
// `channel_label` below, so a board reply was addressed to a task's TITLE.
|
|
110
114
|
channel_id: channelId,
|
|
111
115
|
channel_label: channelLabel,
|
|
112
116
|
sender: s(from.name || from.id),
|
package/lib/org/messaging.mjs
CHANGED
|
@@ -138,8 +138,13 @@ function agentIdOf(cfg) {
|
|
|
138
138
|
*
|
|
139
139
|
* @param {object} o - { cfg, agentRoot, action, channel }
|
|
140
140
|
* @param {object} deps - { ledgerImpl?, catalog? } injectable for tests
|
|
141
|
+
*
|
|
142
|
+
* EXPORTED because room sends are no longer the only attributed outbound:
|
|
143
|
+
* `scripts/daemon/deliver.mjs` posts board/doc/decision comments straight onto
|
|
144
|
+
* their entity thread, bypassing `sendMessage`, and an outbound family missing
|
|
145
|
+
* from the ledger is an outbound family nobody can see.
|
|
141
146
|
*/
|
|
142
|
-
async function attributeMessaging(o, deps = {}) {
|
|
147
|
+
export async function attributeMessaging(o, deps = {}) {
|
|
143
148
|
// Opt-out: an explicit `ledgerImpl: null` (or false) disables attribution.
|
|
144
149
|
if (Object.prototype.hasOwnProperty.call(deps, "ledgerImpl") && !deps.ledgerImpl) return;
|
|
145
150
|
try {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
b2cbf124c99800a0ad1b19d7aea6a88c3c72f07b53838eb44ef6f82d888fda43
|