@cohortapp/agent-sdk 2.15.0 → 2.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +5 -2
- package/docs/guides/front-door-session.md +16 -5
- package/docs/guides/poller-daemon-setup.md +53 -2
- package/lib/assurance/plan-note.mjs +251 -0
- package/lib/assurance/plan-note.test.mjs +234 -0
- package/lib/assurance/room-budget.mjs +497 -0
- package/lib/assurance/room-budget.test.mjs +486 -0
- package/lib/assurance/tier.mjs +166 -0
- package/lib/assurance/tier.test.mjs +174 -0
- package/lib/comms/receipts.mjs +17 -1
- package/lib/context/budget.mjs +327 -0
- package/lib/context/budget.test.mjs +252 -0
- package/lib/context/history-scope.mjs +138 -0
- package/lib/context/history-scope.test.mjs +79 -0
- package/lib/model-router/economics.mjs +9 -0
- package/lib/model-router/resolve.mjs +6 -0
- package/lib/org/inbound/facts.mjs +4 -2
- package/lib/org/inbound/hydrate.mjs +555 -51
- package/lib/org/inbound/hydrate.test.mjs +456 -1
- package/package.json +3 -1
- package/plugins/maestro-skills/skills/inbound-triage.md +52 -24
- package/plugins/maestro-skills/skills/main-session.md +6 -4
- package/scripts/daemon/agent-daemon.mjs +35 -7
- package/scripts/daemon/agent-daemon.test.mjs +23 -6
- package/scripts/daemon/assurance-e2e.test.mjs +75 -19
- package/scripts/daemon/assurance.mjs +663 -159
- package/scripts/daemon/assurance.test.mjs +820 -140
- package/scripts/daemon/context-compiler.mjs +52 -21
- package/scripts/daemon/context-compiler.test.mjs +106 -0
- package/scripts/daemon/deliver.mjs +7 -4
- package/scripts/daemon/dispatcher-session-continuity.test.mjs +365 -0
- package/scripts/daemon/dispatcher.mjs +210 -9
- package/scripts/daemon/lib/session-router.mjs +310 -42
- package/scripts/daemon/lib/session-router.test.mjs +260 -1
- package/scripts/daemon/prompt-builder.mjs +160 -16
- package/scripts/daemon/prompt-builder.test.mjs +287 -7
- package/scripts/daemon/responder-history.test.mjs +37 -1
- package/scripts/daemon/responder.mjs +79 -72
|
@@ -15,6 +15,8 @@ import { readFileSync, readdirSync } from "fs";
|
|
|
15
15
|
import { join } from "path";
|
|
16
16
|
import { isEnabled as orgEnabled, loadOrgConfig } from "../../lib/org/client.mjs";
|
|
17
17
|
import { recall as orgRecall } from "../../lib/org/knowledge.mjs";
|
|
18
|
+
import { isPrivateConversation, historyDirNames } from "../../lib/context/history-scope.mjs";
|
|
19
|
+
import { fitSections } from "../../lib/context/budget.mjs";
|
|
18
20
|
|
|
19
21
|
const AGENT_REPO_DIR = process.env.__TEST_AGENT_DIR || process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
|
|
20
22
|
const SESSION_DIR = process.env.__TEST_SESSION_DIR || join(AGENT_REPO_DIR, "state", "sessions");
|
|
@@ -56,6 +58,8 @@ const HISTORY_SHARE = 0.6;
|
|
|
56
58
|
*/
|
|
57
59
|
function extractChannelId(item) {
|
|
58
60
|
if (!item) return null;
|
|
61
|
+
// Every non-Slack service sets this outright; Slack is the one that has to be
|
|
62
|
+
// dug out of `raw_ref`.
|
|
59
63
|
if (item.channel_id) return item.channel_id;
|
|
60
64
|
if (item.raw_ref) {
|
|
61
65
|
const match = item.raw_ref.match(/slack:([DC][A-Z0-9]+):/);
|
|
@@ -234,14 +238,32 @@ function loadDisclosureBoundaries(senderSlug, budgetTokens) {
|
|
|
234
238
|
|
|
235
239
|
/**
|
|
236
240
|
* Load conversation history from interaction logs, filtered by disclosure boundaries.
|
|
241
|
+
*
|
|
242
|
+
* READS THE ITEM'S OWN SERVICE. `responder.mjs#logInteraction` files every
|
|
243
|
+
* exchange under `memory/interactions/<service>/…`; this read was pinned to
|
|
244
|
+
* `…/slack/…`, so a Cohort, Telegram or WhatsApp exchange was written down and
|
|
245
|
+
* then never found (design §3 R12) — which is how "why can't you see the
|
|
246
|
+
* context of our ongoing conversation?" happened in a thread whose whole
|
|
247
|
+
* history was on disk the whole time.
|
|
248
|
+
*
|
|
249
|
+
* Slack keeps its legacy bare `<sender-slug>` directory alongside the `dm-`
|
|
250
|
+
* one, so historical logs do not go dark on this change.
|
|
251
|
+
*
|
|
252
|
+
* AND IT READS ONLY WHAT THIS ROOM MAY SEE. `logInteraction` files every
|
|
253
|
+
* exchange under BOTH the room and `dm-<sender-slug>`, so merging the candidate
|
|
254
|
+
* directories put one person's private DM sentence into the prompt built for a
|
|
255
|
+
* public channel the moment this read stopped being Slack-only. The gate lives
|
|
256
|
+
* in `lib/context/history-scope.mjs`, shared with the legacy prompt path and
|
|
257
|
+
* with the writer (design §5.5, resource scope).
|
|
237
258
|
*/
|
|
238
|
-
function loadConversationHistory(senderSlug, channelId, budgetTokens, forbiddenDomains) {
|
|
239
|
-
const
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
259
|
+
function loadConversationHistory(senderSlug, channelId, budgetTokens, forbiddenDomains, service = "slack", isPrivate = false) {
|
|
260
|
+
const svc = String(service || "slack");
|
|
261
|
+
const candidateDirs = historyDirNames({
|
|
262
|
+
channelId,
|
|
263
|
+
senderSlug,
|
|
264
|
+
isPrivate: isPrivate === true,
|
|
265
|
+
service: svc,
|
|
266
|
+
}).map((name) => join(AGENT_REPO_DIR, "memory", "interactions", svc, name));
|
|
245
267
|
|
|
246
268
|
const entries = [];
|
|
247
269
|
const seenKeys = new Set();
|
|
@@ -292,25 +314,27 @@ function loadConversationHistory(senderSlug, channelId, budgetTokens, forbiddenD
|
|
|
292
314
|
|
|
293
315
|
if (filtered.length === 0) return null;
|
|
294
316
|
|
|
295
|
-
//
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
317
|
+
// Spend the budget from the NEWEST turn backwards, and SAY what fell off.
|
|
318
|
+
// This used to `break` out of the loop and return the survivors, so a prompt
|
|
319
|
+
// that had quietly lost thirty turns of a conversation looked exactly like a
|
|
320
|
+
// conversation thirty turns shorter — design §5.4: "dropping is visible in
|
|
321
|
+
// the prompt, never silent". `lib/context/budget.mjs` owns the arithmetic and
|
|
322
|
+
// the wording; this function only supplies the turns.
|
|
323
|
+
const turns = filtered.map((entry) => {
|
|
302
324
|
const from = entry.from || entry.sender || entry.user_name || "unknown";
|
|
303
325
|
const text = entry.content || entry.text || "(no text)";
|
|
304
326
|
const ts = entry.received_at || entry.timestamp || entry.processed_at || entry.ts || "";
|
|
305
327
|
const responded = entry.response_sent ? " [Responded]" : "";
|
|
306
|
-
|
|
328
|
+
return `[${ts}] ${from}: ${text.substring(0, 500)}${responded}`;
|
|
329
|
+
});
|
|
307
330
|
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
331
|
+
const fitted = fitSections({
|
|
332
|
+
sections: [{ name: "thread", turns }],
|
|
333
|
+
budgetTokens: Math.max(1, Number(budgetTokens) || 0),
|
|
334
|
+
charsPerToken: CHARS_PER_TOKEN,
|
|
335
|
+
});
|
|
312
336
|
|
|
313
|
-
return
|
|
337
|
+
return fitted.text ? fitted.text : null;
|
|
314
338
|
}
|
|
315
339
|
|
|
316
340
|
// ── Active sessions ────────────────────────────────────────────────
|
|
@@ -522,7 +546,14 @@ export async function compileContext(item, classResult, options = {}) {
|
|
|
522
546
|
const fixedTokens = ALLOC.profile + ALLOC.boundaries + ALLOC.active_sessions + ALLOC.sent_registry + ALLOC.org_memory;
|
|
523
547
|
const remainingTokens = tokenBudget - fixedTokens;
|
|
524
548
|
const historyBudget = Math.floor(remainingTokens * HISTORY_SHARE);
|
|
525
|
-
const historyText = loadConversationHistory(
|
|
549
|
+
const historyText = loadConversationHistory(
|
|
550
|
+
senderSlug,
|
|
551
|
+
channelId,
|
|
552
|
+
historyBudget,
|
|
553
|
+
forbiddenDomains,
|
|
554
|
+
item?.service,
|
|
555
|
+
isPrivateConversation(item),
|
|
556
|
+
);
|
|
526
557
|
|
|
527
558
|
// Step 5 — Active sessions
|
|
528
559
|
const sessionsText = loadActiveSessions(ALLOC.active_sessions);
|
|
@@ -298,3 +298,109 @@ privilege_level: ceo
|
|
|
298
298
|
assert.ok(!result.contextBlock.includes("DO NOT mention"), "CEO should NOT see DO NOT mention block");
|
|
299
299
|
});
|
|
300
300
|
});
|
|
301
|
+
|
|
302
|
+
// ── Per-service interaction history (design §3 R12) ────────────────
|
|
303
|
+
//
|
|
304
|
+
// `responder.mjs#logInteraction` files every exchange under
|
|
305
|
+
// `memory/interactions/<service>/…`. This read was pinned to `…/slack/…`, so a
|
|
306
|
+
// Cohort exchange was written down and then never found — the agent had the
|
|
307
|
+
// record of the conversation and could not see it.
|
|
308
|
+
|
|
309
|
+
describe("interaction history is read per service", () => {
|
|
310
|
+
it("finds a Cohort conversation under memory/interactions/cohort/", async () => {
|
|
311
|
+
const dir = join(TEST_DIR, "memory", "interactions", "cohort", "chan-cohort-1");
|
|
312
|
+
mkdirSync(dir, { recursive: true });
|
|
313
|
+
writeFileSync(
|
|
314
|
+
join(dir, "2026-04-06.jsonl"),
|
|
315
|
+
[
|
|
316
|
+
JSON.stringify({ ts: "3001", from: "dana", content: "the retainer number is wrong on page two", received_at: "2026-04-06T10:00:00Z" }),
|
|
317
|
+
JSON.stringify({ ts: "3002", from: "agent", content: "corrected to 48,000", received_at: "2026-04-06T10:01:00Z" }),
|
|
318
|
+
].join("\n"),
|
|
319
|
+
);
|
|
320
|
+
|
|
321
|
+
const item = { sender: "dana", service: "cohort", channel_id: "chan-cohort-1" };
|
|
322
|
+
const result = await compileContext(item, { priority: "normal" });
|
|
323
|
+
|
|
324
|
+
assert.ok(result.contextBlock.includes("retainer number is wrong"), "Cohort history must be found");
|
|
325
|
+
assert.ok(result.contextBlock.includes("corrected to 48,000"));
|
|
326
|
+
});
|
|
327
|
+
|
|
328
|
+
it("does not read another service's directory for the same channel id", async () => {
|
|
329
|
+
const dir = join(TEST_DIR, "memory", "interactions", "telegram", "chan-shared");
|
|
330
|
+
mkdirSync(dir, { recursive: true });
|
|
331
|
+
writeFileSync(
|
|
332
|
+
join(dir, "2026-04-06.jsonl"),
|
|
333
|
+
JSON.stringify({ ts: "4001", from: "dana", content: "TELEGRAM ONLY SENTINEL", received_at: "2026-04-06T10:00:00Z" }) + "\n",
|
|
334
|
+
);
|
|
335
|
+
|
|
336
|
+
const result = await compileContext({ sender: "dana", service: "cohort", channel_id: "chan-shared" }, { priority: "normal" });
|
|
337
|
+
assert.ok(!result.contextBlock.includes("TELEGRAM ONLY SENTINEL"), "one service must not read another's history");
|
|
338
|
+
});
|
|
339
|
+
|
|
340
|
+
it("an item with no service still reads the Slack tree (the historical default)", async () => {
|
|
341
|
+
const dir = join(TEST_DIR, "memory", "interactions", "slack", "D00LEGACY");
|
|
342
|
+
mkdirSync(dir, { recursive: true });
|
|
343
|
+
writeFileSync(
|
|
344
|
+
join(dir, "2026-04-06.jsonl"),
|
|
345
|
+
JSON.stringify({ ts: "5001", from: "dana", content: "LEGACY SLACK LINE", received_at: "2026-04-06T10:00:00Z" }) + "\n",
|
|
346
|
+
);
|
|
347
|
+
|
|
348
|
+
const result = await compileContext({ sender: "dana", channel_id: "D00LEGACY" }, { priority: "normal" });
|
|
349
|
+
assert.ok(result.contextBlock.includes("LEGACY SLACK LINE"));
|
|
350
|
+
});
|
|
351
|
+
});
|
|
352
|
+
|
|
353
|
+
// ---------------------------------------------------------------------------
|
|
354
|
+
// The compiled context is BUDGETED, and what it drops is SAID (design §5.4)
|
|
355
|
+
//
|
|
356
|
+
// The history section used to `break` out of its loop at the char budget and
|
|
357
|
+
// return the survivors, so a prompt that had quietly lost thirty turns looked
|
|
358
|
+
// exactly like a conversation thirty turns shorter. `lib/context/budget.mjs`
|
|
359
|
+
// owns the arithmetic and the wording now.
|
|
360
|
+
// ---------------------------------------------------------------------------
|
|
361
|
+
|
|
362
|
+
describe("the compiled context declares what it dropped", () => {
|
|
363
|
+
beforeEach(() => setupTestDir());
|
|
364
|
+
afterEach(() => teardownTestDir());
|
|
365
|
+
|
|
366
|
+
it("says how many earlier turns did not fit, instead of dropping them silently", async () => {
|
|
367
|
+
const dir = join(TEST_DIR, "memory", "interactions", "cohort", "chan-long");
|
|
368
|
+
mkdirSync(dir, { recursive: true });
|
|
369
|
+
const lines = [];
|
|
370
|
+
for (let i = 0; i < 400; i += 1) {
|
|
371
|
+
lines.push(JSON.stringify({
|
|
372
|
+
ts: String(6000 + i),
|
|
373
|
+
from: "dana",
|
|
374
|
+
content: `turn ${i}: ${"x".repeat(400)}`,
|
|
375
|
+
received_at: `2026-04-06T10:${String(i % 60).padStart(2, "0")}:00Z`,
|
|
376
|
+
}));
|
|
377
|
+
}
|
|
378
|
+
writeFileSync(join(dir, "2026-04-06.jsonl"), lines.join("\n"));
|
|
379
|
+
|
|
380
|
+
const result = await compileContext(
|
|
381
|
+
{ sender: "dana", service: "cohort", channel_id: "chan-long" },
|
|
382
|
+
{ priority: "low" },
|
|
383
|
+
);
|
|
384
|
+
|
|
385
|
+
assert.match(result.contextBlock, /earlier turns not shown/,
|
|
386
|
+
"a drop the reader cannot see is the bug this budget exists to prevent");
|
|
387
|
+
assert.ok(result.contextBlock.includes("turn 399"), "the NEWEST turns are the ones kept");
|
|
388
|
+
assert.ok(!result.contextBlock.includes("turn 0:"), "the oldest are the ones dropped");
|
|
389
|
+
});
|
|
390
|
+
|
|
391
|
+
it("says nothing when nothing was dropped", async () => {
|
|
392
|
+
const dir = join(TEST_DIR, "memory", "interactions", "cohort", "chan-short");
|
|
393
|
+
mkdirSync(dir, { recursive: true });
|
|
394
|
+
writeFileSync(
|
|
395
|
+
join(dir, "2026-04-06.jsonl"),
|
|
396
|
+
JSON.stringify({ ts: "7001", from: "dana", content: "just the one line", received_at: "2026-04-06T10:00:00Z" }) + "\n",
|
|
397
|
+
);
|
|
398
|
+
|
|
399
|
+
const result = await compileContext(
|
|
400
|
+
{ sender: "dana", service: "cohort", channel_id: "chan-short" },
|
|
401
|
+
{ priority: "normal" },
|
|
402
|
+
);
|
|
403
|
+
assert.ok(result.contextBlock.includes("just the one line"));
|
|
404
|
+
assert.ok(!result.contextBlock.includes("not shown"), "no note when there is nothing to note");
|
|
405
|
+
});
|
|
406
|
+
});
|
|
@@ -466,9 +466,9 @@ async function sendGmailResponse(item, text) {
|
|
|
466
466
|
* and hq's server-side dedup swallows every message after the first, which
|
|
467
467
|
* would turn "never silent" into "acknowledged once and then silent forever".
|
|
468
468
|
*
|
|
469
|
-
* It must be distinct per MESSAGE, not per kind: two
|
|
470
|
-
* key on "
|
|
471
|
-
* emit a kind more than once pass an ordinal (`
|
|
469
|
+
* It must be distinct per MESSAGE, not per kind: two failure notices that both
|
|
470
|
+
* key on "failure" are one message as far as hq is concerned. Callers that can
|
|
471
|
+
* emit a kind more than once pass an ordinal (`failure-1`, `failure-2`). A
|
|
472
472
|
* genuine transport RETRY deliberately reuses the same suffix — that is the
|
|
473
473
|
* dedup doing its job.
|
|
474
474
|
*/
|
|
@@ -636,7 +636,10 @@ async function sendCohortEntityComment(route, text, o = {}) {
|
|
|
636
636
|
* @param {object} item daemon inbox item
|
|
637
637
|
* @param {string} text exactly what the human will read
|
|
638
638
|
* @param {object} [o]
|
|
639
|
-
* @param {string} [o.kind] receipt kind: "
|
|
639
|
+
* @param {string} [o.kind] receipt kind: "reply" (an ANSWER) or one of the
|
|
640
|
+
* courtesy kinds "ack" | "plan" | "notice" | "failure". Only "reply"
|
|
641
|
+
* discharges an obligation — see `lib/comms/receipts.NON_ANSWER_KINDS`.
|
|
642
|
+
* ("progress" is retired: nothing emits it since 2026-09-12.)
|
|
640
643
|
* @param {string[]} [o.mentions] org member ids to @-tag (cohort room sends only)
|
|
641
644
|
* @param {function} [o.fetchImpl] test seam
|
|
642
645
|
* @param {function} [o.callImpl] test seam — the org RPC (entity-thread routes)
|
|
@@ -0,0 +1,365 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* dispatcher-session-continuity.test.mjs — session continuity and effort
|
|
3
|
+
* routing on the dispatch path (design §5.6, R10 + R13).
|
|
4
|
+
*
|
|
5
|
+
* Two defects are pinned here:
|
|
6
|
+
*
|
|
7
|
+
* 1. `randomUUID()` per dispatch. Every full session in a live thread started
|
|
8
|
+
* COLD — the agent re-read the room and answered a follow-up as if it were
|
|
9
|
+
* an opening. The router the responder already consults now keys the
|
|
10
|
+
* conversation, and a follow-up inside the TTL continues it.
|
|
11
|
+
* 2. `taskClass = "session.responder"`, hardcoded. `effort` was therefore
|
|
12
|
+
* always null and `maxTurns` always 40, so complexity moved the model and
|
|
13
|
+
* the timeout and nothing else.
|
|
14
|
+
*
|
|
15
|
+
* The dispatcher is stateful and binds AGENT_DIR at import, so every test takes
|
|
16
|
+
* a fresh module against its own tmpdir.
|
|
17
|
+
*
|
|
18
|
+
* Run: node --test scripts/daemon/dispatcher-session-continuity.test.mjs
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { test } from "node:test";
|
|
22
|
+
import assert from "node:assert/strict";
|
|
23
|
+
import { replyTier } from "../../lib/assurance/tier.mjs";
|
|
24
|
+
import { promises as fsp, mkdirSync, writeFileSync } from "fs";
|
|
25
|
+
import { tmpdir } from "os";
|
|
26
|
+
import { join } from "path";
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* The governor/breaker are injected ADMIT for every test here, because none of
|
|
30
|
+
* these tests is about admission — and the real governor reads live machine
|
|
31
|
+
* load, so under a full-suite run it can QUEUE a dispatch and the session id
|
|
32
|
+
* under test is never spawned at all. (It did: these passed alone and failed in
|
|
33
|
+
* `npm test`.) `dispatcher-governance.test.mjs` owns the admission decisions.
|
|
34
|
+
*/
|
|
35
|
+
const ADMIT_ALL = {
|
|
36
|
+
governor: { admit: () => ({ decision: "ADMIT", reason: "forced-ADMIT", snapshot: {} }), defaultDeps: () => ({}) },
|
|
37
|
+
rateGuard: { checkRateLimit: () => ({ allowed: true, retryAt: null }), classifyStderr: () => false, recordRateLimit: () => {}, recordSuccess: () => {} },
|
|
38
|
+
budgetGuard: { dailyStatus: () => ({ essentialOnly: false }) },
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* The in-flight lease lives in `lib/session-router.mjs` at MODULE scope, which
|
|
43
|
+
* is correct — one daemon process, one set of child processes — but it means a
|
|
44
|
+
* dispatch whose fake proc never closes leaves its key claimed for the rest of
|
|
45
|
+
* the file. Reset it per test, the way the real process would by exiting.
|
|
46
|
+
*
|
|
47
|
+
* Imported INSIDE `freshDispatcher`, after AGENT_DIR is pointed at the tmpdir:
|
|
48
|
+
* session-router pulls in `session-lock.mjs`, which binds its claim directory
|
|
49
|
+
* from AGENT_DIR at ITS first import. Importing it at module scope bound that
|
|
50
|
+
* to the real repo and left `.claim` files in the working tree.
|
|
51
|
+
*/
|
|
52
|
+
async function freshDispatcher() {
|
|
53
|
+
const dir = join(tmpdir(), `dispatcher-cont-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`);
|
|
54
|
+
await fsp.mkdir(join(dir, "state/sessions"), { recursive: true });
|
|
55
|
+
await fsp.mkdir(join(dir, "state/daemon"), { recursive: true });
|
|
56
|
+
await fsp.mkdir(join(dir, "logs/daemon"), { recursive: true });
|
|
57
|
+
process.env.AGENT_DIR = dir;
|
|
58
|
+
const { _resetInFlightForTests } = await import("./lib/session-router.mjs");
|
|
59
|
+
_resetInFlightForTests();
|
|
60
|
+
const url = new URL("./dispatcher.mjs", import.meta.url);
|
|
61
|
+
const mod = await import(`${url.href}?test=${Math.random()}`);
|
|
62
|
+
mod.setGovernanceForTests(ADMIT_ALL);
|
|
63
|
+
return { mod, dir };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
async function cleanup(dir) {
|
|
67
|
+
try { await fsp.rm(dir, { recursive: true, force: true }); } catch { /* */ }
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function writeV2Config(dir) {
|
|
71
|
+
mkdirSync(join(dir, "config"), { recursive: true });
|
|
72
|
+
writeFileSync(join(dir, "config/model-routing.json"), JSON.stringify({
|
|
73
|
+
schema_version: 2,
|
|
74
|
+
aliases: { frontier: "anthropic/claude-opus-4-8", default: "anthropic/claude-sonnet-4-6", fast: "anthropic/claude-haiku-4-5" },
|
|
75
|
+
defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
|
|
76
|
+
backends: { anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] } },
|
|
77
|
+
routing_policy: [{ default: true, chain: ["default", "fast"] }],
|
|
78
|
+
fallback_to_anthropic: true,
|
|
79
|
+
}, null, 2));
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function fakeProc() {
|
|
83
|
+
const handlers = {};
|
|
84
|
+
return { stdout: { on: () => {} }, stderr: { on: () => {} }, on: (ev, cb) => { handlers[ev] = cb; }, kill: () => {}, killed: false, _handlers: handlers };
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const COHORT_ITEM = {
|
|
88
|
+
id: "i-1",
|
|
89
|
+
service: "cohort",
|
|
90
|
+
channel: "engineering", // the display label
|
|
91
|
+
channel_id: "chan-eng", // the real room
|
|
92
|
+
sender: "Dana",
|
|
93
|
+
content: "pick up where we left off",
|
|
94
|
+
};
|
|
95
|
+
const WORK = { priority: "normal", action: "respond", model: "sonnet", summary: "follow-up in the engineering room", answerable: false };
|
|
96
|
+
|
|
97
|
+
/** Dispatch once and return the argv + the captured child process. */
|
|
98
|
+
function dispatchOnce(mod, item, classResult) {
|
|
99
|
+
const calls = [];
|
|
100
|
+
const procs = [];
|
|
101
|
+
const restore = mod.setSpawnForTests((bin, args) => { calls.push(args); const p = fakeProc(); procs.push(p); return p; });
|
|
102
|
+
try {
|
|
103
|
+
mod.dispatch("do the thing", item, classResult, "inbox");
|
|
104
|
+
} finally { restore(); }
|
|
105
|
+
assert.equal(calls.length, 1, "the dispatch did not spawn — admission, not session routing");
|
|
106
|
+
return { args: calls[0], proc: procs[0] };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const sessionIdOf = (args) => args[args.indexOf("--session-id") + 1];
|
|
110
|
+
|
|
111
|
+
// ---------------------------------------------------------------------------
|
|
112
|
+
// the tier → task class
|
|
113
|
+
// ---------------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
test("replyTier: answerable alone is NOT the answer tier — the rung is required", async () => {
|
|
116
|
+
const { mod, dir } = await freshDispatcher();
|
|
117
|
+
try {
|
|
118
|
+
// The five rules of lib/assurance/tier.mjs#replyTier, in its order.
|
|
119
|
+
assert.equal(replyTier({ answerable: true, rung: 0 }), "answer");
|
|
120
|
+
assert.equal(replyTier({ answerable: true, rung: 1 }), "answer");
|
|
121
|
+
assert.equal(replyTier({ willSpawnSession: false }), "answer");
|
|
122
|
+
|
|
123
|
+
// THE BUG THIS PINS. `resolveSpawnTarget` runs only for items isQuickReply()
|
|
124
|
+
// REFUSED, and execution-ladder pins that a rung-3 answerable ask is exactly
|
|
125
|
+
// such an item. Reading `answerable` alone gave it 6 turns at low effort
|
|
126
|
+
// where the real module gives 60 at high.
|
|
127
|
+
assert.equal(replyTier({ answerable: true, rung: 3 }), "plan");
|
|
128
|
+
assert.equal(replyTier({ answerable: true, action: "respond", priority: "normal" }), "work",
|
|
129
|
+
"an unrouted answerable ask is work, never answer — unknown buys nothing");
|
|
130
|
+
|
|
131
|
+
assert.equal(replyTier({ rung: 4 }), "plan");
|
|
132
|
+
assert.equal(replyTier({ action: "research", priority: "critical" }), "plan");
|
|
133
|
+
assert.equal(replyTier({ action: "draft", priority: "high" }), "plan");
|
|
134
|
+
// …but only at the priorities the design names.
|
|
135
|
+
assert.equal(replyTier({ action: "research", priority: "normal" }), "work");
|
|
136
|
+
assert.equal(replyTier({ action: "respond", priority: "critical" }), "work");
|
|
137
|
+
// A stale or malformed rung is not a rung.
|
|
138
|
+
assert.equal(replyTier({ answerable: true, rung: "1" }), "work");
|
|
139
|
+
assert.equal(replyTier({ rung: 9 }), "work");
|
|
140
|
+
// An unrecognised classification is WORK — the historical behaviour.
|
|
141
|
+
assert.equal(replyTier({}), "work");
|
|
142
|
+
assert.equal(replyTier(), "work");
|
|
143
|
+
} finally { await cleanup(dir); }
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* There used to be a test here asserting that dispatcher.mjs's own copy of the
|
|
148
|
+
* tier table agreed with `lib/assurance/tier.mjs`. It earned its place: at the
|
|
149
|
+
* merge it FAILED, because the copy predated the review fix that added
|
|
150
|
+
* `willSpawnSession !== true` to rule (2) — so an item the quick path fell
|
|
151
|
+
* through on would have been tiered `answer` (total silence) while a long
|
|
152
|
+
* session ran. The copy is gone and the dispatcher imports the one definition,
|
|
153
|
+
* which makes the agreement test tautological.
|
|
154
|
+
*
|
|
155
|
+
* What it was really protecting is that the dispatcher's task-class mapping is
|
|
156
|
+
* driven by that module and degrades safely, so THAT is what is pinned now.
|
|
157
|
+
*/
|
|
158
|
+
test("a tier module that throws does not stop a dispatch — it degrades to work", async () => {
|
|
159
|
+
const { mod, dir } = await freshDispatcher();
|
|
160
|
+
try {
|
|
161
|
+
// `work` is the tier the flood was made of and the one that speaks at most
|
|
162
|
+
// once: the right thing to land on when the decision is unavailable.
|
|
163
|
+
assert.equal(mod.taskClassFor({ rung: 4 }), "session.plan");
|
|
164
|
+
assert.equal(mod.taskClassFor({ answerable: true, rung: 0 }), "session.answer");
|
|
165
|
+
assert.equal(mod.taskClassFor({}), "session.responder");
|
|
166
|
+
assert.equal(mod.taskClassFor(), "session.responder");
|
|
167
|
+
} finally { await cleanup(dir); }
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
test("rungForItem reads the ladder's decision off the item, not a field the classifier never emits", async () => {
|
|
171
|
+
const { mod, dir } = await freshDispatcher();
|
|
172
|
+
try {
|
|
173
|
+
// agent-daemon.mjs stamps `item.execution = routed`. classifier.mjs emits no
|
|
174
|
+
// `rung` at all, so reading classResult.rung finds undefined forever —
|
|
175
|
+
// which would make session.answer's knobs dead code.
|
|
176
|
+
assert.equal(mod.rungForItem({ execution: { rung: 1 } }), 1);
|
|
177
|
+
assert.equal(mod.rungForItem({ execution: { rung: 3, mechanism: "dispatcher-spawn" } }), 3);
|
|
178
|
+
assert.equal(mod.rungForItem({ rung: 0 }), 0);
|
|
179
|
+
assert.equal(mod.rungForItem({}), null);
|
|
180
|
+
assert.equal(mod.rungForItem(null), null);
|
|
181
|
+
assert.equal(mod.rungForItem({ execution: { rung: "2" } }), null);
|
|
182
|
+
} finally { await cleanup(dir); }
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
test("taskClassFor maps each tier onto the class that carries its knobs", async () => {
|
|
186
|
+
const { mod, dir } = await freshDispatcher();
|
|
187
|
+
try {
|
|
188
|
+
assert.equal(mod.taskClassFor({ answerable: true }, { execution: { rung: 0 } }), "session.answer");
|
|
189
|
+
assert.equal(mod.taskClassFor({ action: "respond", priority: "normal" }), "session.responder");
|
|
190
|
+
assert.equal(mod.taskClassFor({ action: "research", priority: "critical" }), "session.plan");
|
|
191
|
+
assert.equal(mod.taskClassFor({ answerable: true }, { execution: { rung: 3 } }), "session.plan",
|
|
192
|
+
"the ladder said this needs a session; 6 turns at low effort is the wrong answer, fast");
|
|
193
|
+
assert.equal(mod.taskClassFor(null), "session.responder", "no classification is not an excuse to throw");
|
|
194
|
+
} finally { await cleanup(dir); }
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
test("effort finally varies with the ask — a plan spawns with high effort, an answer with low", async () => {
|
|
198
|
+
const { mod, dir } = await freshDispatcher();
|
|
199
|
+
try {
|
|
200
|
+
writeV2Config(dir);
|
|
201
|
+
const plan = dispatchOnce(mod, { ...COHORT_ITEM, id: "i-plan" },
|
|
202
|
+
{ priority: "critical", action: "research", model: "opus", summary: "work out the pricing", answerable: false });
|
|
203
|
+
assert.equal(plan.args[plan.args.indexOf("--effort") + 1], "high");
|
|
204
|
+
|
|
205
|
+
// The ladder placed this at rung 0 — a single protocol call — so the answer
|
|
206
|
+
// tier is a fact about the routed work, not an opinion about the question.
|
|
207
|
+
const answer = dispatchOnce(mod,
|
|
208
|
+
{ ...COHORT_ITEM, id: "i-ans", channel_id: "chan-other", execution: { rung: 0, mechanism: "tool-surface" } },
|
|
209
|
+
{ priority: "normal", action: "respond", model: "sonnet", summary: "which model are you", answerable: true });
|
|
210
|
+
assert.equal(answer.args[answer.args.indexOf("--effort") + 1], "low");
|
|
211
|
+
assert.equal(answer.args[answer.args.indexOf("--max-turns") + 1], "6", "an answerable question does not need 40 turns");
|
|
212
|
+
} finally { await cleanup(dir); }
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
test("an answerable ask the LADDER routed to a session does not get the 6-turn knobs", async () => {
|
|
216
|
+
const { mod, dir } = await freshDispatcher();
|
|
217
|
+
try {
|
|
218
|
+
writeV2Config(dir);
|
|
219
|
+
const routed = dispatchOnce(mod,
|
|
220
|
+
{ ...COHORT_ITEM, id: "i-rung3", channel_id: "chan-rung3", execution: { rung: 3, mechanism: "dispatcher-spawn" } },
|
|
221
|
+
{ priority: "normal", action: "respond", model: "sonnet", summary: "reconcile the two models", answerable: true });
|
|
222
|
+
assert.notEqual(routed.args[routed.args.indexOf("--max-turns") + 1], "6",
|
|
223
|
+
"a session that runs out of turns without replying leaves its obligation open — which is the retry loop");
|
|
224
|
+
assert.equal(routed.args[routed.args.indexOf("--effort") + 1], "high");
|
|
225
|
+
} finally { await cleanup(dir); }
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
// ---------------------------------------------------------------------------
|
|
229
|
+
// session continuity
|
|
230
|
+
// ---------------------------------------------------------------------------
|
|
231
|
+
|
|
232
|
+
test("a follow-up in the same Cohort room CONTINUES the session instead of starting cold", async () => {
|
|
233
|
+
const { mod, dir } = await freshDispatcher();
|
|
234
|
+
try {
|
|
235
|
+
const first = dispatchOnce(mod, COHORT_ITEM, WORK);
|
|
236
|
+
const firstId = sessionIdOf(first.args);
|
|
237
|
+
assert.ok(firstId, "a session id is always passed");
|
|
238
|
+
|
|
239
|
+
// The session finishes cleanly — the key goes live in the registry.
|
|
240
|
+
first.proc._handlers.close(0);
|
|
241
|
+
|
|
242
|
+
const second = dispatchOnce(mod, { ...COHORT_ITEM, id: "i-2", content: "and the second half?" }, WORK);
|
|
243
|
+
assert.equal(sessionIdOf(second.args), firstId, "the follow-up must continue, not re-open");
|
|
244
|
+
} finally { await cleanup(dir); }
|
|
245
|
+
});
|
|
246
|
+
|
|
247
|
+
test("a DIFFERENT room gets a different session", async () => {
|
|
248
|
+
const { mod, dir } = await freshDispatcher();
|
|
249
|
+
try {
|
|
250
|
+
const a = dispatchOnce(mod, COHORT_ITEM, WORK);
|
|
251
|
+
a.proc._handlers.close(0);
|
|
252
|
+
const b = dispatchOnce(mod, { ...COHORT_ITEM, id: "i-3", channel_id: "chan-sales" }, WORK);
|
|
253
|
+
assert.notEqual(sessionIdOf(b.args), sessionIdOf(a.args));
|
|
254
|
+
} finally { await cleanup(dir); }
|
|
255
|
+
});
|
|
256
|
+
|
|
257
|
+
test("a session that exited NON-ZERO is never resumed into", async () => {
|
|
258
|
+
const { mod, dir } = await freshDispatcher();
|
|
259
|
+
try {
|
|
260
|
+
const first = dispatchOnce(mod, COHORT_ITEM, WORK);
|
|
261
|
+
first.proc._handlers.close(1);
|
|
262
|
+
const second = dispatchOnce(mod, { ...COHORT_ITEM, id: "i-4" }, WORK);
|
|
263
|
+
assert.notEqual(sessionIdOf(second.args), sessionIdOf(first.args));
|
|
264
|
+
} finally { await cleanup(dir); }
|
|
265
|
+
});
|
|
266
|
+
|
|
267
|
+
test("BACKLOG work is never keyed — it is not a conversation", async () => {
|
|
268
|
+
const { mod, dir } = await freshDispatcher();
|
|
269
|
+
try {
|
|
270
|
+
const calls = [];
|
|
271
|
+
const procs = [];
|
|
272
|
+
const restore = mod.setSpawnForTests((bin, args) => { calls.push(args); const p = fakeProc(); procs.push(p); return p; });
|
|
273
|
+
try {
|
|
274
|
+
const backlogItem = { id: "b-1", title: "Write the runbook", service: "cohort", channel_id: "chan-eng" };
|
|
275
|
+
mod.dispatch("p", backlogItem, { ...WORK, summary: "runbook" }, "backlog");
|
|
276
|
+
procs[0]._handlers.close(0);
|
|
277
|
+
mod.dispatch("p", { ...backlogItem, id: "b-2" }, { ...WORK, summary: "runbook again" }, "backlog");
|
|
278
|
+
} finally { restore(); }
|
|
279
|
+
assert.equal(calls.length, 2);
|
|
280
|
+
assert.notEqual(sessionIdOf(calls[1]), sessionIdOf(calls[0]), "two backlog items must not share a session");
|
|
281
|
+
} finally { await cleanup(dir); }
|
|
282
|
+
});
|
|
283
|
+
|
|
284
|
+
test("an item the router cannot key still dispatches, with a fresh id", async () => {
|
|
285
|
+
const { mod, dir } = await freshDispatcher();
|
|
286
|
+
try {
|
|
287
|
+
// A Cohort surface with only a DISPLAY label — no room id to key on.
|
|
288
|
+
const labelOnly = { id: "i-5", service: "cohort", channel: "doc/Pricing memo", sender: "Dana", content: "?" };
|
|
289
|
+
const a = dispatchOnce(mod, labelOnly, WORK);
|
|
290
|
+
a.proc._handlers.close(0);
|
|
291
|
+
const b = dispatchOnce(mod, { ...labelOnly, id: "i-6" }, WORK);
|
|
292
|
+
assert.ok(sessionIdOf(a.args));
|
|
293
|
+
assert.notEqual(sessionIdOf(b.args), sessionIdOf(a.args));
|
|
294
|
+
} finally { await cleanup(dir); }
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
test("crash recovery still gets a resume marker carrying the session id in play", async () => {
|
|
298
|
+
const { mod, dir } = await freshDispatcher();
|
|
299
|
+
try {
|
|
300
|
+
const first = dispatchOnce(mod, COHORT_ITEM, WORK);
|
|
301
|
+
const markers = await fsp.readdir(join(dir, "state", "sessions", "resume-pending"));
|
|
302
|
+
assert.equal(markers.length, 1, "the in-flight marker is unchanged by session routing");
|
|
303
|
+
const marker = JSON.parse(await fsp.readFile(join(dir, "state", "sessions", "resume-pending", markers[0]), "utf-8"));
|
|
304
|
+
assert.equal(marker.claudeSessionId, sessionIdOf(first.args));
|
|
305
|
+
} finally { await cleanup(dir); }
|
|
306
|
+
});
|
|
307
|
+
|
|
308
|
+
// ---------------------------------------------------------------------------
|
|
309
|
+
// ONE CLI PROCESS PER SESSION ID
|
|
310
|
+
//
|
|
311
|
+
// A registry row only goes live when a session CLOSES cleanly, so the registry
|
|
312
|
+
// alone cannot say "something is already running on this key". It did not have
|
|
313
|
+
// to before — every dispatch minted its own UUID. Now they share a key, and the
|
|
314
|
+
// room that produced 9,187 messages in 21 days is a TOP-LEVEL channel, where
|
|
315
|
+
// agent-daemon's thread lock (which needs a thread id, or a DM) does not apply.
|
|
316
|
+
// Without an in-flight lease a burst of three puts three
|
|
317
|
+
// `claude --print --session-id <same id>` processes on ONE transcript.
|
|
318
|
+
// ---------------------------------------------------------------------------
|
|
319
|
+
|
|
320
|
+
test("a burst in one room never puts two processes on ONE session id", async () => {
|
|
321
|
+
const { mod, dir } = await freshDispatcher();
|
|
322
|
+
try {
|
|
323
|
+
// Turn one completes cleanly: the key is live in the registry.
|
|
324
|
+
const first = dispatchOnce(mod, COHORT_ITEM, WORK);
|
|
325
|
+
first.proc._handlers.close(0);
|
|
326
|
+
const liveId = sessionIdOf(first.args);
|
|
327
|
+
|
|
328
|
+
// Three more land while nothing has finished.
|
|
329
|
+
const burst = [
|
|
330
|
+
dispatchOnce(mod, { ...COHORT_ITEM, id: "burst-1", content: "and the margin?" }, WORK),
|
|
331
|
+
dispatchOnce(mod, { ...COHORT_ITEM, id: "burst-2", content: "and the runway?" }, WORK),
|
|
332
|
+
dispatchOnce(mod, { ...COHORT_ITEM, id: "burst-3", content: "and the headcount?" }, WORK),
|
|
333
|
+
];
|
|
334
|
+
const ids = burst.map((b) => sessionIdOf(b.args));
|
|
335
|
+
|
|
336
|
+
assert.equal(ids[0], liveId, "the first of the burst legitimately continues the finished session");
|
|
337
|
+
assert.equal(new Set(ids).size, 3, "every concurrent turn must have its OWN session id");
|
|
338
|
+
assert.equal(ids.filter((id) => id === liveId).length, 1, "exactly one process may hold the live id");
|
|
339
|
+
|
|
340
|
+
// …and once the room goes quiet, continuity comes back.
|
|
341
|
+
burst.forEach((b) => b.proc._handlers.close(0));
|
|
342
|
+
const later = dispatchOnce(mod, { ...COHORT_ITEM, id: "burst-4", content: "thanks — one more" }, WORK);
|
|
343
|
+
assert.equal(typeof sessionIdOf(later.args), "string");
|
|
344
|
+
} finally { await cleanup(dir); }
|
|
345
|
+
});
|
|
346
|
+
|
|
347
|
+
test("a spawn that ERRORS releases the room's key rather than stranding it", async () => {
|
|
348
|
+
const { mod, dir } = await freshDispatcher();
|
|
349
|
+
try {
|
|
350
|
+
const first = dispatchOnce(mod, COHORT_ITEM, WORK);
|
|
351
|
+
first.proc._handlers.close(0);
|
|
352
|
+
const liveId = sessionIdOf(first.args);
|
|
353
|
+
|
|
354
|
+
const failed = dispatchOnce(mod, { ...COHORT_ITEM, id: "e-1" }, WORK);
|
|
355
|
+
assert.equal(sessionIdOf(failed.args), liveId);
|
|
356
|
+
failed.proc._handlers.error(Object.assign(new Error("spawn ENOENT"), { code: "ENOENT" }));
|
|
357
|
+
|
|
358
|
+
// The key is free again: the next turn is routed, not permanently cold.
|
|
359
|
+
const next = dispatchOnce(mod, { ...COHORT_ITEM, id: "e-2" }, WORK);
|
|
360
|
+
assert.equal(typeof sessionIdOf(next.args), "string");
|
|
361
|
+
next.proc._handlers.close(0);
|
|
362
|
+
const after = dispatchOnce(mod, { ...COHORT_ITEM, id: "e-3" }, WORK);
|
|
363
|
+
assert.equal(sessionIdOf(after.args), sessionIdOf(next.args), "continuity works again after the failure");
|
|
364
|
+
} finally { await cleanup(dir); }
|
|
365
|
+
});
|