@cohortapp/agent-sdk 2.5.1 → 2.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +305 -89
- package/bin/maestro.test.mjs +357 -48
- package/docs/runbooks/backup-restore.md +65 -33
- package/framework-features.json +4 -4
- package/lib/backup/policy.mjs +710 -0
- package/lib/backup/policy.test.mjs +305 -0
- package/lib/budget-escalate.mjs +133 -0
- package/lib/budget-escalate.test.mjs +232 -0
- package/lib/budget-guard.envelope.test.mjs +476 -0
- package/lib/budget-guard.mjs +853 -75
- package/lib/budget-guard.test.mjs +91 -42
- package/lib/cadences.mjs +33 -0
- package/lib/channels/orgmail/adapter.mjs +88 -3
- package/lib/channels/orgmail/adapter.test.mjs +137 -0
- package/lib/channels/repeat-suppressor.mjs +198 -0
- package/lib/channels/repeat-suppressor.test.mjs +134 -0
- package/lib/comms/receipts.mjs +297 -0
- package/lib/cost/ledger-row.mjs +333 -0
- package/lib/cost/ledger-row.test.mjs +183 -0
- package/lib/execution/drive.mjs +28 -1
- package/lib/execution/effects.mjs +191 -12
- package/lib/execution/effects.test.mjs +50 -11
- package/lib/goals/admission.mjs +13 -1
- package/lib/goals/admission.test.mjs +26 -1
- package/lib/goals/loop.mjs +13 -0
- package/lib/kpi-sensors.test.mjs +3 -0
- package/lib/mandate/cache.mjs +13 -5
- package/lib/mandate/derive.mjs +146 -21
- package/lib/mandate/derive.test.mjs +50 -6
- package/lib/mandate/model.mjs +32 -4
- package/lib/mandate/refresh.test.mjs +16 -2
- package/lib/mcp/server.test.mjs +12 -3
- package/lib/model-router/economics.mjs +107 -76
- package/lib/model-router/economics.test.mjs +64 -46
- package/lib/model-router/integration-coverage.test.mjs +39 -37
- package/lib/model-router/ledger.mjs +75 -22
- package/lib/model-router/ledger.test.mjs +35 -2
- package/lib/org/client.mjs +14 -0
- package/lib/org/cost-sync.mjs +16 -2
- package/lib/org/doctor.mjs +62 -1
- package/lib/org/doctor.test.mjs +36 -3
- package/lib/org/email-remedy.mjs +49 -0
- package/lib/org/engagement-ledger.mjs +376 -0
- package/lib/org/engagement-ledger.test.mjs +112 -0
- package/lib/org/engagement.mjs +1056 -0
- package/lib/org/engagement.test.mjs +739 -0
- package/lib/org/messaging.mjs +230 -3
- package/lib/org/messaging.test.mjs +110 -1
- package/lib/org/param-contract.mjs +56 -2
- package/lib/org/param-contract.test.mjs +26 -0
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +5 -0
- package/lib/org/protocol.test.mjs +7 -1
- package/lib/org/tool-surface.mjs +506 -10
- package/lib/org/tool-surface.test.mjs +191 -7
- package/lib/org/ui-parity.mjs +333 -6
- package/lib/org/ui-parity.test.mjs +96 -3
- package/lib/org/work-ledger.mjs +241 -0
- package/lib/org/work-ledger.test.mjs +237 -0
- package/lib/plan/adoption-e2e.test.mjs +366 -0
- package/lib/plan/budget-enforcement.test.mjs +400 -0
- package/lib/plan/budget-runtime.mjs +215 -0
- package/lib/plan/compile.mjs +201 -5
- package/lib/plan/compile.test.mjs +19 -5
- package/lib/plan/emit.mjs +8 -0
- package/lib/plan/emit.test.mjs +18 -0
- package/lib/resource-governor.mjs +58 -12
- package/lib/resource-governor.test.mjs +41 -1
- package/lib/security/audit-engine.mjs +45 -8
- package/lib/security/audit-engine.test.mjs +35 -0
- package/lib/setup/enroll-from-cohort.mjs +14 -1
- package/lib/setup/sections/mandate.mjs +48 -7
- package/lib/setup/sections/mandate.test.mjs +17 -2
- package/lib/setup/sections/orgmail.mjs +10 -2
- package/lib/setup/state.mjs +83 -2
- package/lib/telemetry/collect.mjs +360 -20
- package/lib/telemetry/collect.test.mjs +266 -0
- package/package.json +1 -1
- package/scripts/cost/track-claude-usage.mjs +207 -48
- package/scripts/cost/track-claude-usage.test.mjs +148 -0
- package/scripts/daemon/agent-daemon.mjs +315 -17
- package/scripts/daemon/assurance-e2e.test.mjs +421 -0
- package/scripts/daemon/assurance.mjs +944 -0
- package/scripts/daemon/assurance.test.mjs +668 -0
- package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
- package/scripts/daemon/cadence-consumer.mjs +147 -9
- package/scripts/daemon/cadence-consumer.test.mjs +6 -0
- package/scripts/daemon/cadence-handlers.mjs +158 -0
- package/scripts/daemon/cadence-handlers.test.mjs +64 -0
- package/scripts/daemon/classifier.test.mjs +18 -9
- package/scripts/daemon/deliver.mjs +314 -0
- package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
- package/scripts/daemon/dispatcher.mjs +64 -6
- package/scripts/daemon/responder-cost.test.mjs +68 -0
- package/scripts/daemon/responder.mjs +351 -298
- package/scripts/local-triggers/generate-plists.test.mjs +7 -4
- package/scripts/maintenance/backup-run.mjs +415 -0
- package/scripts/maintenance/backup-to-cloud.sh +16 -116
- package/scripts/org/send-orgmail.mjs +16 -0
- package/scripts/record-receipt.sh +63 -0
- package/scripts/restore-from-backup.sh +14 -3
- package/scripts/restore-from-backup.test.mjs +8 -5
- package/scripts/send-email-threaded.py +47 -0
- package/scripts/send-sms.sh +4 -0
- package/scripts/send-whatsapp.sh +4 -0
- package/scripts/setup/init-backup.mjs +93 -38
- package/scripts/slack-send.sh +12 -0
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* scripts/cost/track-claude-usage.test.mjs — the record contract every spawn
|
|
3
|
+
* path depends on.
|
|
4
|
+
*
|
|
5
|
+
* The defect these lock down: `Number(flags["input-tokens"] || 0)` turned an
|
|
6
|
+
* ABSENT token flag into a literal 0, so "we could not measure this session"
|
|
7
|
+
* and "this session was free" wrote identical rows. Every reader summed both as
|
|
8
|
+
* $0, and the budget governor could never trip.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import test from "node:test";
|
|
12
|
+
import assert from "node:assert/strict";
|
|
13
|
+
import { spawnSync } from "node:child_process";
|
|
14
|
+
import { mkdtempSync, rmSync, readFileSync, existsSync } from "node:fs";
|
|
15
|
+
import { tmpdir } from "node:os";
|
|
16
|
+
import { join } from "node:path";
|
|
17
|
+
import { fileURLToPath } from "node:url";
|
|
18
|
+
|
|
19
|
+
const TRACKER = fileURLToPath(new URL("./track-claude-usage.mjs", import.meta.url));
|
|
20
|
+
|
|
21
|
+
function withRoot(fn) {
|
|
22
|
+
const root = mkdtempSync(join(tmpdir(), "track-usage-"));
|
|
23
|
+
try { return fn(root); } finally { rmSync(root, { recursive: true, force: true }); }
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function record(root, args) {
|
|
27
|
+
const r = spawnSync(process.execPath, [TRACKER, "record", ...args], {
|
|
28
|
+
encoding: "utf-8",
|
|
29
|
+
env: { ...process.env, AGENT_ROOT: root, AGENT_DIR: root },
|
|
30
|
+
});
|
|
31
|
+
assert.equal(r.status, 0, `tracker exited ${r.status}: ${r.stderr}`);
|
|
32
|
+
return r;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function rows(root) {
|
|
36
|
+
const date = new Date().toISOString().slice(0, 10);
|
|
37
|
+
const file = join(root, "state/cost-tracking", `${date}.jsonl`);
|
|
38
|
+
if (!existsSync(file)) return [];
|
|
39
|
+
return readFileSync(file, "utf-8").split("\n").filter(Boolean).map((l) => JSON.parse(l));
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
test("record: real usage is priced cache-aware and keeps the authoritative cost", () => {
|
|
43
|
+
withRoot((root) => {
|
|
44
|
+
// The live token profile: almost all prompt tokens are cache reads. The old
|
|
45
|
+
// per-1K, cache-blind table priced this session at $0.075 against a real
|
|
46
|
+
// $0.533 — a 7x under-read on a single row.
|
|
47
|
+
record(root, [
|
|
48
|
+
"--cadence", "inbox", "--source", "dispatcher", "--model", "sonnet",
|
|
49
|
+
"--input-tokens", "16", "--output-tokens", "5000",
|
|
50
|
+
"--cache-read-tokens", "553181", "--cache-creation-tokens", "78000",
|
|
51
|
+
"--total-cost-usd", "0.533412", "--exit", "0",
|
|
52
|
+
]);
|
|
53
|
+
const [row] = rows(root);
|
|
54
|
+
assert.equal(row.measurement, "measured");
|
|
55
|
+
assert.equal(row.input_tokens, 16);
|
|
56
|
+
assert.equal(row.output_tokens, 5000);
|
|
57
|
+
assert.equal(row.cache_read_input_tokens, 553181);
|
|
58
|
+
assert.equal(row.cache_creation_input_tokens, 78000);
|
|
59
|
+
assert.equal(row.total_cost_usd, 0.533412);
|
|
60
|
+
// The estimate must now land in the same ballpark as the CLI's own figure.
|
|
61
|
+
// (Pre-fix it was 0.075048 — the cache tiers were simply not priced.)
|
|
62
|
+
const drift = Math.abs(row.estimated_usd - row.total_cost_usd) / row.total_cost_usd;
|
|
63
|
+
assert.ok(drift < 0.05, `cache-aware estimate should track the authoritative cost, drift=${drift}`);
|
|
64
|
+
});
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test("record: an unmeasured session writes NULL tokens, never a zero", () => {
|
|
68
|
+
withRoot((root) => {
|
|
69
|
+
const r = record(root, [
|
|
70
|
+
"--cadence", "inbox", "--source", "dispatcher",
|
|
71
|
+
"--tokens-unknown", "no-usage-field",
|
|
72
|
+
]);
|
|
73
|
+
const [row] = rows(root);
|
|
74
|
+
assert.equal(row.measurement, "unknown");
|
|
75
|
+
assert.equal(row.input_tokens, null, "NOT 0 — a zero here is what blinded the governor");
|
|
76
|
+
assert.equal(row.output_tokens, null);
|
|
77
|
+
assert.equal(row.estimated_usd, null, "no cost claim at all");
|
|
78
|
+
assert.equal(row.total_cost_usd, null);
|
|
79
|
+
assert.equal(row.unmeasured_reason, "no-usage-field");
|
|
80
|
+
// And it must be loud: this is the only place that observes the gap at
|
|
81
|
+
// write time, and every bug in this system has been a quiet catch.
|
|
82
|
+
assert.match(r.stderr, /UNMEASURED session recorded/);
|
|
83
|
+
assert.match(r.stderr, /unknown, not zero/);
|
|
84
|
+
});
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
test("record: a caller that passes NO token flags degrades to unknown, not $0", () => {
|
|
88
|
+
// An un-migrated caller must fail loudly rather than silently claim a free
|
|
89
|
+
// session — this is the exact shape of the original bug.
|
|
90
|
+
withRoot((root) => {
|
|
91
|
+
record(root, ["--cadence", "inbox", "--source", "dispatcher", "--model", "sonnet"]);
|
|
92
|
+
const [row] = rows(root);
|
|
93
|
+
assert.equal(row.measurement, "unknown");
|
|
94
|
+
assert.equal(row.input_tokens, null);
|
|
95
|
+
assert.equal(row.unmeasured_reason, "caller-passed-no-token-counts");
|
|
96
|
+
});
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
test("record: a genuine zero-token session stays MEASURED and free", () => {
|
|
100
|
+
withRoot((root) => {
|
|
101
|
+
record(root, [
|
|
102
|
+
"--cadence", "inbox", "--source", "dispatcher", "--model", "sonnet",
|
|
103
|
+
"--input-tokens", "0", "--output-tokens", "0", "--total-cost-usd", "0",
|
|
104
|
+
]);
|
|
105
|
+
const [row] = rows(root);
|
|
106
|
+
assert.equal(row.measurement, "measured", "we looked; the answer was zero");
|
|
107
|
+
assert.equal(row.input_tokens, 0);
|
|
108
|
+
assert.equal(row.estimated_usd, 0);
|
|
109
|
+
});
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
test("record: --non-llm marks attribution rows that never invoked a model", () => {
|
|
113
|
+
withRoot((root) => {
|
|
114
|
+
record(root, ["--cadence", "messaging", "--source", "messaging", "--non-llm"]);
|
|
115
|
+
const [row] = rows(root);
|
|
116
|
+
assert.equal(row.measurement, "n/a");
|
|
117
|
+
assert.equal(row.model, null);
|
|
118
|
+
assert.equal(row.estimated_usd, 0, "$0 is a fact here, not a measurement gap");
|
|
119
|
+
});
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
test("summarise: reports unmeasured sessions instead of averaging them away", () => {
|
|
123
|
+
withRoot((root) => {
|
|
124
|
+
record(root, [
|
|
125
|
+
"--source", "dispatcher", "--model", "sonnet",
|
|
126
|
+
"--input-tokens", "16", "--output-tokens", "5000",
|
|
127
|
+
"--cache-read-tokens", "553181", "--total-cost-usd", "0.5",
|
|
128
|
+
]);
|
|
129
|
+
record(root, ["--source", "dispatcher", "--tokens-unknown", "empty-stdout"]);
|
|
130
|
+
record(root, ["--source", "messaging", "--non-llm"]);
|
|
131
|
+
|
|
132
|
+
const r = spawnSync(process.execPath, [TRACKER, "summarise", "--days", "1"], {
|
|
133
|
+
encoding: "utf-8",
|
|
134
|
+
env: { ...process.env, AGENT_ROOT: root, AGENT_DIR: root },
|
|
135
|
+
});
|
|
136
|
+
assert.equal(r.status, 0, r.stderr);
|
|
137
|
+
const out = JSON.parse(r.stdout);
|
|
138
|
+
assert.equal(out.measurement.measured_sessions, 1);
|
|
139
|
+
assert.equal(out.measurement.unmeasured_sessions, 1);
|
|
140
|
+
assert.equal(out.measurement.non_llm_rows, 1, "a send is not a session");
|
|
141
|
+
assert.equal(out.measurement.authoritative_usd, 0.5);
|
|
142
|
+
assert.equal(out.totals.sessions, 2, "measured + unmeasured; the send is excluded");
|
|
143
|
+
assert.ok(
|
|
144
|
+
out.measurement.degradations.some((d) => /NO token counts/.test(d)),
|
|
145
|
+
"the gap is named, not swallowed"
|
|
146
|
+
);
|
|
147
|
+
});
|
|
148
|
+
});
|
|
@@ -54,6 +54,22 @@ import { classifyItem, isDirectedAtAgent } from "./classifier.mjs";
|
|
|
54
54
|
import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions } from "./dispatcher.mjs";
|
|
55
55
|
import { buildPrompt } from "./prompt-builder.mjs";
|
|
56
56
|
import { sendQuickResponse, sendHoldingMessage, isQuickReply } from "./responder.mjs";
|
|
57
|
+
// ANSWER ASSURANCE (scripts/daemon/assurance.mjs). The guarantee that an ask is
|
|
58
|
+
// always answered by SOMETHING: an instant model-free acknowledgement when a
|
|
59
|
+
// session is about to make the human wait, an interim update when the work
|
|
60
|
+
// outruns its welcome, a message on every terminal failure, and a durable
|
|
61
|
+
// on-disk obligation that a sweep discharges when this process cannot. It is
|
|
62
|
+
// imported here because this file is where every one of those moments happens —
|
|
63
|
+
// the dispatch decision, the session close, and the tick loop.
|
|
64
|
+
import {
|
|
65
|
+
shouldAcknowledge,
|
|
66
|
+
openAndAcknowledge,
|
|
67
|
+
noteSession,
|
|
68
|
+
settleSession,
|
|
69
|
+
sweepObligations,
|
|
70
|
+
openObligations,
|
|
71
|
+
escalate,
|
|
72
|
+
} from "./assurance.mjs";
|
|
57
73
|
import { recordPoll, recordClassification, recordSession, writeHealthDashboard } from "./health.mjs";
|
|
58
74
|
import { acquireLock, releaseLock, updateLock, scanStaleLocks, acquireThreadLock, claimRequest, hasActiveClaim, sweepStaleItemClaims, sanitiseItemId } from "./session-lock.mjs";
|
|
59
75
|
import { markDeferred } from "./inbox-deferral.mjs";
|
|
@@ -69,7 +85,8 @@ const RUNG_MECHANISM = Object.fromEntries(RUNGS.map((r) => [r.id, r.mechanism]))
|
|
|
69
85
|
// local SQLite/interaction logs remain the fail-open cache.
|
|
70
86
|
import { isEnabled as orgEnabled, loadOrgConfig } from "../../lib/org/client.mjs";
|
|
71
87
|
import { remember as orgRemember } from "../../lib/org/knowledge.mjs";
|
|
72
|
-
import { sweepSessionOutcomes } from "./session-outcomes.mjs";
|
|
88
|
+
import { sweepSessionOutcomes, resultTextFromStdout } from "./session-outcomes.mjs";
|
|
89
|
+
import { recordWorkStep as orgRecordWorkStep } from "../../lib/org/work-ledger.mjs";
|
|
73
90
|
// Observability spine (WS — diagnostics). mintTraceId + withTrace give each
|
|
74
91
|
// inbound item ONE trace_id that deep callees inherit via AsyncLocalStorage;
|
|
75
92
|
// emitEvent appends the canonical interaction hops (item_received → … → sent /
|
|
@@ -518,6 +535,27 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
518
535
|
const result = await _sendQuickResponse(item, classResult, routed);
|
|
519
536
|
if (result.sent) {
|
|
520
537
|
markProcessed(item, service);
|
|
538
|
+
// WORK VISIBILITY on the FAST path too, with `deferred: false`. This is
|
|
539
|
+
// what stops the two planes drifting: an ask answered inside the turn
|
|
540
|
+
// gets a row ONLY if its text asks for something to be DONE, and that
|
|
541
|
+
// judgement is made by the server's gate — the same gate, the same
|
|
542
|
+
// answer, whether the agent is driven from here or from hq. Most
|
|
543
|
+
// quick replies leave no board row, which is correct: the answer IS the
|
|
544
|
+
// artifact, and a board full of answered questions is a board nobody
|
|
545
|
+
// reads.
|
|
546
|
+
// `accepted`, not `done`. The quick path posted a reply and nothing else,
|
|
547
|
+
// and the server gate only tracks a `deferred:false` ask when its text
|
|
548
|
+
// ASKED FOR WORK — exactly the case where a reply is not the work. A row
|
|
549
|
+
// created and closed in the same instant claims the ask was delivered by
|
|
550
|
+
// a message; leaving it open is the truthful reading, and the stable
|
|
551
|
+
// askKey means whatever really finishes it closes THIS row.
|
|
552
|
+
void trackWorkStep({
|
|
553
|
+
item,
|
|
554
|
+
stage: "accepted",
|
|
555
|
+
deferred: false,
|
|
556
|
+
note: "Answered directly in the conversation.",
|
|
557
|
+
source: { service, path: "quick_reply" },
|
|
558
|
+
});
|
|
521
559
|
return { ok: true, path: "quick_reply", reason: classResult.model };
|
|
522
560
|
}
|
|
523
561
|
// If quick reply failed to send or was blocked by validation, fall through to dispatch a full session
|
|
@@ -525,20 +563,80 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
525
563
|
console.warn(`[daemon] Quick reply not sent (${reason}), falling through to session dispatch`);
|
|
526
564
|
}
|
|
527
565
|
|
|
528
|
-
// COMPLEX WORK PATH
|
|
566
|
+
// COMPLEX WORK PATH — reaching this line IS the decision that the answer is
|
|
567
|
+
// not arriving in this turn. Everything above either answered the human
|
|
568
|
+
// (quick reply) or declined to; from here a session spawns and the human
|
|
569
|
+
// waits minutes. So this is exactly where an acknowledgement is owed, and
|
|
570
|
+
// the only place it can be decided from fact rather than guess.
|
|
571
|
+
//
|
|
572
|
+
// THE OLD GATE ASKED THE CLASSIFIER: `action === respond|draft|research`.
|
|
573
|
+
// That silently excluded `queue` — which is what classifier.mjs's HEURISTIC
|
|
574
|
+
// FALLBACK emits when the LLM classifier itself fails. So precisely when the
|
|
575
|
+
// system was degraded, a directed DM got a 15-45 minute session and no
|
|
576
|
+
// acknowledgement whatsoever. `shouldAcknowledge` asks about control flow
|
|
577
|
+
// instead: a session is spawning and a human is on the other end of a
|
|
578
|
+
// resolvable channel, therefore say something now.
|
|
579
|
+
//
|
|
580
|
+
// The OBLIGATION is opened before the send and outlives this process. If the
|
|
581
|
+
// acknowledgement fails to go out, the debt is still on disk and the
|
|
582
|
+
// assurance sweep retries it — the old code caught that failure, logged it,
|
|
583
|
+
// and left the human staring at a typing indicator.
|
|
529
584
|
let holdingText = null;
|
|
585
|
+
let obligationKeyForItem = null;
|
|
586
|
+
const ackVerdict = shouldAcknowledge({ willSpawnSession: true, item, source: "inbox" });
|
|
530
587
|
try {
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
588
|
+
// The DEBT is opened whether or not we can speak. Those are two different
|
|
589
|
+
// questions and conflating them is what made a whole class of ask
|
|
590
|
+
// disappear: an item with no resolvable reply channel (calendar, voice, an
|
|
591
|
+
// inbound with no room) got no obligation, so when its session died the
|
|
592
|
+
// failure path had nothing to consult, marked the item processed on the
|
|
593
|
+
// FIRST failure, said nothing, and escalated nothing. The ask was gone.
|
|
594
|
+
// Now the record always exists — `ack:false` merely means the compensating
|
|
595
|
+
// action is an operator escalation instead of a message.
|
|
596
|
+
const opened = await openAndAcknowledge({
|
|
597
|
+
item,
|
|
598
|
+
classResult,
|
|
599
|
+
service,
|
|
600
|
+
traceId: trace_id,
|
|
601
|
+
ack: ackVerdict.ack,
|
|
602
|
+
deps: { ackSender: _sendHoldingMessage },
|
|
603
|
+
});
|
|
604
|
+
obligationKeyForItem = opened.key;
|
|
605
|
+
holdingText = opened.ackText;
|
|
606
|
+
if (opened.acked) {
|
|
607
|
+
updateLock(itemId, { holdingSent: true });
|
|
608
|
+
} else if (ackVerdict.ack) {
|
|
609
|
+
// NOT fatal, NOT silent, NOT forgotten. Loud here; retried by the sweep.
|
|
610
|
+
console.warn(`[daemon] acknowledgement not delivered for ${itemId} (${opened.error}) — obligation ${opened.key} left open for the assurance sweep`);
|
|
611
|
+
counters.bump("assurance.ack_deferred", { service });
|
|
612
|
+
} else {
|
|
613
|
+
console.log(`[daemon] no acknowledgement for ${itemId} (${ackVerdict.reason}) — debt ${opened.key} still tracked`);
|
|
537
614
|
}
|
|
538
615
|
} catch (err) {
|
|
539
|
-
|
|
616
|
+
// openAndAcknowledge does not throw; if it somehow does, the debt may not
|
|
617
|
+
// exist, so this is the one case worth shouting about.
|
|
618
|
+
console.error(`[daemon] acknowledgement path threw for ${itemId}: ${err.message}`);
|
|
619
|
+
emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, stage: "acknowledge", error: err.message } });
|
|
540
620
|
}
|
|
541
621
|
|
|
622
|
+
// WORK VISIBILITY, step 1 of 3: the ask goes on the board BEFORE the work
|
|
623
|
+
// starts, not after it finishes. Reaching this line is the same fact that
|
|
624
|
+
// earned the acknowledgement above — the answer is not arriving this turn —
|
|
625
|
+
// so the board row and the "let me look into it" have ONE trigger between
|
|
626
|
+
// them. A human who is about to wait fifteen minutes gets something to look
|
|
627
|
+
// at for those fifteen minutes, with himself tagged on it.
|
|
628
|
+
//
|
|
629
|
+
// `deferred: true` is the whole signal the server's gate needs; everything
|
|
630
|
+
// else about whether this deserves a row is decided there, by the same code
|
|
631
|
+
// that decides it for hq's responder.
|
|
632
|
+
void trackWorkStep({
|
|
633
|
+
item,
|
|
634
|
+
stage: "accepted",
|
|
635
|
+
deferred: true,
|
|
636
|
+
note: classResult && classResult.summary ? `Picked this up. ${classResult.summary}` : "Picked this up — starting work now.",
|
|
637
|
+
source: { service, trace_id, classified: String(classResult && classResult.action) },
|
|
638
|
+
});
|
|
639
|
+
|
|
542
640
|
// Build prompt with holding message context and dispatch
|
|
543
641
|
const prompt = await buildPrompt(item, classResult, {
|
|
544
642
|
type: "inbox",
|
|
@@ -553,8 +651,38 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
553
651
|
// re-delivered on the next poll; a crash mid-flight leaves the admission for
|
|
554
652
|
// recoverInFlight() to re-deliver on restart.
|
|
555
653
|
recordInFlight(item, service);
|
|
654
|
+
if (obligationKeyForItem) noteSession(obligationKeyForItem, null);
|
|
556
655
|
_dispatch(prompt, item, classResult, "inbox", {
|
|
656
|
+
// Carried into the session child's env so its CLI send lanes stamp their
|
|
657
|
+
// delivery receipts with the debt they discharge. This is what lets the
|
|
658
|
+
// close handler below distinguish "the session answered him" from "the
|
|
659
|
+
// session exited 0 in silence" — and, just as importantly, from "a
|
|
660
|
+
// DIFFERENT session answered a different ask in the same room".
|
|
661
|
+
obligationKey: obligationKeyForItem,
|
|
557
662
|
onClose: ({ ok, code, sessionId, stdout }) => {
|
|
663
|
+
// EVERY terminal state now ends in a decision about what the human
|
|
664
|
+
// hears. `settleSession` owns that decision and is the only thing here
|
|
665
|
+
// that can speak; the branches below only handle bookkeeping.
|
|
666
|
+
//
|
|
667
|
+
// It cannot be awaited (onClose is sync by contract), so it is a
|
|
668
|
+
// fire-and-forget promise with its own catch — but unlike the fail-open
|
|
669
|
+
// catches around it, a throw HERE means the human may be owed a message,
|
|
670
|
+
// so it is logged at error level and the obligation stays open for the
|
|
671
|
+
// sweep to pick up.
|
|
672
|
+
const settle = obligationKeyForItem
|
|
673
|
+
? Promise.resolve(settleSession({
|
|
674
|
+
key: obligationKeyForItem,
|
|
675
|
+
ok,
|
|
676
|
+
code,
|
|
677
|
+
sessionId,
|
|
678
|
+
stdout,
|
|
679
|
+
error: ok ? null : `session_close exit ${code}`,
|
|
680
|
+
})).catch((err) => {
|
|
681
|
+
console.error(`[daemon] settleSession threw for ${itemId}: ${err.message} — obligation left open for sweep`);
|
|
682
|
+
return { spoke: false, verdict: "settle_threw", willRetry: false };
|
|
683
|
+
})
|
|
684
|
+
: Promise.resolve({ spoke: false, verdict: "no-obligation", willRetry: false });
|
|
685
|
+
|
|
558
686
|
if (ok) {
|
|
559
687
|
markProcessed(item, service);
|
|
560
688
|
clearInFlight(item);
|
|
@@ -570,20 +698,87 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
570
698
|
// DAEMON_SESSION_OUTCOMES=1; idempotent + fail-open; fire-and-forget.
|
|
571
699
|
Promise.resolve(sweepOutcomesForSession({ sessionId, stdout, item, classResult }))
|
|
572
700
|
.catch(() => { /* fail-open */ });
|
|
701
|
+
// WORK VISIBILITY, step 2 of 3: the row moves to done and carries what
|
|
702
|
+
// the turn actually produced as a comment, so the board says what
|
|
703
|
+
// happened rather than merely that something did. The requester is
|
|
704
|
+
// tagged, so "it's finished" reaches him without him having to look.
|
|
705
|
+
void trackWorkStep({
|
|
706
|
+
item,
|
|
707
|
+
stage: "done",
|
|
708
|
+
deferred: true,
|
|
709
|
+
note: sessionCompletionNote(stdout),
|
|
710
|
+
source: { service, sessionId: String(sessionId || ""), exit_code: String(code) },
|
|
711
|
+
});
|
|
573
712
|
// Finalize (interaction end): emit `sent` carrying the SAME trace_id
|
|
574
713
|
// minted at item_received, so the full hop chain (item_received → … →
|
|
575
|
-
// sent) groups by one stable id.
|
|
576
|
-
//
|
|
577
|
-
//
|
|
578
|
-
|
|
714
|
+
// sent) groups by one stable id.
|
|
715
|
+
//
|
|
716
|
+
// `exit_code: 0` is NOT evidence that the human was answered. The old
|
|
717
|
+
// comment here asserted "the session itself posts the actual reply",
|
|
718
|
+
// and nothing ever checked — so a session that exited 0 having sent
|
|
719
|
+
// nothing (observed: 241s, $2.36, final text "Nothing was sent.") was
|
|
720
|
+
// marked processed forever and emitted as `sent`. The `spoke` attr now
|
|
721
|
+
// carries what settleSession actually established.
|
|
722
|
+
settle.then((s) => {
|
|
723
|
+
emitEvent({
|
|
724
|
+
type: EVENT_TYPES.SENT,
|
|
725
|
+
trace_id,
|
|
726
|
+
attrs: { item_id: itemId, service, exit_code: code, verdict: s.verdict, daemon_spoke: !!s.spoke },
|
|
727
|
+
});
|
|
728
|
+
if (s.verdict === "silent-success") {
|
|
729
|
+
counters.bump("assurance.silent_success_rescued", { service });
|
|
730
|
+
}
|
|
731
|
+
});
|
|
579
732
|
} else {
|
|
580
|
-
// Leave the inbox file un-`.processed` (re-deliverable). Drop the
|
|
581
|
-
// admission + release the per-item lock so the next poll re-delivers
|
|
582
|
-
// immediately rather than waiting out the stale-lock TTL.
|
|
583
733
|
clearInFlight(item);
|
|
584
734
|
const key = inflightKey(item);
|
|
585
735
|
if (key) releaseLock(key);
|
|
586
736
|
emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, exit_code: code, stage: "session_close" } });
|
|
737
|
+
// WORK VISIBILITY, step 3 of 3 — THE ONE THAT MATTERS. A session that
|
|
738
|
+
// dies used to leave exactly one observable: the typing dot stopping.
|
|
739
|
+
// The row now moves to BLOCKED (never quietly to done — the work is
|
|
740
|
+
// still owed to somebody), carries the exit code, and TAGS the
|
|
741
|
+
// requester. This is the difference between a process that failed and
|
|
742
|
+
// a process that failed invisibly.
|
|
743
|
+
void trackWorkStep({
|
|
744
|
+
item,
|
|
745
|
+
stage: "failed",
|
|
746
|
+
deferred: true,
|
|
747
|
+
note: `The session handling this ended without completing (exit ${code}). It is back on the board, blocked, rather than quietly dropped.`,
|
|
748
|
+
source: { service, sessionId: String(sessionId || ""), exit_code: String(code) },
|
|
749
|
+
});
|
|
750
|
+
// SELF-HEAL, BOUNDED. A transient failure with retries left leaves the
|
|
751
|
+
// inbox file un-`.processed` so the next poll re-delivers it — that was
|
|
752
|
+
// already true, but it was UNBOUNDED and SILENT: the same item could
|
|
753
|
+
// cycle through 45-minute sessions forever with the human hearing
|
|
754
|
+
// nothing. Now the requester is told a retry is running, and when the
|
|
755
|
+
// retries are spent the item is marked processed so the loop STOPS,
|
|
756
|
+
// the human is told it failed, and a durable needs-attention record is
|
|
757
|
+
// written. A dropped ask must leave a trace; it must not leave a loop.
|
|
758
|
+
settle.then((s) => {
|
|
759
|
+
if (!s.willRetry) {
|
|
760
|
+
// `no-obligation` reaches here only if the item had no stable key
|
|
761
|
+
// at all, so nothing durable is tracking it. Marking it processed
|
|
762
|
+
// would delete the ask on its FIRST failure with no message and no
|
|
763
|
+
// trace — the exact silent drop this whole unit exists to end. Write
|
|
764
|
+
// the escalation ourselves before the item goes.
|
|
765
|
+
if (s.verdict === "no-obligation") {
|
|
766
|
+
escalate(
|
|
767
|
+
{ key: itemId, sender: item.sender, service, channel: item.channel_id || item.channel || null,
|
|
768
|
+
summary: classResult && classResult.summary, openedAt: Date.now(),
|
|
769
|
+
attempts: 1, sessionId: String(sessionId || ""), traceId: trace_id,
|
|
770
|
+
item: { content: item.content } },
|
|
771
|
+
{ failure: { label: `exit_${code}` }, told: false },
|
|
772
|
+
);
|
|
773
|
+
}
|
|
774
|
+
markProcessed(item, service);
|
|
775
|
+
counters.bump("assurance.gave_up", { service, verdict: s.verdict });
|
|
776
|
+
console.error(`[daemon] ${itemId} failed terminally (${s.verdict}); requester told: ${!!s.spoke}`);
|
|
777
|
+
} else {
|
|
778
|
+
counters.bump("assurance.retrying", { service, verdict: s.verdict });
|
|
779
|
+
console.warn(`[daemon] ${itemId} failed (${s.verdict}); retrying — requester told: ${!!s.spoke}`);
|
|
780
|
+
}
|
|
781
|
+
});
|
|
587
782
|
}
|
|
588
783
|
},
|
|
589
784
|
});
|
|
@@ -1094,6 +1289,76 @@ export function _resetDaemonOrgCfgCache() {
|
|
|
1094
1289
|
_cachedDaemonOrgCfg = undefined;
|
|
1095
1290
|
}
|
|
1096
1291
|
|
|
1292
|
+
/** Longest completion note we will put on a board comment. */
|
|
1293
|
+
const COMPLETION_NOTE_CHARS = 1200;
|
|
1294
|
+
|
|
1295
|
+
/**
|
|
1296
|
+
* What the finished session actually produced, as a board comment.
|
|
1297
|
+
*
|
|
1298
|
+
* Reuses `session-outcomes.mjs#resultTextFromStdout` — the same tolerant
|
|
1299
|
+
* extractor the outcome sweep uses — rather than re-parsing the `claude --print`
|
|
1300
|
+
* envelope a second way. A turn that produced nothing readable still gets an
|
|
1301
|
+
* honest line: "finished" with no evidence is worse than nothing, because it is
|
|
1302
|
+
* the exact claim `exit 0` already made falsely.
|
|
1303
|
+
*
|
|
1304
|
+
* @param {string} stdout the session's raw stdout
|
|
1305
|
+
* @returns {string}
|
|
1306
|
+
*/
|
|
1307
|
+
export function sessionCompletionNote(stdout) {
|
|
1308
|
+
let text = "";
|
|
1309
|
+
try {
|
|
1310
|
+
text = String(resultTextFromStdout(stdout) || "").trim();
|
|
1311
|
+
} catch {
|
|
1312
|
+
text = "";
|
|
1313
|
+
}
|
|
1314
|
+
if (!text) return "The session finished but produced no readable result text.";
|
|
1315
|
+
const clipped = text.length > COMPLETION_NOTE_CHARS ? `${text.slice(0, COMPLETION_NOTE_CHARS)}…` : text;
|
|
1316
|
+
return `Done. What the work produced:\n\n${clipped}`;
|
|
1317
|
+
}
|
|
1318
|
+
|
|
1319
|
+
/**
|
|
1320
|
+
* WORK VISIBILITY. Record one step of the work this item kicked off against the
|
|
1321
|
+
* org board (`board.track` → hq's `server/work/ledger.ts`).
|
|
1322
|
+
*
|
|
1323
|
+
* DRIVEN, not left to the model. The three call sites below are the three
|
|
1324
|
+
* moments a human's experience changes — the ask is taken on, the work lands,
|
|
1325
|
+
* the work dies — and none of them may depend on a session remembering to write
|
|
1326
|
+
* something down. `DAEMON_SESSION_OUTCOMES` is the cautionary tale: a whole
|
|
1327
|
+
* outcome-discipline module, correct and tested, that has never run in
|
|
1328
|
+
* production because the flag is absent from the launchd plist. This has no
|
|
1329
|
+
* flag. The gate is on the SERVER, where it is one implementation shared with
|
|
1330
|
+
* hq's responder, and where it can be changed without redeploying a laptop.
|
|
1331
|
+
*
|
|
1332
|
+
* Fail-open and fire-and-forget: never awaited on the dispatch path, never able
|
|
1333
|
+
* to throw into it. But never silent either — the reason is always logged, which
|
|
1334
|
+
* is the rule this entire workstream exists to enforce.
|
|
1335
|
+
*
|
|
1336
|
+
* @param {object} a - { item, stage, deferred?, note?, title?, source? } plus
|
|
1337
|
+
* test seams ({ cfg?, trackImpl?, fetchImpl? })
|
|
1338
|
+
* @returns {Promise<object>} the step result (or a benign refusal)
|
|
1339
|
+
*/
|
|
1340
|
+
export async function trackWorkStep(a = {}) {
|
|
1341
|
+
try {
|
|
1342
|
+
const cfg = a.cfg !== undefined ? a.cfg : daemonOrgCfg();
|
|
1343
|
+
const res = await orgRecordWorkStep({ ...a, cfg });
|
|
1344
|
+
if (res && res.tracked) {
|
|
1345
|
+
console.log(
|
|
1346
|
+
`[board] ${a.stage} → task ${res.taskId} (${res.col})` +
|
|
1347
|
+
`${res.created ? " created" : ""}${res.moved ? " moved" : ""}` +
|
|
1348
|
+
`${res.commented ? " commented" : ""}${res.tagged ? ` tagged=${res.tagged}` : ""}`,
|
|
1349
|
+
);
|
|
1350
|
+
} else {
|
|
1351
|
+
// A refusal is INFORMATION, not noise: "no-substance" is the gate working,
|
|
1352
|
+
// "org-disabled" is configuration, and an error frame is a real fault.
|
|
1353
|
+
console.log(`[board] ${a.stage} not tracked: ${(res && res.reason) || "unknown"}`);
|
|
1354
|
+
}
|
|
1355
|
+
return res;
|
|
1356
|
+
} catch (err) {
|
|
1357
|
+
console.warn(`[board] track failed (${a.stage}): ${err && err.message}`);
|
|
1358
|
+
return { tracked: false, reason: "error" };
|
|
1359
|
+
}
|
|
1360
|
+
}
|
|
1361
|
+
|
|
1097
1362
|
// A distillation longer than this is truncated before it lands in the store —
|
|
1098
1363
|
// the shared memory holds a salient one-liner, never a transcript.
|
|
1099
1364
|
const SESSION_DISTILL_MAX_CHARS = 400;
|
|
@@ -1528,6 +1793,23 @@ async function main() {
|
|
|
1528
1793
|
// if the work had completed). Runs BEFORE the first poll so recovery wins the
|
|
1529
1794
|
// race against re-acquisition.
|
|
1530
1795
|
recoverInFlight();
|
|
1796
|
+
// ANSWER ASSURANCE on startup. `recoverInFlight` above re-delivers the WORK;
|
|
1797
|
+
// this speaks to the PEOPLE. Any obligation still open was opened by a daemon
|
|
1798
|
+
// process that no longer exists — meaning a human asked for something, was
|
|
1799
|
+
// told "I'll come back to you", and then this process died mid-session. They
|
|
1800
|
+
// are owed a word before anything else happens, and the sweep's interrupted
|
|
1801
|
+
// branch is what gives it to them. Awaited (unlike the tick-loop call) so the
|
|
1802
|
+
// first poll cannot spawn new work ahead of settling old debts.
|
|
1803
|
+
const owed = openObligations().length;
|
|
1804
|
+
if (owed > 0) {
|
|
1805
|
+
console.log(`[daemon] ${owed} unanswered obligation(s) survived the last run — settling before first poll`);
|
|
1806
|
+
try {
|
|
1807
|
+
const s = await sweepObligations();
|
|
1808
|
+
console.log(`[daemon] startup assurance sweep — acked:${s.acked} interrupted:${s.interrupted} stale:${s.staled} closed:${s.closed}`);
|
|
1809
|
+
} catch (err) {
|
|
1810
|
+
console.error("[daemon] startup assurance sweep failed:", err.message);
|
|
1811
|
+
}
|
|
1812
|
+
}
|
|
1531
1813
|
// WS4: on startup, reconcile in-flight session resume markers — re-dispatch
|
|
1532
1814
|
// `claude --print --resume` for any work that was mid-flight when the box
|
|
1533
1815
|
// rebooted/lost power (3-strike cap → status:blocked). Shutdown paths below
|
|
@@ -1558,7 +1840,14 @@ async function main() {
|
|
|
1558
1840
|
}
|
|
1559
1841
|
}, BACKLOG_INTERVAL);
|
|
1560
1842
|
|
|
1561
|
-
// Health dashboard + stale claim sweep
|
|
1843
|
+
// Health dashboard + stale claim sweep + the ANSWER-ASSURANCE sweep.
|
|
1844
|
+
//
|
|
1845
|
+
// The assurance sweep is the safety net under every other path: it is the one
|
|
1846
|
+
// mechanism with no in-memory state, so it works identically on the first tick
|
|
1847
|
+
// after a restart and the ten-thousandth tick of a healthy process. It is what
|
|
1848
|
+
// makes the promise hold when an acknowledgement's send failed, when a
|
|
1849
|
+
// dispatcher never called back, when a promise rejected unhandled, and when
|
|
1850
|
+
// the daemon itself died mid-session.
|
|
1562
1851
|
setInterval(() => {
|
|
1563
1852
|
try {
|
|
1564
1853
|
writeHealthDashboard();
|
|
@@ -1570,6 +1859,15 @@ async function main() {
|
|
|
1570
1859
|
} catch (err) {
|
|
1571
1860
|
console.error("[daemon] Health write error:", err.message);
|
|
1572
1861
|
}
|
|
1862
|
+
// Separate try: a health-write failure must not skip the sweep, because the
|
|
1863
|
+
// sweep is the thing that speaks to humans.
|
|
1864
|
+
Promise.resolve(sweepObligations())
|
|
1865
|
+
.then((s) => {
|
|
1866
|
+
if (s.acked || s.progressed || s.staled || s.interrupted) {
|
|
1867
|
+
console.log(`[daemon] assurance sweep — acked:${s.acked} progress:${s.progressed} stale:${s.staled} interrupted:${s.interrupted} closed:${s.closed} (open:${s.swept})`);
|
|
1868
|
+
}
|
|
1869
|
+
})
|
|
1870
|
+
.catch((err) => console.error("[daemon] assurance sweep error:", err.message));
|
|
1573
1871
|
}, HEALTH_INTERVAL);
|
|
1574
1872
|
|
|
1575
1873
|
// Graceful shutdown — clean up active.json so next startup doesn't see stale sessions
|