@cohortapp/agent-sdk 2.18.7 → 2.18.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/runbooks/fleet-rollout.md +30 -0
- package/lib/comms/send-gate.mjs +28 -4
- package/lib/daemon/reply-debt.mjs +105 -0
- package/lib/org/inbound/directedness.mjs +60 -0
- package/lib/org/inbound/facts.mjs +29 -4
- package/lib/org/inbound/project.mjs +38 -1
- package/lib/org/inbound/surfaces.mjs +68 -0
- package/lib/runtime/adapter.mjs +39 -4
- package/lib/session/first-run.mjs +213 -16
- package/lib/telemetry/alerts.mjs +159 -2
- package/lib/telemetry/collect.mjs +27 -1
- package/package.json +1 -1
- package/policies/ai-disclosure.yaml +47 -0
- package/scripts/daemon/agent-daemon.mjs +206 -13
- package/scripts/daemon/assurance.mjs +5 -0
- package/scripts/daemon/deliver.mjs +79 -11
- package/scripts/daemon/dispatcher.mjs +68 -26
- package/scripts/daemon/lib/session-router.mjs +21 -0
- package/scripts/daemon/responder.mjs +92 -17
- package/scripts/local-triggers/autoupdate.sh +56 -1
- package/scripts/session/supervisor.mjs +70 -13
|
@@ -63,6 +63,9 @@ import { classifyItem, isDirectedAtAgent } from "./classifier.mjs";
|
|
|
63
63
|
// gets a deterministic classification (no LLM triage), no holding ack, and a
|
|
64
64
|
// per-seat jitter so thirteen answers do not land in the same second.
|
|
65
65
|
import { isRollCallItem, rollCallJitterMs } from "../../lib/org/inbound/broadcast.mjs";
|
|
66
|
+
import { isRoomSurface } from "../../lib/org/inbound/surfaces.mjs";
|
|
67
|
+
import { isMembershipReason } from "../../lib/org/inbound/directedness.mjs";
|
|
68
|
+
import { cohortSurfaceOf, cohortSurfaceLabel, cohortSurfaceIsDeclared } from "./deliver.mjs";
|
|
66
69
|
import { itemCollectiveIntent } from "../../lib/org/inbound/collective.mjs";
|
|
67
70
|
import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions, yieldBacklogForReactive, linkActiveSessionBoard } from "./dispatcher.mjs";
|
|
68
71
|
import { boardItemRefFor } from "../../lib/session/current-work.mjs";
|
|
@@ -132,6 +135,9 @@ import { boardPriority } from "./board-mirror.mjs";
|
|
|
132
135
|
import { mintTraceId, withTrace } from "../../lib/diagnostics/trace.mjs";
|
|
133
136
|
import { emitEvent, EVENT_TYPES } from "../../lib/diagnostics/events.mjs";
|
|
134
137
|
import * as counters from "../../lib/diagnostics/counters.mjs";
|
|
138
|
+
// Replies this seat owed and did not send, counted where a human can see them
|
|
139
|
+
// (the presence beat) rather than only in a file on this disk.
|
|
140
|
+
import { REPLY_DEBT_COUNTERS, isScopeFault } from "../../lib/daemon/reply-debt.mjs";
|
|
135
141
|
import { getHookBus } from "../../lib/hooks/bus.mjs";
|
|
136
142
|
// EXECUTION LADDER (lib/execution/**). The reactive front of the chain:
|
|
137
143
|
// intake (what surface is this, and is it mine) → match (which compiled
|
|
@@ -499,18 +505,41 @@ function claudeAvailable() {
|
|
|
499
505
|
* `electedMe:false`. A degraded election must never fall back to respond-to-all —
|
|
500
506
|
* that is the storm this whole change exists to end.
|
|
501
507
|
*
|
|
508
|
+
* ── WHAT A REASON MEANS, AND WHY `decided` IS ON THE VERDICT ────────────────
|
|
509
|
+
* Every path below that is not an election result ends in silence too, because
|
|
510
|
+
* the fail-safe is inverted — but silence-because-hq-picked-someone-else and
|
|
511
|
+
* silence-because-the-election-could-not-run are different events, and for
|
|
512
|
+
* months the caller printed both as "not elected to respond". An operator
|
|
513
|
+
* reading that log saw a decision where there had been an outage, or a missing
|
|
514
|
+
* id, or an item that was never a message. So the verdict now carries `decided`:
|
|
515
|
+
* TRUE only when hq actually ran the election and named responders. The caller
|
|
516
|
+
* logs the two cases in different words.
|
|
517
|
+
*
|
|
502
518
|
* @param {object} item the (cohort) inbox item
|
|
503
519
|
* @param {object} [deps] { electResponder, orgCfg } injectable seams for tests
|
|
504
|
-
* @returns {Promise<{electedMe:boolean, reason:string, election?:string, mode?:string}>}
|
|
520
|
+
* @returns {Promise<{electedMe:boolean, decided:boolean, reason:string, election?:string, mode?:string}>}
|
|
505
521
|
*/
|
|
506
522
|
async function electResponderVerdict(item, deps = {}) {
|
|
507
523
|
const impl = deps.electResponder || electResponder;
|
|
508
524
|
try {
|
|
509
525
|
const cfg = deps.orgCfg !== undefined ? deps.orgCfg : daemonOrgCfg();
|
|
510
|
-
if (!orgEnabled(cfg))
|
|
526
|
+
if (!orgEnabled(cfg)) {
|
|
527
|
+
return { electedMe: false, decided: false, reason: "org integration disabled — no roster to elect against" };
|
|
528
|
+
}
|
|
511
529
|
const channelId = cohortChannelId(item);
|
|
512
530
|
const messageId = cohortMessageId(item);
|
|
513
|
-
|
|
531
|
+
// A ROOM ITEM MISSING ITS IDS IS A DEFECT, AND THE LOG MUST SAY SO. These
|
|
532
|
+
// two used to share the single reason "no-message-id", which was wrong on
|
|
533
|
+
// both counts: it named the message when the CHANNEL was the missing half,
|
|
534
|
+
// and it was the reason 23 doc comments printed in a day — items that have
|
|
535
|
+
// no channel id because they are not room traffic and never should have
|
|
536
|
+
// reached this call at all. That class is now excluded by the caller's
|
|
537
|
+
// surface gate, so anything landing here genuinely is a room item whose
|
|
538
|
+
// projection lost an id.
|
|
539
|
+
if (!channelId || !messageId) {
|
|
540
|
+
const missing = !channelId && !messageId ? "channel id and message id" : (!channelId ? "channel id" : "message id");
|
|
541
|
+
return { electedMe: false, decided: false, reason: `room item carries no ${missing} — the election cannot be asked` };
|
|
542
|
+
}
|
|
514
543
|
const conn = configFromAgent(cfg);
|
|
515
544
|
const res = await impl(
|
|
516
545
|
{ messageId, channelId },
|
|
@@ -518,9 +547,9 @@ async function electResponderVerdict(item, deps = {}) {
|
|
|
518
547
|
);
|
|
519
548
|
if (!res || res.error) {
|
|
520
549
|
const reason = res && res.error
|
|
521
|
-
?
|
|
522
|
-
: "no
|
|
523
|
-
return { electedMe: false, reason: reason.slice(0, 200) };
|
|
550
|
+
? `hq refused the election (${res.error.code || "error"}: ${res.error.message || ""})`
|
|
551
|
+
: "hq returned no election frame";
|
|
552
|
+
return { electedMe: false, decided: false, reason: reason.slice(0, 200) };
|
|
524
553
|
}
|
|
525
554
|
const out = res.result !== undefined ? res.result : res;
|
|
526
555
|
const responders = Array.isArray(out && out.responders) ? out.responders : [];
|
|
@@ -535,14 +564,40 @@ async function electResponderVerdict(item, deps = {}) {
|
|
|
535
564
|
const electedMe =
|
|
536
565
|
!!mySlug &&
|
|
537
566
|
responders.some((r) => String((r && r.slug) || "").trim().toLowerCase() === mySlug);
|
|
567
|
+
// NO SLUG ON RECORD is not an election result: hq may well have elected
|
|
568
|
+
// somebody, this seat simply cannot tell whether it was itself. Reporting it
|
|
569
|
+
// as "hq elected another seat" would be an assertion nothing checked.
|
|
570
|
+
if (!mySlug) {
|
|
571
|
+
return {
|
|
572
|
+
electedMe: false,
|
|
573
|
+
decided: false,
|
|
574
|
+
reason: "this seat has no org member slug on record — cannot tell whether it was elected",
|
|
575
|
+
election: out && out.election,
|
|
576
|
+
mode: out && out.mode,
|
|
577
|
+
};
|
|
578
|
+
}
|
|
579
|
+
const others = responders
|
|
580
|
+
.map((r) => String((r && r.slug) || "").trim())
|
|
581
|
+
.filter(Boolean)
|
|
582
|
+
.slice(0, 5)
|
|
583
|
+
.join(", ");
|
|
538
584
|
return {
|
|
539
585
|
electedMe,
|
|
540
|
-
|
|
586
|
+
decided: true,
|
|
587
|
+
reason: electedMe
|
|
588
|
+
? "hq elected this seat"
|
|
589
|
+
: others
|
|
590
|
+
? `hq elected ${others}`
|
|
591
|
+
: "hq elected nobody",
|
|
541
592
|
election: out && out.election,
|
|
542
593
|
mode: out && out.mode,
|
|
543
594
|
};
|
|
544
595
|
} catch (err) {
|
|
545
|
-
return {
|
|
596
|
+
return {
|
|
597
|
+
electedMe: false,
|
|
598
|
+
decided: false,
|
|
599
|
+
reason: `the election call threw (${err && err.message ? err.message : String(err)})`.slice(0, 200),
|
|
600
|
+
};
|
|
546
601
|
}
|
|
547
602
|
}
|
|
548
603
|
|
|
@@ -802,25 +857,136 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
802
857
|
// twelve seats silent on a message that named all thirteen. The inbound
|
|
803
858
|
// join proved the address (human author, my room, collective marker), so
|
|
804
859
|
// this is not ambient traffic and there is nothing to hold an election over.
|
|
860
|
+
//
|
|
861
|
+
// AND IT NEEDS A ROOM. `messaging.electResponder` takes
|
|
862
|
+
// `{messageId, channelId}`, loads that Message row and that channel's AI
|
|
863
|
+
// roster, and picks one of the seats IN THE ROOM. A document comment, a
|
|
864
|
+
// board comment, an approval or a decision names no channel and has no room
|
|
865
|
+
// roster — there is nobody to elect between. Asking the messaging election
|
|
866
|
+
// about one produced `no-message-id`, and because this gate's failure mode
|
|
867
|
+
// is inverted to fail-SILENT, "this is not a message" came out as "stay
|
|
868
|
+
// quiet": 23 doc comments dropped on this seat in a single day, each one
|
|
869
|
+
// logged as a decision not to answer a message.
|
|
870
|
+
//
|
|
871
|
+
// TWO SEPARATE QUESTIONS, AND THEY MUST NOT BE COLLAPSED INTO ONE. "Is this
|
|
872
|
+
// surface room traffic?" decides whether an election is the RIGHT mechanism;
|
|
873
|
+
// "was the address ambient?" decides whether one is NEEDED; "does the item
|
|
874
|
+
// name a channel?" decides whether one can be ASKED. A personal address on a
|
|
875
|
+
// non-room surface (I own the doc, I am the assignee, the approval is on my
|
|
876
|
+
// desk) skips the election because there is nothing to arbitrate. An AMBIENT
|
|
877
|
+
// address on a non-room surface still wants one — and gets one whenever the
|
|
878
|
+
// item names a room. `isRoomSurface` answers TRUE for an unknown name, so a
|
|
879
|
+
// new ambient channel surface keeps the storm gate by default.
|
|
805
880
|
const rollCall = isRollCallItem(item);
|
|
806
|
-
|
|
881
|
+
const surface = service === "cohort" ? cohortSurfaceOf(item) : "";
|
|
882
|
+
// The name a PERSON reads. `cohortSurfaceOf` answers for ROUTING, and its
|
|
883
|
+
// fallback for a legacy `cohort:<channelId>:<messageId>` raw_ref (or a bare
|
|
884
|
+
// `kind:"message"`, which is what `lib/org/messaging.mjs` stamps) is "dm" —
|
|
885
|
+
// correct as a route, a lie as a label, and it printed
|
|
886
|
+
// `Election on dm in channel/roadmap` for an ambient PUBLIC-channel message
|
|
887
|
+
// in this very lane's test output. `cohortSurfaceLabel` never guesses.
|
|
888
|
+
const surfaceLabel = service === "cohort" ? cohortSurfaceLabel(item) : "";
|
|
889
|
+
const surfaceDeclared = service === "cohort" ? cohortSurfaceIsDeclared(item) : false;
|
|
890
|
+
// THE ADDRESS THE SERVER PROVED OUTRANKS THE LOCAL PROSE HEURISTIC. The
|
|
891
|
+
// comment above says an @mention or a named address "skips the election
|
|
892
|
+
// entirely (`isDirectedAtAgent` already decided)" — but `isDirectedAtAgent`
|
|
893
|
+
// is a SLACK-shaped rule: it matches the seat's configured first name and
|
|
894
|
+
// Slack member id against the message prose, and knows nothing about a
|
|
895
|
+
// Cohort mention row. A Cohort @mention whose prose spells the seat
|
|
896
|
+
// differently (a display name, a surname, a slug) therefore read as
|
|
897
|
+
// UNDIRECTED and went to an election — one surface's gate judging another
|
|
898
|
+
// surface's item, which is the same shape as the doc-comment bug above.
|
|
899
|
+
// `priority_signals.mentions_agent` is the ingest layer's real verdict,
|
|
900
|
+
// stamped by `lib/org/inbound/project.mjs` only for a genuine dm / mention /
|
|
901
|
+
// named / direct address, so it is the one to trust here.
|
|
902
|
+
const provenAddress = service === "cohort" && item.priority_signals && item.priority_signals.mentions_agent === true;
|
|
903
|
+
const undirectedCohort =
|
|
904
|
+
service === "cohort" && !isDm && !rollCall && !provenAddress && !isDirectedAtAgent(item);
|
|
905
|
+
// AMBIENT vs PERSONAL, and whether an election can be ASKED at all. Three
|
|
906
|
+
// facts, kept separate because they answer different questions:
|
|
907
|
+
//
|
|
908
|
+
// `nonRoom` — the surface's reply is not a message posted into a room.
|
|
909
|
+
// `ambient` — the address was proved by MEMBERSHIP or VISIBILITY
|
|
910
|
+
// (`channel`, `participant`, `shared`), so every seat that
|
|
911
|
+
// shares the room or the share holds an equal claim. That is
|
|
912
|
+
// what an election is for.
|
|
913
|
+
// `roomToArbitrate` — the item names a real Cohort channel, so
|
|
914
|
+
// `messaging.electResponder` has a roster to elect from.
|
|
915
|
+
//
|
|
916
|
+
// ONLY THE COMBINATION `nonRoom && ambient && !roomToArbitrate` IS A DEAD
|
|
917
|
+
// END. An ambient non-room item that DOES name a room still goes to the
|
|
918
|
+
// election, exactly as it did before this change — a comment on a
|
|
919
|
+
// chat-attached file (`resolveChatFile` → `yes("file_comment","channel")`)
|
|
920
|
+
// carries its channel id, so hq can arbitrate the room's roster and one
|
|
921
|
+
// seat can answer. Excluding it unconditionally would have made that class
|
|
922
|
+
// permanently unanswerable by anybody, which is a behaviour change this
|
|
923
|
+
// lane never claimed and does not want: if hq cannot resolve the comment id
|
|
924
|
+
// as a Message the election errors, the inverted fail-safe fires, and the
|
|
925
|
+
// silence is the same one — so asking costs nothing and can only turn a
|
|
926
|
+
// never into an answer.
|
|
927
|
+
const nonRoom = undirectedCohort && !isRoomSurface(surface);
|
|
928
|
+
const ambient = undirectedCohort && isMembershipReason(item.direct_reason);
|
|
929
|
+
const roomToArbitrate = Boolean(cohortChannelId(item));
|
|
930
|
+
if (nonRoom && ambient && !roomToArbitrate) {
|
|
931
|
+
// AMBIENT, AND NO MECHANISM CAN REACH IT. A doc comment is addressed to
|
|
932
|
+
// this seat only because the file is SHARED with it — a visibility fact a
|
|
933
|
+
// whole channel can hold at once — but the reply is an RPC against a file
|
|
934
|
+
// id, there is no Message row and no channel id, and
|
|
935
|
+
// `messaging.electResponder` has nothing to be asked about. So the seat
|
|
936
|
+
// stays silent, and the log says THAT: a missing mechanism, not a
|
|
937
|
+
// decision. Arbitration for ambient non-room surfaces is a real gap; it
|
|
938
|
+
// needs a server-side election keyed on the entity, not a local guess.
|
|
939
|
+
console.log(`[daemon] No arbitration exists for an ambient ${surfaceLabel} item (addressed by ${item.direct_reason || "membership"}, it names no room, and the responder election is messaging-only) — staying silent (gap, not a decision; from ${item.sender})`);
|
|
940
|
+
logEvent("classifications", {
|
|
941
|
+
item_id: itemId,
|
|
942
|
+
sender: item.sender,
|
|
943
|
+
service,
|
|
944
|
+
surface,
|
|
945
|
+
surface_declared: surfaceDeclared,
|
|
946
|
+
skipped: true,
|
|
947
|
+
reason: `no_arbitration_for_ambient_surface: ${surface}`,
|
|
948
|
+
direct_reason: item.direct_reason || null,
|
|
949
|
+
summary: classResult.summary,
|
|
950
|
+
});
|
|
951
|
+
markProcessed(item, service);
|
|
952
|
+
return { ok: true, path: "filtered", reason: "no_arbitration_for_ambient_surface" };
|
|
953
|
+
} else if (nonRoom && !ambient) {
|
|
954
|
+
console.log(`[daemon] No election on a ${surfaceLabel} item — it is not room traffic and the inbound join already proved it is addressed here (${item.direct_reason || "addressed"}); handling it on its own surface (from ${item.sender})`);
|
|
955
|
+
} else if (undirectedCohort) {
|
|
807
956
|
const verdict = await _electResponderVerdict(item);
|
|
808
957
|
if (!verdict.electedMe) {
|
|
809
|
-
|
|
958
|
+
// TWO DIFFERENT EVENTS, TWO DIFFERENT SENTENCES. `decided` is true only
|
|
959
|
+
// when hq ran the election and named responders; everything else is the
|
|
960
|
+
// fail-safe firing on an outage or a malformed item, and calling that a
|
|
961
|
+
// decision is how an outage reads as normal operation in the log.
|
|
962
|
+
console.log(
|
|
963
|
+
verdict.decided
|
|
964
|
+
? `[daemon] Election on ${surfaceLabel} in ${item.channel}: ${verdict.reason} — this seat stays silent (from ${item.sender})`
|
|
965
|
+
: `[daemon] Election on ${surfaceLabel} in ${item.channel} could not be decided: ${verdict.reason} — staying silent (fail-safe, not a decision; from ${item.sender})`,
|
|
966
|
+
);
|
|
810
967
|
logEvent("classifications", {
|
|
811
968
|
item_id: itemId,
|
|
812
969
|
sender: item.sender,
|
|
813
970
|
service,
|
|
971
|
+
surface,
|
|
972
|
+
surface_declared: surfaceDeclared,
|
|
814
973
|
skipped: true,
|
|
815
|
-
reason:
|
|
974
|
+
reason: verdict.decided
|
|
975
|
+
? `election_elected_another: ${verdict.reason}`
|
|
976
|
+
: `election_undecided: ${verdict.reason}`,
|
|
977
|
+
election_decided: verdict.decided === true,
|
|
816
978
|
election: verdict.election || null,
|
|
817
979
|
mode: verdict.mode || null,
|
|
818
980
|
summary: classResult.summary,
|
|
819
981
|
});
|
|
820
982
|
markProcessed(item, service);
|
|
821
|
-
return {
|
|
983
|
+
return {
|
|
984
|
+
ok: true,
|
|
985
|
+
path: "filtered",
|
|
986
|
+
reason: verdict.decided ? "election_elected_another" : "election_undecided",
|
|
987
|
+
};
|
|
822
988
|
}
|
|
823
|
-
console.log(`[daemon] Election
|
|
989
|
+
console.log(`[daemon] Election on ${surfaceLabel} in ${item.channel}: ${verdict.reason} (${verdict.election || "elected"}, mode=${verdict.mode || "?"}) — answering ${item.sender}`);
|
|
824
990
|
}
|
|
825
991
|
|
|
826
992
|
// DIRECTED-MESSAGE GATE: In channels and group chats, only respond to
|
|
@@ -1046,6 +1212,32 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
1046
1212
|
// the forbidden one); the escalation is the trace that a human must act.
|
|
1047
1213
|
if (result.permanent) {
|
|
1048
1214
|
const cause = result.code || result.error || "permanent send failure";
|
|
1215
|
+
// A SCOPE REFUSAL IS A CONFIGURATION FAULT AND SOMEBODY ELSE'S TO FIX.
|
|
1216
|
+
// Everything below this line is correct and stays — the ask is not
|
|
1217
|
+
// lost, the obligation is durable, an operator can sweep
|
|
1218
|
+
// state/obligations/needs-attention. But all of it is on THIS DISK. The
|
|
1219
|
+
// measured case (2026-09-22, doc comment on a file this seat may not
|
|
1220
|
+
// comment on) produced three log lines and one JSON file, and nothing
|
|
1221
|
+
// off the box ever said the seat was missing a scope. A person would
|
|
1222
|
+
// have to already suspect this seat to go and look, which is the same
|
|
1223
|
+
// silence in a nicer folder.
|
|
1224
|
+
//
|
|
1225
|
+
// So a scope fault is also COUNTED, and the count rides the presence
|
|
1226
|
+
// beat (lib/telemetry/collect → machine.replyDebt → the
|
|
1227
|
+
// `seat_missing_scope` alert). Counting rather than sending: the only
|
|
1228
|
+
// channel we had was the forbidden one, and the seat cannot know which
|
|
1229
|
+
// other room would reach the right human. What it CAN do is stop being
|
|
1230
|
+
// the only thing that knows.
|
|
1231
|
+
//
|
|
1232
|
+
// Falling through to a session is still refused, for the reason below:
|
|
1233
|
+
// a denied session runs autonomously and improvises unrelated work into
|
|
1234
|
+
// a channel this seat was just told it may not write to.
|
|
1235
|
+
if (isScopeFault(result.code)) {
|
|
1236
|
+
counters.bump(REPLY_DEBT_COUNTERS.scopeRefused, {
|
|
1237
|
+
service, sender: item.sender || "unknown",
|
|
1238
|
+
channel: item.channel_id || item.channel || "none",
|
|
1239
|
+
});
|
|
1240
|
+
}
|
|
1049
1241
|
console.error(`[daemon] Quick reply PERMANENTLY failed for ${item.sender} (${cause}) — NOT spawning a session; opening obligation + escalating`);
|
|
1050
1242
|
let obligationKey = itemId;
|
|
1051
1243
|
try {
|
|
@@ -1061,6 +1253,7 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
1061
1253
|
channel: item.channel_id || item.channel || null,
|
|
1062
1254
|
summary: classResult && classResult.summary, openedAt: Date.now(),
|
|
1063
1255
|
attempts: 1, sessionId: null, traceId: trace_id, lastError: cause,
|
|
1256
|
+
configFault: isScopeFault(result.code) ? "missing-scope" : null,
|
|
1064
1257
|
item: { content: item.content } },
|
|
1065
1258
|
{ failure: { label: cause }, told: false },
|
|
1066
1259
|
);
|
|
@@ -1556,6 +1556,11 @@ export function escalate(rec, o = {}) {
|
|
|
1556
1556
|
failedAt: new Date(Number.isFinite(o.now) ? o.now : Date.now()).toISOString(),
|
|
1557
1557
|
cause: o.failure ? o.failure.label : (rec.lastError || "unknown"),
|
|
1558
1558
|
requesterWasTold: !!o.told,
|
|
1559
|
+
// WHO has to act. A transport failure is this seat's own problem and the
|
|
1560
|
+
// sweep retries it; a scope fault is a configuration the seat cannot
|
|
1561
|
+
// grant itself, so the row says so in a word an operator can grep rather
|
|
1562
|
+
// than leaving them to infer it from `cause`.
|
|
1563
|
+
configFault: rec.configFault || null,
|
|
1559
1564
|
attempts: rec.attempts || 0,
|
|
1560
1565
|
sessionId: rec.sessionId || null,
|
|
1561
1566
|
traceId: rec.traceId || null,
|
|
@@ -34,7 +34,7 @@ import { recordOutbound } from "../../lib/comms/receipts.mjs";
|
|
|
34
34
|
// Pure data (a frozen table, no imports of its own) — the surface vocabulary the
|
|
35
35
|
// inbound projection stamps into `raw_ref`. Imported rather than re-listed so a
|
|
36
36
|
// new surface cannot be added upstream without this file's switch noticing.
|
|
37
|
-
import { SURFACE_NAMES } from "../../lib/org/inbound/surfaces.mjs";
|
|
37
|
+
import { ROOM_SURFACES, SURFACE_NAMES } from "../../lib/org/inbound/surfaces.mjs";
|
|
38
38
|
|
|
39
39
|
const AGENT_REPO_DIR = process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
|
|
40
40
|
|
|
@@ -117,11 +117,17 @@ export function resolveSlackChannel(item) {
|
|
|
117
117
|
* correct case for it.
|
|
118
118
|
*/
|
|
119
119
|
|
|
120
|
-
/**
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
120
|
+
/**
|
|
121
|
+
* Surfaces that genuinely live in a Cohort room — these reply with a message.
|
|
122
|
+
*
|
|
123
|
+
* DERIVED, not re-typed. This was a hand-kept copy of five names, and a
|
|
124
|
+
* hand-kept copy is a list that can be forgotten: `broadcast` had to be added
|
|
125
|
+
* here by hand when the roll-call surface landed, and until it was, the reply
|
|
126
|
+
* had no transport, `canDeliverTo` said no, and the surface was dark at the last
|
|
127
|
+
* step. `surfaces.mjs` now carries `room` as a per-surface fact and this reads
|
|
128
|
+
* it, so a new room surface is routable the moment it is declared.
|
|
129
|
+
*/
|
|
130
|
+
const CHANNEL_SURFACES = ROOM_SURFACES;
|
|
125
131
|
|
|
126
132
|
/**
|
|
127
133
|
* Error frame codes a retry cannot change.
|
|
@@ -174,17 +180,77 @@ function entityFromRawRef(item) {
|
|
|
174
180
|
*/
|
|
175
181
|
export function cohortSurfaceOf(item) {
|
|
176
182
|
if (!item) return "";
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
// otherwise parse as a surface name and route a working DM to nowhere.
|
|
180
|
-
const m = /^cohort:([a-z_]+):/.exec(str(item.raw_ref));
|
|
181
|
-
if (m && SURFACE_NAMES.includes(m[1])) return m[1];
|
|
183
|
+
const declared = declaredSurface(item);
|
|
184
|
+
if (declared) return declared;
|
|
182
185
|
const kind = str(item.kind);
|
|
183
186
|
if (kind === "file_comment") return str(item.channel_id) ? "file_comment" : "doc_comment";
|
|
184
187
|
if (!kind || kind === "message") return "dm";
|
|
185
188
|
return kind;
|
|
186
189
|
}
|
|
187
190
|
|
|
191
|
+
/**
|
|
192
|
+
* The surface name `project.mjs` DECLARED in `raw_ref`, or "" when it declared
|
|
193
|
+
* none.
|
|
194
|
+
*
|
|
195
|
+
* Guarded against the LEGACY raw_ref shape, which is `cohort:<channelId>:
|
|
196
|
+
* <messageId>` — still on disk today, and still what `lib/org/messaging.mjs`
|
|
197
|
+
* stamps. An all-lowercase channel id would otherwise parse as a surface name
|
|
198
|
+
* and route a working DM to nowhere.
|
|
199
|
+
*/
|
|
200
|
+
function declaredSurface(item) {
|
|
201
|
+
const m = /^cohort:([a-z_]+):/.exec(str(item && item.raw_ref));
|
|
202
|
+
return m && SURFACE_NAMES.includes(m[1]) ? m[1] : "";
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* Did `raw_ref` actually DECLARE this item's surface, or was it inferred?
|
|
207
|
+
*
|
|
208
|
+
* Worth recording next to a logged surface: an inferred one is a guess made
|
|
209
|
+
* from `kind` / `channel_kind`, and an operator reading the row later should be
|
|
210
|
+
* able to tell the two apart without re-deriving it.
|
|
211
|
+
*
|
|
212
|
+
* @param {object} item
|
|
213
|
+
* @returns {boolean}
|
|
214
|
+
*/
|
|
215
|
+
export function cohortSurfaceIsDeclared(item) {
|
|
216
|
+
return declaredSurface(item) !== "";
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* How to NAME this item's surface in a line a person reads.
|
|
221
|
+
*
|
|
222
|
+
* ── WHY THIS IS NOT `cohortSurfaceOf` ──
|
|
223
|
+
* Because that function's job is ROUTING, and for routing its fallbacks are
|
|
224
|
+
* right: an item with the legacy `cohort:<channelId>:<messageId>` raw_ref, or
|
|
225
|
+
* bare `kind:"message"`, replies with a message posted into `channel_id`, which
|
|
226
|
+
* is precisely what the `dm` branch does. But "dm" is then a ROUTE, not a
|
|
227
|
+
* claim about where the message was written — and a log line that prints it
|
|
228
|
+
* tells an operator "dm" about a message in a PUBLIC channel. The lane that
|
|
229
|
+
* added these lines to make the log truthful hit exactly that in its own test
|
|
230
|
+
* output: `Election on dm in channel/roadmap`.
|
|
231
|
+
*
|
|
232
|
+
* So the label is derived separately and never guesses: a declared surface is
|
|
233
|
+
* used verbatim; otherwise `channel_kind` (which hq stamps on the narrow reader)
|
|
234
|
+
* names the room; otherwise the item's own `is_dm`; otherwise it says plainly
|
|
235
|
+
* that the surface was never declared, rather than inventing one.
|
|
236
|
+
*
|
|
237
|
+
* @param {object} item
|
|
238
|
+
* @returns {string}
|
|
239
|
+
*/
|
|
240
|
+
export function cohortSurfaceLabel(item) {
|
|
241
|
+
if (!item) return "unknown surface";
|
|
242
|
+
const declared = declaredSurface(item);
|
|
243
|
+
if (declared) return declared;
|
|
244
|
+
const kind = str(item.kind);
|
|
245
|
+
if (kind === "file_comment") return str(item.channel_id) ? "file_comment" : "doc_comment";
|
|
246
|
+
if (kind && kind !== "message") return kind;
|
|
247
|
+
const channelKind = str(item.channel_kind).toUpperCase();
|
|
248
|
+
if (channelKind === "DM") return "dm";
|
|
249
|
+
if (channelKind) return `${channelKind.toLowerCase()} channel message`;
|
|
250
|
+
if (item.is_dm === true) return "dm";
|
|
251
|
+
return "message (surface not declared in raw_ref)";
|
|
252
|
+
}
|
|
253
|
+
|
|
188
254
|
/**
|
|
189
255
|
* Where a reply to this Cohort item goes — a pure function of the item, so it
|
|
190
256
|
* can be asked BEFORE anything is promised (see `canDeliverTo`).
|
|
@@ -766,5 +832,7 @@ export default {
|
|
|
766
832
|
canDeliverTo,
|
|
767
833
|
cohortReplyRoute,
|
|
768
834
|
cohortSurfaceOf,
|
|
835
|
+
cohortSurfaceLabel,
|
|
836
|
+
cohortSurfaceIsDeclared,
|
|
769
837
|
DELIVERABLE_SERVICES,
|
|
770
838
|
};
|
|
@@ -527,11 +527,19 @@ const ACTIVE_PATH = join(AGENT_REPO_DIR, "state", "sessions", "active.json");
|
|
|
527
527
|
// A marker is written to state/sessions/resume-pending/<sessionId>.json right
|
|
528
528
|
// after spawn and DELETED on a clean close. Its presence after a crash/reboot
|
|
529
529
|
// means "this session was mid-flight" — resetActiveSessions() reconciles them
|
|
530
|
-
// (re-dispatching `claude --print --
|
|
531
|
-
//
|
|
532
|
-
//
|
|
533
|
-
//
|
|
534
|
-
//
|
|
530
|
+
// (re-dispatching `claude --print --resume <claudeSessionId> <prompt>`, the
|
|
531
|
+
// same continuation mechanism responder.mjs uses) within a freshness window,
|
|
532
|
+
// instead of blindly wiping the slate. 3 strikes → the queue item is marked
|
|
533
|
+
// status:blocked. This is the "never brick / never drop work" recovery path
|
|
534
|
+
// for the dispatcher.
|
|
535
|
+
//
|
|
536
|
+
// ~~"re-dispatching `claude --print --session-id <claudeSessionId> <prompt>`
|
|
537
|
+
// … NOT `--resume`"~~ — struck 2026-09-25 with the flag itself. A marker's
|
|
538
|
+
// claudeSessionId by construction HAS a transcript, and `--session-id` is
|
|
539
|
+
// REFUSED for such an id (`Error: Session ID <uuid> is already in use.`,
|
|
540
|
+
// exit 1, no model call), so every reconcile exited 1 rather than recovering
|
|
541
|
+
// anything. `--resume` is the continuation. See lib/runtime/adapter.mjs
|
|
542
|
+
// #sessionArgs for the measurement.
|
|
535
543
|
const RESUME_PENDING_DIR = join(AGENT_REPO_DIR, "state", "sessions", "resume-pending");
|
|
536
544
|
// Sessions whose marker is older than this are too stale to resume meaningfully
|
|
537
545
|
// (the work context has moved on); reconcile blocks them rather than re-running.
|
|
@@ -634,10 +642,14 @@ function markItemBlocked(marker, reason) {
|
|
|
634
642
|
* WS4 — instead of *also* blindly discarding any work that was mid-flight when
|
|
635
643
|
* the box rebooted/lost power, when `opts.reconcile` is set we walk
|
|
636
644
|
* state/sessions/resume-pending/ and, for each marker still inside the
|
|
637
|
-
* freshness window, re-spawn `claude --print --
|
|
638
|
-
* <prompt>` (the responder's continuation pattern
|
|
639
|
-
*
|
|
640
|
-
*
|
|
645
|
+
* freshness window, re-spawn `claude --print --resume <claudeSessionId>
|
|
646
|
+
* <prompt>` (the responder's continuation pattern) so the work continues where
|
|
647
|
+
* it left off. After RESUME_MAX_ATTEMPTS (3) the item is marked status:blocked
|
|
648
|
+
* with a reason rather than retried forever.
|
|
649
|
+
*
|
|
650
|
+
* ~~"`--session-id <claudeSessionId> <prompt>` … NOT `--resume`"~~ — struck
|
|
651
|
+
* 2026-09-25: the CLI refuses `--session-id` for an id that already has a
|
|
652
|
+
* transcript, which every marker's id does.
|
|
641
653
|
*
|
|
642
654
|
* Graceful shutdown passes no opts (just clears active.json); only startup
|
|
643
655
|
* reconciles, so we never re-spawn work we're deliberately stopping.
|
|
@@ -736,8 +748,11 @@ export function reconcileResumePending(opts = {}) {
|
|
|
736
748
|
}
|
|
737
749
|
|
|
738
750
|
// No original prompt persisted → there's nothing to re-spawn with (the
|
|
739
|
-
// resume re-issues `--
|
|
740
|
-
//
|
|
751
|
+
// resume re-issues `--resume <id> <prompt>`: the flag continues the
|
|
752
|
+
// transcript, the prompt is what the continuation is FOR, and a bare
|
|
753
|
+
// `--resume` with no prompt is the do-nothing exit 0 that H1 guards
|
|
754
|
+
// against). A marker without a prompt is either a legacy marker (pre-H1)
|
|
755
|
+
// or one whose write
|
|
741
756
|
// dropped the field; we can't truly resume it, so retire it rather than
|
|
742
757
|
// spawn a no-op that the close handler would mistake for "done". (H1)
|
|
743
758
|
if (typeof marker.prompt !== "string" || !marker.prompt) {
|
|
@@ -894,6 +909,11 @@ export function buildResumeSpawn(o = {}) {
|
|
|
894
909
|
model: flag,
|
|
895
910
|
prompt: marker.prompt,
|
|
896
911
|
sessionId: marker.claudeSessionId,
|
|
912
|
+
// Crash recovery by definition continues a transcript the dead session
|
|
913
|
+
// already wrote, so this is `--resume`. It was `--session-id`, which the
|
|
914
|
+
// CLI refuses for an id it has seen — so recovery has been exiting 1 in a
|
|
915
|
+
// tenth of a second rather than recovering anything.
|
|
916
|
+
resumeSession: true,
|
|
897
917
|
permissions: Array.isArray(o.permissionArgs) ? o.permissionArgs : [],
|
|
898
918
|
mcp: {},
|
|
899
919
|
knobs: t,
|
|
@@ -922,19 +942,30 @@ export function buildResumeSpawn(o = {}) {
|
|
|
922
942
|
}
|
|
923
943
|
|
|
924
944
|
/**
|
|
925
|
-
* Real resume spawner. Mirrors responder.mjs's
|
|
926
|
-
*
|
|
927
|
-
*
|
|
928
|
-
*
|
|
945
|
+
* Real resume spawner. Mirrors responder.mjs's session-continuation pattern: a
|
|
946
|
+
* continuation re-spawns `claude --print --resume <sessionId> <prompt>` with
|
|
947
|
+
* the SAME model/flags the original used. The flag rehydrates the transcript;
|
|
948
|
+
* the prompt is what the rehydrated session is asked to carry on with, which
|
|
949
|
+
* is why a resume here is never the bare `--resume` with no prompt.
|
|
950
|
+
*
|
|
951
|
+
* ~~"a continuation re-spawns `claude --print --session-id <sessionId>
|
|
952
|
+
* <prompt>` … it does NOT use `--resume`", and the "Why NOT `--resume` (the
|
|
953
|
+
* previous bug)" paragraph under it~~ — STRUCK 2026-09-25, in the same change
|
|
954
|
+
* that made `resumeSession: true` the flag this function emits. Both halves
|
|
955
|
+
* were false and the file still said them:
|
|
929
956
|
*
|
|
930
|
-
*
|
|
931
|
-
*
|
|
932
|
-
*
|
|
933
|
-
*
|
|
934
|
-
*
|
|
935
|
-
*
|
|
936
|
-
*
|
|
937
|
-
*
|
|
957
|
+
* · `--session-id <id>` STARTS a session under a chosen id. The CLI refuses
|
|
958
|
+
* it for an id that already has a transcript — `Error: Session ID <uuid>
|
|
959
|
+
* is already in use.`, exit 1, ~0.1s, no model call — and a resume marker's
|
|
960
|
+
* id always has one. So crash recovery had never recovered anything.
|
|
961
|
+
* · The bare-`--resume` failure the struck paragraph described (exit 0 having
|
|
962
|
+
* done nothing) is a resume with NO PROMPT. This lane always passes the
|
|
963
|
+
* marker's prompt, so it was never the shape being warned about.
|
|
964
|
+
*
|
|
965
|
+
* Left standing, that paragraph is how the next author reinstates the flag:
|
|
966
|
+
* it reads as a rationale rather than as the belief that cost four replies on
|
|
967
|
+
* this seat, 09-22..24. lib/runtime/adapter.mjs#sessionArgs carries the
|
|
968
|
+
* reproduction.
|
|
938
969
|
*
|
|
939
970
|
* Audit F11: the argv and env are built through the SAME resolveSpawnTarget /
|
|
940
971
|
* runtime-adapter path as spawnSession (buildResumeSpawn), so a retargeted
|
|
@@ -1652,9 +1683,13 @@ function spawnSession(entry) {
|
|
|
1652
1683
|
// session in a live thread started COLD — the agent re-read the room, re-did
|
|
1653
1684
|
// the orientation work, and answered a follow-up as if it were an opening
|
|
1654
1685
|
// (design §3 R10). The router the responder already consults keys the
|
|
1655
|
-
// conversation; a follow-up inside the TTL resumes the same session id
|
|
1656
|
-
//
|
|
1657
|
-
//
|
|
1686
|
+
// conversation; a follow-up inside the TTL resumes the same session id.
|
|
1687
|
+
//
|
|
1688
|
+
// ~~"which in this daemon IS continuation (`--session-id <id> <prompt>`,
|
|
1689
|
+
// never `--resume`)"~~ — struck: it is not, and never was. `--session-id`
|
|
1690
|
+
// STARTS a session under a chosen id and the CLI refuses it for an id that
|
|
1691
|
+
// already has a transcript. Continuation is `--resume`, which is what
|
|
1692
|
+
// `resumeSession` below now selects.
|
|
1658
1693
|
//
|
|
1659
1694
|
// Inbox only. Backlog work is not a conversation, and keying it would put
|
|
1660
1695
|
// unrelated items in one session.
|
|
@@ -1725,6 +1760,13 @@ function spawnSession(entry) {
|
|
|
1725
1760
|
model: engineShape ? engineShape.model : effectiveModelFlag,
|
|
1726
1761
|
prompt,
|
|
1727
1762
|
sessionId: claudeSessionId,
|
|
1763
|
+
// `--resume <id>` when this id names a transcript that already exists (the
|
|
1764
|
+
// router's RESUME decision), `--session-id <id>` when we just minted it.
|
|
1765
|
+
// Passing a used id to `--session-id` is refused outright by the CLI
|
|
1766
|
+
// ("Session ID … is already in use", exit 1 before any model call), which
|
|
1767
|
+
// is what made every routed continuation die — see sessionArgs() in
|
|
1768
|
+
// lib/runtime/adapter.mjs.
|
|
1769
|
+
resumeSession: resumedSessionId != null,
|
|
1728
1770
|
permissions: sessionPermissionArgs({ source: "dispatcher", priority: classResult?.priority }),
|
|
1729
1771
|
mcp: {},
|
|
1730
1772
|
knobs: engineShape ? engineShape.knobs : target,
|
|
@@ -85,6 +85,27 @@ export function _resetInFlightForTests() {
|
|
|
85
85
|
inFlightKeys.clear();
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
+
/**
|
|
89
|
+
* Does this CLI failure mean "the id you asked to START is already taken"?
|
|
90
|
+
*
|
|
91
|
+
* The CLI's exact text is `Error: Session ID <uuid> is already in use.` on
|
|
92
|
+
* stderr with exit 1, emitted in ~0.1s before any model call. It is a PURE
|
|
93
|
+
* addressing fault: the same prompt with a fresh id succeeds. So it is the one
|
|
94
|
+
* spawn failure a reply path may retry blind, and the caller that catches it
|
|
95
|
+
* owes exactly one retry with a new id rather than a dropped reply.
|
|
96
|
+
*
|
|
97
|
+
* Matched loosely on purpose — the caller sees this wrapped as
|
|
98
|
+
* `claude CLI exited 1: Error: Session ID … is already in use.` — but anchored
|
|
99
|
+
* on both halves so an unrelated "in use" message cannot trigger a retry.
|
|
100
|
+
*
|
|
101
|
+
* @param {unknown} err an Error, or the message/stderr text
|
|
102
|
+
* @returns {boolean}
|
|
103
|
+
*/
|
|
104
|
+
export function isSessionIdCollision(err) {
|
|
105
|
+
const text = err instanceof Error ? err.message : typeof err === "string" ? err : "";
|
|
106
|
+
return /session id\b/i.test(text) && /\bis already in use/i.test(text);
|
|
107
|
+
}
|
|
108
|
+
|
|
88
109
|
/**
|
|
89
110
|
* Services whose routing key is `<source>:<channel>[:<thread>]`.
|
|
90
111
|
*
|