@cohortapp/agent-sdk 2.15.0 → 2.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +5 -2
- package/docs/guides/front-door-session.md +16 -5
- package/docs/guides/poller-daemon-setup.md +53 -2
- package/lib/assurance/plan-note.mjs +251 -0
- package/lib/assurance/plan-note.test.mjs +234 -0
- package/lib/assurance/room-budget.mjs +497 -0
- package/lib/assurance/room-budget.test.mjs +486 -0
- package/lib/assurance/tier.mjs +166 -0
- package/lib/assurance/tier.test.mjs +174 -0
- package/lib/comms/receipts.mjs +17 -1
- package/lib/context/budget.mjs +327 -0
- package/lib/context/budget.test.mjs +252 -0
- package/lib/context/history-scope.mjs +138 -0
- package/lib/context/history-scope.test.mjs +79 -0
- package/lib/model-router/economics.mjs +9 -0
- package/lib/model-router/resolve.mjs +6 -0
- package/lib/org/inbound/facts.mjs +4 -2
- package/lib/org/inbound/hydrate.mjs +555 -51
- package/lib/org/inbound/hydrate.test.mjs +456 -1
- package/package.json +3 -1
- package/plugins/maestro-skills/skills/inbound-triage.md +52 -24
- package/plugins/maestro-skills/skills/main-session.md +6 -4
- package/scripts/daemon/agent-daemon.mjs +35 -7
- package/scripts/daemon/agent-daemon.test.mjs +23 -6
- package/scripts/daemon/assurance-e2e.test.mjs +75 -19
- package/scripts/daemon/assurance.mjs +663 -159
- package/scripts/daemon/assurance.test.mjs +820 -140
- package/scripts/daemon/context-compiler.mjs +52 -21
- package/scripts/daemon/context-compiler.test.mjs +106 -0
- package/scripts/daemon/deliver.mjs +7 -4
- package/scripts/daemon/dispatcher-session-continuity.test.mjs +365 -0
- package/scripts/daemon/dispatcher.mjs +210 -9
- package/scripts/daemon/lib/session-router.mjs +310 -42
- package/scripts/daemon/lib/session-router.test.mjs +260 -1
- package/scripts/daemon/prompt-builder.mjs +160 -16
- package/scripts/daemon/prompt-builder.test.mjs +287 -7
- package/scripts/daemon/responder-history.test.mjs +37 -1
- package/scripts/daemon/responder.mjs +79 -72
package/.env.example
CHANGED
|
@@ -226,6 +226,9 @@ GREPTILE_API_KEY=
|
|
|
226
226
|
# Flags that control system behaviour. These are not secrets.
|
|
227
227
|
#
|
|
228
228
|
|
|
229
|
-
#
|
|
230
|
-
#
|
|
229
|
+
# The session context compiler: pre-compile context before each Claude Code
|
|
230
|
+
# session. ON BY DEFAULT in code as well as here — the two used to disagree
|
|
231
|
+
# (code read `=== "1"`, this file shipped `=1`), so which of two materially
|
|
232
|
+
# different context paths ran depended on whether the operator had copied this
|
|
233
|
+
# file. Set to 0 only to fall back to the legacy path in a hurry.
|
|
231
234
|
DAEMON_CONTEXT_COMPILER=1
|
|
@@ -265,11 +265,22 @@ Each line from the feed is one unit of work. The policy the session follows
|
|
|
265
265
|
1. **Answer in the turn** when the ask is answerable in one reply from what
|
|
266
266
|
the agent already knows. `maestro inbox reply <id>` routes through the same
|
|
267
267
|
send-gate and audit as every other outbound path.
|
|
268
|
-
2. Otherwise **
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
268
|
+
2. Otherwise **say nothing yet** — file the ask on the conversation's board
|
|
269
|
+
(`maestro board track <inbox-id> --stage accepted --title … --why …`) and
|
|
270
|
+
either run the work as a dynamic workflow inside the session or spawn a peer
|
|
271
|
+
(`maestro session spawn --name <slug> "<prompt>"`). Report back in-channel
|
|
272
|
+
and `board track --stage done` on completion.
|
|
273
|
+
|
|
274
|
+
~~"Otherwise **acknowledge in-channel in the same turn**"~~ — struck
|
|
275
|
+
2026-09-12. That reflex was 28.8 % of every message this fleet sent, and 9 %
|
|
276
|
+
of it was never followed by a substantive reply at all. The daemon's
|
|
277
|
+
assurance sweep owns the ONE interim an ask may earn: nothing for the first
|
|
278
|
+
90 s, then at most one line, and at most one line per channel per 15 minutes
|
|
279
|
+
FROM THIS SEAT however many asks are in flight there — the budget is durable
|
|
280
|
+
under this agent's `AGENT_DIR`, and seats share no store
|
|
281
|
+
(`docs/guides/poller-daemon-setup.md` §2.6a). If the work is big enough that the human needs to know its shape,
|
|
282
|
+
post the PLAN — what you will do, in what order, what comes back — as the
|
|
283
|
+
first step of doing it, not as a promise in front of it.
|
|
273
284
|
3. **Handoffs** (`{"type":"handoff", …}`) are cadence ticks the daemon has
|
|
274
285
|
handed over; the session runs the rendered prompt and `maestro session ack
|
|
275
286
|
<tickId>`. A handoff not acked within 30 minutes is re-enqueued by the
|
|
@@ -203,7 +203,55 @@ Handles quick replies without spawning a full Claude session:
|
|
|
203
203
|
|
|
204
204
|
- `isQuickReply()` detects items that can be answered immediately (greetings, acknowledgments)
|
|
205
205
|
- `sendQuickResponse()` sends a fast reply via the appropriate channel
|
|
206
|
-
- `sendHoldingMessage()`
|
|
206
|
+
- `sendHoldingMessage()` composes the one-line interim, when one is owed at all
|
|
207
|
+
|
|
208
|
+
### 2.6a Acknowledgement discipline (`assurance.mjs` + `lib/assurance/`)
|
|
209
|
+
|
|
210
|
+
**Silence is the default.** ~~"`sendHoldingMessage()` sends 'Got it, working on
|
|
211
|
+
this' for complex requests."~~ — struck 2026-09-12. Measured over fourteen days
|
|
212
|
+
in `org_default_adaptic`: of 10,667 agent messages, 3,069 (28.8 %) were opening
|
|
213
|
+
acknowledgements and 961 (9.0 %) were timed progress nags — 37.8 % carrying no
|
|
214
|
+
content — and 272 of those acknowledgements were never followed by a substantive
|
|
215
|
+
reply inside an hour. Nothing is sent when an obligation opens any more.
|
|
216
|
+
|
|
217
|
+
What governs an interim now:
|
|
218
|
+
|
|
219
|
+
| Gate | Rule |
|
|
220
|
+
|---|---|
|
|
221
|
+
| **Tier** (`lib/assurance/tier.mjs`) | `answer` ⇒ never anything (the reply is the acknowledgement). `work` ⇒ nothing for `ACK_AFTER_MS`, then at most one. `plan` ⇒ one message, and it is the plan. **The plan tier currently degrades to the work path.** `assurance.notePlan` needs the session's first assistant turn, and `dispatcher.mjs` spawns with `--output-format json`, so stdout arrives as one object at exit — after the answer. Streaming it is a dispatcher change and lands with WP-2; until then a plan-tier ask gets the same single interim as a work-tier one, never silence. |
|
|
222
|
+
| **Time** (`ASSURANCE_ACK_AFTER_MS`, 90 s) | Below it the typing indicator is the acknowledgement. |
|
|
223
|
+
| **Room** (`lib/assurance/room-budget.mjs`, `ASSURANCE_ROOM_INTERIM_WINDOW_MS`, 15 min) | At most ONE interim per `(service, channel)` per window **per seat**, however many obligations are open in it. Over budget ⇒ the debt is still opened and tracked; only the message is withheld. Durable at `state/obligations/room-budget.json`, which is under `AGENT_DIR` — seats share no store, so a room with N seats in it has a ceiling of N per window, not one. This gate is the backstop on what survives the three above it, not the thing that removed the measured traffic. |
|
|
224
|
+
| **Once** | `interimSaid` latches per obligation and is INHERITED by a retry that reopens the same key. |
|
|
225
|
+
| **Copy** | `sanitiseAckText` rejects the generic openers the ack prompt already forbade (`on it`, `looking into`, `checking`, `one moment`, `got it`, `will do`, `working on it`, `taking a look`, `digging in`) and anything under 20 characters. Rejection ⇒ null ⇒ send nothing. **That list is a floor, not the policy** — it is a phrase blocklist sized against the measured population, and "Picking this up now." is just as empty and passes it. Widen it when a new phrasing is measured in production; do not read a gap in it as permission. |
|
|
226
|
+
|
|
227
|
+
**There are no timed progress updates.** `composeProgress`,
|
|
228
|
+
`ASSURANCE_PROGRESS_AFTER_MS`, `ASSURANCE_PROGRESS_EVERY_MS` and
|
|
229
|
+
`ASSURANCE_PROGRESS_MAX` are deleted. A message that is a pure function of
|
|
230
|
+
elapsed minutes tells a reader nothing the timestamp does not, and two
|
|
231
|
+
per-obligation timers over overlapping work put "Still on this — 5 minutes in"
|
|
232
|
+
beside "Still going — 15 minutes in" at the same instant in a live channel. Long
|
|
233
|
+
work either reports CONTENT — `lib/assurance/plan-note.mjs` renders at most
|
|
234
|
+
three bullets of the session's OWN stated intent, or returns null and nothing is
|
|
235
|
+
posted — or it stays quiet until the answer.
|
|
236
|
+
|
|
237
|
+
**Outcome notices are unaffected by the room budget** — suppressing one
|
|
238
|
+
re-creates the original bug, a person never told their work died. They are
|
|
239
|
+
deduplicated twice instead, on questions the interim budget does not ask:
|
|
240
|
+
|
|
241
|
+
- **once per `(channel, obligationKey, notice)`** (`assurance.noticeAlreadySaid`,
|
|
242
|
+
inherited across a retry that reopens the same key) — this is the retry loop:
|
|
243
|
+
notice → retry → fresh record → notice again;
|
|
244
|
+
- **once per `(room, notice, exact sentence)`** (`room-budget.claimRoomNotice`)
|
|
245
|
+
— this is the SIBLING case the first cannot reach. Twenty separate asks in one
|
|
246
|
+
channel whose sessions all die the same way are twenty separate records, and
|
|
247
|
+
the stale sentence says nothing about which ask it is, so the reader gets one
|
|
248
|
+
line twenty times and cannot tell them apart.
|
|
249
|
+
|
|
250
|
+
Only a BYTE-IDENTICAL repeat is withheld. A different cause, a different
|
|
251
|
+
excerpt, "a retry is running" versus "I have stopped" — all distinct text, all
|
|
252
|
+
said, at any volume. The needs-attention escalation is written per obligation
|
|
253
|
+
regardless, so an operator still sees every dead ask even when the room only
|
|
254
|
+
heard about the first.
|
|
207
255
|
|
|
208
256
|
### 2.7 Session Lock (`session-lock.mjs`)
|
|
209
257
|
|
|
@@ -231,7 +279,10 @@ Pre-compiles session context to reduce prompt size:
|
|
|
231
279
|
|
|
232
280
|
- Reads recent interactions, queue state, and active items
|
|
233
281
|
- Compresses context to fit within token limits
|
|
234
|
-
-
|
|
282
|
+
- **On by default.** Set `DAEMON_CONTEXT_COMPILER=0` in `.env` to fall back to
|
|
283
|
+
the legacy per-prompt history path. The code used to default this OFF while
|
|
284
|
+
`.env.example` shipped `=1`, so which path ran depended on whether the
|
|
285
|
+
operator had copied the example env; both now agree.
|
|
235
286
|
|
|
236
287
|
---
|
|
237
288
|
|
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/assurance/plan-note.mjs — the only interim that is allowed to exist,
|
|
3
|
+
* and the copy rules that govern every interim.
|
|
4
|
+
*
|
|
5
|
+
* WHY A PLAN AND NOT A HOLDING LINE
|
|
6
|
+
*
|
|
7
|
+
* The research is unambiguous and it does not say what the old code assumed.
|
|
8
|
+
* Nielsen: under a second, show nothing; under ten, a state change rather than
|
|
9
|
+
* a message; beyond ten, show WORK. The labour-illusion result is about showing
|
|
10
|
+
* work, not about promising to return — and every source agrees that the broken
|
|
11
|
+
* promise is what destroys trust. Production bore that out precisely: of 3,069
|
|
12
|
+
* opening acknowledgements, 272 were never followed by a substantive reply
|
|
13
|
+
* within the hour.
|
|
14
|
+
*
|
|
15
|
+
* So the replaced mechanism is not "a better ack". It is: when the work is big
|
|
16
|
+
* enough that a human deserves to hear something before the answer, the one
|
|
17
|
+
* thing they hear is the PLAN the model itself stated — what will be done, in
|
|
18
|
+
* what order — rendered from the session's own first turn.
|
|
19
|
+
*
|
|
20
|
+
* THE HARD RULE: NEVER FABRICATE A PLAN.
|
|
21
|
+
*
|
|
22
|
+
* Every bullet is text the model wrote. Markers are normalised and long lines
|
|
23
|
+
* are clipped; nothing is invented, summarised into something unsaid, or filled
|
|
24
|
+
* in from a template. No first turn, or a first turn that states no intent,
|
|
25
|
+
* returns `null` — and null means the agent posts NOTHING. A rendered template
|
|
26
|
+
* over an absent session would be the same content-free message this work
|
|
27
|
+
* package exists to delete, only longer.
|
|
28
|
+
*
|
|
29
|
+
* WHY THE COPY RULES LIVE HERE
|
|
30
|
+
*
|
|
31
|
+
* After `composeProgress` was deleted, this is the only module left that
|
|
32
|
+
* COMPOSES interim copy. Two rules apply to every interim the system can emit,
|
|
33
|
+
* so they live with the composer rather than in two drifting copies:
|
|
34
|
+
*
|
|
35
|
+
* GENERIC_OPENER — the set the ack system prompt already forbade and the
|
|
36
|
+
* sanitiser did not enforce; 3,069 of them shipped.
|
|
37
|
+
* `assurance.sanitiseAckText` imports this, so the prompt
|
|
38
|
+
* and the filter cannot disagree again.
|
|
39
|
+
*
|
|
40
|
+
* IT IS A FLOOR, NOT A DEFINITION OF "CONTENT-FREE".
|
|
41
|
+
* This is a phrase blocklist, and a phrase blocklist can
|
|
42
|
+
* only refuse phrasings someone has written down. It
|
|
43
|
+
* refuses every opener in the measured population, which
|
|
44
|
+
* is what it was sized against; "Picking this up now.",
|
|
45
|
+
* "Beginning work on this now." and "Ack — starting on
|
|
46
|
+
* it." are all just as empty and all pass it. That is
|
|
47
|
+
* tolerable ONLY because the blocklist is the last and
|
|
48
|
+
* weakest of four gates, not the policy: the tier decides
|
|
49
|
+
* whether an interim may exist at all, nothing is said
|
|
50
|
+
* before ACK_AFTER_MS, `interimSaid` latches one per ask,
|
|
51
|
+
* the room budget bounds the seat, and two rejected
|
|
52
|
+
* generations latch `ackSuppressed` permanently. A line
|
|
53
|
+
* that slips through is ONE line. Widen the set when a
|
|
54
|
+
* new phrasing is measured in production — do not treat
|
|
55
|
+
* the absence of a phrasing here as permission.
|
|
56
|
+
* FORWARD_PROMISE — "I'll come back", "I'll follow up", "as soon as". A
|
|
57
|
+
* promise is only permissible when a durable obligation
|
|
58
|
+
* exists and the sweep can close it; a plan bullet carries
|
|
59
|
+
* no such backing, so a bullet that is only a promise is
|
|
60
|
+
* dropped.
|
|
61
|
+
*
|
|
62
|
+
* PURE. No clock, no disk, no environment, no network.
|
|
63
|
+
*
|
|
64
|
+
* @module lib/assurance/plan-note
|
|
65
|
+
*/
|
|
66
|
+
|
|
67
|
+
"use strict";
|
|
68
|
+
|
|
69
|
+
/** At most three. A fourth step is a document, not a message. */
|
|
70
|
+
export const PLAN_MAX_BULLETS = 3;
|
|
71
|
+
|
|
72
|
+
/** Longest a single bullet may be before it is clipped. */
|
|
73
|
+
export const PLAN_BULLET_MAX_CHARS = 160;
|
|
74
|
+
|
|
75
|
+
/** Shortest a line may be and still be a step rather than a noise. */
|
|
76
|
+
const PLAN_BULLET_MIN_CHARS = 15;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* The generic openers the ack system prompt bans ("Never a generic 'on it' or
|
|
80
|
+
* 'looking into it'") and which shipped 3,069 times because nothing enforced
|
|
81
|
+
* it. Anchored at the start: a line that OPENS this way is a reflex whatever
|
|
82
|
+
* follows it.
|
|
83
|
+
*/
|
|
84
|
+
export const GENERIC_OPENER =
|
|
85
|
+
/^(on it|looking (now|into)|checking|one moment|got it|will do|working on it|taking a look|digging in)\b/i;
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* A commitment to come back later. Permitted only where a durable obligation
|
|
89
|
+
* backs it and the sweep can discharge it — which a plan bullet never is.
|
|
90
|
+
*/
|
|
91
|
+
export const FORWARD_PROMISE =
|
|
92
|
+
/\b(i'?(ll|m going to)\s+(come back|follow up|get back|circle back|report back|update you|let you know|revert)|i will\s+(come back|follow up|get back|circle back|report back|update you|let you know|revert)|as soon as|the moment it'?s done)\b/i;
|
|
93
|
+
|
|
94
|
+
/** @param {unknown} text @returns {boolean} */
|
|
95
|
+
export function containsForwardPromise(text) {
|
|
96
|
+
return typeof text === "string" && FORWARD_PROMISE.test(text);
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Text blocks out of one assistant message's content array. */
|
|
100
|
+
function textOfContent(content) {
|
|
101
|
+
if (typeof content === "string") return content;
|
|
102
|
+
if (!Array.isArray(content)) return "";
|
|
103
|
+
return content
|
|
104
|
+
.filter((b) => b && typeof b === "object" && b.type === "text" && typeof b.text === "string")
|
|
105
|
+
.map((b) => b.text)
|
|
106
|
+
.join("\n");
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** One event/object → its assistant text, or "". */
|
|
110
|
+
function textOfEvent(ev) {
|
|
111
|
+
if (!ev || typeof ev !== "object" || Array.isArray(ev)) return "";
|
|
112
|
+
if (ev.type === "assistant") {
|
|
113
|
+
const msg = ev.message && typeof ev.message === "object" ? ev.message : ev;
|
|
114
|
+
return textOfContent(msg.content).trim();
|
|
115
|
+
}
|
|
116
|
+
// The `--output-format json` result object: one field carrying the run's text.
|
|
117
|
+
if (typeof ev.result === "string") return ev.result.trim();
|
|
118
|
+
if (ev.message && typeof ev.message === "object") return textOfContent(ev.message.content).trim();
|
|
119
|
+
return "";
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** Parse, tolerantly. A banner line before the JSON must not lose the payload. */
|
|
123
|
+
function parseLoose(raw) {
|
|
124
|
+
const t = String(raw).trim();
|
|
125
|
+
if (!t) return null;
|
|
126
|
+
try { return JSON.parse(t); } catch { /* fall through to the tolerant paths */ }
|
|
127
|
+
// Newline-delimited stream-json: take every line that parses.
|
|
128
|
+
const lines = t.split("\n").map((l) => l.trim()).filter(Boolean);
|
|
129
|
+
if (lines.length > 1) {
|
|
130
|
+
const evs = [];
|
|
131
|
+
for (const l of lines) { try { evs.push(JSON.parse(l)); } catch { /* a prose line is not an event */ } }
|
|
132
|
+
if (evs.length) return evs;
|
|
133
|
+
}
|
|
134
|
+
const start = t.indexOf("{");
|
|
135
|
+
const end = t.lastIndexOf("}");
|
|
136
|
+
if (start !== -1 && end > start) {
|
|
137
|
+
try { return JSON.parse(t.slice(start, end + 1)); } catch { /* genuinely not JSON */ }
|
|
138
|
+
}
|
|
139
|
+
return null;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* The session's FIRST assistant turn, as plain text — or null.
|
|
144
|
+
*
|
|
145
|
+
* First and not last, deliberately: the first turn is where a model states what
|
|
146
|
+
* it is about to do. The last turn is the answer, and the answer is the
|
|
147
|
+
* session's to deliver, not this module's to pre-empt.
|
|
148
|
+
*
|
|
149
|
+
* @param {unknown} input raw stdout, a parsed result object, a stream-json
|
|
150
|
+
* event, an array of events, or plain prose
|
|
151
|
+
* @returns {string|null}
|
|
152
|
+
*/
|
|
153
|
+
export function firstAssistantText(input) {
|
|
154
|
+
if (input == null) return null;
|
|
155
|
+
|
|
156
|
+
if (Array.isArray(input)) {
|
|
157
|
+
for (const ev of input) {
|
|
158
|
+
const t = textOfEvent(ev);
|
|
159
|
+
if (t) return t;
|
|
160
|
+
}
|
|
161
|
+
return null;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
if (typeof input === "object") {
|
|
165
|
+
const t = textOfEvent(input);
|
|
166
|
+
return t || null;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
if (typeof input !== "string") return null;
|
|
170
|
+
const trimmed = input.trim();
|
|
171
|
+
if (!trimmed) return null;
|
|
172
|
+
|
|
173
|
+
const parsed = parseLoose(trimmed);
|
|
174
|
+
if (parsed != null) {
|
|
175
|
+
const t = firstAssistantText(parsed);
|
|
176
|
+
if (t) return t;
|
|
177
|
+
// It parsed as JSON but carried no assistant text. Returning the raw JSON
|
|
178
|
+
// as "prose" would put a serialised object in front of a human.
|
|
179
|
+
return null;
|
|
180
|
+
}
|
|
181
|
+
// It LOOKS like a payload and will not parse — a truncated or corrupt stdout.
|
|
182
|
+
// Prose never opens with a brace; showing this to a human would be showing
|
|
183
|
+
// them our own broken plumbing.
|
|
184
|
+
if (/^[[{]/.test(trimmed)) return null;
|
|
185
|
+
return trimmed;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** Strip a list marker without touching the model's words. */
|
|
189
|
+
function stripMarker(line) {
|
|
190
|
+
return line.replace(/^\s*(?:[-*•–—]|\d{1,2}[.)])\s+/, "").trim();
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Does this line STATE an intent?
|
|
195
|
+
*
|
|
196
|
+
* Two shapes qualify and nothing else does: an explicit list item, or a
|
|
197
|
+
* first-person/sequencing opener. A question, an answer, an aside and a
|
|
198
|
+
* pleasantry all fail, which is why a turn that is not a plan renders nothing.
|
|
199
|
+
*/
|
|
200
|
+
const INTENT_OPENER =
|
|
201
|
+
/^(i'?ll|i will|i'?m going to|i am going to|i plan to|i'?m starting|first|then|next|after that|finally|start(ing)? by|step \d)\b/i;
|
|
202
|
+
|
|
203
|
+
function isStatedStep(rawLine) {
|
|
204
|
+
const line = String(rawLine || "").trim();
|
|
205
|
+
if (!line) return false;
|
|
206
|
+
const listItem = /^\s*(?:[-*•–—]|\d{1,2}[.)])\s+/.test(line);
|
|
207
|
+
const body = stripMarker(line);
|
|
208
|
+
if (!body) return false;
|
|
209
|
+
if (body.endsWith("?")) return false; // a question back is not a plan
|
|
210
|
+
if (GENERIC_OPENER.test(body)) return false; // "On it" in a numbered list is still "On it"
|
|
211
|
+
if (containsForwardPromise(body)) return false; // no promise without a mechanism
|
|
212
|
+
if (body.length < PLAN_BULLET_MIN_CHARS) return false;
|
|
213
|
+
return listItem || INTENT_OPENER.test(body);
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
function clip(s) {
|
|
217
|
+
return s.length <= PLAN_BULLET_MAX_CHARS ? s : `${s.slice(0, PLAN_BULLET_MAX_CHARS).trimEnd()}…`;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Render the plan the model stated, or null.
|
|
222
|
+
*
|
|
223
|
+
* @param {unknown} firstTurn whatever the dispatcher can give us of the
|
|
224
|
+
* session's first assistant turn
|
|
225
|
+
* @returns {string|null} the message to post, or null ⇒ post nothing
|
|
226
|
+
*/
|
|
227
|
+
export function renderPlanNote(firstTurn) {
|
|
228
|
+
const text = firstAssistantText(firstTurn);
|
|
229
|
+
if (!text) return null;
|
|
230
|
+
|
|
231
|
+
const bullets = [];
|
|
232
|
+
for (const line of text.split("\n")) {
|
|
233
|
+
if (bullets.length >= PLAN_MAX_BULLETS) break;
|
|
234
|
+
if (!isStatedStep(line)) continue;
|
|
235
|
+
const body = clip(stripMarker(line));
|
|
236
|
+
if (!bullets.includes(body)) bullets.push(body);
|
|
237
|
+
}
|
|
238
|
+
if (bullets.length === 0) return null;
|
|
239
|
+
|
|
240
|
+
return [`Here's the plan:`, ...bullets.map((b) => `- ${b}`)].join("\n");
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
export default {
|
|
244
|
+
renderPlanNote,
|
|
245
|
+
firstAssistantText,
|
|
246
|
+
containsForwardPromise,
|
|
247
|
+
PLAN_MAX_BULLETS,
|
|
248
|
+
PLAN_BULLET_MAX_CHARS,
|
|
249
|
+
GENERIC_OPENER,
|
|
250
|
+
FORWARD_PROMISE,
|
|
251
|
+
};
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* plan-note.test.mjs — if we say we are working, show work; otherwise say nothing.
|
|
3
|
+
*
|
|
4
|
+
* The rule this pins is the hard one: NEVER FABRICATE A PLAN. Every bullet must
|
|
5
|
+
* be text the model itself wrote in its own first turn. When there is no first
|
|
6
|
+
* turn, or nothing in it that states an intent, the answer is null and null
|
|
7
|
+
* means the agent posts nothing at all. A rendered template over an absent
|
|
8
|
+
* session is exactly the content-free message this whole work package exists to
|
|
9
|
+
* delete — it would merely be a longer one.
|
|
10
|
+
*
|
|
11
|
+
* Run: node --test lib/assurance/plan-note.test.mjs
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { test, describe } from "node:test";
|
|
15
|
+
import assert from "node:assert/strict";
|
|
16
|
+
|
|
17
|
+
import {
|
|
18
|
+
renderPlanNote,
|
|
19
|
+
firstAssistantText,
|
|
20
|
+
PLAN_MAX_BULLETS,
|
|
21
|
+
FORWARD_PROMISE,
|
|
22
|
+
containsForwardPromise,
|
|
23
|
+
} from "./plan-note.mjs";
|
|
24
|
+
|
|
25
|
+
const STEPS = [
|
|
26
|
+
"First I'll pull the July ledger rows and the forecast we signed off in June.",
|
|
27
|
+
"Then I'll reconcile them line by line and mark every variance over 5%.",
|
|
28
|
+
"Finally I'll write up the three biggest gaps with the numbers behind each.",
|
|
29
|
+
].join("\n");
|
|
30
|
+
|
|
31
|
+
// ---------------------------------------------------------------------------
|
|
32
|
+
// firstAssistantText — every shape the CLI can hand us
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
describe("firstAssistantText — reading the session's own first turn", () => {
|
|
36
|
+
test("a --output-format json result object", () => {
|
|
37
|
+
assert.match(firstAssistantText(JSON.stringify({ type: "result", result: STEPS })), /pull the July ledger/);
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
test("an already-parsed result object", () => {
|
|
41
|
+
assert.match(firstAssistantText({ type: "result", result: STEPS }), /pull the July ledger/);
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
test("a stream-json assistant message", () => {
|
|
45
|
+
const msg = { type: "assistant", message: { content: [{ type: "text", text: STEPS }] } };
|
|
46
|
+
assert.match(firstAssistantText(msg), /pull the July ledger/);
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
test("an array of stream-json events takes the FIRST assistant turn, not the last", () => {
|
|
50
|
+
const events = [
|
|
51
|
+
{ type: "system", subtype: "init" },
|
|
52
|
+
{ type: "assistant", message: { content: [{ type: "text", text: STEPS }] } },
|
|
53
|
+
{ type: "assistant", message: { content: [{ type: "text", text: "Done — all three variances written up." }] } },
|
|
54
|
+
];
|
|
55
|
+
assert.match(firstAssistantText(events), /pull the July ledger/);
|
|
56
|
+
assert.doesNotMatch(firstAssistantText(events), /Done —/);
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
test("newline-delimited stream-json text", () => {
|
|
60
|
+
const ndjson = [
|
|
61
|
+
JSON.stringify({ type: "system", subtype: "init" }),
|
|
62
|
+
JSON.stringify({ type: "assistant", message: { content: [{ type: "text", text: STEPS }] } }),
|
|
63
|
+
].join("\n");
|
|
64
|
+
assert.match(firstAssistantText(ndjson), /pull the July ledger/);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test("a tool_use block is not text and is skipped", () => {
|
|
68
|
+
const msg = { type: "assistant", message: { content: [{ type: "tool_use", name: "Read", input: {} }, { type: "text", text: STEPS }] } };
|
|
69
|
+
assert.match(firstAssistantText(msg), /pull the July ledger/);
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test("plain prose is taken as itself", () => {
|
|
73
|
+
assert.equal(firstAssistantText(STEPS), STEPS);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test("absent, empty and malformed input read as nothing", () => {
|
|
77
|
+
for (const bad of [null, undefined, "", " ", 12, [], {}, { type: "result" }, "{not json", { type: "assistant", message: null }]) {
|
|
78
|
+
assert.equal(firstAssistantText(bad), null, `${JSON.stringify(bad)} must read as nothing`);
|
|
79
|
+
}
|
|
80
|
+
});
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
// ---------------------------------------------------------------------------
|
|
84
|
+
// renderPlanNote
|
|
85
|
+
// ---------------------------------------------------------------------------
|
|
86
|
+
|
|
87
|
+
describe("renderPlanNote — at most three bullets of STATED intent, or null", () => {
|
|
88
|
+
test("three stated steps render as three bullets, in the model's own words", () => {
|
|
89
|
+
const note = renderPlanNote({ type: "assistant", message: { content: [{ type: "text", text: STEPS }] } });
|
|
90
|
+
assert.ok(note, "a real first turn with real steps must produce a note");
|
|
91
|
+
const bullets = note.split("\n").filter((l) => l.startsWith("- "));
|
|
92
|
+
assert.equal(bullets.length, 3);
|
|
93
|
+
assert.match(bullets[0], /pull the July ledger/);
|
|
94
|
+
assert.match(bullets[1], /reconcile them line by line/);
|
|
95
|
+
assert.match(bullets[2], /three biggest gaps/);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
test("a numbered list is read as steps", () => {
|
|
99
|
+
const text = "Here's how I'll approach it:\n1. Pull the deal memos from the data room.\n2. Cross-check each against the signed term sheet.\n3. Flag anything the term sheet does not cover.";
|
|
100
|
+
const note = renderPlanNote(text);
|
|
101
|
+
assert.equal(note.split("\n").filter((l) => l.startsWith("- ")).length, 3);
|
|
102
|
+
assert.doesNotMatch(note, /^\s*1\./m, "the list markers are normalised away, not doubled");
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
test("more than three steps are CLIPPED to three, never summarised into something unsaid", () => {
|
|
106
|
+
const text = ["1. Pull the ledger rows for July.", "2. Reconcile against the forecast.", "3. Mark variances over five percent.", "4. Write the summary.", "5. Post it in the channel."].join("\n");
|
|
107
|
+
const note = renderPlanNote(text);
|
|
108
|
+
const bullets = note.split("\n").filter((l) => l.startsWith("- "));
|
|
109
|
+
assert.equal(bullets.length, PLAN_MAX_BULLETS);
|
|
110
|
+
assert.match(bullets[0], /Pull the ledger rows/);
|
|
111
|
+
assert.match(bullets[2], /variances over five percent/);
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
test("NEVER FABRICATE: every bullet is a substring of the model's own turn", () => {
|
|
115
|
+
const text = "1. Pull the deal memos from the data room.\n2. Cross-check each against the signed term sheet.";
|
|
116
|
+
const note = renderPlanNote(text);
|
|
117
|
+
for (const b of note.split("\n").filter((l) => l.startsWith("- "))) {
|
|
118
|
+
const body = b.slice(2).replace(/\.$/, "");
|
|
119
|
+
assert.ok(text.includes(body), `"${body}" is not text the model wrote`);
|
|
120
|
+
}
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("no first turn ⇒ null ⇒ nothing is posted", () => {
|
|
124
|
+
for (const bad of [null, undefined, "", {}, [], "{broken", { type: "result" }]) {
|
|
125
|
+
assert.equal(renderPlanNote(bad), null, `${JSON.stringify(bad)} must render nothing`);
|
|
126
|
+
}
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
test("a first turn with no stated intent ⇒ null, not a template", () => {
|
|
130
|
+
// Chatter, a question back, an answer — none of these is a plan, and
|
|
131
|
+
// dressing any of them up as one is the fabrication this rule forbids.
|
|
132
|
+
for (const text of [
|
|
133
|
+
"Sure, happy to look at that.",
|
|
134
|
+
"Which quarter did you mean — Q2 or Q3?",
|
|
135
|
+
"The board call is at 14:00 UK on Thursday.",
|
|
136
|
+
"Hmm.",
|
|
137
|
+
]) {
|
|
138
|
+
assert.equal(renderPlanNote(text), null, `"${text}" is not a plan`);
|
|
139
|
+
}
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
test("a single stated step is still a plan", () => {
|
|
143
|
+
const note = renderPlanNote("I'll reconcile the July ledger against the signed forecast and list every variance over 5%.");
|
|
144
|
+
assert.ok(note);
|
|
145
|
+
assert.equal(note.split("\n").filter((l) => l.startsWith("- ")).length, 1);
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
test("a generic opener is not a step and is dropped", () => {
|
|
149
|
+
// "On it" inside a numbered list is still "On it".
|
|
150
|
+
const note = renderPlanNote("1. On it.\n2. Looking into this now.\n3. Pull the July ledger rows and reconcile them against the forecast.");
|
|
151
|
+
assert.ok(note);
|
|
152
|
+
const bullets = note.split("\n").filter((l) => l.startsWith("- "));
|
|
153
|
+
assert.equal(bullets.length, 1);
|
|
154
|
+
assert.match(bullets[0], /July ledger/);
|
|
155
|
+
});
|
|
156
|
+
|
|
157
|
+
test("…and a turn that is ONLY generic openers renders nothing", () => {
|
|
158
|
+
assert.equal(renderPlanNote("1. On it.\n2. Checking.\n3. One moment."), null);
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
test("the note names no working agent and wears no assistant costume", () => {
|
|
162
|
+
const note = renderPlanNote(STEPS);
|
|
163
|
+
assert.doesNotMatch(note, /\b(Odette|Coh|the agent|automation)\b/i);
|
|
164
|
+
assert.doesNotMatch(note, /^(sure|certainly|absolutely|happy to|of course)\b/i);
|
|
165
|
+
assert.doesNotMatch(note, /let me know if|here's a comprehensive/i);
|
|
166
|
+
assert.doesNotMatch(note, /[\u{1F300}-\u{1FAFF}]/u, "no emoji");
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
test("bullets are clipped, so one runaway paragraph cannot become a wall of text", () => {
|
|
170
|
+
const long = `I'll ${"reconcile every single line of the ledger ".repeat(20)}now.`;
|
|
171
|
+
const note = renderPlanNote(long);
|
|
172
|
+
for (const line of note.split("\n")) assert.ok(line.length <= 200, `line of ${line.length} chars is not a bullet`);
|
|
173
|
+
});
|
|
174
|
+
|
|
175
|
+
test("the renderer is pure — same input, same output, input untouched", () => {
|
|
176
|
+
const events = [{ type: "assistant", message: { content: [{ type: "text", text: STEPS }] } }];
|
|
177
|
+
const before = JSON.stringify(events);
|
|
178
|
+
const first = renderPlanNote(events);
|
|
179
|
+
for (let i = 0; i < 20; i++) assert.equal(renderPlanNote(events), first);
|
|
180
|
+
assert.equal(JSON.stringify(events), before);
|
|
181
|
+
});
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
// ---------------------------------------------------------------------------
|
|
185
|
+
// The copy rule the plan shares with every other interim
|
|
186
|
+
// ---------------------------------------------------------------------------
|
|
187
|
+
|
|
188
|
+
describe("containsForwardPromise — no promise without a mechanism", () => {
|
|
189
|
+
test("it recognises the promises the production sample was full of", () => {
|
|
190
|
+
for (const t of [
|
|
191
|
+
"Still on this — 5 minutes in. I'll come back as soon as I've got something.",
|
|
192
|
+
"Still going — 15 minutes in. I'll follow up the moment it's done.",
|
|
193
|
+
"I'll get back to you shortly.",
|
|
194
|
+
"I will report back as soon as I have it.",
|
|
195
|
+
"I'll let you know once it's done.",
|
|
196
|
+
"I'll circle back on this.",
|
|
197
|
+
]) {
|
|
198
|
+
assert.equal(containsForwardPromise(t), true, `"${t}" is a forward promise`);
|
|
199
|
+
}
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
test("it does not flag a statement of what is being done now", () => {
|
|
203
|
+
for (const t of [
|
|
204
|
+
"Pulling the July ledger rows and reconciling them against the forecast.",
|
|
205
|
+
"I'm reconciling July against the forecast now.",
|
|
206
|
+
"Three variances so far, all in payroll.",
|
|
207
|
+
"Done — pushed as a41f0c9.",
|
|
208
|
+
]) {
|
|
209
|
+
assert.equal(containsForwardPromise(t), false, `"${t}" is not a promise`);
|
|
210
|
+
}
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
test("a bullet that is only a promise is dropped from the plan", () => {
|
|
214
|
+
const note = renderPlanNote("1. Pull the July ledger rows and reconcile against the forecast.\n2. I'll come back as soon as I've got something.");
|
|
215
|
+
const bullets = note.split("\n").filter((l) => l.startsWith("- "));
|
|
216
|
+
assert.equal(bullets.length, 1);
|
|
217
|
+
assert.doesNotMatch(note, FORWARD_PROMISE);
|
|
218
|
+
});
|
|
219
|
+
|
|
220
|
+
test("a rendered plan NEVER carries a forward promise", () => {
|
|
221
|
+
for (const text of [
|
|
222
|
+
STEPS,
|
|
223
|
+
"1. Draft the memo. 2. I'll follow up with the numbers.",
|
|
224
|
+
"I'll pull the rows, then I'll come back as soon as I've got something.",
|
|
225
|
+
]) {
|
|
226
|
+
const note = renderPlanNote(text);
|
|
227
|
+
if (note != null) assert.doesNotMatch(note, FORWARD_PROMISE, `"${note}" promises a return the sweep did not agree to`);
|
|
228
|
+
}
|
|
229
|
+
});
|
|
230
|
+
|
|
231
|
+
test("garbage in never throws", () => {
|
|
232
|
+
for (const bad of [null, undefined, 12, {}, []]) assert.equal(containsForwardPromise(bad), false);
|
|
233
|
+
});
|
|
234
|
+
});
|