@cohortapp/agent-sdk 2.15.0 → 2.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +5 -2
- package/docs/guides/front-door-session.md +16 -5
- package/docs/guides/poller-daemon-setup.md +53 -2
- package/lib/assurance/plan-note.mjs +251 -0
- package/lib/assurance/plan-note.test.mjs +234 -0
- package/lib/assurance/room-budget.mjs +497 -0
- package/lib/assurance/room-budget.test.mjs +486 -0
- package/lib/assurance/tier.mjs +166 -0
- package/lib/assurance/tier.test.mjs +174 -0
- package/lib/comms/receipts.mjs +17 -1
- package/lib/context/budget.mjs +327 -0
- package/lib/context/budget.test.mjs +252 -0
- package/lib/context/history-scope.mjs +138 -0
- package/lib/context/history-scope.test.mjs +79 -0
- package/lib/model-router/economics.mjs +9 -0
- package/lib/model-router/resolve.mjs +6 -0
- package/lib/org/inbound/facts.mjs +4 -2
- package/lib/org/inbound/hydrate.mjs +555 -51
- package/lib/org/inbound/hydrate.test.mjs +456 -1
- package/package.json +3 -1
- package/plugins/maestro-skills/skills/inbound-triage.md +52 -24
- package/plugins/maestro-skills/skills/main-session.md +6 -4
- package/scripts/daemon/agent-daemon.mjs +35 -7
- package/scripts/daemon/agent-daemon.test.mjs +23 -6
- package/scripts/daemon/assurance-e2e.test.mjs +75 -19
- package/scripts/daemon/assurance.mjs +663 -159
- package/scripts/daemon/assurance.test.mjs +820 -140
- package/scripts/daemon/context-compiler.mjs +52 -21
- package/scripts/daemon/context-compiler.test.mjs +106 -0
- package/scripts/daemon/deliver.mjs +7 -4
- package/scripts/daemon/dispatcher-session-continuity.test.mjs +365 -0
- package/scripts/daemon/dispatcher.mjs +210 -9
- package/scripts/daemon/lib/session-router.mjs +310 -42
- package/scripts/daemon/lib/session-router.test.mjs +260 -1
- package/scripts/daemon/prompt-builder.mjs +160 -16
- package/scripts/daemon/prompt-builder.test.mjs +287 -7
- package/scripts/daemon/responder-history.test.mjs +37 -1
- package/scripts/daemon/responder.mjs +79 -72
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* tier.test.mjs — the reply tier, across the whole classifier matrix.
|
|
3
|
+
*
|
|
4
|
+
* The tier is the ONE decision that says whether a human hears anything before
|
|
5
|
+
* the answer. Getting it wrong in the loud direction is the 3,069 generic acks
|
|
6
|
+
* and 961 clock-derived nags measured in production on 2026-09-12; getting it
|
|
7
|
+
* wrong in the quiet direction is the silence the obligation ledger was built
|
|
8
|
+
* to end. So every cell of the matrix is pinned here rather than sampled.
|
|
9
|
+
*
|
|
10
|
+
* Run: node --test lib/assurance/tier.test.mjs
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { test, describe } from "node:test";
|
|
14
|
+
import assert from "node:assert/strict";
|
|
15
|
+
|
|
16
|
+
import {
|
|
17
|
+
replyTier,
|
|
18
|
+
TIERS,
|
|
19
|
+
ANSWER_MAX_RUNG,
|
|
20
|
+
PLAN_MIN_RUNG,
|
|
21
|
+
PLAN_ACTIONS,
|
|
22
|
+
PLAN_PRIORITIES,
|
|
23
|
+
} from "./tier.mjs";
|
|
24
|
+
|
|
25
|
+
/** Every value `classifier.mjs` can emit — the real enums, not a sample. */
|
|
26
|
+
const ACTIONS = ["respond", "draft", "research", "queue", "archive", "ignore"];
|
|
27
|
+
const PRIORITIES = ["critical", "high", "normal", "ignore"];
|
|
28
|
+
/** Every rung in `lib/execution/route.RUNGS`, plus "not routed yet". */
|
|
29
|
+
const RUNGS = [0, 1, 2, 3, 4, 5, null];
|
|
30
|
+
|
|
31
|
+
describe("replyTier — the three tiers and nothing else", () => {
|
|
32
|
+
test("every cell of the matrix returns one of exactly three tiers", () => {
|
|
33
|
+
for (const action of ACTIONS) {
|
|
34
|
+
for (const priority of PRIORITIES) {
|
|
35
|
+
for (const rung of RUNGS) {
|
|
36
|
+
for (const answerable of [true, false, null, undefined]) {
|
|
37
|
+
for (const willSpawnSession of [true, false]) {
|
|
38
|
+
const t = replyTier({ answerable, action, priority, rung, willSpawnSession });
|
|
39
|
+
assert.ok(TIERS.includes(t), `{${action}/${priority}/rung:${rung}/answerable:${answerable}/spawn:${willSpawnSession}} → ${t}`);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test("no session spawning ⇒ answer tier, whatever the classifier said", () => {
|
|
48
|
+
// The reply arrives in this turn. An interim in front of an answer the
|
|
49
|
+
// human is about to read is the definition of content-free traffic.
|
|
50
|
+
for (const action of ACTIONS) {
|
|
51
|
+
for (const priority of PRIORITIES) {
|
|
52
|
+
for (const rung of RUNGS) {
|
|
53
|
+
assert.equal(
|
|
54
|
+
replyTier({ answerable: false, action, priority, rung, willSpawnSession: false }),
|
|
55
|
+
"answer",
|
|
56
|
+
`${action}/${priority}/rung:${rung} with no session must be "answer"`,
|
|
57
|
+
);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
test("answerable at rung 0 or 1 is the answer tier — the reply IS the acknowledgement", () => {
|
|
64
|
+
// `willSpawnSession` UNKNOWN: the caller is asking hypothetically (which is
|
|
65
|
+
// how WP-2's effort router will use it), so the classifier verdict decides.
|
|
66
|
+
for (const rung of [0, ANSWER_MAX_RUNG]) {
|
|
67
|
+
assert.equal(replyTier({ answerable: true, action: "respond", priority: "high", rung }), "answer");
|
|
68
|
+
assert.equal(replyTier({ answerable: true, action: "respond", priority: "high", rung, willSpawnSession: false }), "answer");
|
|
69
|
+
}
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test("REGRESSION: an ASSERTED session disqualifies the answer tier, however answerable the item looked", () => {
|
|
73
|
+
// THE QUICK-PATH FALL-THROUGH. agent-daemon.mjs tries a quick reply first;
|
|
74
|
+
// when that reply fails transiently or is blocked by validation it falls
|
|
75
|
+
// THROUGH to a full session dispatch and asks for a tier with
|
|
76
|
+
// willSpawnSession:true while classResult.answerable is still true. The
|
|
77
|
+
// answer tier's whole premise — "the reply arrives in this turn" — is
|
|
78
|
+
// exactly what the fall-through has falsified.
|
|
79
|
+
//
|
|
80
|
+
// Latent rather than live only because `rung` is null today (R13). The
|
|
81
|
+
// moment WP-2 routes it, that item would have landed in `answer`,
|
|
82
|
+
// shouldAcknowledge would have returned ack:false, the sweep would have
|
|
83
|
+
// read the durable tier as never-speak, and a 15-45 minute session would
|
|
84
|
+
// have run with the human hearing NOTHING at all. Pinned before WP-2 lands
|
|
85
|
+
// on top of it.
|
|
86
|
+
for (const rung of [0, ANSWER_MAX_RUNG]) {
|
|
87
|
+
assert.equal(
|
|
88
|
+
replyTier({ answerable: true, action: "respond", priority: "high", rung, willSpawnSession: true }),
|
|
89
|
+
"work",
|
|
90
|
+
`rung ${rung}: a spawning session speaks once at most — it is never silent`,
|
|
91
|
+
);
|
|
92
|
+
}
|
|
93
|
+
// …and the plan tier still outranks it, because rung 3+ is a project.
|
|
94
|
+
assert.equal(replyTier({ answerable: true, action: "respond", priority: "high", rung: 4, willSpawnSession: true }), "plan");
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
test("a routed rung of 3 or above is the plan tier, whatever the action", () => {
|
|
98
|
+
for (const rung of RUNGS.filter((r) => r != null && r >= PLAN_MIN_RUNG)) {
|
|
99
|
+
for (const action of ACTIONS) {
|
|
100
|
+
assert.equal(
|
|
101
|
+
replyTier({ answerable: false, action, priority: "normal", rung, willSpawnSession: true }),
|
|
102
|
+
"plan",
|
|
103
|
+
`rung ${rung} / ${action} must be "plan"`,
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
test("research or draft at critical/high priority is the plan tier even with no routed rung", () => {
|
|
110
|
+
for (const action of PLAN_ACTIONS) {
|
|
111
|
+
for (const priority of PLAN_PRIORITIES) {
|
|
112
|
+
assert.equal(replyTier({ answerable: false, action, priority, rung: null, willSpawnSession: true }), "plan");
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
test("…and the same actions at normal priority are only work", () => {
|
|
118
|
+
for (const action of PLAN_ACTIONS) {
|
|
119
|
+
for (const priority of ["normal", "ignore"]) {
|
|
120
|
+
assert.equal(replyTier({ answerable: false, action, priority, rung: null, willSpawnSession: true }), "work");
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
});
|
|
124
|
+
|
|
125
|
+
test("the ordinary directed ask — respond/high, unrouted — is work: silence, then at most one line", () => {
|
|
126
|
+
// This is the shape of the overwhelming majority of the 10,667 measured
|
|
127
|
+
// agent messages. It must NOT be plan tier, or the flood returns wearing a
|
|
128
|
+
// better costume.
|
|
129
|
+
assert.equal(replyTier({ answerable: false, action: "respond", priority: "high", rung: null, willSpawnSession: true }), "work");
|
|
130
|
+
assert.equal(replyTier({ answerable: false, action: "queue", priority: "critical", rung: null, willSpawnSession: true }), "work");
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
test("an ABSENT rung is never read as the cheapest rung", () => {
|
|
134
|
+
// `lib/backlog` already learned this: absence is not rung 0. A missing rung
|
|
135
|
+
// may not buy the answer tier's silence-on-the-strength-of-a-quick-path,
|
|
136
|
+
// and may not buy the plan tier's licence to speak either.
|
|
137
|
+
assert.equal(replyTier({ answerable: true, action: "respond", priority: "high", rung: null, willSpawnSession: true }), "work");
|
|
138
|
+
assert.equal(replyTier({ answerable: true, action: "respond", priority: "high", rung: undefined, willSpawnSession: true }), "work");
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
test("a non-numeric or out-of-range rung is treated as unrouted, never as a licence to speak", () => {
|
|
142
|
+
for (const rung of ["3", "team", NaN, Infinity, -1, 6, {}, []]) {
|
|
143
|
+
assert.equal(
|
|
144
|
+
replyTier({ answerable: false, action: "respond", priority: "normal", rung, willSpawnSession: true }),
|
|
145
|
+
"work",
|
|
146
|
+
`rung ${JSON.stringify(rung)} must not be trusted`,
|
|
147
|
+
);
|
|
148
|
+
}
|
|
149
|
+
});
|
|
150
|
+
|
|
151
|
+
test("answerable only counts when it is strictly true", () => {
|
|
152
|
+
// classifier.mjs coerces a missing/odd field to false for the same reason:
|
|
153
|
+
// never guess "answerable".
|
|
154
|
+
for (const answerable of ["true", 1, {}, null, undefined]) {
|
|
155
|
+
assert.equal(replyTier({ answerable, action: "respond", priority: "high", rung: 0, willSpawnSession: true }), "work");
|
|
156
|
+
}
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
test("garbage in gives work out — the tier never throws and never invents silence", () => {
|
|
160
|
+
// Fail-open direction: the tier that can still speak, never the tier that
|
|
161
|
+
// cannot. A classifier failure must not silently mute the agent.
|
|
162
|
+
assert.equal(replyTier(), "work");
|
|
163
|
+
assert.equal(replyTier(null), "work");
|
|
164
|
+
assert.equal(replyTier({}), "work");
|
|
165
|
+
assert.equal(replyTier({ action: 12, priority: [], rung: "x", willSpawnSession: "yes" }), "work");
|
|
166
|
+
});
|
|
167
|
+
|
|
168
|
+
test("the tier is a pure function — same input, same answer, no ambient reads", () => {
|
|
169
|
+
const args = { answerable: false, action: "research", priority: "critical", rung: null, willSpawnSession: true };
|
|
170
|
+
const first = replyTier(args);
|
|
171
|
+
for (let i = 0; i < 50; i++) assert.equal(replyTier(args), first);
|
|
172
|
+
assert.deepEqual(args, { answerable: false, action: "research", priority: "critical", rung: null, willSpawnSession: true }, "the argument is not mutated");
|
|
173
|
+
});
|
|
174
|
+
});
|
package/lib/comms/receipts.mjs
CHANGED
|
@@ -211,12 +211,28 @@ export function spokeSince(q = {}) {
|
|
|
211
211
|
* @param {string} [q.agentRoot]
|
|
212
212
|
* @returns {{heard:boolean, basis:string}}
|
|
213
213
|
*/
|
|
214
|
+
/**
|
|
215
|
+
* Receipt kinds that are COURTESIES, not answers.
|
|
216
|
+
*
|
|
217
|
+
* A receipt of one of these proves the agent said something in the room; it
|
|
218
|
+
* does not prove the ask was answered, and discharging a debt on one closes an
|
|
219
|
+
* unanswered question as answered.
|
|
220
|
+
*
|
|
221
|
+
* `progress` is retained although nothing emits it any more (deleted
|
|
222
|
+
* 2026-09-12 — see `scripts/daemon/assurance.mjs`): receipts already on disk
|
|
223
|
+
* carry it, and a historical progress ping must not start discharging debts on
|
|
224
|
+
* the day it stops being written. `notice` is the interrupted/orphan apology —
|
|
225
|
+
* information about the MACHINE, not about the ask. `plan` is the plan-tier
|
|
226
|
+
* interim.
|
|
227
|
+
*/
|
|
228
|
+
export const NON_ANSWER_KINDS = Object.freeze(["ack", "progress", "notice", "failure", "plan"]);
|
|
229
|
+
|
|
214
230
|
export function spokeFor(q = {}) {
|
|
215
231
|
const rec = q.obligation || {};
|
|
216
232
|
const ch = channelKey(rec.channel);
|
|
217
233
|
const since = Number.isFinite(rec.openedAt) ? rec.openedAt : q.now;
|
|
218
234
|
if (!ch || !Number.isFinite(since)) return { heard: false, basis: "no-channel" };
|
|
219
|
-
const exclude = Array.isArray(q.excludeKinds) ? q.excludeKinds :
|
|
235
|
+
const exclude = Array.isArray(q.excludeKinds) ? q.excludeKinds : NON_ANSWER_KINDS;
|
|
220
236
|
const recs = readReceiptsSince(since, q).filter(
|
|
221
237
|
(r) => r.channel === ch && !exclude.includes(r.kind),
|
|
222
238
|
);
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/context/budget.mjs — the declared token budget for an inbound prompt.
|
|
3
|
+
*
|
|
4
|
+
* Every bound in this repo up to now has been a row count or a `.slice(n)` char
|
|
5
|
+
* clamp (design §5.4, R8). Both are proxies: a row count cannot tell a one-word
|
|
6
|
+
* turn from a pasted stack trace, and a char clamp spends the allowance on
|
|
7
|
+
* whichever section happens to be assembled first. So the prompt that reaches
|
|
8
|
+
* the model is whatever the assembly order happened to produce, and the parts
|
|
9
|
+
* that fall off do so SILENTLY — the model reads a thread with a hole in it and
|
|
10
|
+
* cannot tell the hole from a conversation that never happened.
|
|
11
|
+
*
|
|
12
|
+
* This module states the budget instead of discovering it:
|
|
13
|
+
*
|
|
14
|
+
* 1. a budget PER TIER (§5.1) — an Answer gets 6k input tokens, Work 12k,
|
|
15
|
+
* Plan 20k. Context rot is measured: accuracy falls with input length well
|
|
16
|
+
* before the window limit, so a bigger prompt is not a better one.
|
|
17
|
+
* 2. sections filled in PRIORITY order ({@link SECTION_ORDER}) — the trigger
|
|
18
|
+
* is the thing being answered and is never dropped whole; the backlog is
|
|
19
|
+
* the first thing to go.
|
|
20
|
+
* 3. every drop VISIBLE — `… 14 earlier turns not shown`, in the prompt, where
|
|
21
|
+
* the model reads it. A section that did not fit at all still leaves a line
|
|
22
|
+
* saying so.
|
|
23
|
+
*
|
|
24
|
+
* PURE. No clock, no env, no fs, no network: tokens are estimated from
|
|
25
|
+
* character length at {@link CHARS_PER_TOKEN}, the same ratio
|
|
26
|
+
* `context-compiler.mjs` already uses, so the two agree. A caller that has a
|
|
27
|
+
* real tokeniser passes `estimate` and this module uses it instead.
|
|
28
|
+
*
|
|
29
|
+
* Char clamps stay where they are as a second belt — this is the first one.
|
|
30
|
+
*
|
|
31
|
+
* @module lib/context/budget
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
"use strict";
|
|
35
|
+
|
|
36
|
+
/** Characters per token. The estimate `scripts/daemon/context-compiler.mjs` uses. */
|
|
37
|
+
export const CHARS_PER_TOKEN = 4;
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Input-token budget per reply tier (design §5.4). A tier is the ack/effort
|
|
41
|
+
* tier from `lib/assurance/tier.mjs` — answer | work | plan.
|
|
42
|
+
*/
|
|
43
|
+
export const TIER_BUDGETS = Object.freeze({
|
|
44
|
+
answer: 6000,
|
|
45
|
+
work: 12000,
|
|
46
|
+
plan: 20000,
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* FILL priority. Earlier wins the budget; later is dropped first.
|
|
51
|
+
*
|
|
52
|
+
* trigger the message being answered — without it there is nothing to answer
|
|
53
|
+
* thread the conversation it sits in
|
|
54
|
+
* entity the anchored artifact: the task card, the doc body, the decision
|
|
55
|
+
* digest the distilled record of this thread
|
|
56
|
+
* memory recalled facts from elsewhere
|
|
57
|
+
* backlog what else is open — useful, never load-bearing
|
|
58
|
+
*/
|
|
59
|
+
export const FILL_PRIORITY = Object.freeze([
|
|
60
|
+
"trigger",
|
|
61
|
+
"thread",
|
|
62
|
+
"entity",
|
|
63
|
+
"digest",
|
|
64
|
+
"memory",
|
|
65
|
+
"backlog",
|
|
66
|
+
]);
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* RENDER order — a DIFFERENT question, and conflating the two costs the cache.
|
|
70
|
+
*
|
|
71
|
+
* Fill priority asks "who wins the budget"; the trigger wins it, because a
|
|
72
|
+
* prompt without the message being answered is not a smaller prompt but a
|
|
73
|
+
* broken one. Render order asks "what byte comes first", and design §4 is
|
|
74
|
+
* explicit about the consequence: "any byte change before a breakpoint
|
|
75
|
+
* invalidates everything after it". Rendering the trigger first — the ONE
|
|
76
|
+
* section that is different on every single inbound — gives the block a
|
|
77
|
+
* zero-length stable prefix, so WP-4's `cache_control` breakpoint would sit
|
|
78
|
+
* after volatile content and every read would be a silent miss. Prompt caching
|
|
79
|
+
* is the highest-ROI latency lever in the research pass; this is what it needs.
|
|
80
|
+
*
|
|
81
|
+
* So: stable first, volatile last, trigger last of all. `thread` sits second to
|
|
82
|
+
* last because it is append-only and rendered oldest-first, which makes its own
|
|
83
|
+
* prefix stable between turns even as it grows.
|
|
84
|
+
*/
|
|
85
|
+
export const RENDER_ORDER = Object.freeze([
|
|
86
|
+
"entity",
|
|
87
|
+
"digest",
|
|
88
|
+
"backlog",
|
|
89
|
+
"memory",
|
|
90
|
+
"thread",
|
|
91
|
+
"trigger",
|
|
92
|
+
]);
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* @deprecated Read {@link FILL_PRIORITY} or {@link RENDER_ORDER} by name — the
|
|
96
|
+
* whole point of the split is that "the section order" is two different orders.
|
|
97
|
+
* Kept as an alias so an existing importer does not silently get the wrong one.
|
|
98
|
+
*/
|
|
99
|
+
export const SECTION_ORDER = FILL_PRIORITY;
|
|
100
|
+
|
|
101
|
+
/** Sections that must render SOMETHING rather than be dropped whole. */
|
|
102
|
+
const NEVER_DROPPED_WHOLE = new Set(["trigger"]);
|
|
103
|
+
|
|
104
|
+
/** A blank-safe string. */
|
|
105
|
+
function s(v) {
|
|
106
|
+
return v == null ? "" : String(v);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** The declared budget for a tier; an unknown tier takes the middle one. */
|
|
110
|
+
export function budgetFor(tier) {
|
|
111
|
+
const key = s(tier).toLowerCase();
|
|
112
|
+
return Object.prototype.hasOwnProperty.call(TIER_BUDGETS, key) ? TIER_BUDGETS[key] : TIER_BUDGETS.work;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Estimated tokens for a string, rounded up so a partial token still costs one. */
|
|
116
|
+
export function estimateTokens(text, charsPerToken = CHARS_PER_TOKEN) {
|
|
117
|
+
const t = s(text);
|
|
118
|
+
if (!t) return 0;
|
|
119
|
+
const per = Number.isFinite(charsPerToken) && charsPerToken > 0 ? charsPerToken : CHARS_PER_TOKEN;
|
|
120
|
+
return Math.ceil(t.length / per);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/** Where a section sorts in a declared order; unknown names sort last, stably. */
|
|
124
|
+
function rankIn(order, name) {
|
|
125
|
+
const i = order.indexOf(name);
|
|
126
|
+
return i === -1 ? order.length : i;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/** `1 earlier turn` / `14 earlier turns` — the count is the point, so it leads. */
|
|
130
|
+
function turnNote(n) {
|
|
131
|
+
return `… ${n} earlier turn${n === 1 ? "" : "s"} not shown`;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** `… 1,204 characters not shown`. */
|
|
135
|
+
function charNote(n) {
|
|
136
|
+
return `… ${n.toLocaleString("en-US")} character${n === 1 ? "" : "s"} not shown`;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* @typedef {Object} SectionInput
|
|
141
|
+
* @property {string} name one of {@link SECTION_ORDER}, or any name (sorts last)
|
|
142
|
+
* @property {string} [title] heading rendered above the body
|
|
143
|
+
* @property {string} [text] a whole-text section
|
|
144
|
+
* @property {string[]} [turns] a turn-structured section, OLDEST FIRST
|
|
145
|
+
*/
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* @typedef {Object} FittedSection
|
|
149
|
+
* @property {string} name
|
|
150
|
+
* @property {boolean} included false == dropped whole (still rendered as a note)
|
|
151
|
+
* @property {string} text what was kept, drop note included
|
|
152
|
+
* @property {number} keptTurns
|
|
153
|
+
* @property {number} droppedTurns
|
|
154
|
+
* @property {number} droppedChars
|
|
155
|
+
* @property {number} tokens
|
|
156
|
+
*/
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Fit the sections into the tier's budget.
|
|
160
|
+
*
|
|
161
|
+
* Turn-structured sections spend the budget from the NEWEST turn backwards —
|
|
162
|
+
* the turns nearest the message being answered are the ones that explain it
|
|
163
|
+
* (the same rule `lib/org/inbound/hydrate.mjs#renderThread` follows, and for the
|
|
164
|
+
* same reason: clipping the joined string keeps the oldest turns and drops the
|
|
165
|
+
* newest, which is the exact inverse of what context is for).
|
|
166
|
+
*
|
|
167
|
+
* NEVER THROWS. A malformed section is skipped; a missing budget takes the
|
|
168
|
+
* tier's. The prompt path must not be able to fail on its own budgeting.
|
|
169
|
+
*
|
|
170
|
+
* @param {Object} o
|
|
171
|
+
* @param {SectionInput[]} o.sections
|
|
172
|
+
* @param {string} [o.tier] answer | work | plan
|
|
173
|
+
* @param {number} [o.budgetTokens] override the tier's budget
|
|
174
|
+
* @param {number} [o.charsPerToken]
|
|
175
|
+
* @param {(text:string)=>number} [o.estimate] a real tokeniser, when the caller has one
|
|
176
|
+
* @returns {{tier:string, budgetTokens:number, usedTokens:number,
|
|
177
|
+
* sections:FittedSection[], drops:Array<object>, text:string}}
|
|
178
|
+
*/
|
|
179
|
+
export function fitSections(o = {}) {
|
|
180
|
+
const tier = s(o.tier).toLowerCase() || "work";
|
|
181
|
+
const budgetTokens = Number.isFinite(o.budgetTokens) && o.budgetTokens > 0
|
|
182
|
+
? Math.floor(o.budgetTokens)
|
|
183
|
+
: budgetFor(tier);
|
|
184
|
+
const charsPerToken = Number.isFinite(o.charsPerToken) && o.charsPerToken > 0 ? o.charsPerToken : CHARS_PER_TOKEN;
|
|
185
|
+
const tokensOf = typeof o.estimate === "function" ? o.estimate : (t) => estimateTokens(t, charsPerToken);
|
|
186
|
+
|
|
187
|
+
const input = (Array.isArray(o.sections) ? o.sections : [])
|
|
188
|
+
.map((sec, i) => ({ sec, i }))
|
|
189
|
+
.filter(({ sec }) => sec && typeof sec === "object" && s(sec.name))
|
|
190
|
+
.sort((a, b) => rankIn(FILL_PRIORITY, s(a.sec.name)) - rankIn(FILL_PRIORITY, s(b.sec.name)) || a.i - b.i);
|
|
191
|
+
|
|
192
|
+
const out = [];
|
|
193
|
+
const drops = [];
|
|
194
|
+
let used = 0;
|
|
195
|
+
|
|
196
|
+
for (const { sec } of input) {
|
|
197
|
+
const name = s(sec.name);
|
|
198
|
+
const title = s(sec.title);
|
|
199
|
+
const headCost = title ? tokensOf(`${title}\n`) : 0;
|
|
200
|
+
const room = budgetTokens - used - headCost;
|
|
201
|
+
|
|
202
|
+
const fitted = Array.isArray(sec.turns)
|
|
203
|
+
? fitTurns(sec.turns, room, tokensOf)
|
|
204
|
+
: fitText(s(sec.text), room, tokensOf, charsPerToken);
|
|
205
|
+
|
|
206
|
+
// A section with no content at all is simply absent — there is nothing to
|
|
207
|
+
// report the loss of, and a note about an empty section is noise.
|
|
208
|
+
if (fitted.empty) continue;
|
|
209
|
+
|
|
210
|
+
if (!fitted.body && !NEVER_DROPPED_WHOLE.has(name)) {
|
|
211
|
+
// Dropped whole. It still shows: an absence the reader cannot see is the
|
|
212
|
+
// failure this module exists to prevent.
|
|
213
|
+
const note = `… ${name} not shown`;
|
|
214
|
+
const cost = tokensOf(note);
|
|
215
|
+
if (used + cost <= budgetTokens) {
|
|
216
|
+
out.push({ name, included: false, text: note, keptTurns: 0, droppedTurns: fitted.droppedTurns, droppedChars: fitted.droppedChars, tokens: cost });
|
|
217
|
+
used += cost;
|
|
218
|
+
} else {
|
|
219
|
+
out.push({ name, included: false, text: "", keptTurns: 0, droppedTurns: fitted.droppedTurns, droppedChars: fitted.droppedChars, tokens: 0 });
|
|
220
|
+
}
|
|
221
|
+
drops.push({ section: name, turns: fitted.droppedTurns, chars: fitted.droppedChars, note });
|
|
222
|
+
continue;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// The trigger is the thing being answered: it always renders something,
|
|
226
|
+
// even when the budget cannot pay for it. A prompt with no trigger is not a
|
|
227
|
+
// smaller prompt, it is a broken one.
|
|
228
|
+
let body = fitted.body;
|
|
229
|
+
if (!body && NEVER_DROPPED_WHOLE.has(name)) {
|
|
230
|
+
const minChars = Math.max(1, Math.floor(Math.max(room, 1) * charsPerToken));
|
|
231
|
+
const whole = Array.isArray(sec.turns) ? sec.turns.filter(Boolean).map(s).join("\n") : s(sec.text);
|
|
232
|
+
body = whole.slice(0, minChars);
|
|
233
|
+
const lost = whole.length - body.length;
|
|
234
|
+
if (lost > 0) {
|
|
235
|
+
body += `\n${charNote(lost)}`;
|
|
236
|
+
drops.push({ section: name, turns: 0, chars: lost, note: charNote(lost) });
|
|
237
|
+
}
|
|
238
|
+
} else if (fitted.droppedTurns > 0) {
|
|
239
|
+
drops.push({ section: name, turns: fitted.droppedTurns, chars: 0, note: turnNote(fitted.droppedTurns) });
|
|
240
|
+
} else if (fitted.droppedChars > 0) {
|
|
241
|
+
drops.push({ section: name, turns: 0, chars: fitted.droppedChars, note: charNote(fitted.droppedChars) });
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
const text = title ? `${title}\n${body}` : body;
|
|
245
|
+
const cost = tokensOf(text);
|
|
246
|
+
out.push({
|
|
247
|
+
name,
|
|
248
|
+
included: true,
|
|
249
|
+
text,
|
|
250
|
+
keptTurns: fitted.keptTurns,
|
|
251
|
+
droppedTurns: fitted.droppedTurns,
|
|
252
|
+
droppedChars: fitted.droppedChars,
|
|
253
|
+
tokens: cost,
|
|
254
|
+
});
|
|
255
|
+
used += cost;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
// Filled in priority order, RENDERED in cache order. The two are separate
|
|
259
|
+
// arrays for the reason RENDER_ORDER states, and this is the only place the
|
|
260
|
+
// difference is applied.
|
|
261
|
+
const rendered = out
|
|
262
|
+
.map((sec, i) => ({ sec, i }))
|
|
263
|
+
.sort((a, b) => rankIn(RENDER_ORDER, a.sec.name) - rankIn(RENDER_ORDER, b.sec.name) || a.i - b.i)
|
|
264
|
+
.map(({ sec }) => sec);
|
|
265
|
+
|
|
266
|
+
return {
|
|
267
|
+
tier,
|
|
268
|
+
budgetTokens,
|
|
269
|
+
usedTokens: used,
|
|
270
|
+
sections: rendered,
|
|
271
|
+
drops,
|
|
272
|
+
text: rendered.map((sec) => sec.text).filter(Boolean).join("\n\n"),
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** Keep the newest turns that fit; report the rest as a visible note. */
|
|
277
|
+
function fitTurns(turns, room, tokensOf) {
|
|
278
|
+
const rows = (Array.isArray(turns) ? turns : []).filter((t) => s(t).trim()).map((t) => s(t));
|
|
279
|
+
if (rows.length === 0) return { empty: true, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: 0 };
|
|
280
|
+
if (room <= 0) return { empty: false, body: "", keptTurns: 0, droppedTurns: rows.length, droppedChars: 0 };
|
|
281
|
+
|
|
282
|
+
const kept = [];
|
|
283
|
+
let used = 0;
|
|
284
|
+
for (let i = rows.length - 1; i >= 0; i -= 1) {
|
|
285
|
+
const cost = tokensOf(rows[i]) + (kept.length ? 1 : 0);
|
|
286
|
+
// The note itself has to fit, or the drop would be the silent kind.
|
|
287
|
+
const reserve = i > 0 ? tokensOf(turnNote(i)) + 1 : 0;
|
|
288
|
+
if (used + cost + reserve > room) break;
|
|
289
|
+
kept.unshift(rows[i]);
|
|
290
|
+
used += cost;
|
|
291
|
+
}
|
|
292
|
+
const dropped = rows.length - kept.length;
|
|
293
|
+
if (kept.length === 0) return { empty: false, body: "", keptTurns: 0, droppedTurns: dropped, droppedChars: 0 };
|
|
294
|
+
const body = dropped > 0 ? [turnNote(dropped), ...kept].join("\n") : kept.join("\n");
|
|
295
|
+
return { empty: false, body, keptTurns: kept.length, droppedTurns: dropped, droppedChars: 0 };
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/** Keep the head that fits; report the tail as a visible note. */
|
|
299
|
+
function fitText(text, room, tokensOf, charsPerToken) {
|
|
300
|
+
const t = s(text).trim();
|
|
301
|
+
if (!t) return { empty: true, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: 0 };
|
|
302
|
+
if (room <= 0) return { empty: false, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: t.length };
|
|
303
|
+
if (tokensOf(t) <= room) return { empty: false, body: t, keptTurns: 1, droppedTurns: 0, droppedChars: 0 };
|
|
304
|
+
|
|
305
|
+
// Reserve room for the note, then take the largest head that still fits.
|
|
306
|
+
let chars = Math.max(0, Math.floor(room * charsPerToken));
|
|
307
|
+
let head = t.slice(0, chars);
|
|
308
|
+
let note = charNote(t.length - head.length);
|
|
309
|
+
while (head.length > 0 && tokensOf(`${head}\n${note}`) > room) {
|
|
310
|
+
chars = Math.floor(chars * 0.8);
|
|
311
|
+
head = t.slice(0, chars);
|
|
312
|
+
note = charNote(t.length - head.length);
|
|
313
|
+
}
|
|
314
|
+
if (!head) return { empty: false, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: t.length };
|
|
315
|
+
return { empty: false, body: `${head}\n${note}`, keptTurns: 1, droppedTurns: 0, droppedChars: t.length - head.length };
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
export default {
|
|
319
|
+
TIER_BUDGETS,
|
|
320
|
+
FILL_PRIORITY,
|
|
321
|
+
RENDER_ORDER,
|
|
322
|
+
SECTION_ORDER,
|
|
323
|
+
CHARS_PER_TOKEN,
|
|
324
|
+
budgetFor,
|
|
325
|
+
estimateTokens,
|
|
326
|
+
fitSections,
|
|
327
|
+
};
|