@nebutra/agent-runtime 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (179) hide show
  1. package/.turbo/turbo-build.log +115 -0
  2. package/.turbo/turbo-test.log +44 -0
  3. package/.turbo/turbo-typecheck.log +4 -0
  4. package/CHANGELOG.md +253 -0
  5. package/LICENSE +676 -0
  6. package/README.md +50 -0
  7. package/dist/adapters/dispatcher-sse.d.ts +68 -0
  8. package/dist/adapters/dispatcher-sse.js +11 -0
  9. package/dist/adapters/dispatcher-sse.js.map +1 -0
  10. package/dist/adapters/index.d.ts +12 -0
  11. package/dist/adapters/index.js +21 -0
  12. package/dist/adapters/index.js.map +1 -0
  13. package/dist/adapters/mcp-catalog.d.ts +58 -0
  14. package/dist/adapters/mcp-catalog.js +9 -0
  15. package/dist/adapters/mcp-catalog.js.map +1 -0
  16. package/dist/adapters/prisma-rollout.d.ts +60 -0
  17. package/dist/adapters/prisma-rollout.js +7 -0
  18. package/dist/adapters/prisma-rollout.js.map +1 -0
  19. package/dist/chunk-24ZXP7FI.js +93 -0
  20. package/dist/chunk-24ZXP7FI.js.map +1 -0
  21. package/dist/chunk-2DA6Q6TN.js +126 -0
  22. package/dist/chunk-2DA6Q6TN.js.map +1 -0
  23. package/dist/chunk-37BBB2P2.js +73 -0
  24. package/dist/chunk-37BBB2P2.js.map +1 -0
  25. package/dist/chunk-57W3AR43.js +52 -0
  26. package/dist/chunk-57W3AR43.js.map +1 -0
  27. package/dist/chunk-5N4644PB.js +67 -0
  28. package/dist/chunk-5N4644PB.js.map +1 -0
  29. package/dist/chunk-5YS7WAPS.js +177 -0
  30. package/dist/chunk-5YS7WAPS.js.map +1 -0
  31. package/dist/chunk-6EGG2OZC.js +13 -0
  32. package/dist/chunk-6EGG2OZC.js.map +1 -0
  33. package/dist/chunk-7BUOF367.js +126 -0
  34. package/dist/chunk-7BUOF367.js.map +1 -0
  35. package/dist/chunk-BJBBR3QA.js +121 -0
  36. package/dist/chunk-BJBBR3QA.js.map +1 -0
  37. package/dist/chunk-CGRCUKGT.js +73 -0
  38. package/dist/chunk-CGRCUKGT.js.map +1 -0
  39. package/dist/chunk-FUG5DT2C.js +75 -0
  40. package/dist/chunk-FUG5DT2C.js.map +1 -0
  41. package/dist/chunk-LO24VOA3.js +199 -0
  42. package/dist/chunk-LO24VOA3.js.map +1 -0
  43. package/dist/chunk-MUF7ZZTO.js +57 -0
  44. package/dist/chunk-MUF7ZZTO.js.map +1 -0
  45. package/dist/chunk-NN7DATXA.js +46 -0
  46. package/dist/chunk-NN7DATXA.js.map +1 -0
  47. package/dist/chunk-PGGWSUTM.js +33 -0
  48. package/dist/chunk-PGGWSUTM.js.map +1 -0
  49. package/dist/chunk-RDKYDMXT.js +135 -0
  50. package/dist/chunk-RDKYDMXT.js.map +1 -0
  51. package/dist/chunk-YYFPDBJG.js +63 -0
  52. package/dist/chunk-YYFPDBJG.js.map +1 -0
  53. package/dist/chunk-ZMYX5VBU.js +135 -0
  54. package/dist/chunk-ZMYX5VBU.js.map +1 -0
  55. package/dist/chunk-ZTSKS42I.js +131 -0
  56. package/dist/chunk-ZTSKS42I.js.map +1 -0
  57. package/dist/commands.d.ts +74 -0
  58. package/dist/commands.js +10 -0
  59. package/dist/commands.js.map +1 -0
  60. package/dist/definitions.d.ts +94 -0
  61. package/dist/definitions.js +15 -0
  62. package/dist/definitions.js.map +1 -0
  63. package/dist/dispatcher.d.ts +50 -0
  64. package/dist/dispatcher.js +8 -0
  65. package/dist/dispatcher.js.map +1 -0
  66. package/dist/durable-turn.d.ts +58 -0
  67. package/dist/durable-turn.js +9 -0
  68. package/dist/durable-turn.js.map +1 -0
  69. package/dist/hook-pipeline.d.ts +114 -0
  70. package/dist/hook-pipeline.js +13 -0
  71. package/dist/hook-pipeline.js.map +1 -0
  72. package/dist/index.d.ts +1874 -0
  73. package/dist/index.js +3117 -0
  74. package/dist/index.js.map +1 -0
  75. package/dist/loop.d.ts +78 -0
  76. package/dist/loop.js +9 -0
  77. package/dist/loop.js.map +1 -0
  78. package/dist/mcp-bridge.d.ts +48 -0
  79. package/dist/mcp-bridge.js +8 -0
  80. package/dist/mcp-bridge.js.map +1 -0
  81. package/dist/model.d.ts +154 -0
  82. package/dist/model.js +9 -0
  83. package/dist/model.js.map +1 -0
  84. package/dist/policy.d.ts +130 -0
  85. package/dist/policy.js +23 -0
  86. package/dist/policy.js.map +1 -0
  87. package/dist/protocol.d.ts +170 -0
  88. package/dist/protocol.js +15 -0
  89. package/dist/protocol.js.map +1 -0
  90. package/dist/rollout-store-persistent.d.ts +48 -0
  91. package/dist/rollout-store-persistent.js +9 -0
  92. package/dist/rollout-store-persistent.js.map +1 -0
  93. package/dist/rollout.d.ts +82 -0
  94. package/dist/rollout.js +15 -0
  95. package/dist/rollout.js.map +1 -0
  96. package/dist/sandbox.d.ts +65 -0
  97. package/dist/sandbox.js +15 -0
  98. package/dist/sandbox.js.map +1 -0
  99. package/dist/skills.d.ts +93 -0
  100. package/dist/skills.js +10 -0
  101. package/dist/skills.js.map +1 -0
  102. package/dist/subagents.d.ts +129 -0
  103. package/dist/subagents.js +21 -0
  104. package/dist/subagents.js.map +1 -0
  105. package/dist/tools.d.ts +77 -0
  106. package/dist/tools.js +9 -0
  107. package/dist/tools.js.map +1 -0
  108. package/package.json +74 -0
  109. package/src/adapters/dispatcher-sse.test.ts +218 -0
  110. package/src/adapters/dispatcher-sse.ts +222 -0
  111. package/src/adapters/index.ts +18 -0
  112. package/src/adapters/mcp-catalog.test.ts +213 -0
  113. package/src/adapters/mcp-catalog.ts +188 -0
  114. package/src/adapters/prisma-rollout.test.ts +153 -0
  115. package/src/adapters/prisma-rollout.ts +104 -0
  116. package/src/agent-runtime.test.ts +176 -0
  117. package/src/artifact-stream.test.ts +330 -0
  118. package/src/artifact-stream.ts +453 -0
  119. package/src/channel-gateway.test.ts +432 -0
  120. package/src/channel-gateway.ts +357 -0
  121. package/src/code-review.test.ts +501 -0
  122. package/src/code-review.ts +495 -0
  123. package/src/command-suggestions.test.ts +251 -0
  124. package/src/command-suggestions.ts +338 -0
  125. package/src/commands.test.ts +184 -0
  126. package/src/commands.ts +140 -0
  127. package/src/commit-message.test.ts +249 -0
  128. package/src/commit-message.ts +180 -0
  129. package/src/context-compaction.test.ts +522 -0
  130. package/src/context-compaction.ts +434 -0
  131. package/src/definitions.test.ts +78 -0
  132. package/src/definitions.ts +190 -0
  133. package/src/deployment-status.test.ts +215 -0
  134. package/src/deployment-status.ts +227 -0
  135. package/src/design-context.test.ts +195 -0
  136. package/src/design-context.ts +198 -0
  137. package/src/dispatcher.test.ts +234 -0
  138. package/src/dispatcher.ts +189 -0
  139. package/src/durable-turn.test.ts +209 -0
  140. package/src/durable-turn.ts +135 -0
  141. package/src/edit-planner.test.ts +204 -0
  142. package/src/edit-planner.ts +325 -0
  143. package/src/fuzzy-match.test.ts +311 -0
  144. package/src/fuzzy-match.ts +444 -0
  145. package/src/hook-pipeline.test.ts +279 -0
  146. package/src/hook-pipeline.ts +373 -0
  147. package/src/inbound-admission.test.ts +394 -0
  148. package/src/inbound-admission.ts +246 -0
  149. package/src/index.ts +47 -0
  150. package/src/loop.test.ts +161 -0
  151. package/src/loop.ts +211 -0
  152. package/src/mcp-bridge.test.ts +165 -0
  153. package/src/mcp-bridge.ts +76 -0
  154. package/src/memory-provider.test.ts +232 -0
  155. package/src/memory-provider.ts +257 -0
  156. package/src/model.ts +168 -0
  157. package/src/permission-ruleset.test.ts +301 -0
  158. package/src/permission-ruleset.ts +200 -0
  159. package/src/policy.ts +151 -0
  160. package/src/project-repo.test.ts +232 -0
  161. package/src/project-repo.ts +311 -0
  162. package/src/protocol.ts +159 -0
  163. package/src/rollout-store-persistent.test.ts +217 -0
  164. package/src/rollout-store-persistent.ts +166 -0
  165. package/src/rollout.ts +150 -0
  166. package/src/sandbox.ts +113 -0
  167. package/src/session-share.test.ts +360 -0
  168. package/src/session-share.ts +310 -0
  169. package/src/skill-distillation.test.ts +177 -0
  170. package/src/skill-distillation.ts +369 -0
  171. package/src/skills.test.ts +277 -0
  172. package/src/skills.ts +255 -0
  173. package/src/subagents.test.ts +290 -0
  174. package/src/subagents.ts +332 -0
  175. package/src/tools.ts +126 -0
  176. package/src/workbench.test.ts +0 -0
  177. package/src/workbench.ts +0 -0
  178. package/tsconfig.json +12 -0
  179. package/tsup.config.ts +33 -0
@@ -0,0 +1,434 @@
1
+ /**
2
+ * In-session context compaction — a faithful re-expression of a coding agent's
3
+ * mid-turn "the transcript no longer fits, shrink it and keep going" recovery
4
+ * into Sailor's grammar: TypeScript, deterministic, model-agnostic, no datastore
5
+ * and no tokenizer-vendor lock-in.
6
+ *
7
+ * ── What this is (and is NOT) ───────────────────────────────────────────────
8
+ * This module shrinks the CURRENT, in-flight transcript when a turn overflows
9
+ * the model's context window. It is NOT cross-session memory: nothing here
10
+ * persists, recalls, or indexes past sessions. The only output is a single
11
+ * summary string that the caller substitutes for the old transcript so the
12
+ * SAME turn can be retried within budget. Memory is a different subsystem;
13
+ * conflating the two is the classic mistake this header exists to prevent.
14
+ *
15
+ * ── Mental model: recursive map-reduce ──────────────────────────────────────
16
+ * A transcript is a list of whole {@link Message}s. We never split a message —
17
+ * the unit of summarization is always a complete message so a tool call and its
18
+ * output, or a reasoning block, are never torn in half.
19
+ *
20
+ * MAP {@link split} greedily packs whole messages into chunks each ≤ the
21
+ * token budget, then {@link summarizeChunks} summarizes every chunk
22
+ * (bounded concurrency) into a partial summary.
23
+ * REDUCE {@link reduce} binary-pairs the partial summaries, summarizes each
24
+ * pair, and recurses on the shorter list — at most {@link REDUCE_DEPTH}
25
+ * levels — until the combined summary fits the budget. The final text
26
+ * is hard-capped at {@link OUTPUT_TOKEN_CAP} (a model that ignores the
27
+ * length instruction is truncated, not trusted).
28
+ *
29
+ * ── The numeric contract (constants ARE the design — exported) ──────────────
30
+ * • {@link TOKEN_CORRECTION} = 1.3 — char/4 and similar cheap heuristics
31
+ * UNDERCOUNT real BPE tokens by ~15-30% on code/markup-heavy transcripts.
32
+ * Multiplying the base estimate by 1.3 buys headroom so a "fits" decision
33
+ * does not itself overflow. We round UP everywhere for the same reason:
34
+ * over-counting wastes a little budget, under-counting reintroduces the
35
+ * exact overflow we are recovering from.
36
+ * • {@link BUDGET_RATIO} = 0.6 — the summary must coexist with the system
37
+ * prompt, the model's reply, fresh tool output, and decode headroom in the
38
+ * SAME retried turn. Spending only 60% of the usable window on the recap
39
+ * leaves room for all of that.
40
+ * • {@link MIN_BUDGET} = 1000 — below ~1k tokens a "summary" degenerates into
41
+ * lossy garbage; we would rather over-summarize than emit something useless,
42
+ * so the budget is clamped up to a floor that can still hold a coherent map
43
+ * of paths/commands/decisions.
44
+ * • {@link CONCURRENCY} = 3 — chunk summaries are independent and fan out, but
45
+ * an unbounded fan-out melts rate limits mid-recovery (the worst possible
46
+ * moment). 3 is the empirical "fast but polite" point for hosted models.
47
+ * • {@link REDUCE_DEPTH} = 3 — binary reduction shrinks the summary list
48
+ * geometrically; 3 levels collapse up to 8 partials into one, which covers
49
+ * realistic overflowed transcripts while bounding worst-case model calls so
50
+ * a pathological "never shrinks" model cannot recurse forever.
51
+ * • {@link OUTPUT_TOKEN_CAP} = 2048 — the recap must itself be small enough to
52
+ * leave the bulk of the retried turn for actual work; also the hard backstop
53
+ * against a model that ignores the length instruction.
54
+ * • {@link CLIP_TOOL_CHARS} = 2000 / {@link CLIP_TEXT_CHARS} = 16000 — raw
55
+ * tool dumps (lockfiles, build logs) and giant pasted blobs are mostly noise
56
+ * for a recap; clipping them BEFORE summarization keeps the model focused on
57
+ * signal and keeps chunk sizing predictable. Text gets a larger allowance
58
+ * than tool I/O because prose carries more decision-relevant signal per char.
59
+ *
60
+ * ── Fail-closed sentinel (DOCUMENTED CHOICE) ────────────────────────────────
61
+ * If ANY chunk or reduce-group summarization rejects, we must NOT splice a
62
+ * partial or empty recap into the turn — that silently DROPS context the agent
63
+ * still needs and corrupts the run in a way that is nearly impossible to debug
64
+ * after the fact. Instead {@link compactTranscript} resolves with the sentinel
65
+ * string `"compact"`. The contract with the caller: seeing `"compact"` back
66
+ * means "compaction could not be completed safely — retry the WHOLE compaction"
67
+ * (the same token the eligibility check keys on), never "here is your summary".
68
+ * Failing loud-but-recoverable beats failing silent-and-lossy.
69
+ *
70
+ * ── Determinism & immutability ──────────────────────────────────────────────
71
+ * No Date, no Math.random, no network, no I/O. Output is a pure function of the
72
+ * transcript plus the injected `summarize` and `base` estimator. The transcript
73
+ * is never mutated; every transformation builds new arrays/strings.
74
+ */
75
+
76
+ import { z } from "zod";
77
+
78
+ /** Tokenizers undercount real BPE by ~15-30%; scale the base estimate up. */
79
+ export const TOKEN_CORRECTION = 1.3;
80
+
81
+ /** Fraction of the usable window the recap is allowed to occupy. */
82
+ export const BUDGET_RATIO = 0.6;
83
+
84
+ /** Hard floor: below this a summary degenerates into lossy noise. */
85
+ export const MIN_BUDGET = 1000;
86
+
87
+ /** Max simultaneous in-flight chunk summaries (rate-limit politeness). */
88
+ export const CONCURRENCY = 3;
89
+
90
+ /** Max binary-reduce recursion levels (bounds worst-case model calls). */
91
+ export const REDUCE_DEPTH = 3;
92
+
93
+ /** Hard cap on the final combined summary size, in corrected tokens. */
94
+ export const OUTPUT_TOKEN_CAP = 2048;
95
+
96
+ /** Per-tool-part input/output clip length before summarization. */
97
+ export const CLIP_TOOL_CHARS = 2000;
98
+
99
+ /** Per-text-part clip length before summarization. */
100
+ export const CLIP_TEXT_CHARS = 16000;
101
+
102
+ /** Appended whenever any field is clipped, so the loss is visible to the model. */
103
+ const TRUNCATION_MARKER = " …[truncated]";
104
+
105
+ /** A text fragment of a message. */
106
+ export interface TextPart {
107
+ readonly type: "text";
108
+ readonly text: string;
109
+ }
110
+
111
+ /** A tool invocation fragment: the call inputs and the tool's raw output. */
112
+ export interface ToolPart {
113
+ readonly type: "tool";
114
+ readonly name: string;
115
+ readonly input: string;
116
+ readonly output: string;
117
+ }
118
+
119
+ export type Part = TextPart | ToolPart;
120
+
121
+ /** One conversational turn fragment owned by a single role. */
122
+ export interface Message {
123
+ readonly role: "user" | "assistant" | "tool" | "system";
124
+ readonly parts: readonly Part[];
125
+ }
126
+
127
+ /** The ordered list of messages making up the current turn's context. */
128
+ export type Transcript = readonly Message[];
129
+
130
+ /**
131
+ * Caller-injected model seam. Real callers back this with their LLM; tests pass
132
+ * a deterministic fake. The module never talks to a network itself.
133
+ */
134
+ export type Summarize = (prompt: string) => Promise<string>;
135
+
136
+ /** Caller-injected cheap token heuristic; defaults to chars/4. */
137
+ export type BaseEstimator = (text: string) => number;
138
+
139
+ /** The eligibility signal this module keys on (also the failure sentinel). */
140
+ export const COMPACT_SENTINEL = "compact";
141
+
142
+ const defaultBase: BaseEstimator = (s) => Math.ceil(s.length / 4);
143
+
144
+ const textPartSchema = z.object({
145
+ type: z.literal("text"),
146
+ text: z.string(),
147
+ });
148
+
149
+ const toolPartSchema = z.object({
150
+ type: z.literal("tool"),
151
+ name: z.string(),
152
+ input: z.string(),
153
+ output: z.string(),
154
+ });
155
+
156
+ const messageSchema = z.object({
157
+ role: z.enum(["user", "assistant", "tool", "system"]),
158
+ parts: z.array(z.discriminatedUnion("type", [textPartSchema, toolPartSchema])),
159
+ });
160
+
161
+ const compactInputSchema = z.object({
162
+ transcript: z.array(messageSchema),
163
+ usableTokens: z.number().int().positive(),
164
+ });
165
+
166
+ /**
167
+ * Corrected token estimate: `ceil(base(text) * TOKEN_CORRECTION)`. The base
168
+ * estimator is injectable so callers can swap a real tokenizer in; we still
169
+ * apply the correction multiplier on top because even real tokenizers diverge
170
+ * from the model's server-side counting. Always rounds UP — see header.
171
+ */
172
+ export function estimateTokens(text: string, base: BaseEstimator = defaultBase): number {
173
+ return Math.ceil(base(text) * TOKEN_CORRECTION);
174
+ }
175
+
176
+ /**
177
+ * Budget = 60% of the usable window, floored, then clamped UP to
178
+ * {@link MIN_BUDGET} so a tiny window can never produce an incoherent recap.
179
+ */
180
+ export function computeBudget(usableTokens: number): number {
181
+ return Math.max(MIN_BUDGET, Math.floor(usableTokens * BUDGET_RATIO));
182
+ }
183
+
184
+ /** Clip a string to `limit`, appending a visible marker only if it shrank. */
185
+ function clip(value: string, limit: number): string {
186
+ if (value.length <= limit) {
187
+ return value;
188
+ }
189
+ return value.slice(0, limit) + TRUNCATION_MARKER;
190
+ }
191
+
192
+ /** Render one part to its XML-ish line(s), clipping at the part-type limit. */
193
+ function renderPart(part: Part): string {
194
+ if (part.type === "text") {
195
+ return `<text>${clip(part.text, CLIP_TEXT_CHARS)}</text>`;
196
+ }
197
+ const input = clip(part.input, CLIP_TOOL_CHARS);
198
+ const output = clip(part.output, CLIP_TOOL_CHARS);
199
+ return `<tool name="${part.name}"><input>${input}</input><output>${output}</output></tool>`;
200
+ }
201
+
202
+ /**
203
+ * Produce a stable, XML-ish `<conversation>` rendering of a single message,
204
+ * clipping oversized text/tool fields so a giant blob can neither dominate a
205
+ * chunk nor blow the model's own window during summarization. Pure: the input
206
+ * message is never mutated (new strings only).
207
+ */
208
+ export function renderClipped(msg: Message): string {
209
+ const body = msg.parts.map(renderPart).join("");
210
+ return `<conversation><message role="${msg.role}">${body}</message></conversation>`;
211
+ }
212
+
213
+ /**
214
+ * MAP step. Greedily pack WHOLE messages into chunks whose corrected estimate
215
+ * is ≤ `budget`. A message that alone exceeds `budget` becomes its own chunk
216
+ * (oversize fallback) — it is summarized later through the same clipping render
217
+ * path, so it can never wedge the packer. Pure: builds new arrays only.
218
+ */
219
+ export function split(
220
+ transcript: Transcript,
221
+ budget: number,
222
+ estimate: (text: string) => number,
223
+ ): Message[][] {
224
+ const chunks: Message[][] = [];
225
+ let current: Message[] = [];
226
+ let currentTokens = 0;
227
+
228
+ for (const msg of transcript) {
229
+ const size = estimate(renderClipped(msg));
230
+
231
+ if (size > budget) {
232
+ // Flush whatever is buffered, then this message stands alone.
233
+ if (current.length > 0) {
234
+ chunks.push(current);
235
+ current = [];
236
+ currentTokens = 0;
237
+ }
238
+ chunks.push([msg]);
239
+ continue;
240
+ }
241
+
242
+ if (current.length > 0 && currentTokens + size > budget) {
243
+ chunks.push(current);
244
+ current = [];
245
+ currentTokens = 0;
246
+ }
247
+
248
+ current = [...current, msg];
249
+ currentTokens += size;
250
+ }
251
+
252
+ if (current.length > 0) {
253
+ chunks.push(current);
254
+ }
255
+
256
+ return chunks;
257
+ }
258
+
259
+ /**
260
+ * The summarization rubric. Embedded verbatim so the prompt provably instructs
261
+ * preservation of the five things a resumed agent cannot reconstruct on its
262
+ * own: file paths, exact commands, errors, decisions made, and unresolved /
263
+ * outstanding tasks. Tests assert each rubric item is present.
264
+ */
265
+ function buildPrompt(rendered: string): string {
266
+ return [
267
+ "Summarize the following conversation transcript so the work can continue",
268
+ "without the original messages. You MUST preserve, verbatim where possible:",
269
+ " 1. Every file path that was read, written, or referenced.",
270
+ " 2. Every exact command that was run (and its key result).",
271
+ " 3. Every error or failure encountered.",
272
+ " 4. Every decision made and the reasoning behind it.",
273
+ " 5. Every unresolved / outstanding task still to be done.",
274
+ `Keep the summary under ${OUTPUT_TOKEN_CAP} tokens. Be terse; drop pleasantries.`,
275
+ "",
276
+ rendered,
277
+ ].join("\n");
278
+ }
279
+
280
+ /**
281
+ * Hand-rolled bounded-concurrency map (no p-limit dependency). Runs `worker`
282
+ * over `items` with at most {@link CONCURRENCY} in flight, preserving result
283
+ * order. Any rejection propagates (caught by the orchestrator → fail-closed).
284
+ */
285
+ async function mapLimited<T, R>(
286
+ items: readonly T[],
287
+ worker: (item: T, index: number) => Promise<R>,
288
+ ): Promise<R[]> {
289
+ const results = new Array<R>(items.length);
290
+ let cursor = 0;
291
+
292
+ async function run(): Promise<void> {
293
+ while (cursor < items.length) {
294
+ const index = cursor;
295
+ cursor += 1;
296
+ results[index] = await worker(items[index] as T, index);
297
+ }
298
+ }
299
+
300
+ const lanes = Math.min(CONCURRENCY, items.length);
301
+ await Promise.all(Array.from({ length: lanes }, () => run()));
302
+ return results;
303
+ }
304
+
305
+ /**
306
+ * MAP fan-out: render each chunk (clipping applied), then summarize all chunks
307
+ * under the concurrency limit. Order-preserving so the reduce step keeps
308
+ * chronological coherence.
309
+ */
310
+ async function summarizeChunks(
311
+ chunks: readonly Message[][],
312
+ summarize: Summarize,
313
+ ): Promise<string[]> {
314
+ return mapLimited(chunks, async (chunk) => {
315
+ const rendered = chunk.map(renderClipped).join("\n");
316
+ return summarize(buildPrompt(rendered));
317
+ });
318
+ }
319
+
320
+ /** Cap a final combined summary at {@link OUTPUT_TOKEN_CAP}, marking any cut. */
321
+ function capOutput(summary: string, estimate: (text: string) => number): string {
322
+ if (estimate(summary) <= OUTPUT_TOKEN_CAP) {
323
+ return summary;
324
+ }
325
+ // Walk down a char budget that maps to the token cap. Estimation is
326
+ // monotonic in length, so a linear shrink converges deterministically.
327
+ let cut = summary;
328
+ while (cut.length > 0 && estimate(cut + TRUNCATION_MARKER) > OUTPUT_TOKEN_CAP) {
329
+ // Shrink proportionally toward the cap, but always make progress.
330
+ const ratio = OUTPUT_TOKEN_CAP / Math.max(1, estimate(cut));
331
+ const nextLen = Math.min(cut.length - 1, Math.floor(cut.length * ratio));
332
+ cut = cut.slice(0, Math.max(0, nextLen));
333
+ }
334
+ return cut + TRUNCATION_MARKER;
335
+ }
336
+
337
+ /**
338
+ * REDUCE step. Binary-pair the partial summaries, summarize each pair, recurse
339
+ * on the (now shorter) list. Stops when the combined text fits `budget` OR
340
+ * after {@link REDUCE_DEPTH} levels (a model that never shrinks must not loop
341
+ * forever). The terminal combined summary is always {@link capOutput}-bounded.
342
+ */
343
+ async function reduce(
344
+ summaries: readonly string[],
345
+ budget: number,
346
+ estimate: (text: string) => number,
347
+ summarize: Summarize,
348
+ ): Promise<string> {
349
+ let level = summaries;
350
+ let depth = 0;
351
+
352
+ // A lone summary that already fits needs no reduction at all.
353
+ if (level.length === 1 && estimate(level[0] as string) <= budget) {
354
+ return capOutput(level[0] as string, estimate);
355
+ }
356
+
357
+ while (depth < REDUCE_DEPTH) {
358
+ const combined = level.join("\n\n");
359
+ if (level.length === 1 || estimate(combined) <= budget) {
360
+ return capOutput(level.length === 1 ? (level[0] as string) : combined, estimate);
361
+ }
362
+
363
+ const groups: string[][] = [];
364
+ for (let i = 0; i < level.length; i += 2) {
365
+ groups.push(level.slice(i, i + 2));
366
+ }
367
+
368
+ const next = await mapLimited(groups, async (group) => {
369
+ if (group.length === 1) {
370
+ return group[0] as string;
371
+ }
372
+ return summarize(buildPrompt(group.join("\n\n")));
373
+ });
374
+
375
+ level = next;
376
+ depth += 1;
377
+ }
378
+
379
+ // Depth exhausted: emit the best combination we have, hard-capped.
380
+ return capOutput(level.length === 1 ? (level[0] as string) : level.join("\n\n"), estimate);
381
+ }
382
+
383
+ /** Input to {@link compactTranscript}. `base` is optional (exactOptional). */
384
+ export interface CompactTranscriptInput {
385
+ readonly transcript: Transcript;
386
+ readonly usableTokens: number;
387
+ readonly summarize: Summarize;
388
+ readonly base?: BaseEstimator | undefined;
389
+ }
390
+
391
+ /**
392
+ * Orchestrate split → summarizeChunks → reduce and return the final recap.
393
+ *
394
+ * FAIL CLOSED: if any chunk or reduce-group summarization rejects, this resolves
395
+ * with the {@link COMPACT_SENTINEL} string `"compact"` rather than a partial or
396
+ * empty summary — never silently dropping context. The caller treats `"compact"`
397
+ * as "retry the whole compaction" (see module header).
398
+ *
399
+ * Deterministic given the injected `summarize` and `base`. The transcript is
400
+ * never mutated.
401
+ */
402
+ export async function compactTranscript(input: CompactTranscriptInput): Promise<string> {
403
+ const { transcript, usableTokens } = compactInputSchema.parse({
404
+ transcript: input.transcript,
405
+ usableTokens: input.usableTokens,
406
+ });
407
+
408
+ const base = input.base ?? defaultBase;
409
+ const estimate = (text: string): number => estimateTokens(text, base);
410
+ const budget = computeBudget(usableTokens);
411
+
412
+ try {
413
+ const chunks = split(transcript, budget, estimate);
414
+ if (chunks.length === 0) {
415
+ return "";
416
+ }
417
+
418
+ const partials = await summarizeChunks(chunks, input.summarize);
419
+ return await reduce(partials, budget, estimate, input.summarize);
420
+ } catch {
421
+ // Any summarization failure → loud-but-recoverable sentinel, never a
422
+ // partial/empty recap that would silently corrupt the retried turn.
423
+ return COMPACT_SENTINEL;
424
+ }
425
+ }
426
+
427
+ /**
428
+ * Eligibility gate. Compaction runs only when the model stopped specifically
429
+ * because it wants a compaction or because the context overflowed; every other
430
+ * stop reason (end_turn, tool_use, …) is left untouched.
431
+ */
432
+ export function shouldCompact(stop: { reason: string }): boolean {
433
+ return stop.reason === "compact" || stop.reason === "context_overflow";
434
+ }
@@ -0,0 +1,78 @@
1
+ import { describe, expect, it } from "vitest";
2
+ import {
3
+ type Definition,
4
+ DefinitionResolver,
5
+ parseFrontmatter,
6
+ substituteArguments,
7
+ } from "./definitions";
8
+
9
+ const def = (
10
+ over: Partial<Definition> & Pick<Definition, "slug" | "tenantId" | "sourceTier">,
11
+ ): Definition => ({
12
+ frontmatter: {
13
+ name: over.slug,
14
+ description: "",
15
+ allowedTools: [],
16
+ disallowedTools: [],
17
+ argNames: [],
18
+ modelInvocable: true,
19
+ userInvocable: true,
20
+ executionMode: "inline",
21
+ paths: [],
22
+ },
23
+ bodyRef: "ref",
24
+ ...over,
25
+ });
26
+
27
+ describe("parseFrontmatter", () => {
28
+ it("parses fields + kebab aliases + inverted disable-model-invocation", () => {
29
+ const { frontmatter, body } = parseFrontmatter(
30
+ `---\nname: deploy\ndescription: ship it\nallowed-tools: [Bash, Edit]\ndisable-model-invocation: true\ncontext: fork\n---\nrun the deploy`,
31
+ );
32
+ expect(frontmatter.name).toBe("deploy");
33
+ expect(frontmatter.allowedTools).toEqual(["Bash", "Edit"]);
34
+ expect(frontmatter.modelInvocable).toBe(false);
35
+ expect(frontmatter.executionMode).toBe("fork");
36
+ expect(body.trim()).toBe("run the deploy");
37
+ });
38
+ it("throws when the frontmatter block is missing", () => {
39
+ expect(() => parseFrontmatter("no frontmatter here")).toThrow(/frontmatter/);
40
+ });
41
+ });
42
+
43
+ describe("DefinitionResolver", () => {
44
+ it("higher tier overrides lower on slug collision", () => {
45
+ const r = new DefinitionResolver([
46
+ def({ slug: "x", tenantId: "t", sourceTier: "bundled" }),
47
+ def({ slug: "x", tenantId: "t", sourceTier: "plugin" }),
48
+ ]);
49
+ const out = r.resolve({ tenantId: "t" });
50
+ expect(out).toHaveLength(1);
51
+ expect(out[0]?.sourceTier).toBe("plugin");
52
+ });
53
+ it("isolates tenants — fail closed", () => {
54
+ const r = new DefinitionResolver([def({ slug: "x", tenantId: "t_a", sourceTier: "builtin" })]);
55
+ expect(r.resolve({ tenantId: "t_b" })).toHaveLength(0);
56
+ expect(() => r.resolve({ tenantId: "" })).toThrow();
57
+ });
58
+ it("dual gate: availability (plan) ∧ enabled", () => {
59
+ const r = new DefinitionResolver([
60
+ def({ slug: "pro", tenantId: "t", sourceTier: "builtin", availabilityPlans: ["pro"] }),
61
+ def({ slug: "off", tenantId: "t", sourceTier: "builtin", enabled: false }),
62
+ ]);
63
+ expect(r.resolve({ tenantId: "t" }).map((d) => d.slug)).toEqual([]);
64
+ expect(r.resolve({ tenantId: "t", plan: "pro" }).map((d) => d.slug)).toEqual(["pro"]);
65
+ });
66
+ });
67
+
68
+ describe("substituteArguments", () => {
69
+ it("substitutes named args then variables, blanks unknowns", () => {
70
+ expect(
71
+ substituteArguments(
72
+ "deploy ${env} as ${SESSION} ${missing}",
73
+ { env: "prod" },
74
+ { SESSION: "s1" },
75
+ ),
76
+ ).toBe("deploy prod as s1 ");
77
+ });
78
+ });