@nebutra/agent-runtime 0.2.1 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -676
- package/README.md +33 -10
- package/dist/adapters/index.d.ts +24 -5
- package/dist/adapters/index.js +27 -0
- package/dist/adapters/index.js.map +1 -1
- package/dist/{chunk-KCNN4QUQ.js → chunk-5UGIUYJR.js} +4 -4
- package/dist/chunk-7ELA4SWE.js +73 -0
- package/dist/chunk-7ELA4SWE.js.map +1 -0
- package/dist/chunk-Q62VKHIT.js +178 -0
- package/dist/chunk-Q62VKHIT.js.map +1 -0
- package/dist/chunk-VXILZAXK.js +541 -0
- package/dist/chunk-VXILZAXK.js.map +1 -0
- package/dist/command-exec.d.ts +45 -0
- package/dist/command-exec.js +15 -0
- package/dist/command-exec.js.map +1 -0
- package/dist/index.d.ts +3 -2
- package/dist/index.js +81 -28
- package/dist/index.js.map +1 -1
- package/dist/orchestration.d.ts +42 -1
- package/dist/orchestration.js +7 -3
- package/dist/pulsar.js +2 -2
- package/dist/sandbox-DfcRttXt.d.ts +219 -0
- package/dist/sandbox.d.ts +2 -64
- package/dist/sandbox.js +25 -3
- package/package.json +81 -30
- package/.turbo/turbo-build.log +0 -130
- package/.turbo/turbo-test.log +0 -47
- package/.turbo/turbo-typecheck.log +0 -4
- package/CHANGELOG.md +0 -250
- package/dist/chunk-MUF7ZZTO.js +0 -57
- package/dist/chunk-MUF7ZZTO.js.map +0 -1
- package/dist/chunk-MX2WL43P.js +0 -90
- package/dist/chunk-MX2WL43P.js.map +0 -1
- package/examples/pulsar-quickstart.ts +0 -35
- package/examples/resume-branch.ts +0 -45
- package/examples/subagent-fanout.ts +0 -20
- package/src/adapters/dispatcher-sse.test.ts +0 -218
- package/src/adapters/dispatcher-sse.ts +0 -222
- package/src/adapters/index.ts +0 -18
- package/src/adapters/mcp-catalog.test.ts +0 -213
- package/src/adapters/mcp-catalog.ts +0 -188
- package/src/adapters/prisma-rollout.test.ts +0 -153
- package/src/adapters/prisma-rollout.ts +0 -104
- package/src/agent-runtime.test.ts +0 -176
- package/src/artifact-stream.test.ts +0 -330
- package/src/artifact-stream.ts +0 -453
- package/src/channel-gateway.test.ts +0 -432
- package/src/channel-gateway.ts +0 -357
- package/src/cli.ts +0 -49
- package/src/code-review.test.ts +0 -501
- package/src/code-review.ts +0 -495
- package/src/command-suggestions.test.ts +0 -251
- package/src/command-suggestions.ts +0 -338
- package/src/commands.test.ts +0 -184
- package/src/commands.ts +0 -140
- package/src/commit-message.test.ts +0 -249
- package/src/commit-message.ts +0 -180
- package/src/context-compaction.test.ts +0 -522
- package/src/context-compaction.ts +0 -438
- package/src/definitions.test.ts +0 -78
- package/src/definitions.ts +0 -190
- package/src/deployment-status.test.ts +0 -215
- package/src/deployment-status.ts +0 -227
- package/src/design-context.test.ts +0 -195
- package/src/design-context.ts +0 -198
- package/src/dispatcher.test.ts +0 -234
- package/src/dispatcher.ts +0 -189
- package/src/durable-turn.test.ts +0 -209
- package/src/durable-turn.ts +0 -135
- package/src/edit-planner.test.ts +0 -204
- package/src/edit-planner.ts +0 -325
- package/src/fuzzy-match.test.ts +0 -311
- package/src/fuzzy-match.ts +0 -444
- package/src/hook-pipeline.test.ts +0 -279
- package/src/hook-pipeline.ts +0 -373
- package/src/inbound-admission.test.ts +0 -394
- package/src/inbound-admission.ts +0 -246
- package/src/index.ts +0 -49
- package/src/loop.test.ts +0 -161
- package/src/loop.ts +0 -211
- package/src/mcp-bridge.test.ts +0 -165
- package/src/mcp-bridge.ts +0 -81
- package/src/memory-provider.test.ts +0 -232
- package/src/memory-provider.ts +0 -257
- package/src/model.ts +0 -168
- package/src/orchestration.test.ts +0 -53
- package/src/orchestration.ts +0 -146
- package/src/permission-ruleset.test.ts +0 -301
- package/src/permission-ruleset.ts +0 -1
- package/src/policy.ts +0 -151
- package/src/project-repo.test.ts +0 -232
- package/src/project-repo.ts +0 -311
- package/src/protocol.ts +0 -159
- package/src/pulsar.test.ts +0 -156
- package/src/pulsar.ts +0 -322
- package/src/rollout-store-persistent.test.ts +0 -217
- package/src/rollout-store-persistent.ts +0 -166
- package/src/rollout.ts +0 -150
- package/src/sandbox.ts +0 -113
- package/src/session-share.test.ts +0 -360
- package/src/session-share.ts +0 -310
- package/src/skill-distillation.test.ts +0 -177
- package/src/skill-distillation.ts +0 -369
- package/src/skills.test.ts +0 -277
- package/src/skills.ts +0 -277
- package/src/subagents.test.ts +0 -290
- package/src/subagents.ts +0 -332
- package/src/tools.test.ts +0 -12
- package/src/tools.ts +0 -129
- package/src/workbench.test.ts +0 -0
- package/src/workbench.ts +0 -0
- package/tsconfig.json +0 -12
- package/tsup.config.ts +0 -36
- /package/dist/{chunk-KCNN4QUQ.js.map → chunk-5UGIUYJR.js.map} +0 -0
|
@@ -1,438 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* In-session context compaction — a faithful re-expression of a coding agent's
|
|
3
|
-
* mid-turn "the transcript no longer fits, shrink it and keep going" recovery
|
|
4
|
-
* into Sailor's grammar: TypeScript, deterministic, model-agnostic, no datastore
|
|
5
|
-
* and no tokenizer-vendor lock-in.
|
|
6
|
-
*
|
|
7
|
-
* ── What this is (and is NOT) ───────────────────────────────────────────────
|
|
8
|
-
* This module shrinks the CURRENT, in-flight transcript when a turn overflows
|
|
9
|
-
* the model's context window. It is NOT cross-session memory: nothing here
|
|
10
|
-
* persists, recalls, or indexes past sessions. The only output is a single
|
|
11
|
-
* summary string that the caller substitutes for the old transcript so the
|
|
12
|
-
* SAME turn can be retried within budget. Memory is a different subsystem;
|
|
13
|
-
* conflating the two is the classic mistake this header exists to prevent.
|
|
14
|
-
*
|
|
15
|
-
* ── Mental model: recursive map-reduce ──────────────────────────────────────
|
|
16
|
-
* A transcript is a list of whole {@link Message}s. We never split a message —
|
|
17
|
-
* the unit of summarization is always a complete message so a tool call and its
|
|
18
|
-
* output, or a reasoning block, are never torn in half.
|
|
19
|
-
*
|
|
20
|
-
* MAP {@link split} greedily packs whole messages into chunks each ≤ the
|
|
21
|
-
* token budget, then {@link summarizeChunks} summarizes every chunk
|
|
22
|
-
* (bounded concurrency) into a partial summary.
|
|
23
|
-
* REDUCE {@link reduce} binary-pairs the partial summaries, summarizes each
|
|
24
|
-
* pair, and recurses on the shorter list — at most {@link REDUCE_DEPTH}
|
|
25
|
-
* levels — until the combined summary fits the budget. The final text
|
|
26
|
-
* is hard-capped at {@link OUTPUT_TOKEN_CAP} (a model that ignores the
|
|
27
|
-
* length instruction is truncated, not trusted).
|
|
28
|
-
*
|
|
29
|
-
* ── The numeric contract (constants ARE the design — exported) ──────────────
|
|
30
|
-
* • {@link TOKEN_CORRECTION} = 1.3 — char/4 and similar cheap heuristics
|
|
31
|
-
* UNDERCOUNT real BPE tokens by ~15-30% on code/markup-heavy transcripts.
|
|
32
|
-
* Multiplying the base estimate by 1.3 buys headroom so a "fits" decision
|
|
33
|
-
* does not itself overflow. We round UP everywhere for the same reason:
|
|
34
|
-
* over-counting wastes a little budget, under-counting reintroduces the
|
|
35
|
-
* exact overflow we are recovering from.
|
|
36
|
-
* • {@link BUDGET_RATIO} = 0.6 — the summary must coexist with the system
|
|
37
|
-
* prompt, the model's reply, fresh tool output, and decode headroom in the
|
|
38
|
-
* SAME retried turn. Spending only 60% of the usable window on the recap
|
|
39
|
-
* leaves room for all of that.
|
|
40
|
-
* • {@link MIN_BUDGET} = 1000 — below ~1k tokens a "summary" degenerates into
|
|
41
|
-
* lossy garbage; we would rather over-summarize than emit something useless,
|
|
42
|
-
* so the budget is clamped up to a floor that can still hold a coherent map
|
|
43
|
-
* of paths/commands/decisions.
|
|
44
|
-
* • {@link CONCURRENCY} = 3 — chunk summaries are independent and fan out, but
|
|
45
|
-
* an unbounded fan-out melts rate limits mid-recovery (the worst possible
|
|
46
|
-
* moment). 3 is the empirical "fast but polite" point for hosted models.
|
|
47
|
-
* • {@link REDUCE_DEPTH} = 3 — binary reduction shrinks the summary list
|
|
48
|
-
* geometrically; 3 levels collapse up to 8 partials into one, which covers
|
|
49
|
-
* realistic overflowed transcripts while bounding worst-case model calls so
|
|
50
|
-
* a pathological "never shrinks" model cannot recurse forever.
|
|
51
|
-
* • {@link OUTPUT_TOKEN_CAP} = 2048 — the recap must itself be small enough to
|
|
52
|
-
* leave the bulk of the retried turn for actual work; also the hard backstop
|
|
53
|
-
* against a model that ignores the length instruction.
|
|
54
|
-
* • {@link CLIP_TOOL_CHARS} = 2000 / {@link CLIP_TEXT_CHARS} = 16000 — raw
|
|
55
|
-
* tool dumps (lockfiles, build logs) and giant pasted blobs are mostly noise
|
|
56
|
-
* for a recap; clipping them BEFORE summarization keeps the model focused on
|
|
57
|
-
* signal and keeps chunk sizing predictable. Text gets a larger allowance
|
|
58
|
-
* than tool I/O because prose carries more decision-relevant signal per char.
|
|
59
|
-
*
|
|
60
|
-
* ── Fail-closed sentinel (DOCUMENTED CHOICE) ────────────────────────────────
|
|
61
|
-
* If ANY chunk or reduce-group summarization rejects, we must NOT splice a
|
|
62
|
-
* partial or empty recap into the turn — that silently DROPS context the agent
|
|
63
|
-
* still needs and corrupts the run in a way that is nearly impossible to debug
|
|
64
|
-
* after the fact. Instead {@link compactTranscript} resolves with the sentinel
|
|
65
|
-
* string `"compact"`. The contract with the caller: seeing `"compact"` back
|
|
66
|
-
* means "compaction could not be completed safely — retry the WHOLE compaction"
|
|
67
|
-
* (the same token the eligibility check keys on), never "here is your summary".
|
|
68
|
-
* Failing loud-but-recoverable beats failing silent-and-lossy.
|
|
69
|
-
*
|
|
70
|
-
* ── Determinism & immutability ──────────────────────────────────────────────
|
|
71
|
-
* No Date, no Math.random, no network, no I/O. Output is a pure function of the
|
|
72
|
-
* transcript plus the injected `summarize` and `base` estimator. The transcript
|
|
73
|
-
* is never mutated; every transformation builds new arrays/strings.
|
|
74
|
-
*/
|
|
75
|
-
|
|
76
|
-
import {
|
|
77
|
-
type BaseTokenEstimator,
|
|
78
|
-
estimateTokens as estimatePrimitiveTokens,
|
|
79
|
-
} from "@nebutra/ai-primitives";
|
|
80
|
-
import { z } from "zod";
|
|
81
|
-
|
|
82
|
-
/** Tokenizers undercount real BPE by ~15-30%; scale the base estimate up. */
|
|
83
|
-
export const TOKEN_CORRECTION = 1.3;
|
|
84
|
-
|
|
85
|
-
/** Fraction of the usable window the recap is allowed to occupy. */
|
|
86
|
-
export const BUDGET_RATIO = 0.6;
|
|
87
|
-
|
|
88
|
-
/** Hard floor: below this a summary degenerates into lossy noise. */
|
|
89
|
-
export const MIN_BUDGET = 1000;
|
|
90
|
-
|
|
91
|
-
/** Max simultaneous in-flight chunk summaries (rate-limit politeness). */
|
|
92
|
-
export const CONCURRENCY = 3;
|
|
93
|
-
|
|
94
|
-
/** Max binary-reduce recursion levels (bounds worst-case model calls). */
|
|
95
|
-
export const REDUCE_DEPTH = 3;
|
|
96
|
-
|
|
97
|
-
/** Hard cap on the final combined summary size, in corrected tokens. */
|
|
98
|
-
export const OUTPUT_TOKEN_CAP = 2048;
|
|
99
|
-
|
|
100
|
-
/** Per-tool-part input/output clip length before summarization. */
|
|
101
|
-
export const CLIP_TOOL_CHARS = 2000;
|
|
102
|
-
|
|
103
|
-
/** Per-text-part clip length before summarization. */
|
|
104
|
-
export const CLIP_TEXT_CHARS = 16000;
|
|
105
|
-
|
|
106
|
-
/** Appended whenever any field is clipped, so the loss is visible to the model. */
|
|
107
|
-
const TRUNCATION_MARKER = " …[truncated]";
|
|
108
|
-
|
|
109
|
-
/** A text fragment of a message. */
|
|
110
|
-
export interface TextPart {
|
|
111
|
-
readonly type: "text";
|
|
112
|
-
readonly text: string;
|
|
113
|
-
}
|
|
114
|
-
|
|
115
|
-
/** A tool invocation fragment: the call inputs and the tool's raw output. */
|
|
116
|
-
export interface ToolPart {
|
|
117
|
-
readonly type: "tool";
|
|
118
|
-
readonly name: string;
|
|
119
|
-
readonly input: string;
|
|
120
|
-
readonly output: string;
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
export type Part = TextPart | ToolPart;
|
|
124
|
-
|
|
125
|
-
/** One conversational turn fragment owned by a single role. */
|
|
126
|
-
export interface Message {
|
|
127
|
-
readonly role: "user" | "assistant" | "tool" | "system";
|
|
128
|
-
readonly parts: readonly Part[];
|
|
129
|
-
}
|
|
130
|
-
|
|
131
|
-
/** The ordered list of messages making up the current turn's context. */
|
|
132
|
-
export type Transcript = readonly Message[];
|
|
133
|
-
|
|
134
|
-
/**
|
|
135
|
-
* Caller-injected model seam. Real callers back this with their LLM; tests pass
|
|
136
|
-
* a deterministic fake. The module never talks to a network itself.
|
|
137
|
-
*/
|
|
138
|
-
export type Summarize = (prompt: string) => Promise<string>;
|
|
139
|
-
|
|
140
|
-
/** Caller-injected cheap token heuristic; defaults to chars/4. */
|
|
141
|
-
export type BaseEstimator = BaseTokenEstimator;
|
|
142
|
-
|
|
143
|
-
/** The eligibility signal this module keys on (also the failure sentinel). */
|
|
144
|
-
export const COMPACT_SENTINEL = "compact";
|
|
145
|
-
|
|
146
|
-
const defaultBase: BaseEstimator = (s) => Math.ceil(s.length / 4);
|
|
147
|
-
|
|
148
|
-
const textPartSchema = z.object({
|
|
149
|
-
type: z.literal("text"),
|
|
150
|
-
text: z.string(),
|
|
151
|
-
});
|
|
152
|
-
|
|
153
|
-
const toolPartSchema = z.object({
|
|
154
|
-
type: z.literal("tool"),
|
|
155
|
-
name: z.string(),
|
|
156
|
-
input: z.string(),
|
|
157
|
-
output: z.string(),
|
|
158
|
-
});
|
|
159
|
-
|
|
160
|
-
const messageSchema = z.object({
|
|
161
|
-
role: z.enum(["user", "assistant", "tool", "system"]),
|
|
162
|
-
parts: z.array(z.discriminatedUnion("type", [textPartSchema, toolPartSchema])),
|
|
163
|
-
});
|
|
164
|
-
|
|
165
|
-
const compactInputSchema = z.object({
|
|
166
|
-
transcript: z.array(messageSchema),
|
|
167
|
-
usableTokens: z.number().int().positive(),
|
|
168
|
-
});
|
|
169
|
-
|
|
170
|
-
/**
|
|
171
|
-
* Corrected token estimate: `ceil(base(text) * TOKEN_CORRECTION)`. The base
|
|
172
|
-
* estimator is injectable so callers can swap a real tokenizer in; we still
|
|
173
|
-
* apply the correction multiplier on top because even real tokenizers diverge
|
|
174
|
-
* from the model's server-side counting. Always rounds UP — see header.
|
|
175
|
-
*/
|
|
176
|
-
export function estimateTokens(text: string, base: BaseEstimator = defaultBase): number {
|
|
177
|
-
return estimatePrimitiveTokens(text, { base, correction: TOKEN_CORRECTION });
|
|
178
|
-
}
|
|
179
|
-
|
|
180
|
-
/**
|
|
181
|
-
* Budget = 60% of the usable window, floored, then clamped UP to
|
|
182
|
-
* {@link MIN_BUDGET} so a tiny window can never produce an incoherent recap.
|
|
183
|
-
*/
|
|
184
|
-
export function computeBudget(usableTokens: number): number {
|
|
185
|
-
return Math.max(MIN_BUDGET, Math.floor(usableTokens * BUDGET_RATIO));
|
|
186
|
-
}
|
|
187
|
-
|
|
188
|
-
/** Clip a string to `limit`, appending a visible marker only if it shrank. */
|
|
189
|
-
function clip(value: string, limit: number): string {
|
|
190
|
-
if (value.length <= limit) {
|
|
191
|
-
return value;
|
|
192
|
-
}
|
|
193
|
-
return value.slice(0, limit) + TRUNCATION_MARKER;
|
|
194
|
-
}
|
|
195
|
-
|
|
196
|
-
/** Render one part to its XML-ish line(s), clipping at the part-type limit. */
|
|
197
|
-
function renderPart(part: Part): string {
|
|
198
|
-
if (part.type === "text") {
|
|
199
|
-
return `<text>${clip(part.text, CLIP_TEXT_CHARS)}</text>`;
|
|
200
|
-
}
|
|
201
|
-
const input = clip(part.input, CLIP_TOOL_CHARS);
|
|
202
|
-
const output = clip(part.output, CLIP_TOOL_CHARS);
|
|
203
|
-
return `<tool name="${part.name}"><input>${input}</input><output>${output}</output></tool>`;
|
|
204
|
-
}
|
|
205
|
-
|
|
206
|
-
/**
|
|
207
|
-
* Produce a stable, XML-ish `<conversation>` rendering of a single message,
|
|
208
|
-
* clipping oversized text/tool fields so a giant blob can neither dominate a
|
|
209
|
-
* chunk nor blow the model's own window during summarization. Pure: the input
|
|
210
|
-
* message is never mutated (new strings only).
|
|
211
|
-
*/
|
|
212
|
-
export function renderClipped(msg: Message): string {
|
|
213
|
-
const body = msg.parts.map(renderPart).join("");
|
|
214
|
-
return `<conversation><message role="${msg.role}">${body}</message></conversation>`;
|
|
215
|
-
}
|
|
216
|
-
|
|
217
|
-
/**
|
|
218
|
-
* MAP step. Greedily pack WHOLE messages into chunks whose corrected estimate
|
|
219
|
-
* is ≤ `budget`. A message that alone exceeds `budget` becomes its own chunk
|
|
220
|
-
* (oversize fallback) — it is summarized later through the same clipping render
|
|
221
|
-
* path, so it can never wedge the packer. Pure: builds new arrays only.
|
|
222
|
-
*/
|
|
223
|
-
export function split(
|
|
224
|
-
transcript: Transcript,
|
|
225
|
-
budget: number,
|
|
226
|
-
estimate: (text: string) => number,
|
|
227
|
-
): Message[][] {
|
|
228
|
-
const chunks: Message[][] = [];
|
|
229
|
-
let current: Message[] = [];
|
|
230
|
-
let currentTokens = 0;
|
|
231
|
-
|
|
232
|
-
for (const msg of transcript) {
|
|
233
|
-
const size = estimate(renderClipped(msg));
|
|
234
|
-
|
|
235
|
-
if (size > budget) {
|
|
236
|
-
// Flush whatever is buffered, then this message stands alone.
|
|
237
|
-
if (current.length > 0) {
|
|
238
|
-
chunks.push(current);
|
|
239
|
-
current = [];
|
|
240
|
-
currentTokens = 0;
|
|
241
|
-
}
|
|
242
|
-
chunks.push([msg]);
|
|
243
|
-
continue;
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
if (current.length > 0 && currentTokens + size > budget) {
|
|
247
|
-
chunks.push(current);
|
|
248
|
-
current = [];
|
|
249
|
-
currentTokens = 0;
|
|
250
|
-
}
|
|
251
|
-
|
|
252
|
-
current = [...current, msg];
|
|
253
|
-
currentTokens += size;
|
|
254
|
-
}
|
|
255
|
-
|
|
256
|
-
if (current.length > 0) {
|
|
257
|
-
chunks.push(current);
|
|
258
|
-
}
|
|
259
|
-
|
|
260
|
-
return chunks;
|
|
261
|
-
}
|
|
262
|
-
|
|
263
|
-
/**
|
|
264
|
-
* The summarization rubric. Embedded verbatim so the prompt provably instructs
|
|
265
|
-
* preservation of the five things a resumed agent cannot reconstruct on its
|
|
266
|
-
* own: file paths, exact commands, errors, decisions made, and unresolved /
|
|
267
|
-
* outstanding tasks. Tests assert each rubric item is present.
|
|
268
|
-
*/
|
|
269
|
-
function buildPrompt(rendered: string): string {
|
|
270
|
-
return [
|
|
271
|
-
"Summarize the following conversation transcript so the work can continue",
|
|
272
|
-
"without the original messages. You MUST preserve, verbatim where possible:",
|
|
273
|
-
" 1. Every file path that was read, written, or referenced.",
|
|
274
|
-
" 2. Every exact command that was run (and its key result).",
|
|
275
|
-
" 3. Every error or failure encountered.",
|
|
276
|
-
" 4. Every decision made and the reasoning behind it.",
|
|
277
|
-
" 5. Every unresolved / outstanding task still to be done.",
|
|
278
|
-
`Keep the summary under ${OUTPUT_TOKEN_CAP} tokens. Be terse; drop pleasantries.`,
|
|
279
|
-
"",
|
|
280
|
-
rendered,
|
|
281
|
-
].join("\n");
|
|
282
|
-
}
|
|
283
|
-
|
|
284
|
-
/**
|
|
285
|
-
* Hand-rolled bounded-concurrency map (no p-limit dependency). Runs `worker`
|
|
286
|
-
* over `items` with at most {@link CONCURRENCY} in flight, preserving result
|
|
287
|
-
* order. Any rejection propagates (caught by the orchestrator → fail-closed).
|
|
288
|
-
*/
|
|
289
|
-
async function mapLimited<T, R>(
|
|
290
|
-
items: readonly T[],
|
|
291
|
-
worker: (item: T, index: number) => Promise<R>,
|
|
292
|
-
): Promise<R[]> {
|
|
293
|
-
const results = new Array<R>(items.length);
|
|
294
|
-
let cursor = 0;
|
|
295
|
-
|
|
296
|
-
async function run(): Promise<void> {
|
|
297
|
-
while (cursor < items.length) {
|
|
298
|
-
const index = cursor;
|
|
299
|
-
cursor += 1;
|
|
300
|
-
results[index] = await worker(items[index] as T, index);
|
|
301
|
-
}
|
|
302
|
-
}
|
|
303
|
-
|
|
304
|
-
const lanes = Math.min(CONCURRENCY, items.length);
|
|
305
|
-
await Promise.all(Array.from({ length: lanes }, () => run()));
|
|
306
|
-
return results;
|
|
307
|
-
}
|
|
308
|
-
|
|
309
|
-
/**
|
|
310
|
-
* MAP fan-out: render each chunk (clipping applied), then summarize all chunks
|
|
311
|
-
* under the concurrency limit. Order-preserving so the reduce step keeps
|
|
312
|
-
* chronological coherence.
|
|
313
|
-
*/
|
|
314
|
-
async function summarizeChunks(
|
|
315
|
-
chunks: readonly Message[][],
|
|
316
|
-
summarize: Summarize,
|
|
317
|
-
): Promise<string[]> {
|
|
318
|
-
return mapLimited(chunks, async (chunk) => {
|
|
319
|
-
const rendered = chunk.map(renderClipped).join("\n");
|
|
320
|
-
return summarize(buildPrompt(rendered));
|
|
321
|
-
});
|
|
322
|
-
}
|
|
323
|
-
|
|
324
|
-
/** Cap a final combined summary at {@link OUTPUT_TOKEN_CAP}, marking any cut. */
|
|
325
|
-
function capOutput(summary: string, estimate: (text: string) => number): string {
|
|
326
|
-
if (estimate(summary) <= OUTPUT_TOKEN_CAP) {
|
|
327
|
-
return summary;
|
|
328
|
-
}
|
|
329
|
-
// Walk down a char budget that maps to the token cap. Estimation is
|
|
330
|
-
// monotonic in length, so a linear shrink converges deterministically.
|
|
331
|
-
let cut = summary;
|
|
332
|
-
while (cut.length > 0 && estimate(cut + TRUNCATION_MARKER) > OUTPUT_TOKEN_CAP) {
|
|
333
|
-
// Shrink proportionally toward the cap, but always make progress.
|
|
334
|
-
const ratio = OUTPUT_TOKEN_CAP / Math.max(1, estimate(cut));
|
|
335
|
-
const nextLen = Math.min(cut.length - 1, Math.floor(cut.length * ratio));
|
|
336
|
-
cut = cut.slice(0, Math.max(0, nextLen));
|
|
337
|
-
}
|
|
338
|
-
return cut + TRUNCATION_MARKER;
|
|
339
|
-
}
|
|
340
|
-
|
|
341
|
-
/**
|
|
342
|
-
* REDUCE step. Binary-pair the partial summaries, summarize each pair, recurse
|
|
343
|
-
* on the (now shorter) list. Stops when the combined text fits `budget` OR
|
|
344
|
-
* after {@link REDUCE_DEPTH} levels (a model that never shrinks must not loop
|
|
345
|
-
* forever). The terminal combined summary is always {@link capOutput}-bounded.
|
|
346
|
-
*/
|
|
347
|
-
async function reduce(
|
|
348
|
-
summaries: readonly string[],
|
|
349
|
-
budget: number,
|
|
350
|
-
estimate: (text: string) => number,
|
|
351
|
-
summarize: Summarize,
|
|
352
|
-
): Promise<string> {
|
|
353
|
-
let level = summaries;
|
|
354
|
-
let depth = 0;
|
|
355
|
-
|
|
356
|
-
// A lone summary that already fits needs no reduction at all.
|
|
357
|
-
if (level.length === 1 && estimate(level[0] as string) <= budget) {
|
|
358
|
-
return capOutput(level[0] as string, estimate);
|
|
359
|
-
}
|
|
360
|
-
|
|
361
|
-
while (depth < REDUCE_DEPTH) {
|
|
362
|
-
const combined = level.join("\n\n");
|
|
363
|
-
if (level.length === 1 || estimate(combined) <= budget) {
|
|
364
|
-
return capOutput(level.length === 1 ? (level[0] as string) : combined, estimate);
|
|
365
|
-
}
|
|
366
|
-
|
|
367
|
-
const groups: string[][] = [];
|
|
368
|
-
for (let i = 0; i < level.length; i += 2) {
|
|
369
|
-
groups.push(level.slice(i, i + 2));
|
|
370
|
-
}
|
|
371
|
-
|
|
372
|
-
const next = await mapLimited(groups, async (group) => {
|
|
373
|
-
if (group.length === 1) {
|
|
374
|
-
return group[0] as string;
|
|
375
|
-
}
|
|
376
|
-
return summarize(buildPrompt(group.join("\n\n")));
|
|
377
|
-
});
|
|
378
|
-
|
|
379
|
-
level = next;
|
|
380
|
-
depth += 1;
|
|
381
|
-
}
|
|
382
|
-
|
|
383
|
-
// Depth exhausted: emit the best combination we have, hard-capped.
|
|
384
|
-
return capOutput(level.length === 1 ? (level[0] as string) : level.join("\n\n"), estimate);
|
|
385
|
-
}
|
|
386
|
-
|
|
387
|
-
/** Input to {@link compactTranscript}. `base` is optional (exactOptional). */
|
|
388
|
-
export interface CompactTranscriptInput {
|
|
389
|
-
readonly transcript: Transcript;
|
|
390
|
-
readonly usableTokens: number;
|
|
391
|
-
readonly summarize: Summarize;
|
|
392
|
-
readonly base?: BaseEstimator | undefined;
|
|
393
|
-
}
|
|
394
|
-
|
|
395
|
-
/**
|
|
396
|
-
* Orchestrate split → summarizeChunks → reduce and return the final recap.
|
|
397
|
-
*
|
|
398
|
-
* FAIL CLOSED: if any chunk or reduce-group summarization rejects, this resolves
|
|
399
|
-
* with the {@link COMPACT_SENTINEL} string `"compact"` rather than a partial or
|
|
400
|
-
* empty summary — never silently dropping context. The caller treats `"compact"`
|
|
401
|
-
* as "retry the whole compaction" (see module header).
|
|
402
|
-
*
|
|
403
|
-
* Deterministic given the injected `summarize` and `base`. The transcript is
|
|
404
|
-
* never mutated.
|
|
405
|
-
*/
|
|
406
|
-
export async function compactTranscript(input: CompactTranscriptInput): Promise<string> {
|
|
407
|
-
const { transcript, usableTokens } = compactInputSchema.parse({
|
|
408
|
-
transcript: input.transcript,
|
|
409
|
-
usableTokens: input.usableTokens,
|
|
410
|
-
});
|
|
411
|
-
|
|
412
|
-
const base = input.base ?? defaultBase;
|
|
413
|
-
const estimate = (text: string): number => estimateTokens(text, base);
|
|
414
|
-
const budget = computeBudget(usableTokens);
|
|
415
|
-
|
|
416
|
-
try {
|
|
417
|
-
const chunks = split(transcript, budget, estimate);
|
|
418
|
-
if (chunks.length === 0) {
|
|
419
|
-
return "";
|
|
420
|
-
}
|
|
421
|
-
|
|
422
|
-
const partials = await summarizeChunks(chunks, input.summarize);
|
|
423
|
-
return await reduce(partials, budget, estimate, input.summarize);
|
|
424
|
-
} catch {
|
|
425
|
-
// Any summarization failure → loud-but-recoverable sentinel, never a
|
|
426
|
-
// partial/empty recap that would silently corrupt the retried turn.
|
|
427
|
-
return COMPACT_SENTINEL;
|
|
428
|
-
}
|
|
429
|
-
}
|
|
430
|
-
|
|
431
|
-
/**
|
|
432
|
-
* Eligibility gate. Compaction runs only when the model stopped specifically
|
|
433
|
-
* because it wants a compaction or because the context overflowed; every other
|
|
434
|
-
* stop reason (end_turn, tool_use, …) is left untouched.
|
|
435
|
-
*/
|
|
436
|
-
export function shouldCompact(stop: { reason: string }): boolean {
|
|
437
|
-
return stop.reason === "compact" || stop.reason === "context_overflow";
|
|
438
|
-
}
|
package/src/definitions.test.ts
DELETED
|
@@ -1,78 +0,0 @@
|
|
|
1
|
-
import { describe, expect, it } from "vitest";
|
|
2
|
-
import {
|
|
3
|
-
type Definition,
|
|
4
|
-
DefinitionResolver,
|
|
5
|
-
parseFrontmatter,
|
|
6
|
-
substituteArguments,
|
|
7
|
-
} from "./definitions";
|
|
8
|
-
|
|
9
|
-
const def = (
|
|
10
|
-
over: Partial<Definition> & Pick<Definition, "slug" | "tenantId" | "sourceTier">,
|
|
11
|
-
): Definition => ({
|
|
12
|
-
frontmatter: {
|
|
13
|
-
name: over.slug,
|
|
14
|
-
description: "",
|
|
15
|
-
allowedTools: [],
|
|
16
|
-
disallowedTools: [],
|
|
17
|
-
argNames: [],
|
|
18
|
-
modelInvocable: true,
|
|
19
|
-
userInvocable: true,
|
|
20
|
-
executionMode: "inline",
|
|
21
|
-
paths: [],
|
|
22
|
-
},
|
|
23
|
-
bodyRef: "ref",
|
|
24
|
-
...over,
|
|
25
|
-
});
|
|
26
|
-
|
|
27
|
-
describe("parseFrontmatter", () => {
|
|
28
|
-
it("parses fields + kebab aliases + inverted disable-model-invocation", () => {
|
|
29
|
-
const { frontmatter, body } = parseFrontmatter(
|
|
30
|
-
`---\nname: deploy\ndescription: ship it\nallowed-tools: [Bash, Edit]\ndisable-model-invocation: true\ncontext: fork\n---\nrun the deploy`,
|
|
31
|
-
);
|
|
32
|
-
expect(frontmatter.name).toBe("deploy");
|
|
33
|
-
expect(frontmatter.allowedTools).toEqual(["Bash", "Edit"]);
|
|
34
|
-
expect(frontmatter.modelInvocable).toBe(false);
|
|
35
|
-
expect(frontmatter.executionMode).toBe("fork");
|
|
36
|
-
expect(body.trim()).toBe("run the deploy");
|
|
37
|
-
});
|
|
38
|
-
it("throws when the frontmatter block is missing", () => {
|
|
39
|
-
expect(() => parseFrontmatter("no frontmatter here")).toThrow(/frontmatter/);
|
|
40
|
-
});
|
|
41
|
-
});
|
|
42
|
-
|
|
43
|
-
describe("DefinitionResolver", () => {
|
|
44
|
-
it("higher tier overrides lower on slug collision", () => {
|
|
45
|
-
const r = new DefinitionResolver([
|
|
46
|
-
def({ slug: "x", tenantId: "t", sourceTier: "bundled" }),
|
|
47
|
-
def({ slug: "x", tenantId: "t", sourceTier: "plugin" }),
|
|
48
|
-
]);
|
|
49
|
-
const out = r.resolve({ tenantId: "t" });
|
|
50
|
-
expect(out).toHaveLength(1);
|
|
51
|
-
expect(out[0]?.sourceTier).toBe("plugin");
|
|
52
|
-
});
|
|
53
|
-
it("isolates tenants — fail closed", () => {
|
|
54
|
-
const r = new DefinitionResolver([def({ slug: "x", tenantId: "t_a", sourceTier: "builtin" })]);
|
|
55
|
-
expect(r.resolve({ tenantId: "t_b" })).toHaveLength(0);
|
|
56
|
-
expect(() => r.resolve({ tenantId: "" })).toThrow();
|
|
57
|
-
});
|
|
58
|
-
it("dual gate: availability (plan) ∧ enabled", () => {
|
|
59
|
-
const r = new DefinitionResolver([
|
|
60
|
-
def({ slug: "pro", tenantId: "t", sourceTier: "builtin", availabilityPlans: ["pro"] }),
|
|
61
|
-
def({ slug: "off", tenantId: "t", sourceTier: "builtin", enabled: false }),
|
|
62
|
-
]);
|
|
63
|
-
expect(r.resolve({ tenantId: "t" }).map((d) => d.slug)).toEqual([]);
|
|
64
|
-
expect(r.resolve({ tenantId: "t", plan: "pro" }).map((d) => d.slug)).toEqual(["pro"]);
|
|
65
|
-
});
|
|
66
|
-
});
|
|
67
|
-
|
|
68
|
-
describe("substituteArguments", () => {
|
|
69
|
-
it("substitutes named args then variables, blanks unknowns", () => {
|
|
70
|
-
expect(
|
|
71
|
-
substituteArguments(
|
|
72
|
-
"deploy ${env} as ${SESSION} ${missing}",
|
|
73
|
-
{ env: "prod" },
|
|
74
|
-
{ SESSION: "s1" },
|
|
75
|
-
),
|
|
76
|
-
).toBe("deploy prod as s1 ");
|
|
77
|
-
});
|
|
78
|
-
});
|