@nebutra/agent-runtime 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/LICENSE +21 -676
  2. package/README.md +2 -0
  3. package/dist/adapters/dispatcher-sse.js +1 -0
  4. package/dist/adapters/index.d.ts +24 -5
  5. package/dist/adapters/index.js +28 -0
  6. package/dist/adapters/index.js.map +1 -1
  7. package/dist/adapters/mcp-catalog.js +1 -0
  8. package/dist/adapters/prisma-rollout.js +1 -0
  9. package/dist/chunk-424PT5DM.js +23 -0
  10. package/dist/chunk-424PT5DM.js.map +1 -0
  11. package/dist/{chunk-NN7DATXA.js → chunk-4Y25ZTKI.js} +3 -3
  12. package/dist/chunk-4Y25ZTKI.js.map +1 -0
  13. package/dist/{chunk-BJBBR3QA.js → chunk-D4YAPLOW.js} +4 -4
  14. package/dist/chunk-D4YAPLOW.js.map +1 -0
  15. package/dist/{chunk-ZMYX5VBU.js → chunk-GQZKYWFT.js} +23 -7
  16. package/dist/chunk-GQZKYWFT.js.map +1 -0
  17. package/dist/chunk-KCNN4QUQ.js +255 -0
  18. package/dist/chunk-KCNN4QUQ.js.map +1 -0
  19. package/dist/{chunk-PGGWSUTM.js → chunk-NI4EDT4T.js} +2 -2
  20. package/dist/chunk-NI4EDT4T.js.map +1 -0
  21. package/dist/chunk-Q62VKHIT.js +178 -0
  22. package/dist/chunk-Q62VKHIT.js.map +1 -0
  23. package/dist/{chunk-MUF7ZZTO.js → chunk-R5HOSQUW.js} +2 -1
  24. package/dist/{chunk-MUF7ZZTO.js.map → chunk-R5HOSQUW.js.map} +1 -1
  25. package/dist/{chunk-5N4644PB.js → chunk-SD2ZJ7XG.js} +4 -4
  26. package/dist/cli.d.ts +2 -0
  27. package/dist/cli.js +52 -0
  28. package/dist/cli.js.map +1 -0
  29. package/dist/commands.js +1 -0
  30. package/dist/definitions.js +1 -0
  31. package/dist/dispatcher.js +1 -0
  32. package/dist/durable-turn.js +3 -2
  33. package/dist/hook-pipeline.js +1 -0
  34. package/dist/index.d.ts +8 -85
  35. package/dist/index.js +249 -134
  36. package/dist/index.js.map +1 -1
  37. package/dist/loop.d.ts +2 -2
  38. package/dist/loop.js +3 -2
  39. package/dist/mcp-bridge.d.ts +3 -3
  40. package/dist/mcp-bridge.js +3 -2
  41. package/dist/model.js +1 -0
  42. package/dist/orchestration.d.ts +84 -0
  43. package/dist/orchestration.js +16 -0
  44. package/dist/orchestration.js.map +1 -0
  45. package/dist/policy.js +1 -0
  46. package/dist/protocol.js +1 -0
  47. package/dist/pulsar.d.ts +78 -0
  48. package/dist/pulsar.js +17 -0
  49. package/dist/pulsar.js.map +1 -0
  50. package/dist/rollout-store-persistent.js +1 -0
  51. package/dist/rollout.js +1 -0
  52. package/dist/sandbox.js +2 -1
  53. package/dist/skills.js +2 -1
  54. package/dist/subagents.js +1 -0
  55. package/dist/tools.d.ts +4 -3
  56. package/dist/tools.js +5 -3
  57. package/package.json +84 -27
  58. package/.turbo/turbo-build.log +0 -115
  59. package/.turbo/turbo-test.log +0 -44
  60. package/.turbo/turbo-typecheck.log +0 -4
  61. package/CHANGELOG.md +0 -253
  62. package/dist/chunk-BJBBR3QA.js.map +0 -1
  63. package/dist/chunk-NN7DATXA.js.map +0 -1
  64. package/dist/chunk-PGGWSUTM.js.map +0 -1
  65. package/dist/chunk-ZMYX5VBU.js.map +0 -1
  66. package/src/adapters/dispatcher-sse.test.ts +0 -218
  67. package/src/adapters/dispatcher-sse.ts +0 -222
  68. package/src/adapters/index.ts +0 -18
  69. package/src/adapters/mcp-catalog.test.ts +0 -213
  70. package/src/adapters/mcp-catalog.ts +0 -188
  71. package/src/adapters/prisma-rollout.test.ts +0 -153
  72. package/src/adapters/prisma-rollout.ts +0 -104
  73. package/src/agent-runtime.test.ts +0 -176
  74. package/src/artifact-stream.test.ts +0 -330
  75. package/src/artifact-stream.ts +0 -453
  76. package/src/channel-gateway.test.ts +0 -432
  77. package/src/channel-gateway.ts +0 -357
  78. package/src/code-review.test.ts +0 -501
  79. package/src/code-review.ts +0 -495
  80. package/src/command-suggestions.test.ts +0 -251
  81. package/src/command-suggestions.ts +0 -338
  82. package/src/commands.test.ts +0 -184
  83. package/src/commands.ts +0 -140
  84. package/src/commit-message.test.ts +0 -249
  85. package/src/commit-message.ts +0 -180
  86. package/src/context-compaction.test.ts +0 -522
  87. package/src/context-compaction.ts +0 -434
  88. package/src/definitions.test.ts +0 -78
  89. package/src/definitions.ts +0 -190
  90. package/src/deployment-status.test.ts +0 -215
  91. package/src/deployment-status.ts +0 -227
  92. package/src/design-context.test.ts +0 -195
  93. package/src/design-context.ts +0 -198
  94. package/src/dispatcher.test.ts +0 -234
  95. package/src/dispatcher.ts +0 -189
  96. package/src/durable-turn.test.ts +0 -209
  97. package/src/durable-turn.ts +0 -135
  98. package/src/edit-planner.test.ts +0 -204
  99. package/src/edit-planner.ts +0 -325
  100. package/src/fuzzy-match.test.ts +0 -311
  101. package/src/fuzzy-match.ts +0 -444
  102. package/src/hook-pipeline.test.ts +0 -279
  103. package/src/hook-pipeline.ts +0 -373
  104. package/src/inbound-admission.test.ts +0 -394
  105. package/src/inbound-admission.ts +0 -246
  106. package/src/index.ts +0 -47
  107. package/src/loop.test.ts +0 -161
  108. package/src/loop.ts +0 -211
  109. package/src/mcp-bridge.test.ts +0 -165
  110. package/src/mcp-bridge.ts +0 -76
  111. package/src/memory-provider.test.ts +0 -232
  112. package/src/memory-provider.ts +0 -257
  113. package/src/model.ts +0 -168
  114. package/src/permission-ruleset.test.ts +0 -301
  115. package/src/permission-ruleset.ts +0 -200
  116. package/src/policy.ts +0 -151
  117. package/src/project-repo.test.ts +0 -232
  118. package/src/project-repo.ts +0 -311
  119. package/src/protocol.ts +0 -159
  120. package/src/rollout-store-persistent.test.ts +0 -217
  121. package/src/rollout-store-persistent.ts +0 -166
  122. package/src/rollout.ts +0 -150
  123. package/src/sandbox.ts +0 -113
  124. package/src/session-share.test.ts +0 -360
  125. package/src/session-share.ts +0 -310
  126. package/src/skill-distillation.test.ts +0 -177
  127. package/src/skill-distillation.ts +0 -369
  128. package/src/skills.test.ts +0 -277
  129. package/src/skills.ts +0 -255
  130. package/src/subagents.test.ts +0 -290
  131. package/src/subagents.ts +0 -332
  132. package/src/tools.ts +0 -126
  133. package/src/workbench.test.ts +0 -0
  134. package/src/workbench.ts +0 -0
  135. package/tsconfig.json +0 -12
  136. package/tsup.config.ts +0 -33
  137. /package/dist/{chunk-5N4644PB.js.map → chunk-SD2ZJ7XG.js.map} +0 -0
@@ -1,180 +0,0 @@
1
- /**
2
- * Conventional-Commits message generation — a faithful re-expression of a
3
- * coding agent's "write my commit message from staged changes" capability into
4
- * Sailor's grammar: TypeScript, multi-tenant, no model-vendor lock-in.
5
- *
6
- * ── This is a WRAP, not an agent ────────────────────────────────────────────
7
- * The small-model invocation is INJECTED via the {@link CompletionModel} port.
8
- * This module does NOT implement an agent loop, a transport, a token budget, or
9
- * tenancy — those travel inside the injected port (the real impl is the runtime
10
- * model seam; tests pass a deterministic fake). The delta this module owns is
11
- * purely DOMAIN SHAPING:
12
- * • the Conventional Commits system prompt,
13
- * • the git-context → user-prompt rendering,
14
- * • the "regenerate, but DIFFERENT" negative-constraint contract,
15
- * • output de-formatting + bounded retry with a FAIL-CLOSED terminal error.
16
- *
17
- * Everything here is pure/deterministic given the injected model. Inputs are
18
- * Zod-validated at the boundary; nothing is mutated in place.
19
- */
20
-
21
- import { z } from "zod";
22
-
23
- /** Sampling temperature handed to the injected model (slightly creative, stable). */
24
- export const DEFAULT_TEMPERATURE = 0.3;
25
- /** Per-attempt abort budget (ms) handed to the injected model. */
26
- export const DEFAULT_ABORT_MS = 30_000;
27
- /** Total attempts before failing closed (initial try + retries inclusive). */
28
- export const MAX_RETRIES = 3;
29
-
30
- const fileSchema = z.object({
31
- status: z.string(),
32
- path: z.string(),
33
- diff: z.string(),
34
- });
35
-
36
- const gitContextSchema = z.object({
37
- branch: z.string(),
38
- recentCommits: z.array(z.string()),
39
- files: z.array(fileSchema),
40
- });
41
-
42
- /** Staged git context the message is derived from. Zod-validated public input. */
43
- export type GitContext = z.infer<typeof gitContextSchema>;
44
-
45
- /**
46
- * When `previous` is set this is a "regenerate, but materially DIFFERENT from
47
- * the previous message" request — see {@link buildCommitPrompt}. Built
48
- * conditionally to honor `exactOptionalPropertyTypes`.
49
- */
50
- export interface GenerateOptions {
51
- readonly previous?: string | undefined;
52
- }
53
-
54
- /**
55
- * Injected small-model seam. The real implementation is the runtime's model
56
- * invocation (carrying tenancy/auth); tests inject a deterministic fake. This
57
- * module consumes the port and never implements model plumbing itself.
58
- */
59
- export interface CompletionModel {
60
- complete(p: {
61
- system: string;
62
- user: string;
63
- temperature: number;
64
- abortMs: number;
65
- }): Promise<string>;
66
- }
67
-
68
- /** Raised when every attempt is exhausted. Fail closed — never a fake message. */
69
- export class CommitMessageError extends Error {
70
- constructor(message = "failed to generate a commit message") {
71
- super(message);
72
- this.name = "CommitMessageError";
73
- }
74
- }
75
-
76
- const SYSTEM_PROMPT = [
77
- "You write a single git commit message that follows the Conventional Commits specification.",
78
- "",
79
- "Format: type(scope): subject",
80
- " - scope is optional; omit the parentheses entirely when there is no scope.",
81
- " - allowed types: feat, fix, docs, style, refactor, perf, test, build, ci, chore.",
82
- " - subject: imperative mood, lower-case, concise, no trailing period.",
83
- " - an optional body MAY follow after one blank line to explain the why.",
84
- "",
85
- "Return ONLY the commit message. No code fences, no quotes, no commentary, no preamble.",
86
- ].join("\n");
87
-
88
- function renderFiles(files: GitContext["files"]): string {
89
- return files.map((f) => `[${f.status}] ${f.path}\n${f.diff}`).join("\n\n");
90
- }
91
-
92
- /**
93
- * Builds the `{ system, user }` pair. The system prompt encodes the
94
- * Conventional Commits spec and the "only the message" instruction. The user
95
- * prompt renders branch + recent commits + per-file status/path/diff. When
96
- * `opts.previous` is set, a NEGATIVE CONSTRAINT block is appended so a
97
- * regenerate request yields a materially different message. Pure.
98
- */
99
- export function buildCommitPrompt(
100
- ctx: GitContext,
101
- opts?: GenerateOptions,
102
- ): { system: string; user: string } {
103
- const sections = [
104
- `Branch:\n${ctx.branch}`,
105
- `Recent commits:\n${ctx.recentCommits.join("\n")}`,
106
- `Staged changes:\n${renderFiles(ctx.files)}`,
107
- ];
108
-
109
- if (opts?.previous !== undefined) {
110
- sections.push(
111
- [
112
- "NEGATIVE CONSTRAINT:",
113
- "the following message was already generated and rejected;",
114
- "produce a materially different message:",
115
- opts.previous,
116
- ].join("\n"),
117
- );
118
- }
119
-
120
- return { system: SYSTEM_PROMPT, user: sections.join("\n\n") };
121
- }
122
-
123
- const FENCE = /^```[^\n]*\n([\s\S]*?)\n?```$/;
124
-
125
- /**
126
- * Removes surrounding triple-backtick fences (with optional language tag) and
127
- * surrounding single/double quotes, then trims. Pure — defensive against models
128
- * that wrap output despite the system instruction.
129
- */
130
- export function stripFormatting(raw: string): string {
131
- let s = raw.trim();
132
- const fenced = FENCE.exec(s);
133
- if (fenced) {
134
- s = (fenced[1] ?? "").trim();
135
- }
136
- if (
137
- s.length >= 2 &&
138
- ((s.startsWith('"') && s.endsWith('"')) || (s.startsWith("'") && s.endsWith("'")))
139
- ) {
140
- s = s.slice(1, -1).trim();
141
- }
142
- return s;
143
- }
144
-
145
- /**
146
- * Generates a Conventional-Commits message from staged git context by wrapping
147
- * the injected {@link CompletionModel}. Zod-validates `ctx`; calls the model at
148
- * {@link DEFAULT_TEMPERATURE}/{@link DEFAULT_ABORT_MS}; retries up to
149
- * {@link MAX_RETRIES} total attempts on rejection OR empty/whitespace output;
150
- * returns the de-formatted message. FAIL CLOSED: if every attempt is exhausted
151
- * it throws {@link CommitMessageError} — never an empty or fabricated message.
152
- */
153
- export async function generateCommitMessage(args: {
154
- ctx: GitContext;
155
- model: CompletionModel;
156
- opts?: GenerateOptions;
157
- }): Promise<string> {
158
- const ctx = gitContextSchema.parse(args.ctx);
159
- const prompt = buildCommitPrompt(ctx, args.opts !== undefined ? args.opts : undefined);
160
-
161
- for (let attempt = 0; attempt < MAX_RETRIES; attempt += 1) {
162
- try {
163
- const raw = await args.model.complete({
164
- system: prompt.system,
165
- user: prompt.user,
166
- temperature: DEFAULT_TEMPERATURE,
167
- abortMs: DEFAULT_ABORT_MS,
168
- });
169
- const message = stripFormatting(raw);
170
- if (message.length > 0) {
171
- return message;
172
- }
173
- } catch {
174
- // Swallow per-attempt failure; the bounded loop decides the terminal
175
- // outcome and fails closed below.
176
- }
177
- }
178
-
179
- throw new CommitMessageError();
180
- }
@@ -1,522 +0,0 @@
1
- import { describe, expect, it } from "vitest";
2
- import {
3
- BUDGET_RATIO,
4
- CLIP_TEXT_CHARS,
5
- CLIP_TOOL_CHARS,
6
- CONCURRENCY,
7
- compactTranscript,
8
- computeBudget,
9
- estimateTokens,
10
- type Message,
11
- MIN_BUDGET,
12
- OUTPUT_TOKEN_CAP,
13
- REDUCE_DEPTH,
14
- renderClipped,
15
- shouldCompact,
16
- split,
17
- TOKEN_CORRECTION,
18
- type Transcript,
19
- } from "./context-compaction.js";
20
-
21
- /**
22
- * House-rule fidelity: NO mocking libraries. Every "model" is a hand-written
23
- * deterministic fake whose behaviour is fully observable from the test, so the
24
- * suite stays reproducible without Date/random/network.
25
- */
26
-
27
- /** A deterministic base estimator: 1 token per 4 chars (the documented default). */
28
- const charBase = (s: string): number => Math.ceil(s.length / 4);
29
-
30
- /** Echoing summarizer — returns a short, deterministic, prompt-derived string. */
31
- const echoSummarize =
32
- (label = "S") =>
33
- async (prompt: string): Promise<string> =>
34
- `${label}[len=${prompt.length}]`;
35
-
36
- /**
37
- * A summarizer that records every prompt it ever saw AND tracks the maximum
38
- * number of calls that were ever in flight simultaneously, so concurrency
39
- * limiting can be asserted without timers or mocks. Each call yields control
40
- * to the microtask queue a fixed number of times before resolving, which lets
41
- * the scheduler interleave the limiter's batch.
42
- */
43
- class TrackingSummarizer {
44
- readonly prompts: string[] = [];
45
- inFlight = 0;
46
- maxInFlight = 0;
47
- #counter = 0;
48
-
49
- constructor(private readonly settle = 4) {}
50
-
51
- summarize = async (prompt: string): Promise<string> => {
52
- this.prompts.push(prompt);
53
- this.inFlight += 1;
54
- this.maxInFlight = Math.max(this.maxInFlight, this.inFlight);
55
- for (let i = 0; i < this.settle; i += 1) {
56
- await Promise.resolve();
57
- }
58
- this.inFlight -= 1;
59
- this.#counter += 1;
60
- return `sum#${this.#counter}`;
61
- };
62
- }
63
-
64
- /** A summarizer that rejects on the Nth (1-based) call, otherwise echoes. */
65
- const failOnCall = (n: number) => {
66
- let calls = 0;
67
- return async (_prompt: string): Promise<string> => {
68
- calls += 1;
69
- if (calls === n) {
70
- throw new Error("model overloaded");
71
- }
72
- return `ok#${calls}`;
73
- };
74
- };
75
-
76
- const text = (role: Message["role"], body: string): Message => ({
77
- role,
78
- parts: [{ type: "text", text: body }],
79
- });
80
-
81
- describe("exported design constants", () => {
82
- it("freezes the documented numeric contract", () => {
83
- expect(TOKEN_CORRECTION).toBe(1.3);
84
- expect(BUDGET_RATIO).toBe(0.6);
85
- expect(MIN_BUDGET).toBe(1000);
86
- expect(CONCURRENCY).toBe(3);
87
- expect(REDUCE_DEPTH).toBe(3);
88
- expect(OUTPUT_TOKEN_CAP).toBe(2048);
89
- expect(CLIP_TOOL_CHARS).toBe(2000);
90
- expect(CLIP_TEXT_CHARS).toBe(16000);
91
- });
92
- });
93
-
94
- describe("estimateTokens", () => {
95
- it("applies the 1.3x correction over the default char estimator", () => {
96
- // 8 chars → base ceil(8/4)=2 → 2 * 1.3 = 2.6 → ceil → 3
97
- expect(estimateTokens("abcdefgh")).toBe(3);
98
- });
99
-
100
- it("uses an injected base estimator when provided", () => {
101
- // base returns 10 → 10 * 1.3 = 13
102
- expect(estimateTokens("ignored", () => 10)).toBe(13);
103
- });
104
-
105
- it("always rounds up so it never undercounts", () => {
106
- // base 1 → 1.3 → ceil → 2
107
- expect(estimateTokens("abcd", () => 1)).toBe(2);
108
- });
109
-
110
- it("returns 0 for empty text under the default estimator", () => {
111
- expect(estimateTokens("")).toBe(0);
112
- });
113
- });
114
-
115
- describe("computeBudget", () => {
116
- it("takes 60% of the usable token window", () => {
117
- expect(computeBudget(10_000)).toBe(6000);
118
- });
119
-
120
- it("floors fractional budgets", () => {
121
- // 3333 * 0.6 = 1999.8 → floor → 1999 (> MIN_BUDGET)
122
- expect(computeBudget(3333)).toBe(1999);
123
- });
124
-
125
- it("never returns below MIN_BUDGET", () => {
126
- expect(computeBudget(0)).toBe(MIN_BUDGET);
127
- expect(computeBudget(100)).toBe(MIN_BUDGET);
128
- });
129
-
130
- it("uses MIN_BUDGET exactly at the crossover", () => {
131
- // 1666 * 0.6 = 999.6 → floor 999 → clamped up to 1000
132
- expect(computeBudget(1666)).toBe(MIN_BUDGET);
133
- });
134
- });
135
-
136
- describe("shouldCompact", () => {
137
- it("is true for the explicit compact stop reason", () => {
138
- expect(shouldCompact({ reason: "compact" })).toBe(true);
139
- });
140
-
141
- it("is true for context_overflow", () => {
142
- expect(shouldCompact({ reason: "context_overflow" })).toBe(true);
143
- });
144
-
145
- it("is false for unrelated stop reasons", () => {
146
- expect(shouldCompact({ reason: "end_turn" })).toBe(false);
147
- expect(shouldCompact({ reason: "tool_use" })).toBe(false);
148
- expect(shouldCompact({ reason: "" })).toBe(false);
149
- });
150
- });
151
-
152
- describe("renderClipped", () => {
153
- it("emits an XML-ish conversation envelope tagged by role", () => {
154
- const out = renderClipped(text("user", "hello world"));
155
- expect(out).toContain("<conversation>");
156
- expect(out).toContain("</conversation>");
157
- expect(out).toContain('role="user"');
158
- expect(out).toContain("hello world");
159
- });
160
-
161
- it("renders tool parts with name, input and output", () => {
162
- const msg: Message = {
163
- role: "tool",
164
- parts: [{ type: "tool", name: "grep", input: "foo", output: "bar" }],
165
- };
166
- const out = renderClipped(msg);
167
- expect(out).toContain("grep");
168
- expect(out).toContain("foo");
169
- expect(out).toContain("bar");
170
- });
171
-
172
- it("clips overlong text to CLIP_TEXT_CHARS and appends a truncation marker", () => {
173
- const big = "x".repeat(CLIP_TEXT_CHARS + 500);
174
- const out = renderClipped(text("assistant", big));
175
- expect(out).toContain("x".repeat(CLIP_TEXT_CHARS));
176
- expect(out).not.toContain("x".repeat(CLIP_TEXT_CHARS + 1));
177
- expect(out.toLowerCase()).toContain("truncat");
178
- });
179
-
180
- it("clips tool input and output to CLIP_TOOL_CHARS independently", () => {
181
- const msg: Message = {
182
- role: "tool",
183
- parts: [
184
- {
185
- type: "tool",
186
- name: "shell",
187
- input: "i".repeat(CLIP_TOOL_CHARS + 100),
188
- output: "o".repeat(CLIP_TOOL_CHARS + 100),
189
- },
190
- ],
191
- };
192
- const out = renderClipped(msg);
193
- expect(out).toContain("i".repeat(CLIP_TOOL_CHARS));
194
- expect(out).not.toContain("i".repeat(CLIP_TOOL_CHARS + 1));
195
- expect(out).toContain("o".repeat(CLIP_TOOL_CHARS));
196
- expect(out).not.toContain("o".repeat(CLIP_TOOL_CHARS + 1));
197
- });
198
-
199
- it("does not append a marker when nothing was clipped", () => {
200
- const out = renderClipped(text("user", "short"));
201
- expect(out.toLowerCase()).not.toContain("truncat");
202
- });
203
-
204
- it("does not mutate the input message", () => {
205
- const msg = text("user", "y".repeat(CLIP_TEXT_CHARS + 10));
206
- const snapshot = JSON.stringify(msg);
207
- renderClipped(msg);
208
- expect(JSON.stringify(msg)).toBe(snapshot);
209
- });
210
- });
211
-
212
- describe("split", () => {
213
- it("greedily packs whole messages into budget-sized chunks", () => {
214
- // split() estimates the RENDERED (clipped, XML-enveloped) message, not the
215
- // raw text. Derive the budget from the real rendered size so the test
216
- // pins packing behaviour, not a guessed envelope length. Uniform "user"
217
- // role keeps every message the same size.
218
- const t: Transcript = [
219
- text("user", "aaaa"),
220
- text("user", "bbbb"),
221
- text("user", "cccc"),
222
- text("user", "dddd"),
223
- text("user", "eeee"),
224
- ];
225
- const per = estimateTokens(renderClipped(text("user", "aaaa")));
226
- // Budget holds exactly 2 messages (2*per) but not a 3rd.
227
- const chunks = split(t, per * 2, estimateTokens);
228
- // 2 + 2 + 1 messages → 3 chunks
229
- expect(chunks).toHaveLength(3);
230
- expect(chunks[0]).toHaveLength(2);
231
- expect(chunks[1]).toHaveLength(2);
232
- expect(chunks[2]).toHaveLength(1);
233
- });
234
-
235
- it("never splits a single message across chunks", () => {
236
- const t: Transcript = [text("user", "aaaa"), text("user", "bbbb")];
237
- const per = estimateTokens(renderClipped(text("user", "aaaa")));
238
- // Budget holds exactly one message → one message per chunk.
239
- const chunks = split(t, per, estimateTokens);
240
- expect(chunks).toHaveLength(2);
241
- expect(chunks.every((c) => c.length === 1)).toBe(true);
242
- });
243
-
244
- it("makes an oversize single message its own chunk", () => {
245
- const small = text("user", "aaaa");
246
- const huge = text("assistant", "z".repeat(4000));
247
- const t: Transcript = [small, huge, text("user", "bbbb")];
248
- const per = estimateTokens(renderClipped(small));
249
- // Budget holds the small messages but the huge one alone exceeds it.
250
- const chunks = split(t, per, estimateTokens);
251
- // small alone, huge alone (oversize fallback), bbbb alone
252
- expect(chunks).toHaveLength(3);
253
- const oversize = chunks[1] ?? [];
254
- expect(oversize).toHaveLength(1);
255
- expect(oversize[0]).toBe(huge);
256
- });
257
-
258
- it("returns no chunks for an empty transcript", () => {
259
- expect(split([], 100, estimateTokens)).toEqual([]);
260
- });
261
-
262
- it("does not mutate the source transcript", () => {
263
- const t: Transcript = [text("user", "aaaa"), text("user", "bbbb")];
264
- const snapshot = JSON.stringify(t);
265
- split(t, 1, estimateTokens);
266
- expect(JSON.stringify(t)).toBe(snapshot);
267
- });
268
- });
269
-
270
- describe("compactTranscript — input validation", () => {
271
- it("rejects a non-array transcript", async () => {
272
- await expect(
273
- // @ts-expect-error — exercising the zod boundary
274
- compactTranscript({ transcript: "nope", usableTokens: 4000, summarize: echoSummarize() }),
275
- ).rejects.toThrow();
276
- });
277
-
278
- it("rejects a non-positive usable token budget", async () => {
279
- await expect(
280
- compactTranscript({
281
- transcript: [text("user", "hi")],
282
- usableTokens: 0,
283
- summarize: echoSummarize(),
284
- }),
285
- ).rejects.toThrow();
286
- });
287
-
288
- it("rejects a malformed message part", async () => {
289
- await expect(
290
- compactTranscript({
291
- // @ts-expect-error — exercising the zod boundary
292
- transcript: [{ role: "user", parts: [{ type: "text" }] }],
293
- usableTokens: 4000,
294
- summarize: echoSummarize(),
295
- }),
296
- ).rejects.toThrow();
297
- });
298
-
299
- it("accepts a well-formed tool part", async () => {
300
- const out = await compactTranscript({
301
- transcript: [
302
- { role: "tool", parts: [{ type: "tool", name: "ls", input: "-la", output: "files" }] },
303
- ],
304
- usableTokens: 4000,
305
- summarize: echoSummarize(),
306
- });
307
- expect(typeof out).toBe("string");
308
- });
309
- });
310
-
311
- describe("summarization prompt rubric", () => {
312
- it("instructs preservation of paths, commands, errors, decisions and open tasks", async () => {
313
- const tracker = new TrackingSummarizer();
314
- await compactTranscript({
315
- transcript: [text("user", "do the thing"), text("assistant", "done")],
316
- usableTokens: 4000,
317
- summarize: tracker.summarize,
318
- });
319
- const joined = tracker.prompts.join("\n").toLowerCase();
320
- expect(joined).toContain("file path");
321
- expect(joined).toContain("command");
322
- expect(joined).toContain("error");
323
- expect(joined).toContain("decision");
324
- // "outstanding" / "unresolved" tasks
325
- expect(joined).toMatch(/unresolved|outstanding/);
326
- });
327
- });
328
-
329
- describe("concurrency limiter", () => {
330
- it("never runs more than CONCURRENCY chunk summaries at once", async () => {
331
- const tracker = new TrackingSummarizer(6);
332
- // Each message alone exceeds the (MIN_BUDGET-floored) budget, so split()
333
- // emits one chunk per message → 20 independent chunk summaries that the
334
- // limiter must fan out at most CONCURRENCY at a time.
335
- const t: Transcript = Array.from({ length: 20 }, (_, i) =>
336
- text("user", `m${i}-${"x".repeat(5000)}`),
337
- );
338
- await compactTranscript({
339
- transcript: t,
340
- usableTokens: 2000,
341
- summarize: tracker.summarize,
342
- });
343
- expect(tracker.prompts.length).toBeGreaterThan(CONCURRENCY);
344
- expect(tracker.maxInFlight).toBeLessThanOrEqual(CONCURRENCY);
345
- expect(tracker.maxInFlight).toBeGreaterThan(1);
346
- });
347
- });
348
-
349
- describe("reduce — recursive map-reduce", () => {
350
- it("collapses many partial summaries into a single string", async () => {
351
- const t: Transcript = Array.from({ length: 30 }, (_, i) =>
352
- text("user", `message number ${i} with some body`),
353
- );
354
- const out = await compactTranscript({
355
- transcript: t,
356
- usableTokens: 2500,
357
- summarize: echoSummarize("R"),
358
- });
359
- expect(typeof out).toBe("string");
360
- expect(out.length).toBeGreaterThan(0);
361
- });
362
-
363
- it("bounds recursion to REDUCE_DEPTH levels even when summaries stay large", async () => {
364
- let calls = 0;
365
- // A summarizer that NEVER shrinks: always returns a big block so the
366
- // reducer can never satisfy the budget and must stop at REDUCE_DEPTH.
367
- const neverShrinks = async (_p: string): Promise<string> => {
368
- calls += 1;
369
- return "B".repeat(5000);
370
- };
371
- const t: Transcript = Array.from({ length: 16 }, (_, i) =>
372
- text("user", `chunk-seed-${i}-${"q".repeat(40)}`),
373
- );
374
- const out = await compactTranscript({
375
- transcript: t,
376
- usableTokens: 2000,
377
- summarize: neverShrinks,
378
- });
379
- // Terminates (does not recurse forever) and applies the output cap.
380
- expect(calls).toBeGreaterThan(0);
381
- expect(calls).toBeLessThan(1000);
382
- expect(estimateTokens(out)).toBeLessThanOrEqual(OUTPUT_TOKEN_CAP);
383
- });
384
-
385
- it("caps the final combined summary at OUTPUT_TOKEN_CAP and marks the truncation", async () => {
386
- // Single chunk whose summary massively overshoots the cap.
387
- const overshoot = async (_p: string): Promise<string> => "Z".repeat(40_000);
388
- const out = await compactTranscript({
389
- transcript: [text("user", "small input")],
390
- usableTokens: 4000,
391
- summarize: overshoot,
392
- });
393
- expect(estimateTokens(out)).toBeLessThanOrEqual(OUTPUT_TOKEN_CAP);
394
- expect(out.toLowerCase()).toContain("truncat");
395
- });
396
-
397
- it("passes a single small summary straight through without re-summarizing", async () => {
398
- let calls = 0;
399
- const counting = async (_p: string): Promise<string> => {
400
- calls += 1;
401
- return "tiny";
402
- };
403
- await compactTranscript({
404
- transcript: [text("user", "hi")],
405
- usableTokens: 4000,
406
- summarize: counting,
407
- });
408
- // Exactly one map call; no reduce pass needed for a lone fitting summary.
409
- expect(calls).toBe(1);
410
- });
411
- });
412
-
413
- describe("fail-closed sentinel", () => {
414
- it('returns "compact" when a chunk (MAP) summarization rejects', async () => {
415
- // Oversized messages → one chunk each → multiple MAP summarize calls; the
416
- // 2nd rejects, so the orchestrator must fail closed rather than splice a
417
- // partial recap built from only the 1st chunk.
418
- const t: Transcript = Array.from({ length: 12 }, (_, i) =>
419
- text("user", `m${i}-${"x".repeat(5000)}`),
420
- );
421
- const out = await compactTranscript({
422
- transcript: t,
423
- usableTokens: 2000,
424
- summarize: failOnCall(2),
425
- });
426
- expect(out).toBe("compact");
427
- });
428
-
429
- it('returns "compact" when a reduce-group summarization rejects', async () => {
430
- // 6 oversized messages → 6 chunks → 6 MAP calls (1-6) that all succeed and
431
- // return LARGE partials so the budget is NOT satisfied and a real reduce
432
- // pass runs. Reduce groups 6 partials into 3 pairs → calls 7,8,9; call 7
433
- // rejects, exercising a reduce-group failure specifically.
434
- let calls = 0;
435
- const bigThenFail = async (_p: string): Promise<string> => {
436
- calls += 1;
437
- if (calls === 7) {
438
- throw new Error("model overloaded mid-reduce");
439
- }
440
- return "D".repeat(8000);
441
- };
442
- const t: Transcript = Array.from({ length: 6 }, (_, i) =>
443
- text("user", `seed-${i}-${"y".repeat(5000)}`),
444
- );
445
- const out = await compactTranscript({
446
- transcript: t,
447
- usableTokens: 2000,
448
- summarize: bigThenFail,
449
- });
450
- expect(calls).toBeGreaterThanOrEqual(7);
451
- expect(out).toBe("compact");
452
- });
453
-
454
- it("never returns an empty or partial summary on failure", async () => {
455
- const out = await compactTranscript({
456
- transcript: [text("user", "a"), text("user", "b"), text("user", "c")],
457
- usableTokens: 1500,
458
- summarize: async () => {
459
- throw new Error("always down");
460
- },
461
- });
462
- expect(out).toBe("compact");
463
- });
464
- });
465
-
466
- describe("end-to-end happy path", () => {
467
- it("produces a non-empty deterministic summary for a realistic transcript", async () => {
468
- const transcript: Transcript = [
469
- text("system", "You are a coding agent."),
470
- text("user", "Run the test suite at packages/foo and report failures."),
471
- {
472
- role: "assistant",
473
- parts: [{ type: "text", text: "Running `pnpm test` now." }],
474
- },
475
- {
476
- role: "tool",
477
- parts: [
478
- {
479
- type: "tool",
480
- name: "shell",
481
- input: "pnpm test",
482
- output: "1 failing: expected 3 received 4",
483
- },
484
- ],
485
- },
486
- text("assistant", "Decision: patch the off-by-one in sum(). Outstanding: re-run suite."),
487
- ];
488
- const first = await compactTranscript({
489
- transcript,
490
- usableTokens: 8000,
491
- summarize: echoSummarize("E"),
492
- base: charBase,
493
- });
494
- const second = await compactTranscript({
495
- transcript,
496
- usableTokens: 8000,
497
- summarize: echoSummarize("E"),
498
- base: charBase,
499
- });
500
- expect(first.length).toBeGreaterThan(0);
501
- expect(first).not.toBe("compact");
502
- // Deterministic given the same injected summarize + base.
503
- expect(first).toBe(second);
504
- });
505
-
506
- it("does not mutate the caller's transcript end-to-end", async () => {
507
- const transcript: Transcript = [
508
- text("user", "p".repeat(CLIP_TEXT_CHARS + 50)),
509
- {
510
- role: "tool",
511
- parts: [{ type: "tool", name: "x", input: "i", output: "o" }],
512
- },
513
- ];
514
- const snapshot = JSON.stringify(transcript);
515
- await compactTranscript({
516
- transcript,
517
- usableTokens: 3000,
518
- summarize: echoSummarize(),
519
- });
520
- expect(JSON.stringify(transcript)).toBe(snapshot);
521
- });
522
- });