@nebutra/agent-runtime 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -676
- package/README.md +2 -0
- package/dist/adapters/dispatcher-sse.js +1 -0
- package/dist/adapters/index.d.ts +24 -5
- package/dist/adapters/index.js +28 -0
- package/dist/adapters/index.js.map +1 -1
- package/dist/adapters/mcp-catalog.js +1 -0
- package/dist/adapters/prisma-rollout.js +1 -0
- package/dist/chunk-424PT5DM.js +23 -0
- package/dist/chunk-424PT5DM.js.map +1 -0
- package/dist/{chunk-NN7DATXA.js → chunk-4Y25ZTKI.js} +3 -3
- package/dist/chunk-4Y25ZTKI.js.map +1 -0
- package/dist/{chunk-BJBBR3QA.js → chunk-D4YAPLOW.js} +4 -4
- package/dist/chunk-D4YAPLOW.js.map +1 -0
- package/dist/{chunk-ZMYX5VBU.js → chunk-GQZKYWFT.js} +23 -7
- package/dist/chunk-GQZKYWFT.js.map +1 -0
- package/dist/chunk-KCNN4QUQ.js +255 -0
- package/dist/chunk-KCNN4QUQ.js.map +1 -0
- package/dist/{chunk-PGGWSUTM.js → chunk-NI4EDT4T.js} +2 -2
- package/dist/chunk-NI4EDT4T.js.map +1 -0
- package/dist/chunk-Q62VKHIT.js +178 -0
- package/dist/chunk-Q62VKHIT.js.map +1 -0
- package/dist/{chunk-MUF7ZZTO.js → chunk-R5HOSQUW.js} +2 -1
- package/dist/{chunk-MUF7ZZTO.js.map → chunk-R5HOSQUW.js.map} +1 -1
- package/dist/{chunk-5N4644PB.js → chunk-SD2ZJ7XG.js} +4 -4
- package/dist/cli.d.ts +2 -0
- package/dist/cli.js +52 -0
- package/dist/cli.js.map +1 -0
- package/dist/commands.js +1 -0
- package/dist/definitions.js +1 -0
- package/dist/dispatcher.js +1 -0
- package/dist/durable-turn.js +3 -2
- package/dist/hook-pipeline.js +1 -0
- package/dist/index.d.ts +8 -85
- package/dist/index.js +249 -134
- package/dist/index.js.map +1 -1
- package/dist/loop.d.ts +2 -2
- package/dist/loop.js +3 -2
- package/dist/mcp-bridge.d.ts +3 -3
- package/dist/mcp-bridge.js +3 -2
- package/dist/model.js +1 -0
- package/dist/orchestration.d.ts +84 -0
- package/dist/orchestration.js +16 -0
- package/dist/orchestration.js.map +1 -0
- package/dist/policy.js +1 -0
- package/dist/protocol.js +1 -0
- package/dist/pulsar.d.ts +78 -0
- package/dist/pulsar.js +17 -0
- package/dist/pulsar.js.map +1 -0
- package/dist/rollout-store-persistent.js +1 -0
- package/dist/rollout.js +1 -0
- package/dist/sandbox.js +2 -1
- package/dist/skills.js +2 -1
- package/dist/subagents.js +1 -0
- package/dist/tools.d.ts +4 -3
- package/dist/tools.js +5 -3
- package/package.json +84 -27
- package/.turbo/turbo-build.log +0 -115
- package/.turbo/turbo-test.log +0 -44
- package/.turbo/turbo-typecheck.log +0 -4
- package/CHANGELOG.md +0 -253
- package/dist/chunk-BJBBR3QA.js.map +0 -1
- package/dist/chunk-NN7DATXA.js.map +0 -1
- package/dist/chunk-PGGWSUTM.js.map +0 -1
- package/dist/chunk-ZMYX5VBU.js.map +0 -1
- package/src/adapters/dispatcher-sse.test.ts +0 -218
- package/src/adapters/dispatcher-sse.ts +0 -222
- package/src/adapters/index.ts +0 -18
- package/src/adapters/mcp-catalog.test.ts +0 -213
- package/src/adapters/mcp-catalog.ts +0 -188
- package/src/adapters/prisma-rollout.test.ts +0 -153
- package/src/adapters/prisma-rollout.ts +0 -104
- package/src/agent-runtime.test.ts +0 -176
- package/src/artifact-stream.test.ts +0 -330
- package/src/artifact-stream.ts +0 -453
- package/src/channel-gateway.test.ts +0 -432
- package/src/channel-gateway.ts +0 -357
- package/src/code-review.test.ts +0 -501
- package/src/code-review.ts +0 -495
- package/src/command-suggestions.test.ts +0 -251
- package/src/command-suggestions.ts +0 -338
- package/src/commands.test.ts +0 -184
- package/src/commands.ts +0 -140
- package/src/commit-message.test.ts +0 -249
- package/src/commit-message.ts +0 -180
- package/src/context-compaction.test.ts +0 -522
- package/src/context-compaction.ts +0 -434
- package/src/definitions.test.ts +0 -78
- package/src/definitions.ts +0 -190
- package/src/deployment-status.test.ts +0 -215
- package/src/deployment-status.ts +0 -227
- package/src/design-context.test.ts +0 -195
- package/src/design-context.ts +0 -198
- package/src/dispatcher.test.ts +0 -234
- package/src/dispatcher.ts +0 -189
- package/src/durable-turn.test.ts +0 -209
- package/src/durable-turn.ts +0 -135
- package/src/edit-planner.test.ts +0 -204
- package/src/edit-planner.ts +0 -325
- package/src/fuzzy-match.test.ts +0 -311
- package/src/fuzzy-match.ts +0 -444
- package/src/hook-pipeline.test.ts +0 -279
- package/src/hook-pipeline.ts +0 -373
- package/src/inbound-admission.test.ts +0 -394
- package/src/inbound-admission.ts +0 -246
- package/src/index.ts +0 -47
- package/src/loop.test.ts +0 -161
- package/src/loop.ts +0 -211
- package/src/mcp-bridge.test.ts +0 -165
- package/src/mcp-bridge.ts +0 -76
- package/src/memory-provider.test.ts +0 -232
- package/src/memory-provider.ts +0 -257
- package/src/model.ts +0 -168
- package/src/permission-ruleset.test.ts +0 -301
- package/src/permission-ruleset.ts +0 -200
- package/src/policy.ts +0 -151
- package/src/project-repo.test.ts +0 -232
- package/src/project-repo.ts +0 -311
- package/src/protocol.ts +0 -159
- package/src/rollout-store-persistent.test.ts +0 -217
- package/src/rollout-store-persistent.ts +0 -166
- package/src/rollout.ts +0 -150
- package/src/sandbox.ts +0 -113
- package/src/session-share.test.ts +0 -360
- package/src/session-share.ts +0 -310
- package/src/skill-distillation.test.ts +0 -177
- package/src/skill-distillation.ts +0 -369
- package/src/skills.test.ts +0 -277
- package/src/skills.ts +0 -255
- package/src/subagents.test.ts +0 -290
- package/src/subagents.ts +0 -332
- package/src/tools.ts +0 -126
- package/src/workbench.test.ts +0 -0
- package/src/workbench.ts +0 -0
- package/tsconfig.json +0 -12
- package/tsup.config.ts +0 -33
- /package/dist/{chunk-5N4644PB.js.map → chunk-SD2ZJ7XG.js.map} +0 -0
package/src/commit-message.ts
DELETED
|
@@ -1,180 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Conventional-Commits message generation — a faithful re-expression of a
|
|
3
|
-
* coding agent's "write my commit message from staged changes" capability into
|
|
4
|
-
* Sailor's grammar: TypeScript, multi-tenant, no model-vendor lock-in.
|
|
5
|
-
*
|
|
6
|
-
* ── This is a WRAP, not an agent ────────────────────────────────────────────
|
|
7
|
-
* The small-model invocation is INJECTED via the {@link CompletionModel} port.
|
|
8
|
-
* This module does NOT implement an agent loop, a transport, a token budget, or
|
|
9
|
-
* tenancy — those travel inside the injected port (the real impl is the runtime
|
|
10
|
-
* model seam; tests pass a deterministic fake). The delta this module owns is
|
|
11
|
-
* purely DOMAIN SHAPING:
|
|
12
|
-
* • the Conventional Commits system prompt,
|
|
13
|
-
* • the git-context → user-prompt rendering,
|
|
14
|
-
* • the "regenerate, but DIFFERENT" negative-constraint contract,
|
|
15
|
-
* • output de-formatting + bounded retry with a FAIL-CLOSED terminal error.
|
|
16
|
-
*
|
|
17
|
-
* Everything here is pure/deterministic given the injected model. Inputs are
|
|
18
|
-
* Zod-validated at the boundary; nothing is mutated in place.
|
|
19
|
-
*/
|
|
20
|
-
|
|
21
|
-
import { z } from "zod";
|
|
22
|
-
|
|
23
|
-
/** Sampling temperature handed to the injected model (slightly creative, stable). */
|
|
24
|
-
export const DEFAULT_TEMPERATURE = 0.3;
|
|
25
|
-
/** Per-attempt abort budget (ms) handed to the injected model. */
|
|
26
|
-
export const DEFAULT_ABORT_MS = 30_000;
|
|
27
|
-
/** Total attempts before failing closed (initial try + retries inclusive). */
|
|
28
|
-
export const MAX_RETRIES = 3;
|
|
29
|
-
|
|
30
|
-
const fileSchema = z.object({
|
|
31
|
-
status: z.string(),
|
|
32
|
-
path: z.string(),
|
|
33
|
-
diff: z.string(),
|
|
34
|
-
});
|
|
35
|
-
|
|
36
|
-
const gitContextSchema = z.object({
|
|
37
|
-
branch: z.string(),
|
|
38
|
-
recentCommits: z.array(z.string()),
|
|
39
|
-
files: z.array(fileSchema),
|
|
40
|
-
});
|
|
41
|
-
|
|
42
|
-
/** Staged git context the message is derived from. Zod-validated public input. */
|
|
43
|
-
export type GitContext = z.infer<typeof gitContextSchema>;
|
|
44
|
-
|
|
45
|
-
/**
|
|
46
|
-
* When `previous` is set this is a "regenerate, but materially DIFFERENT from
|
|
47
|
-
* the previous message" request — see {@link buildCommitPrompt}. Built
|
|
48
|
-
* conditionally to honor `exactOptionalPropertyTypes`.
|
|
49
|
-
*/
|
|
50
|
-
export interface GenerateOptions {
|
|
51
|
-
readonly previous?: string | undefined;
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
/**
|
|
55
|
-
* Injected small-model seam. The real implementation is the runtime's model
|
|
56
|
-
* invocation (carrying tenancy/auth); tests inject a deterministic fake. This
|
|
57
|
-
* module consumes the port and never implements model plumbing itself.
|
|
58
|
-
*/
|
|
59
|
-
export interface CompletionModel {
|
|
60
|
-
complete(p: {
|
|
61
|
-
system: string;
|
|
62
|
-
user: string;
|
|
63
|
-
temperature: number;
|
|
64
|
-
abortMs: number;
|
|
65
|
-
}): Promise<string>;
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
/** Raised when every attempt is exhausted. Fail closed — never a fake message. */
|
|
69
|
-
export class CommitMessageError extends Error {
|
|
70
|
-
constructor(message = "failed to generate a commit message") {
|
|
71
|
-
super(message);
|
|
72
|
-
this.name = "CommitMessageError";
|
|
73
|
-
}
|
|
74
|
-
}
|
|
75
|
-
|
|
76
|
-
const SYSTEM_PROMPT = [
|
|
77
|
-
"You write a single git commit message that follows the Conventional Commits specification.",
|
|
78
|
-
"",
|
|
79
|
-
"Format: type(scope): subject",
|
|
80
|
-
" - scope is optional; omit the parentheses entirely when there is no scope.",
|
|
81
|
-
" - allowed types: feat, fix, docs, style, refactor, perf, test, build, ci, chore.",
|
|
82
|
-
" - subject: imperative mood, lower-case, concise, no trailing period.",
|
|
83
|
-
" - an optional body MAY follow after one blank line to explain the why.",
|
|
84
|
-
"",
|
|
85
|
-
"Return ONLY the commit message. No code fences, no quotes, no commentary, no preamble.",
|
|
86
|
-
].join("\n");
|
|
87
|
-
|
|
88
|
-
function renderFiles(files: GitContext["files"]): string {
|
|
89
|
-
return files.map((f) => `[${f.status}] ${f.path}\n${f.diff}`).join("\n\n");
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
/**
|
|
93
|
-
* Builds the `{ system, user }` pair. The system prompt encodes the
|
|
94
|
-
* Conventional Commits spec and the "only the message" instruction. The user
|
|
95
|
-
* prompt renders branch + recent commits + per-file status/path/diff. When
|
|
96
|
-
* `opts.previous` is set, a NEGATIVE CONSTRAINT block is appended so a
|
|
97
|
-
* regenerate request yields a materially different message. Pure.
|
|
98
|
-
*/
|
|
99
|
-
export function buildCommitPrompt(
|
|
100
|
-
ctx: GitContext,
|
|
101
|
-
opts?: GenerateOptions,
|
|
102
|
-
): { system: string; user: string } {
|
|
103
|
-
const sections = [
|
|
104
|
-
`Branch:\n${ctx.branch}`,
|
|
105
|
-
`Recent commits:\n${ctx.recentCommits.join("\n")}`,
|
|
106
|
-
`Staged changes:\n${renderFiles(ctx.files)}`,
|
|
107
|
-
];
|
|
108
|
-
|
|
109
|
-
if (opts?.previous !== undefined) {
|
|
110
|
-
sections.push(
|
|
111
|
-
[
|
|
112
|
-
"NEGATIVE CONSTRAINT:",
|
|
113
|
-
"the following message was already generated and rejected;",
|
|
114
|
-
"produce a materially different message:",
|
|
115
|
-
opts.previous,
|
|
116
|
-
].join("\n"),
|
|
117
|
-
);
|
|
118
|
-
}
|
|
119
|
-
|
|
120
|
-
return { system: SYSTEM_PROMPT, user: sections.join("\n\n") };
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
const FENCE = /^```[^\n]*\n([\s\S]*?)\n?```$/;
|
|
124
|
-
|
|
125
|
-
/**
|
|
126
|
-
* Removes surrounding triple-backtick fences (with optional language tag) and
|
|
127
|
-
* surrounding single/double quotes, then trims. Pure — defensive against models
|
|
128
|
-
* that wrap output despite the system instruction.
|
|
129
|
-
*/
|
|
130
|
-
export function stripFormatting(raw: string): string {
|
|
131
|
-
let s = raw.trim();
|
|
132
|
-
const fenced = FENCE.exec(s);
|
|
133
|
-
if (fenced) {
|
|
134
|
-
s = (fenced[1] ?? "").trim();
|
|
135
|
-
}
|
|
136
|
-
if (
|
|
137
|
-
s.length >= 2 &&
|
|
138
|
-
((s.startsWith('"') && s.endsWith('"')) || (s.startsWith("'") && s.endsWith("'")))
|
|
139
|
-
) {
|
|
140
|
-
s = s.slice(1, -1).trim();
|
|
141
|
-
}
|
|
142
|
-
return s;
|
|
143
|
-
}
|
|
144
|
-
|
|
145
|
-
/**
|
|
146
|
-
* Generates a Conventional-Commits message from staged git context by wrapping
|
|
147
|
-
* the injected {@link CompletionModel}. Zod-validates `ctx`; calls the model at
|
|
148
|
-
* {@link DEFAULT_TEMPERATURE}/{@link DEFAULT_ABORT_MS}; retries up to
|
|
149
|
-
* {@link MAX_RETRIES} total attempts on rejection OR empty/whitespace output;
|
|
150
|
-
* returns the de-formatted message. FAIL CLOSED: if every attempt is exhausted
|
|
151
|
-
* it throws {@link CommitMessageError} — never an empty or fabricated message.
|
|
152
|
-
*/
|
|
153
|
-
export async function generateCommitMessage(args: {
|
|
154
|
-
ctx: GitContext;
|
|
155
|
-
model: CompletionModel;
|
|
156
|
-
opts?: GenerateOptions;
|
|
157
|
-
}): Promise<string> {
|
|
158
|
-
const ctx = gitContextSchema.parse(args.ctx);
|
|
159
|
-
const prompt = buildCommitPrompt(ctx, args.opts !== undefined ? args.opts : undefined);
|
|
160
|
-
|
|
161
|
-
for (let attempt = 0; attempt < MAX_RETRIES; attempt += 1) {
|
|
162
|
-
try {
|
|
163
|
-
const raw = await args.model.complete({
|
|
164
|
-
system: prompt.system,
|
|
165
|
-
user: prompt.user,
|
|
166
|
-
temperature: DEFAULT_TEMPERATURE,
|
|
167
|
-
abortMs: DEFAULT_ABORT_MS,
|
|
168
|
-
});
|
|
169
|
-
const message = stripFormatting(raw);
|
|
170
|
-
if (message.length > 0) {
|
|
171
|
-
return message;
|
|
172
|
-
}
|
|
173
|
-
} catch {
|
|
174
|
-
// Swallow per-attempt failure; the bounded loop decides the terminal
|
|
175
|
-
// outcome and fails closed below.
|
|
176
|
-
}
|
|
177
|
-
}
|
|
178
|
-
|
|
179
|
-
throw new CommitMessageError();
|
|
180
|
-
}
|
|
@@ -1,522 +0,0 @@
|
|
|
1
|
-
import { describe, expect, it } from "vitest";
|
|
2
|
-
import {
|
|
3
|
-
BUDGET_RATIO,
|
|
4
|
-
CLIP_TEXT_CHARS,
|
|
5
|
-
CLIP_TOOL_CHARS,
|
|
6
|
-
CONCURRENCY,
|
|
7
|
-
compactTranscript,
|
|
8
|
-
computeBudget,
|
|
9
|
-
estimateTokens,
|
|
10
|
-
type Message,
|
|
11
|
-
MIN_BUDGET,
|
|
12
|
-
OUTPUT_TOKEN_CAP,
|
|
13
|
-
REDUCE_DEPTH,
|
|
14
|
-
renderClipped,
|
|
15
|
-
shouldCompact,
|
|
16
|
-
split,
|
|
17
|
-
TOKEN_CORRECTION,
|
|
18
|
-
type Transcript,
|
|
19
|
-
} from "./context-compaction.js";
|
|
20
|
-
|
|
21
|
-
/**
|
|
22
|
-
* House-rule fidelity: NO mocking libraries. Every "model" is a hand-written
|
|
23
|
-
* deterministic fake whose behaviour is fully observable from the test, so the
|
|
24
|
-
* suite stays reproducible without Date/random/network.
|
|
25
|
-
*/
|
|
26
|
-
|
|
27
|
-
/** A deterministic base estimator: 1 token per 4 chars (the documented default). */
|
|
28
|
-
const charBase = (s: string): number => Math.ceil(s.length / 4);
|
|
29
|
-
|
|
30
|
-
/** Echoing summarizer — returns a short, deterministic, prompt-derived string. */
|
|
31
|
-
const echoSummarize =
|
|
32
|
-
(label = "S") =>
|
|
33
|
-
async (prompt: string): Promise<string> =>
|
|
34
|
-
`${label}[len=${prompt.length}]`;
|
|
35
|
-
|
|
36
|
-
/**
|
|
37
|
-
* A summarizer that records every prompt it ever saw AND tracks the maximum
|
|
38
|
-
* number of calls that were ever in flight simultaneously, so concurrency
|
|
39
|
-
* limiting can be asserted without timers or mocks. Each call yields control
|
|
40
|
-
* to the microtask queue a fixed number of times before resolving, which lets
|
|
41
|
-
* the scheduler interleave the limiter's batch.
|
|
42
|
-
*/
|
|
43
|
-
class TrackingSummarizer {
|
|
44
|
-
readonly prompts: string[] = [];
|
|
45
|
-
inFlight = 0;
|
|
46
|
-
maxInFlight = 0;
|
|
47
|
-
#counter = 0;
|
|
48
|
-
|
|
49
|
-
constructor(private readonly settle = 4) {}
|
|
50
|
-
|
|
51
|
-
summarize = async (prompt: string): Promise<string> => {
|
|
52
|
-
this.prompts.push(prompt);
|
|
53
|
-
this.inFlight += 1;
|
|
54
|
-
this.maxInFlight = Math.max(this.maxInFlight, this.inFlight);
|
|
55
|
-
for (let i = 0; i < this.settle; i += 1) {
|
|
56
|
-
await Promise.resolve();
|
|
57
|
-
}
|
|
58
|
-
this.inFlight -= 1;
|
|
59
|
-
this.#counter += 1;
|
|
60
|
-
return `sum#${this.#counter}`;
|
|
61
|
-
};
|
|
62
|
-
}
|
|
63
|
-
|
|
64
|
-
/** A summarizer that rejects on the Nth (1-based) call, otherwise echoes. */
|
|
65
|
-
const failOnCall = (n: number) => {
|
|
66
|
-
let calls = 0;
|
|
67
|
-
return async (_prompt: string): Promise<string> => {
|
|
68
|
-
calls += 1;
|
|
69
|
-
if (calls === n) {
|
|
70
|
-
throw new Error("model overloaded");
|
|
71
|
-
}
|
|
72
|
-
return `ok#${calls}`;
|
|
73
|
-
};
|
|
74
|
-
};
|
|
75
|
-
|
|
76
|
-
const text = (role: Message["role"], body: string): Message => ({
|
|
77
|
-
role,
|
|
78
|
-
parts: [{ type: "text", text: body }],
|
|
79
|
-
});
|
|
80
|
-
|
|
81
|
-
describe("exported design constants", () => {
|
|
82
|
-
it("freezes the documented numeric contract", () => {
|
|
83
|
-
expect(TOKEN_CORRECTION).toBe(1.3);
|
|
84
|
-
expect(BUDGET_RATIO).toBe(0.6);
|
|
85
|
-
expect(MIN_BUDGET).toBe(1000);
|
|
86
|
-
expect(CONCURRENCY).toBe(3);
|
|
87
|
-
expect(REDUCE_DEPTH).toBe(3);
|
|
88
|
-
expect(OUTPUT_TOKEN_CAP).toBe(2048);
|
|
89
|
-
expect(CLIP_TOOL_CHARS).toBe(2000);
|
|
90
|
-
expect(CLIP_TEXT_CHARS).toBe(16000);
|
|
91
|
-
});
|
|
92
|
-
});
|
|
93
|
-
|
|
94
|
-
describe("estimateTokens", () => {
|
|
95
|
-
it("applies the 1.3x correction over the default char estimator", () => {
|
|
96
|
-
// 8 chars → base ceil(8/4)=2 → 2 * 1.3 = 2.6 → ceil → 3
|
|
97
|
-
expect(estimateTokens("abcdefgh")).toBe(3);
|
|
98
|
-
});
|
|
99
|
-
|
|
100
|
-
it("uses an injected base estimator when provided", () => {
|
|
101
|
-
// base returns 10 → 10 * 1.3 = 13
|
|
102
|
-
expect(estimateTokens("ignored", () => 10)).toBe(13);
|
|
103
|
-
});
|
|
104
|
-
|
|
105
|
-
it("always rounds up so it never undercounts", () => {
|
|
106
|
-
// base 1 → 1.3 → ceil → 2
|
|
107
|
-
expect(estimateTokens("abcd", () => 1)).toBe(2);
|
|
108
|
-
});
|
|
109
|
-
|
|
110
|
-
it("returns 0 for empty text under the default estimator", () => {
|
|
111
|
-
expect(estimateTokens("")).toBe(0);
|
|
112
|
-
});
|
|
113
|
-
});
|
|
114
|
-
|
|
115
|
-
describe("computeBudget", () => {
|
|
116
|
-
it("takes 60% of the usable token window", () => {
|
|
117
|
-
expect(computeBudget(10_000)).toBe(6000);
|
|
118
|
-
});
|
|
119
|
-
|
|
120
|
-
it("floors fractional budgets", () => {
|
|
121
|
-
// 3333 * 0.6 = 1999.8 → floor → 1999 (> MIN_BUDGET)
|
|
122
|
-
expect(computeBudget(3333)).toBe(1999);
|
|
123
|
-
});
|
|
124
|
-
|
|
125
|
-
it("never returns below MIN_BUDGET", () => {
|
|
126
|
-
expect(computeBudget(0)).toBe(MIN_BUDGET);
|
|
127
|
-
expect(computeBudget(100)).toBe(MIN_BUDGET);
|
|
128
|
-
});
|
|
129
|
-
|
|
130
|
-
it("uses MIN_BUDGET exactly at the crossover", () => {
|
|
131
|
-
// 1666 * 0.6 = 999.6 → floor 999 → clamped up to 1000
|
|
132
|
-
expect(computeBudget(1666)).toBe(MIN_BUDGET);
|
|
133
|
-
});
|
|
134
|
-
});
|
|
135
|
-
|
|
136
|
-
describe("shouldCompact", () => {
|
|
137
|
-
it("is true for the explicit compact stop reason", () => {
|
|
138
|
-
expect(shouldCompact({ reason: "compact" })).toBe(true);
|
|
139
|
-
});
|
|
140
|
-
|
|
141
|
-
it("is true for context_overflow", () => {
|
|
142
|
-
expect(shouldCompact({ reason: "context_overflow" })).toBe(true);
|
|
143
|
-
});
|
|
144
|
-
|
|
145
|
-
it("is false for unrelated stop reasons", () => {
|
|
146
|
-
expect(shouldCompact({ reason: "end_turn" })).toBe(false);
|
|
147
|
-
expect(shouldCompact({ reason: "tool_use" })).toBe(false);
|
|
148
|
-
expect(shouldCompact({ reason: "" })).toBe(false);
|
|
149
|
-
});
|
|
150
|
-
});
|
|
151
|
-
|
|
152
|
-
describe("renderClipped", () => {
|
|
153
|
-
it("emits an XML-ish conversation envelope tagged by role", () => {
|
|
154
|
-
const out = renderClipped(text("user", "hello world"));
|
|
155
|
-
expect(out).toContain("<conversation>");
|
|
156
|
-
expect(out).toContain("</conversation>");
|
|
157
|
-
expect(out).toContain('role="user"');
|
|
158
|
-
expect(out).toContain("hello world");
|
|
159
|
-
});
|
|
160
|
-
|
|
161
|
-
it("renders tool parts with name, input and output", () => {
|
|
162
|
-
const msg: Message = {
|
|
163
|
-
role: "tool",
|
|
164
|
-
parts: [{ type: "tool", name: "grep", input: "foo", output: "bar" }],
|
|
165
|
-
};
|
|
166
|
-
const out = renderClipped(msg);
|
|
167
|
-
expect(out).toContain("grep");
|
|
168
|
-
expect(out).toContain("foo");
|
|
169
|
-
expect(out).toContain("bar");
|
|
170
|
-
});
|
|
171
|
-
|
|
172
|
-
it("clips overlong text to CLIP_TEXT_CHARS and appends a truncation marker", () => {
|
|
173
|
-
const big = "x".repeat(CLIP_TEXT_CHARS + 500);
|
|
174
|
-
const out = renderClipped(text("assistant", big));
|
|
175
|
-
expect(out).toContain("x".repeat(CLIP_TEXT_CHARS));
|
|
176
|
-
expect(out).not.toContain("x".repeat(CLIP_TEXT_CHARS + 1));
|
|
177
|
-
expect(out.toLowerCase()).toContain("truncat");
|
|
178
|
-
});
|
|
179
|
-
|
|
180
|
-
it("clips tool input and output to CLIP_TOOL_CHARS independently", () => {
|
|
181
|
-
const msg: Message = {
|
|
182
|
-
role: "tool",
|
|
183
|
-
parts: [
|
|
184
|
-
{
|
|
185
|
-
type: "tool",
|
|
186
|
-
name: "shell",
|
|
187
|
-
input: "i".repeat(CLIP_TOOL_CHARS + 100),
|
|
188
|
-
output: "o".repeat(CLIP_TOOL_CHARS + 100),
|
|
189
|
-
},
|
|
190
|
-
],
|
|
191
|
-
};
|
|
192
|
-
const out = renderClipped(msg);
|
|
193
|
-
expect(out).toContain("i".repeat(CLIP_TOOL_CHARS));
|
|
194
|
-
expect(out).not.toContain("i".repeat(CLIP_TOOL_CHARS + 1));
|
|
195
|
-
expect(out).toContain("o".repeat(CLIP_TOOL_CHARS));
|
|
196
|
-
expect(out).not.toContain("o".repeat(CLIP_TOOL_CHARS + 1));
|
|
197
|
-
});
|
|
198
|
-
|
|
199
|
-
it("does not append a marker when nothing was clipped", () => {
|
|
200
|
-
const out = renderClipped(text("user", "short"));
|
|
201
|
-
expect(out.toLowerCase()).not.toContain("truncat");
|
|
202
|
-
});
|
|
203
|
-
|
|
204
|
-
it("does not mutate the input message", () => {
|
|
205
|
-
const msg = text("user", "y".repeat(CLIP_TEXT_CHARS + 10));
|
|
206
|
-
const snapshot = JSON.stringify(msg);
|
|
207
|
-
renderClipped(msg);
|
|
208
|
-
expect(JSON.stringify(msg)).toBe(snapshot);
|
|
209
|
-
});
|
|
210
|
-
});
|
|
211
|
-
|
|
212
|
-
describe("split", () => {
|
|
213
|
-
it("greedily packs whole messages into budget-sized chunks", () => {
|
|
214
|
-
// split() estimates the RENDERED (clipped, XML-enveloped) message, not the
|
|
215
|
-
// raw text. Derive the budget from the real rendered size so the test
|
|
216
|
-
// pins packing behaviour, not a guessed envelope length. Uniform "user"
|
|
217
|
-
// role keeps every message the same size.
|
|
218
|
-
const t: Transcript = [
|
|
219
|
-
text("user", "aaaa"),
|
|
220
|
-
text("user", "bbbb"),
|
|
221
|
-
text("user", "cccc"),
|
|
222
|
-
text("user", "dddd"),
|
|
223
|
-
text("user", "eeee"),
|
|
224
|
-
];
|
|
225
|
-
const per = estimateTokens(renderClipped(text("user", "aaaa")));
|
|
226
|
-
// Budget holds exactly 2 messages (2*per) but not a 3rd.
|
|
227
|
-
const chunks = split(t, per * 2, estimateTokens);
|
|
228
|
-
// 2 + 2 + 1 messages → 3 chunks
|
|
229
|
-
expect(chunks).toHaveLength(3);
|
|
230
|
-
expect(chunks[0]).toHaveLength(2);
|
|
231
|
-
expect(chunks[1]).toHaveLength(2);
|
|
232
|
-
expect(chunks[2]).toHaveLength(1);
|
|
233
|
-
});
|
|
234
|
-
|
|
235
|
-
it("never splits a single message across chunks", () => {
|
|
236
|
-
const t: Transcript = [text("user", "aaaa"), text("user", "bbbb")];
|
|
237
|
-
const per = estimateTokens(renderClipped(text("user", "aaaa")));
|
|
238
|
-
// Budget holds exactly one message → one message per chunk.
|
|
239
|
-
const chunks = split(t, per, estimateTokens);
|
|
240
|
-
expect(chunks).toHaveLength(2);
|
|
241
|
-
expect(chunks.every((c) => c.length === 1)).toBe(true);
|
|
242
|
-
});
|
|
243
|
-
|
|
244
|
-
it("makes an oversize single message its own chunk", () => {
|
|
245
|
-
const small = text("user", "aaaa");
|
|
246
|
-
const huge = text("assistant", "z".repeat(4000));
|
|
247
|
-
const t: Transcript = [small, huge, text("user", "bbbb")];
|
|
248
|
-
const per = estimateTokens(renderClipped(small));
|
|
249
|
-
// Budget holds the small messages but the huge one alone exceeds it.
|
|
250
|
-
const chunks = split(t, per, estimateTokens);
|
|
251
|
-
// small alone, huge alone (oversize fallback), bbbb alone
|
|
252
|
-
expect(chunks).toHaveLength(3);
|
|
253
|
-
const oversize = chunks[1] ?? [];
|
|
254
|
-
expect(oversize).toHaveLength(1);
|
|
255
|
-
expect(oversize[0]).toBe(huge);
|
|
256
|
-
});
|
|
257
|
-
|
|
258
|
-
it("returns no chunks for an empty transcript", () => {
|
|
259
|
-
expect(split([], 100, estimateTokens)).toEqual([]);
|
|
260
|
-
});
|
|
261
|
-
|
|
262
|
-
it("does not mutate the source transcript", () => {
|
|
263
|
-
const t: Transcript = [text("user", "aaaa"), text("user", "bbbb")];
|
|
264
|
-
const snapshot = JSON.stringify(t);
|
|
265
|
-
split(t, 1, estimateTokens);
|
|
266
|
-
expect(JSON.stringify(t)).toBe(snapshot);
|
|
267
|
-
});
|
|
268
|
-
});
|
|
269
|
-
|
|
270
|
-
describe("compactTranscript — input validation", () => {
|
|
271
|
-
it("rejects a non-array transcript", async () => {
|
|
272
|
-
await expect(
|
|
273
|
-
// @ts-expect-error — exercising the zod boundary
|
|
274
|
-
compactTranscript({ transcript: "nope", usableTokens: 4000, summarize: echoSummarize() }),
|
|
275
|
-
).rejects.toThrow();
|
|
276
|
-
});
|
|
277
|
-
|
|
278
|
-
it("rejects a non-positive usable token budget", async () => {
|
|
279
|
-
await expect(
|
|
280
|
-
compactTranscript({
|
|
281
|
-
transcript: [text("user", "hi")],
|
|
282
|
-
usableTokens: 0,
|
|
283
|
-
summarize: echoSummarize(),
|
|
284
|
-
}),
|
|
285
|
-
).rejects.toThrow();
|
|
286
|
-
});
|
|
287
|
-
|
|
288
|
-
it("rejects a malformed message part", async () => {
|
|
289
|
-
await expect(
|
|
290
|
-
compactTranscript({
|
|
291
|
-
// @ts-expect-error — exercising the zod boundary
|
|
292
|
-
transcript: [{ role: "user", parts: [{ type: "text" }] }],
|
|
293
|
-
usableTokens: 4000,
|
|
294
|
-
summarize: echoSummarize(),
|
|
295
|
-
}),
|
|
296
|
-
).rejects.toThrow();
|
|
297
|
-
});
|
|
298
|
-
|
|
299
|
-
it("accepts a well-formed tool part", async () => {
|
|
300
|
-
const out = await compactTranscript({
|
|
301
|
-
transcript: [
|
|
302
|
-
{ role: "tool", parts: [{ type: "tool", name: "ls", input: "-la", output: "files" }] },
|
|
303
|
-
],
|
|
304
|
-
usableTokens: 4000,
|
|
305
|
-
summarize: echoSummarize(),
|
|
306
|
-
});
|
|
307
|
-
expect(typeof out).toBe("string");
|
|
308
|
-
});
|
|
309
|
-
});
|
|
310
|
-
|
|
311
|
-
describe("summarization prompt rubric", () => {
|
|
312
|
-
it("instructs preservation of paths, commands, errors, decisions and open tasks", async () => {
|
|
313
|
-
const tracker = new TrackingSummarizer();
|
|
314
|
-
await compactTranscript({
|
|
315
|
-
transcript: [text("user", "do the thing"), text("assistant", "done")],
|
|
316
|
-
usableTokens: 4000,
|
|
317
|
-
summarize: tracker.summarize,
|
|
318
|
-
});
|
|
319
|
-
const joined = tracker.prompts.join("\n").toLowerCase();
|
|
320
|
-
expect(joined).toContain("file path");
|
|
321
|
-
expect(joined).toContain("command");
|
|
322
|
-
expect(joined).toContain("error");
|
|
323
|
-
expect(joined).toContain("decision");
|
|
324
|
-
// "outstanding" / "unresolved" tasks
|
|
325
|
-
expect(joined).toMatch(/unresolved|outstanding/);
|
|
326
|
-
});
|
|
327
|
-
});
|
|
328
|
-
|
|
329
|
-
describe("concurrency limiter", () => {
|
|
330
|
-
it("never runs more than CONCURRENCY chunk summaries at once", async () => {
|
|
331
|
-
const tracker = new TrackingSummarizer(6);
|
|
332
|
-
// Each message alone exceeds the (MIN_BUDGET-floored) budget, so split()
|
|
333
|
-
// emits one chunk per message → 20 independent chunk summaries that the
|
|
334
|
-
// limiter must fan out at most CONCURRENCY at a time.
|
|
335
|
-
const t: Transcript = Array.from({ length: 20 }, (_, i) =>
|
|
336
|
-
text("user", `m${i}-${"x".repeat(5000)}`),
|
|
337
|
-
);
|
|
338
|
-
await compactTranscript({
|
|
339
|
-
transcript: t,
|
|
340
|
-
usableTokens: 2000,
|
|
341
|
-
summarize: tracker.summarize,
|
|
342
|
-
});
|
|
343
|
-
expect(tracker.prompts.length).toBeGreaterThan(CONCURRENCY);
|
|
344
|
-
expect(tracker.maxInFlight).toBeLessThanOrEqual(CONCURRENCY);
|
|
345
|
-
expect(tracker.maxInFlight).toBeGreaterThan(1);
|
|
346
|
-
});
|
|
347
|
-
});
|
|
348
|
-
|
|
349
|
-
describe("reduce — recursive map-reduce", () => {
|
|
350
|
-
it("collapses many partial summaries into a single string", async () => {
|
|
351
|
-
const t: Transcript = Array.from({ length: 30 }, (_, i) =>
|
|
352
|
-
text("user", `message number ${i} with some body`),
|
|
353
|
-
);
|
|
354
|
-
const out = await compactTranscript({
|
|
355
|
-
transcript: t,
|
|
356
|
-
usableTokens: 2500,
|
|
357
|
-
summarize: echoSummarize("R"),
|
|
358
|
-
});
|
|
359
|
-
expect(typeof out).toBe("string");
|
|
360
|
-
expect(out.length).toBeGreaterThan(0);
|
|
361
|
-
});
|
|
362
|
-
|
|
363
|
-
it("bounds recursion to REDUCE_DEPTH levels even when summaries stay large", async () => {
|
|
364
|
-
let calls = 0;
|
|
365
|
-
// A summarizer that NEVER shrinks: always returns a big block so the
|
|
366
|
-
// reducer can never satisfy the budget and must stop at REDUCE_DEPTH.
|
|
367
|
-
const neverShrinks = async (_p: string): Promise<string> => {
|
|
368
|
-
calls += 1;
|
|
369
|
-
return "B".repeat(5000);
|
|
370
|
-
};
|
|
371
|
-
const t: Transcript = Array.from({ length: 16 }, (_, i) =>
|
|
372
|
-
text("user", `chunk-seed-${i}-${"q".repeat(40)}`),
|
|
373
|
-
);
|
|
374
|
-
const out = await compactTranscript({
|
|
375
|
-
transcript: t,
|
|
376
|
-
usableTokens: 2000,
|
|
377
|
-
summarize: neverShrinks,
|
|
378
|
-
});
|
|
379
|
-
// Terminates (does not recurse forever) and applies the output cap.
|
|
380
|
-
expect(calls).toBeGreaterThan(0);
|
|
381
|
-
expect(calls).toBeLessThan(1000);
|
|
382
|
-
expect(estimateTokens(out)).toBeLessThanOrEqual(OUTPUT_TOKEN_CAP);
|
|
383
|
-
});
|
|
384
|
-
|
|
385
|
-
it("caps the final combined summary at OUTPUT_TOKEN_CAP and marks the truncation", async () => {
|
|
386
|
-
// Single chunk whose summary massively overshoots the cap.
|
|
387
|
-
const overshoot = async (_p: string): Promise<string> => "Z".repeat(40_000);
|
|
388
|
-
const out = await compactTranscript({
|
|
389
|
-
transcript: [text("user", "small input")],
|
|
390
|
-
usableTokens: 4000,
|
|
391
|
-
summarize: overshoot,
|
|
392
|
-
});
|
|
393
|
-
expect(estimateTokens(out)).toBeLessThanOrEqual(OUTPUT_TOKEN_CAP);
|
|
394
|
-
expect(out.toLowerCase()).toContain("truncat");
|
|
395
|
-
});
|
|
396
|
-
|
|
397
|
-
it("passes a single small summary straight through without re-summarizing", async () => {
|
|
398
|
-
let calls = 0;
|
|
399
|
-
const counting = async (_p: string): Promise<string> => {
|
|
400
|
-
calls += 1;
|
|
401
|
-
return "tiny";
|
|
402
|
-
};
|
|
403
|
-
await compactTranscript({
|
|
404
|
-
transcript: [text("user", "hi")],
|
|
405
|
-
usableTokens: 4000,
|
|
406
|
-
summarize: counting,
|
|
407
|
-
});
|
|
408
|
-
// Exactly one map call; no reduce pass needed for a lone fitting summary.
|
|
409
|
-
expect(calls).toBe(1);
|
|
410
|
-
});
|
|
411
|
-
});
|
|
412
|
-
|
|
413
|
-
describe("fail-closed sentinel", () => {
|
|
414
|
-
it('returns "compact" when a chunk (MAP) summarization rejects', async () => {
|
|
415
|
-
// Oversized messages → one chunk each → multiple MAP summarize calls; the
|
|
416
|
-
// 2nd rejects, so the orchestrator must fail closed rather than splice a
|
|
417
|
-
// partial recap built from only the 1st chunk.
|
|
418
|
-
const t: Transcript = Array.from({ length: 12 }, (_, i) =>
|
|
419
|
-
text("user", `m${i}-${"x".repeat(5000)}`),
|
|
420
|
-
);
|
|
421
|
-
const out = await compactTranscript({
|
|
422
|
-
transcript: t,
|
|
423
|
-
usableTokens: 2000,
|
|
424
|
-
summarize: failOnCall(2),
|
|
425
|
-
});
|
|
426
|
-
expect(out).toBe("compact");
|
|
427
|
-
});
|
|
428
|
-
|
|
429
|
-
it('returns "compact" when a reduce-group summarization rejects', async () => {
|
|
430
|
-
// 6 oversized messages → 6 chunks → 6 MAP calls (1-6) that all succeed and
|
|
431
|
-
// return LARGE partials so the budget is NOT satisfied and a real reduce
|
|
432
|
-
// pass runs. Reduce groups 6 partials into 3 pairs → calls 7,8,9; call 7
|
|
433
|
-
// rejects, exercising a reduce-group failure specifically.
|
|
434
|
-
let calls = 0;
|
|
435
|
-
const bigThenFail = async (_p: string): Promise<string> => {
|
|
436
|
-
calls += 1;
|
|
437
|
-
if (calls === 7) {
|
|
438
|
-
throw new Error("model overloaded mid-reduce");
|
|
439
|
-
}
|
|
440
|
-
return "D".repeat(8000);
|
|
441
|
-
};
|
|
442
|
-
const t: Transcript = Array.from({ length: 6 }, (_, i) =>
|
|
443
|
-
text("user", `seed-${i}-${"y".repeat(5000)}`),
|
|
444
|
-
);
|
|
445
|
-
const out = await compactTranscript({
|
|
446
|
-
transcript: t,
|
|
447
|
-
usableTokens: 2000,
|
|
448
|
-
summarize: bigThenFail,
|
|
449
|
-
});
|
|
450
|
-
expect(calls).toBeGreaterThanOrEqual(7);
|
|
451
|
-
expect(out).toBe("compact");
|
|
452
|
-
});
|
|
453
|
-
|
|
454
|
-
it("never returns an empty or partial summary on failure", async () => {
|
|
455
|
-
const out = await compactTranscript({
|
|
456
|
-
transcript: [text("user", "a"), text("user", "b"), text("user", "c")],
|
|
457
|
-
usableTokens: 1500,
|
|
458
|
-
summarize: async () => {
|
|
459
|
-
throw new Error("always down");
|
|
460
|
-
},
|
|
461
|
-
});
|
|
462
|
-
expect(out).toBe("compact");
|
|
463
|
-
});
|
|
464
|
-
});
|
|
465
|
-
|
|
466
|
-
describe("end-to-end happy path", () => {
|
|
467
|
-
it("produces a non-empty deterministic summary for a realistic transcript", async () => {
|
|
468
|
-
const transcript: Transcript = [
|
|
469
|
-
text("system", "You are a coding agent."),
|
|
470
|
-
text("user", "Run the test suite at packages/foo and report failures."),
|
|
471
|
-
{
|
|
472
|
-
role: "assistant",
|
|
473
|
-
parts: [{ type: "text", text: "Running `pnpm test` now." }],
|
|
474
|
-
},
|
|
475
|
-
{
|
|
476
|
-
role: "tool",
|
|
477
|
-
parts: [
|
|
478
|
-
{
|
|
479
|
-
type: "tool",
|
|
480
|
-
name: "shell",
|
|
481
|
-
input: "pnpm test",
|
|
482
|
-
output: "1 failing: expected 3 received 4",
|
|
483
|
-
},
|
|
484
|
-
],
|
|
485
|
-
},
|
|
486
|
-
text("assistant", "Decision: patch the off-by-one in sum(). Outstanding: re-run suite."),
|
|
487
|
-
];
|
|
488
|
-
const first = await compactTranscript({
|
|
489
|
-
transcript,
|
|
490
|
-
usableTokens: 8000,
|
|
491
|
-
summarize: echoSummarize("E"),
|
|
492
|
-
base: charBase,
|
|
493
|
-
});
|
|
494
|
-
const second = await compactTranscript({
|
|
495
|
-
transcript,
|
|
496
|
-
usableTokens: 8000,
|
|
497
|
-
summarize: echoSummarize("E"),
|
|
498
|
-
base: charBase,
|
|
499
|
-
});
|
|
500
|
-
expect(first.length).toBeGreaterThan(0);
|
|
501
|
-
expect(first).not.toBe("compact");
|
|
502
|
-
// Deterministic given the same injected summarize + base.
|
|
503
|
-
expect(first).toBe(second);
|
|
504
|
-
});
|
|
505
|
-
|
|
506
|
-
it("does not mutate the caller's transcript end-to-end", async () => {
|
|
507
|
-
const transcript: Transcript = [
|
|
508
|
-
text("user", "p".repeat(CLIP_TEXT_CHARS + 50)),
|
|
509
|
-
{
|
|
510
|
-
role: "tool",
|
|
511
|
-
parts: [{ type: "tool", name: "x", input: "i", output: "o" }],
|
|
512
|
-
},
|
|
513
|
-
];
|
|
514
|
-
const snapshot = JSON.stringify(transcript);
|
|
515
|
-
await compactTranscript({
|
|
516
|
-
transcript,
|
|
517
|
-
usableTokens: 3000,
|
|
518
|
-
summarize: echoSummarize(),
|
|
519
|
-
});
|
|
520
|
-
expect(JSON.stringify(transcript)).toBe(snapshot);
|
|
521
|
-
});
|
|
522
|
-
});
|