pi-jev-lens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/presend.ts ADDED
@@ -0,0 +1,217 @@
1
+ import type { TypeSafeClient } from "@typesafe-ai/sdk";
2
+ import type { Config } from "./config.ts";
3
+ import { truncate } from "./text.ts";
4
+ import { relevantView, splitBlocks, splitSections, type Block, type Candidates, type View, type ViewKind } from "./views.ts";
5
+
6
+ export interface PresendState {
7
+ task: { first_user_request: string; latest_user_message: string };
8
+ agent: { text_before_call: string; tool: string; args: string };
9
+ result: { kind: string; is_error: boolean; total_lines: number; total_chars: number };
10
+ views: Record<string, { lines: number; chars: number; preview: string }>;
11
+ }
12
+
13
+ export interface PresendDecision {
14
+ view: ViewKind;
15
+ needsFull: number;
16
+ probabilities: Record<string, number>;
17
+ confidence: number;
18
+ }
19
+
20
+ export interface PresendClassifier {
21
+ choose(state: PresendState, kinds: ViewKind[], signal?: AbortSignal): Promise<{ choice: ViewKind; probabilities: Record<string, number>; confidence: number; needsFull: number }>;
22
+ /** Second step: for each block, P(the agent will need its body). */
23
+ expand(state: ExpandState, signal?: AbortSignal): Promise<number[]>;
24
+ }
25
+
26
+ export interface ExpandState {
27
+ task: PresendState["task"];
28
+ agent: PresendState["agent"];
29
+ file: { kind: string; total_lines: number };
30
+ blocks: { index: number; signature: string; lines: string; preview: string }[];
31
+ }
32
+
33
+ export function buildExpandState(base: PresendState, blocks: Block[], text: string): ExpandState {
34
+ const lines = text.split("\n");
35
+ return {
36
+ task: base.task,
37
+ agent: base.agent,
38
+ file: { kind: base.result.kind, total_lines: base.result.total_lines },
39
+ blocks: blocks.map((b, index) => ({ index, signature: b.name, lines: `${b.from}-${b.to}`, preview: truncate(lines.slice(b.from - 1, Math.min(b.to, b.from + 2)).join("\n"), 240) })),
40
+ };
41
+ }
42
+
43
+ export function expandQuestions(n: number, prompts: PromptVariant = DEFAULT_PROMPTS, kind = "code") {
44
+ const q: Record<string, { type: "noul"; instructions: string; criteria: { true: string; false: string } }> = {};
45
+ const [instructions, t, f] = kind === "command" ? [prompts.sectionInstructions, prompts.sectionTrue, prompts.sectionFalse] : [prompts.expandInstructions, prompts.expandTrue, prompts.expandFalse];
46
+ for (let i = 0; i < n; i++) {
47
+ q[`b${i}`] = { type: "noul", instructions: instructions.replaceAll("{i}", String(i)), criteria: { true: t, false: f } };
48
+ }
49
+ return q;
50
+ }
51
+
52
+ const VIEW_DESCRIPTIONS: Record<ViewKind, string> = {
53
+ full: "The complete, unmodified output. Needed when the agent will edit or quote exact text, when details anywhere in the output matter, or when nothing else clearly suffices.",
54
+ outline: "Structure only: imports, exports, signatures, class and function headers, headings, doc comments, with line numbers. Enough to understand what a file offers and where things are, not enough to edit a body verbatim.",
55
+ focus: "Only the lines that mention the identifiers from the task and the tool call, with a few lines of context, line-numbered. Enough when the agent is looking for specific names.",
56
+ signals: "Command output reduced to errors, warnings, failing tests and the final summary lines with context, line-numbered. Enough for reacting to a failed or passed run.",
57
+ sample: "Header plus a sample of rows and the total count, for tabular or log-like data. Enough to learn the shape of the data, not its contents.",
58
+ head_tail: "The first and last lines only. Enough to see what the output is and how it ends.",
59
+ relevant: "Outline plus the full bodies of the blocks the agent will need.",
60
+ matches: "Search output (grep, rg) reduced to the first matches of every file with the count of further matches per file, plus any non-match lines. Enough to see which files and lines are involved; not enough to read every match.",
61
+ log: "Script or server output with repeated lines collapsed: the first two and the last occurrence of every repeated line pattern, plus errors and the final lines. Enough to follow what happened; not enough to see every iteration.",
62
+ tree: "A directory listing reduced to the first few entries of every directory, with the number of omitted entries per directory. Enough to learn the project layout, not enough to find one specific file in a large directory.",
63
+ testlog: "A test run reduced to the failing tests with their assertion and traceback, the short summary and the final counts. Passing tests and decoration are dropped. Enough for reacting to a test run; not enough to see the output of passing tests.",
64
+ sections: "The first line of every section of the output (grep match groups, JSON objects, paragraphs, command markers), line-numbered. The bodies of the sections the agent needs are added in a second step. Enough when only some parts of a long mixed output matter.",
65
+ };
66
+
67
+ export function buildPresendState(
68
+ cfg: Pick<Config, "stateHeadChars">,
69
+ input: { firstUser: string; latestUser: string; agentText: string; toolName: string; args: unknown; isError: boolean; cands: Candidates; totalLines: number; totalChars: number },
70
+ ): PresendState {
71
+ const views: PresendState["views"] = {};
72
+ for (const v of input.cands.views) {
73
+ views[v.kind] = { lines: v.lines, chars: v.chars, preview: truncate(v.text, v.kind === "full" ? Math.min(1500, cfg.stateHeadChars) : 900) };
74
+ }
75
+ return {
76
+ task: { first_user_request: truncate(input.firstUser, 600), latest_user_message: truncate(input.latestUser, 400) },
77
+ agent: { text_before_call: truncate(input.agentText, 600), tool: input.toolName, args: truncate(JSON.stringify(input.args ?? {}), 300) },
78
+ result: { kind: input.cands.kind, is_error: input.isError, total_lines: input.totalLines, total_chars: input.totalChars },
79
+ views,
80
+ };
81
+ }
82
+
83
+ /** Everything the autoresearch loop may vary: prompt texts and view descriptions. Defaults = current best. */
84
+ export interface PromptVariant {
85
+ viewInstructions: string;
86
+ viewDescriptions: Partial<Record<ViewKind, string>>;
87
+ needsFullInstructions: string;
88
+ needsFullTrue: string;
89
+ needsFullFalse: string;
90
+ expandInstructions: string;
91
+ expandTrue: string;
92
+ expandFalse: string;
93
+ /** Second step for command output: per section, will the agent need its contents. */
94
+ sectionInstructions: string;
95
+ sectionTrue: string;
96
+ sectionFalse: string;
97
+ }
98
+
99
+ export const DEFAULT_PROMPTS: PromptVariant = {
100
+ viewInstructions:
101
+ "A coding agent working on `task` just called `agent.tool` with `agent.args` (its reasoning right before the call is `agent.text_before_call`). The output is large. `views` lists candidate presentations of the same output with a preview of each. Which view is the smallest one that still gives the agent everything it needs for its next step? Prefer smaller views only when the agent's purpose is clearly served by them; when in doubt, choose full.",
102
+ viewDescriptions: {},
103
+ needsFullInstructions:
104
+ "Will the agent's next step require the exact, complete text of this output, for example to make an edit whose old text must match, to copy code, or to check details that could be anywhere in it?",
105
+ needsFullTrue: "The agent asked for this to modify it, copy from it, or review it line by line; the task is about the contents of this specific output.",
106
+ needsFullFalse: "The agent is orienting itself, checking structure, looking for where something lives, confirming an outcome, or sampling data.",
107
+ expandInstructions: "The agent working on `task` just read this file (`agent.args`) for the reason in `agent.text_before_call`. Will it need the full body of block `blocks[{i}]` (not just its signature) for its next step?",
108
+ expandTrue: "The task or the agent's stated purpose concerns this block: it will edit it, call it in a specific way, explain its logic, or debug it.",
109
+ expandFalse: "The block is unrelated to the task, or knowing its signature and existence is enough.",
110
+ sectionInstructions: "The agent working on `task` just ran the command `agent.args` for the reason in `agent.text_before_call`. The output is split into sections listed in `blocks`. Will the agent need the contents of section `blocks[{i}]` (not just its first line) for its next step?",
111
+ sectionTrue: "The section holds the result, error, value or match the agent ran the command to see, or something it will quote, compare or act on.",
112
+ sectionFalse: "The section is boilerplate, an unrelated match, setup or progress output, or its first line already tells the agent what it needs.",
113
+ };
114
+
115
+ export function presendQuestions(kinds: ViewKind[], prompts: PromptVariant = DEFAULT_PROMPTS) {
116
+ const criteria: Record<string, string> = {};
117
+ for (const k of kinds) criteria[k] = prompts.viewDescriptions[k] ?? VIEW_DESCRIPTIONS[k];
118
+ return {
119
+ view: { type: "choice" as const, instructions: prompts.viewInstructions, criteria },
120
+ needs_full: { type: "noul" as const, instructions: prompts.needsFullInstructions, criteria: { true: prompts.needsFullTrue, false: prompts.needsFullFalse } },
121
+ };
122
+ }
123
+
124
+ export class JevPresend implements PresendClassifier {
125
+ constructor(private client: TypeSafeClient, private model: string, private prompts: PromptVariant = DEFAULT_PROMPTS) {}
126
+ async expand(state: ExpandState, signal?: AbortSignal): Promise<number[]> {
127
+ const r = await this.client.systemOne({ state: state as never, questions: expandQuestions(state.blocks.length, this.prompts, state.file.kind), model: this.model }, { signal, timeout: 15000 });
128
+ return state.blocks.map((_, i) => (r.answers[`b${i}`] as { noul: number }).noul);
129
+ }
130
+ async choose(state: PresendState, kinds: ViewKind[], signal?: AbortSignal) {
131
+ const r = await this.client.systemOne({ state: state as never, questions: presendQuestions(kinds, this.prompts), model: this.model }, { signal, timeout: 15000 });
132
+ return { choice: r.answers.view.choice as ViewKind, probabilities: r.answers.view.probabilities as Record<string, number>, confidence: r.answers.view.confidence, needsFull: r.answers.needs_full.noul };
133
+ }
134
+ }
135
+
136
+ export class MockPresend implements PresendClassifier {
137
+ async expand(state: ExpandState): Promise<number[]> {
138
+ // deterministic: blocks whose signature mentions a term from the task
139
+ const words = (state.task.first_user_request + " " + state.agent.args).toLowerCase();
140
+ return state.blocks.map((b) => (b.signature.toLowerCase().split(/[^a-z0-9_]+/).some((w) => w.length > 4 && words.includes(w)) ? 0.9 : 0.1));
141
+ }
142
+ constructor(private pick: (state: PresendState, kinds: ViewKind[]) => ViewKind = (s, kinds) => (s.result.kind === "data" && kinds.includes("sample") ? "sample" : kinds.includes("outline") ? "outline" : "full")) {}
143
+ async choose(state: PresendState, kinds: ViewKind[]) {
144
+ const choice = this.pick(state, kinds);
145
+ const probabilities: Record<string, number> = {};
146
+ for (const k of kinds) probabilities[k] = k === choice ? 0.9 : 0.1 / Math.max(1, kinds.length - 1);
147
+ return { choice, probabilities, confidence: 0.9, needsFull: choice === "full" ? 0.9 : 0.1 };
148
+ }
149
+ }
150
+
151
+ /**
152
+ * Turn jev's answers into a view, erring on the side of sending more:
153
+ * full when needsFull is likely, when full itself carries real mass, or when the chosen view is not confident.
154
+ */
155
+ export function decideView(
156
+ answer: { choice: ViewKind; probabilities: Record<string, number>; confidence: number; needsFull: number },
157
+ cands: Candidates,
158
+ cfg: Pick<Config, "presendNeedsFullAbove" | "presendFullMassAbove" | "presendMinConfidence" | "presendCodeNeedsFullAbove" | "presendCommandNeedsFullAbove"> & Partial<Pick<Config, "presendCodePolicy" | "presendCommandPolicy">>,
159
+ ): View {
160
+ const full = cands.views[0];
161
+ if (cands.kind === "command" && cfg.presendCommandPolicy === "sections") {
162
+ // sections-first: when jev would send full but does not think exact full text is needed, send the
163
+ // section headers and let the second step put back the sections it needs (full if that reaches 90 %).
164
+ const sections = cands.views.find((v) => v.kind === "sections");
165
+ if (sections && answer.choice === "full" && answer.needsFull <= cfg.presendCommandNeedsFullAbove) return sections;
166
+ }
167
+ if (cands.kind === "code" && cfg.presendCodePolicy === "outline") {
168
+ // outline-first: structure now, bodies via the expansion step, everything else via recall.
169
+ // Only when the expansion step can run (2+ blocks): an outline nobody can expand was edited from at once.
170
+ const outline = cands.views.find((v) => v.kind === "outline");
171
+ const blocks = (cands as { blocks?: Block[] }).blocks;
172
+ if (outline && blocks && blocks.length >= 2) return outline;
173
+ if (outline && !(blocks && blocks.length >= 2)) return full;
174
+ }
175
+ const needsFullAbove = cands.kind === "code" ? Math.min(cfg.presendNeedsFullAbove, cfg.presendCodeNeedsFullAbove) : cands.kind === "command" ? cfg.presendCommandNeedsFullAbove : cfg.presendNeedsFullAbove;
176
+ if (answer.needsFull > needsFullAbove) return full;
177
+ // For code, "focus" alone is a locating aid; if it wins, upgrade to outline so structure comes along (the second step may expand bodies).
178
+ if (cands.kind === "code" && answer.choice === "focus") { const outline = cands.views.find((v) => v.kind === "outline"); if (outline) return outline; }
179
+ if ((answer.probabilities.full ?? 0) > cfg.presendFullMassAbove) return full;
180
+ if (answer.confidence < cfg.presendMinConfidence) return full;
181
+ return cands.views.find((v) => v.kind === answer.choice) ?? full;
182
+ }
183
+
184
+ /**
185
+ * Second node of the pre-send graph: when a code file was reduced to its outline (or command output to its
186
+ * section headers), ask jev which block bodies the agent will need and put those back. Returns undefined when not applicable.
187
+ */
188
+ export async function expandRelevantBlocks(
189
+ presend: PresendClassifier,
190
+ base: PresendState,
191
+ text: string,
192
+ cands: Candidates,
193
+ chosen: View,
194
+ threshold: number,
195
+ signal?: AbortSignal,
196
+ precomputed?: Block[],
197
+ /** Command output only: send full when no section reaches this probability (0 = headers alone are allowed). */
198
+ floor = 0,
199
+ ): Promise<{ view: View; blocks: Block[]; probs: number[] } | undefined> {
200
+ const isCode = cands.kind === "code" && (chosen.kind === "outline" || chosen.kind === "focus");
201
+ const isCommand = cands.kind === "command" && chosen.kind === "sections";
202
+ if (!isCode && !isCommand) return undefined;
203
+ const blocks = precomputed && precomputed.length >= 2 ? precomputed : isCommand ? splitSections(text) : splitBlocks(text);
204
+ if (blocks.length < 2) return undefined;
205
+ const probs = await presend.expand(buildExpandState(base, blocks, text), signal);
206
+ const expand = new Set<number>();
207
+ probs.forEach((p, i) => { if (p > threshold) expand.add(i); });
208
+ // the import/constants header is small and often edited (new imports): always keep it whole when short
209
+ if (isCode && blocks[0]?.name.startsWith("(header") && blocks[0].to - blocks[0].from < 20) expand.add(0);
210
+ // Command output where every section scores low is "cannot tell", not "nothing needed" (docs read for
211
+ // orientation score flat and low): headers alone would drop what the agent came for, so send it all.
212
+ if (isCommand && Math.max(...probs) < floor) return { view: cands.views[0], blocks, probs };
213
+ const outline = cands.views.find((v) => v.kind === (isCommand ? "sections" : "outline"));
214
+ const view = relevantView(text, cands.kind, blocks, expand, outline?.included);
215
+ if (view.chars >= cands.views[0].chars * 0.9) return { view: cands.views[0], blocks, probs };
216
+ return { view, blocks, probs };
217
+ }
@@ -0,0 +1,90 @@
1
+ /** A deliberately small shell subset, not a general shell parser. Unknown syntax fails closed. */
2
+ export function displayedFiles(command: string): string[] | undefined {
3
+ if (command.length > 16384 || /[<>`$\\()#\r]/.test(command)) return undefined;
4
+ const tokens: { value: string; operator?: boolean }[] = [];
5
+ for (let i = 0; i < command.length;) {
6
+ const c = command[i];
7
+ if (c === " " || c === "\t") { i++; continue; }
8
+ if (";|&\n".includes(c)) {
9
+ let value = c;
10
+ if (command[i + 1] === c && (c === "&" || c === "|")) { value += c; i++; }
11
+ if (value === "&" || value === "||") return undefined;
12
+ tokens.push({ value, operator: true }); i++; continue;
13
+ }
14
+ if (c === "'" || c === '"') {
15
+ const end = command.indexOf(c, i + 1);
16
+ if (end < 0 || (end + 1 < command.length && !/[\s;|&]/.test(command[end + 1]))) return undefined;
17
+ const value = command.slice(i + 1, end);
18
+ // Quoted wildcard/brace paths need literal matching, not shell expansion.
19
+ if (/[{}*?\[\]\n]/.test(value)) return undefined;
20
+ tokens.push({ value }); i = end + 1; continue;
21
+ }
22
+ let end = i;
23
+ while (end < command.length && !/[\s;|&]/.test(command[end])) end++;
24
+ const value = command.slice(i, end);
25
+ if (!value || /['"]/.test(value)) return undefined;
26
+ tokens.push({ value }); i = end;
27
+ }
28
+ const files: string[] = [];
29
+ let stage: string[] = [], filter = false;
30
+ const finish = () => {
31
+ const paths = displayStage(stage);
32
+ if (!paths || (filter ? paths.length !== 0 : paths.length === 0)) return false;
33
+ files.push(...paths);
34
+ stage = [];
35
+ return files.length <= 256;
36
+ };
37
+ for (const token of tokens) {
38
+ if (!token.operator) { stage.push(token.value); continue; }
39
+ if (!stage.length) {
40
+ if (token.value === "\n" && !filter) continue;
41
+ return undefined;
42
+ }
43
+ if (!finish()) return undefined;
44
+ filter = token.value === "|";
45
+ }
46
+ if (stage.length) { if (!finish()) return undefined; }
47
+ else if (filter || tokens.at(-1)?.value === "&&") return undefined;
48
+ return files.length ? files : undefined;
49
+ }
50
+
51
+ /** Only content-preserving cat, line-limited head/tail, and sed -n range-p are recognized. */
52
+ function displayStage(words: string[]): string[] | undefined {
53
+ const [executable, ...args] = words;
54
+ if (!executable || !/^(?:(?:\/usr)?\/bin\/)?(?:cat|head|tail|sed)$/.test(executable)) return undefined;
55
+ const cmd = executable.split("/").pop();
56
+ let i = 0;
57
+ if (cmd === "sed") {
58
+ if (args[i++] !== "-n" || !/^\d+(?:,(?:\d+|\$))?p$/.test(args[i++] ?? "")) return undefined;
59
+ } else if (cmd === "head" || cmd === "tail") {
60
+ if (args[i] === "-n") { i++; if (!/^\d+$/.test(args[i++] ?? "")) return undefined; }
61
+ else if (/^-\d+$/.test(args[i] ?? "")) i++;
62
+ }
63
+ if (args[i] === "--") i++;
64
+ const paths: string[] = [];
65
+ for (const arg of args.slice(i)) {
66
+ if (!arg || arg.startsWith("-") || /[~!]/.test(arg)) return undefined;
67
+ const expanded = expandBraces(arg);
68
+ if (!expanded) return undefined;
69
+ paths.push(...expanded);
70
+ }
71
+ return paths;
72
+ }
73
+
74
+ function expandBraces(word: string): string[] | undefined {
75
+ let pending = [word];
76
+ for (;;) {
77
+ const next: string[] = [];
78
+ let expanded = false;
79
+ for (const item of pending) {
80
+ const m = /^(.*?)\{([^{}]+)\}(.*)$/.exec(item);
81
+ if (!m) { if (/[{}]/.test(item)) return undefined; next.push(item); continue; }
82
+ if (!m[2].includes(",")) return undefined;
83
+ for (const alt of m[2].split(",")) next.push(m[1] + alt + m[3]);
84
+ expanded = true;
85
+ if (next.length > 256) return undefined;
86
+ }
87
+ if (!expanded) return next;
88
+ pending = next;
89
+ }
90
+ }
package/src/text.ts ADDED
@@ -0,0 +1,68 @@
1
+ import type { AgentMessage } from "./pi-types.ts";
2
+
3
+ export function contentText(content: unknown): string {
4
+ if (typeof content === "string") return content;
5
+ if (!Array.isArray(content)) return "";
6
+ const parts: string[] = [];
7
+ for (const block of content) {
8
+ if (!block || typeof block !== "object") continue;
9
+ const b = block as { type?: string; text?: string; thinking?: string; name?: string; arguments?: unknown };
10
+ if (b.type === "text" && typeof b.text === "string") parts.push(b.text);
11
+ }
12
+ return parts.join("\n");
13
+ }
14
+
15
+ export function toolCallsOf(message: AgentMessage): { name: string; arguments: unknown }[] {
16
+ if (message.role !== "assistant") return [];
17
+ const out: { name: string; arguments: unknown }[] = [];
18
+ for (const block of message.content) {
19
+ if (block.type === "toolCall") out.push({ name: block.name, arguments: block.arguments });
20
+ }
21
+ return out;
22
+ }
23
+
24
+ export function head(s: string, n: number): string {
25
+ return s.length <= n ? s : s.slice(0, n);
26
+ }
27
+
28
+ export function tail(s: string, n: number): string {
29
+ return s.length <= n ? "" : s.slice(-n);
30
+ }
31
+
32
+ export function truncate(s: string, n: number): string {
33
+ return s.length <= n ? s : `${s.slice(0, n)}…`;
34
+ }
35
+
36
+ /** Rough token estimate matching pi's heuristic (chars / 4). */
37
+ export function estimateTokensOfText(s: string): number {
38
+ return Math.ceil(s.length / 4);
39
+ }
40
+
41
+ /** Short, stable description of a tool call for stubs and memory pointers. */
42
+ export function describeToolCall(toolName: string, args: unknown, outputChars: number, lines: number): string {
43
+ const a = (args ?? {}) as Record<string, unknown>;
44
+ const pick = (k: string): string | undefined => (typeof a[k] === "string" ? (a[k] as string) : undefined);
45
+ let what = "";
46
+ switch (toolName) {
47
+ case "read":
48
+ what = pick("path") ?? "";
49
+ break;
50
+ case "bash":
51
+ what = truncate((pick("command") ?? "").replace(/\s+/g, " "), 80);
52
+ break;
53
+ case "grep":
54
+ what = `${pick("pattern") ?? ""} in ${pick("path") ?? "."}`;
55
+ break;
56
+ case "find":
57
+ case "ls":
58
+ what = pick("path") ?? pick("pattern") ?? "";
59
+ break;
60
+ case "edit":
61
+ case "write":
62
+ what = pick("path") ?? "";
63
+ break;
64
+ default:
65
+ what = truncate(JSON.stringify(a), 80);
66
+ }
67
+ return `${toolName} ${what}`.trim() + ` (${lines} lines, ${outputChars} chars)`;
68
+ }
@@ -0,0 +1,220 @@
1
+ /**
2
+ * Tree-sitter backed structure for code views: exact top-level blocks (functions, classes,
3
+ * methods, top-level assignments) with signature lines, for the languages that ship in
4
+ * tree-sitter-wasms. Falls back to undefined when the language is unknown or parsing fails,
5
+ * in which case views.ts uses its regex heuristics.
6
+ */
7
+ import { existsSync } from "node:fs";
8
+ import { createRequire } from "node:module";
9
+ import { dirname, extname, join } from "node:path";
10
+ import type { Block } from "./views.ts";
11
+
12
+ const require = createRequire(import.meta.url);
13
+
14
+ /** Grammars shipped by @vscode/tree-sitter-wasm (ABI-compatible with web-tree-sitter 0.27). */
15
+ const LANG_BY_EXT: Record<string, string> = {
16
+ ".js": "javascript", ".mjs": "javascript", ".cjs": "javascript", ".jsx": "javascript",
17
+ ".ts": "typescript", ".mts": "typescript", ".cts": "typescript", ".tsx": "tsx",
18
+ ".py": "python", ".pyx": "python", ".go": "go", ".rs": "rust", ".java": "java", ".rb": "ruby",
19
+ ".c": "c", ".h": "c", ".cc": "cpp", ".cpp": "cpp", ".hpp": "cpp", ".cs": "c-sharp", ".php": "php",
20
+ ".sh": "bash", ".bash": "bash", ".css": "css",
21
+ ".kt": "kotlin", ".kts": "kotlin",
22
+ };
23
+
24
+ /** Extra grammar packages: language → wasm path resolver (the VS Code bundle has no Kotlin). */
25
+ const EXTRA_WASM: Record<string, () => string | undefined> = {
26
+ kotlin: () => {
27
+ try {
28
+ const dir = dirname(require.resolve("@binclusive/tree-sitter-kotlin-wasm/package.json"));
29
+ const { readdirSync } = require("node:fs") as typeof import("node:fs");
30
+ const walk = (d: string): string | undefined => { for (const f of readdirSync(d, { withFileTypes: true })) { const p = join(d, f.name); if (f.isDirectory() && f.name !== "node_modules") { const r = walk(p); if (r) return r; } else if (f.name.endsWith(".wasm")) return p; } return undefined; };
31
+ return walk(dir);
32
+ } catch { return undefined; }
33
+ },
34
+ };
35
+
36
+ /** Node types that count as top-level blocks, per language family. */
37
+ const BLOCK_TYPES = new Set([
38
+ "function_declaration", "function_definition", "generator_function_declaration", "class_declaration", "class_definition",
39
+ "method_definition", "method_declaration", "abstract_class_declaration", "interface_declaration", "type_alias_declaration",
40
+ "enum_declaration", "module", "internal_module", "lexical_declaration", "variable_declaration", "export_statement",
41
+ "decorated_definition", "function_item", "impl_item", "struct_item", "enum_item", "trait_item", "mod_item", "const_item", "static_item",
42
+ "type_item", "func_literal", "method_declaration", "type_declaration", "var_declaration", "const_declaration",
43
+ "class_specifier", "struct_specifier", "namespace_definition", "template_declaration", "preproc_function_def",
44
+ "function_signature", "singleton_method", "module", "class", "method", "object_declaration", "property_declaration",
45
+ "companion_object", "constructor_declaration", "record_declaration", "annotation_type_declaration", "macro_definition", "extern_crate_declaration",
46
+ ]);
47
+ const HEADER_TYPES = new Set(["import_statement", "import_declaration", "import_from_statement", "package_clause", "package_declaration", "use_declaration", "preproc_include", "require_call", "using_directive", "comment", "expression_statement", "attribute_item", "mod_item"]);
48
+
49
+ type TS = typeof import("web-tree-sitter");
50
+ let ts: TS | undefined;
51
+ let inited: Promise<void> | undefined;
52
+ const languages = new Map<string, Promise<import("web-tree-sitter").Language | undefined>>();
53
+
54
+ function wasmDir(): string | undefined {
55
+ try {
56
+ return join(dirname(require.resolve("@vscode/tree-sitter-wasm/package.json")), "wasm");
57
+ } catch {
58
+ return undefined;
59
+ }
60
+ }
61
+
62
+ async function init(): Promise<TS | undefined> {
63
+ if (ts) return ts;
64
+ if (!inited) {
65
+ inited = (async () => {
66
+ try {
67
+ const mod = (await import("web-tree-sitter")) as TS;
68
+ await mod.Parser.init();
69
+ ts = mod;
70
+ } catch {
71
+ ts = undefined;
72
+ }
73
+ })();
74
+ }
75
+ await inited;
76
+ return ts;
77
+ }
78
+
79
+ async function language(name: string) {
80
+ if (!languages.has(name)) {
81
+ languages.set(name, (async () => {
82
+ const mod = await init();
83
+ const dir = wasmDir();
84
+ if (!mod || !dir) return undefined;
85
+ const file = EXTRA_WASM[name]?.() ?? join(dir, `tree-sitter-${name}.wasm`);
86
+ if (!file || !existsSync(file)) return undefined;
87
+ try { return await mod.Language.load(file); } catch { return undefined; }
88
+ })());
89
+ }
90
+ return languages.get(name)!;
91
+ }
92
+
93
+ export function languageForPath(path: string): string | undefined {
94
+ return LANG_BY_EXT[extname(path).toLowerCase()];
95
+ }
96
+
97
+ /**
98
+ * Top-level blocks from the syntax tree: each named child of the root that is a declaration
99
+ * becomes a block spanning its full line range (including a directly preceding comment).
100
+ * Leading imports and other non-block statements are folded into a header block.
101
+ */
102
+ export async function treeSitterBlocks(path: string, text: string, maxBlocks = 48): Promise<Block[] | undefined> {
103
+ const lang = languageForPath(path);
104
+ if (!lang) return undefined;
105
+ const mod = await init();
106
+ const L = await language(lang);
107
+ if (!mod || !L) return undefined;
108
+ const parser = new mod.Parser();
109
+ parser.setLanguage(L);
110
+ const tree = parser.parse(text);
111
+ if (!tree) return undefined;
112
+ const lines = text.split("\n");
113
+ const root = tree.rootNode;
114
+ // Python and Ruby put everything under "module"/"program"; unwrap one level when the root has a single block child.
115
+ let children = root.namedChildren;
116
+ if (children.length === 1 && (children[0].type === "module" || children[0].type === "program")) children = children[0].namedChildren;
117
+ const blocks: Block[] = [];
118
+ let pendingComment: number | undefined;
119
+ let headerEnd = 0;
120
+ for (const c of children) {
121
+ if (!c) continue;
122
+ const startLine = c.startPosition.row;
123
+ const endLine = c.endPosition.row;
124
+ if (c.type === "comment") { if (pendingComment === undefined) pendingComment = startLine; continue; }
125
+ let node = c;
126
+ // export const x = ...; export default class ...; decorated defs
127
+ if ((c.type === "export_statement" || c.type === "decorated_definition") && c.namedChildren.length) {
128
+ const inner = c.namedChildren.find((n) => n && BLOCK_TYPES.has(n.type));
129
+ if (inner) node = inner;
130
+ }
131
+ // Multi-line top-level assignments (config dicts, tables, constants) are blocks too.
132
+ const isBigAssignment = (node.type === "expression_statement" || node.type === "assignment") && endLine - startLine >= 3;
133
+ // one-line declarations (type aliases, Kotlin data classes, Rust consts) are blocks too: they belong in the outline
134
+ const isBlock = BLOCK_TYPES.has(node.type) || isBigAssignment;
135
+ if (!isBlock) { pendingComment = undefined; if (blocks.length === 0) headerEnd = endLine; continue; }
136
+ const from = (pendingComment ?? startLine) + 1;
137
+ pendingComment = undefined;
138
+ const sig = lines[startLine].trim().slice(0, 120);
139
+ // Large classes: expose their methods as blocks so the second step can pick individual bodies.
140
+ const body = node.namedChildren.find((n) => n && (n.type === "class_body" || n.type === "block" || n.type === "declaration_list" || n.type === "field_declaration_list"));
141
+ const methods = body ? body.namedChildren.filter((n) => n && (n.type === "method_definition" || n.type === "function_definition" || n.type === "method_declaration" || n.type === "constructor_declaration" || n.type === "decorated_definition" || n.type === "function_item" || n.type === "function_declaration" || n.type === "companion_object" || n.type === "property_declaration")) : [];
142
+ if (endLine - startLine > 40 && methods.length >= 2) {
143
+ blocks.push({ name: sig, from, to: methods[0]!.startPosition.row });
144
+ for (let k = 0; k < methods.length; k++) {
145
+ const mm = methods[k]!;
146
+ const mEnd = k + 1 < methods.length ? methods[k + 1]!.startPosition.row : endLine + 1;
147
+ blocks.push({ name: `${sig.replace(/[{:]\s*$/, "")} › ${lines[mm.startPosition.row].trim().slice(0, 80)}`, from: mm.startPosition.row + 1, to: mEnd });
148
+ }
149
+ continue;
150
+ }
151
+ blocks.push({ name: sig, from, to: endLine + 1 });
152
+ }
153
+ tree.delete();
154
+ parser.delete();
155
+ if (blocks.length < 2) return blocks.length === 0 ? [] : undefined;
156
+ // fill gaps so the block list partitions the file
157
+ const out: Block[] = [];
158
+ if (blocks[0].from > 1) out.push({ name: "(header: imports, constants)", from: 1, to: blocks[0].from - 1 });
159
+ for (let i = 0; i < blocks.length; i++) {
160
+ const b = { ...blocks[i] };
161
+ const next = blocks[i + 1];
162
+ if (next && next.from > b.to + 1) b.to = next.from - 1;
163
+ if (!next && b.to < lines.length) b.to = lines.length;
164
+ out.push(b);
165
+ }
166
+ void headerEnd;
167
+ return out.slice(0, maxBlocks);
168
+ }
169
+
170
+ const SIGNATURE_TYPES = new Set(["function_declaration", "function_definition", "method_definition", "method_declaration", "constructor_declaration", "function_item", "decorated_definition", "class_declaration", "class_definition", "interface_declaration", "struct_item", "enum_item", "trait_item", "impl_item", "object_declaration", "companion_object", "type_alias_declaration", "type_item", "record_declaration", "enum_declaration", "abstract_class_declaration", "singleton_method", "method", "class", "module", "func_literal", "generator_function_declaration", "lexical_declaration", "property_declaration"]);
171
+
172
+ /** 0-based rows of every declaration signature in the tree, at any nesting depth (methods in small classes, nested functions). */
173
+ async function signatureRows(path: string, text: string): Promise<number[] | undefined> {
174
+ const lang = languageForPath(path);
175
+ if (!lang) return undefined;
176
+ const mod = await init();
177
+ const L = await language(lang);
178
+ if (!mod || !L) return undefined;
179
+ const parser = new mod.Parser();
180
+ parser.setLanguage(L);
181
+ const tree = parser.parse(text);
182
+ if (!tree) return undefined;
183
+ const rows = new Set<number>();
184
+ const walk = (n: import("web-tree-sitter").Node, depth: number) => {
185
+ if (depth > 6) return;
186
+ for (const c of n.namedChildren) {
187
+ if (!c) continue;
188
+ if (SIGNATURE_TYPES.has(c.type)) {
189
+ // property/lexical declarations only when they hold a function (arrow functions, lambdas) or are top-level
190
+ if ((c.type === "lexical_declaration" || c.type === "property_declaration") && depth > 0 && !/=>|lambda|fun\b|function\b/.test(text.split("\n")[c.startPosition.row])) continue;
191
+ rows.add(c.startPosition.row);
192
+ }
193
+ walk(c, depth + 1);
194
+ }
195
+ };
196
+ walk(tree.rootNode, 0);
197
+ tree.delete();
198
+ parser.delete();
199
+ return [...rows].sort((a, b) => a - b);
200
+ }
201
+
202
+ /** Signature lines (block starts) as an outline index list, 0-based. */
203
+ export async function treeSitterOutline(path: string, text: string): Promise<number[] | undefined> {
204
+ const sigs = await signatureRows(path, text);
205
+ if (!sigs) return undefined;
206
+ const blocks = (await treeSitterBlocks(path, text)) ?? [];
207
+ const idx: number[] = [...sigs];
208
+ const lines = text.split("\n");
209
+ if (blocks.length === 0) for (let i = 0; i < Math.min(lines.length, 60); i++) if (/^\s*(import|from|require|use|package|#include|using)\b/.test(lines[i])) idx.push(i);
210
+ for (const b of blocks) {
211
+ if (b.name.startsWith("(header")) {
212
+ // imports plus any one-line declarations (Kotlin data classes, type aliases, constants) that were too short to be blocks
213
+ for (let i = b.from - 1; i < b.to; i++) if (/^\s*(import|from|require|use|package|#include|using)\b/.test(lines[i]) || /^(export\s+)?(const|let|var|val|type|typealias|data class|sealed class|enum class|class|interface|object|fun|def|pub|static|final)\b/.test(lines[i])) idx.push(i);
214
+ continue;
215
+ }
216
+ // the signature line is the first non-comment line of the block
217
+ for (let i = b.from - 1; i < b.to; i++) { if (!/^\s*(\/\/|\/\*|\*|#|"""|''')/.test(lines[i]) && lines[i].trim()) { idx.push(i); break; } }
218
+ }
219
+ return idx;
220
+ }
package/src/types.ts ADDED
@@ -0,0 +1,45 @@
1
+ export type Bucket = "keep" | "trim" | "forget";
2
+
3
+ export interface Probabilities {
4
+ needed: number;
5
+ outcomeOnly: number;
6
+ durable: number;
7
+ }
8
+
9
+ export interface Decision {
10
+ /** toolCallId of the tool result this decision is about. */
11
+ id: string;
12
+ toolName: string;
13
+ bucket: Bucket;
14
+ durable: boolean;
15
+ p: Probabilities;
16
+ /** One-line description used in the stub, fixed at decision time so the stub never changes. */
17
+ summary: string;
18
+ tokensBefore: number;
19
+ decidedAt: number;
20
+ /** "pending" until first applied in a context call; then frozen forever. */
21
+ status: "pending" | "applied";
22
+ appliedAtCall?: number;
23
+ /** Why the decision was applied (rolling, cold-cache, compaction, forced). */
24
+ appliedReason?: string;
25
+ }
26
+
27
+ export interface DurableNote {
28
+ source: "user" | "agent" | "tool";
29
+ text: string;
30
+ p: number;
31
+ at: number;
32
+ }
33
+
34
+ export interface CallStats {
35
+ call: number;
36
+ at: number;
37
+ messages: number;
38
+ tokensOriginal: number;
39
+ tokensSent: number;
40
+ tokensPruned: number;
41
+ appliedNow: number;
42
+ frozen: number;
43
+ pendingHeld: number;
44
+ coldCache: boolean;
45
+ }