@davesheffer/hunch 1.41.6 → 1.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli/index.js +402 -154
- package/dist/constitution/experiment.d.ts +3 -3
- package/dist/constitution/g3.d.ts +1 -1
- package/dist/core/footprint.d.ts +15 -0
- package/dist/core/footprint.js +167 -0
- package/dist/core/groundingLag.d.ts +15 -0
- package/dist/core/groundingLag.js +27 -0
- package/dist/core/hookText.d.ts +4 -0
- package/dist/core/hookText.js +8 -0
- package/dist/core/pipeline.d.ts +28 -0
- package/dist/core/pipeline.js +50 -0
- package/dist/core/shellwrites.d.ts +7 -0
- package/dist/core/shellwrites.js +131 -0
- package/dist/core/siblingfix.d.ts +124 -0
- package/dist/core/siblingfix.js +814 -0
- package/dist/core/taskReportHook.d.ts +1 -1
- package/dist/core/taskReportHook.js +19 -3
- package/dist/extractors/git.d.ts +31 -0
- package/dist/extractors/git.js +180 -1
- package/dist/extractors/nativeTreeSitter.d.ts +2 -1
- package/dist/extractors/nativeTreeSitter.js +128 -30
- package/dist/integrations/claudemd.d.ts +20 -2
- package/dist/integrations/claudemd.js +79 -53
- package/dist/integrations/providers.d.ts +9 -5
- package/dist/integrations/providers.js +34 -22
- package/dist/integrations/team.d.ts +24 -4
- package/dist/integrations/team.js +154 -16
- package/dist/integrations/worktree.d.ts +3 -2
- package/dist/integrations/worktree.js +7 -4
- package/dist/mcp/server.d.ts +489 -0
- package/dist/mcp/server.js +208 -62
- package/dist/mcp/taskReportTools.js +12 -9
- package/dist/mcp/toolset.d.ts +11 -1
- package/dist/mcp/toolset.js +28 -8
- package/dist/store/hunchStore.d.ts +25 -1
- package/dist/store/hunchStore.js +146 -21
- package/dist/store/jsonStore.js +25 -4
- package/package.json +1 -1
- package/server.json +2 -2
|
@@ -59,8 +59,8 @@ export declare const ExperimentCaseBankSchema: z.ZodObject<{
|
|
|
59
59
|
id: z.ZodString;
|
|
60
60
|
content_hash: z.ZodString;
|
|
61
61
|
experiment: z.ZodEnum<{
|
|
62
|
-
"EXP-03": "EXP-03";
|
|
63
62
|
"EXP-01": "EXP-01";
|
|
63
|
+
"EXP-03": "EXP-03";
|
|
64
64
|
}>;
|
|
65
65
|
preregistration_id: z.ZodString;
|
|
66
66
|
preregistration_hash: z.ZodString;
|
|
@@ -137,8 +137,8 @@ export declare const ExperimentRunSchema: z.ZodObject<{
|
|
|
137
137
|
id: z.ZodString;
|
|
138
138
|
content_hash: z.ZodString;
|
|
139
139
|
experiment: z.ZodEnum<{
|
|
140
|
-
"EXP-03": "EXP-03";
|
|
141
140
|
"EXP-01": "EXP-01";
|
|
141
|
+
"EXP-03": "EXP-03";
|
|
142
142
|
}>;
|
|
143
143
|
preregistration_id: z.ZodString;
|
|
144
144
|
preregistration_hash: z.ZodString;
|
|
@@ -210,8 +210,8 @@ export declare const ExperimentOutcomeSchema: z.ZodObject<{
|
|
|
210
210
|
run_id: z.ZodString;
|
|
211
211
|
assignment_id: z.ZodString;
|
|
212
212
|
experiment: z.ZodEnum<{
|
|
213
|
-
"EXP-03": "EXP-03";
|
|
214
213
|
"EXP-01": "EXP-01";
|
|
214
|
+
"EXP-03": "EXP-03";
|
|
215
215
|
}>;
|
|
216
216
|
arm: z.ZodString;
|
|
217
217
|
status: z.ZodEnum<{
|
|
@@ -42,8 +42,8 @@ export declare const ExperimentPreregistrationSchema: z.ZodObject<{
|
|
|
42
42
|
id: z.ZodString;
|
|
43
43
|
content_hash: z.ZodString;
|
|
44
44
|
experiment: z.ZodEnum<{
|
|
45
|
-
"EXP-03": "EXP-03";
|
|
46
45
|
"EXP-01": "EXP-01";
|
|
46
|
+
"EXP-03": "EXP-03";
|
|
47
47
|
}>;
|
|
48
48
|
revision: z.ZodNumber;
|
|
49
49
|
hypothesis: z.ZodString;
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
export interface FootprintSurface {
|
|
2
|
+
id: string;
|
|
3
|
+
chars: number;
|
|
4
|
+
est_tokens: number;
|
|
5
|
+
detail?: Record<string, number>;
|
|
6
|
+
}
|
|
7
|
+
export interface FootprintReport {
|
|
8
|
+
schema: "hunch.footprint/1";
|
|
9
|
+
estimate: "chars/4";
|
|
10
|
+
surfaces: FootprintSurface[];
|
|
11
|
+
unmeasured: string[];
|
|
12
|
+
}
|
|
13
|
+
export declare function measureFootprint(root: string, opts?: {
|
|
14
|
+
target?: string;
|
|
15
|
+
}): Promise<FootprintReport>;
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
// Context footprint (#372): how much text Hunch injects into an agent's
|
|
2
|
+
// context, measured deterministically from the same code paths the product
|
|
3
|
+
// serves. Tokens are an estimate (characters / 4), not a tokenizer.
|
|
4
|
+
import { execFileSync } from "node:child_process";
|
|
5
|
+
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
|
6
|
+
import { tmpdir } from "node:os";
|
|
7
|
+
import { join } from "node:path";
|
|
8
|
+
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
|
|
9
|
+
import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
|
|
10
|
+
import { buildServer } from "../mcp/server.js";
|
|
11
|
+
import { renderHunchSection } from "../integrations/claudemd.js";
|
|
12
|
+
import { HunchStore } from "../store/hunchStore.js";
|
|
13
|
+
import { hunchPaths } from "./paths.js";
|
|
14
|
+
import { PIPELINE_LOOP } from "./pipeline.js";
|
|
15
|
+
import { HOOK_REMINDER } from "./hookText.js";
|
|
16
|
+
import { taskInstruction } from "./taskReportHook.js";
|
|
17
|
+
const CONTEXT_BUDGET = 1500;
|
|
18
|
+
const estTokens = (chars) => Math.ceil(chars / 4);
|
|
19
|
+
const surface = (id, chars, detail) => ({ id, chars, est_tokens: estTokens(chars), ...(detail ? { detail } : {}) });
|
|
20
|
+
const jsonChars = (v) => (v === undefined ? 0 : JSON.stringify(v).length);
|
|
21
|
+
/** A host shows one channel of a tool result: Claude Code shows only
|
|
22
|
+
* structuredContent, text-only hosts only content. The larger one is the cost. */
|
|
23
|
+
function hostVisible(id, result) {
|
|
24
|
+
const content = jsonChars(result.content), structured = jsonChars(result.structuredContent);
|
|
25
|
+
return surface(id, Math.max(content, structured), { content_chars: content, structured_chars: structured });
|
|
26
|
+
}
|
|
27
|
+
/** The hunch_task start and finish results for a task with one delivered lesson.
|
|
28
|
+
* Driven in a throwaway store, never `root`: a task writes a ledger row, and
|
|
29
|
+
* `taskRecords: false` keeps finish from writing or committing a graph record. */
|
|
30
|
+
async function measureTaskLifecycle() {
|
|
31
|
+
const dir = mkdtempSync(join(tmpdir(), "hunch-footprint-"));
|
|
32
|
+
try {
|
|
33
|
+
// A real task runs in a git repo; the source snapshot reads git.
|
|
34
|
+
try {
|
|
35
|
+
execFileSync("git", ["init", "-q", dir], { stdio: "ignore" });
|
|
36
|
+
}
|
|
37
|
+
catch { /* measured without git */ }
|
|
38
|
+
const store = new HunchStore(hunchPaths(dir));
|
|
39
|
+
try {
|
|
40
|
+
store.json.ensureDirs();
|
|
41
|
+
writeFileSync(join(dir, ".hunch", "local.json"), JSON.stringify({ taskRecords: false, autoCommit: false }));
|
|
42
|
+
store.json.put("constraints", {
|
|
43
|
+
id: "con_footprint_sample", type: "architecture", statement: "Tool results stay machine-readable.",
|
|
44
|
+
scope: ["src/sample.ts"], severity: "blocking", enforcement: "advisory_v1", match: null, forbids: null,
|
|
45
|
+
rationale: "Orchestrators must not parse prose.", source_decision: null, violations: [], status: "active",
|
|
46
|
+
valid_from: "2026-01-01T00:00:00.000Z", valid_to: null,
|
|
47
|
+
provenance: { source: "human_confirmed", confidence: 1, evidence: [] },
|
|
48
|
+
});
|
|
49
|
+
store.reindex();
|
|
50
|
+
}
|
|
51
|
+
finally {
|
|
52
|
+
store.close();
|
|
53
|
+
}
|
|
54
|
+
const server = buildServer(dir);
|
|
55
|
+
const [ct, st] = InMemoryTransport.createLinkedPair();
|
|
56
|
+
const client = new Client({ name: "hunch-footprint", version: "1" });
|
|
57
|
+
await Promise.all([server.connect(st), client.connect(ct)]);
|
|
58
|
+
try {
|
|
59
|
+
const start = await client.callTool({ name: "hunch_task", arguments: { action: "start", title: "Assistant task" } });
|
|
60
|
+
const taskId = start.structuredContent?.task?.task_id;
|
|
61
|
+
if (start.isError || !taskId)
|
|
62
|
+
throw new Error("hunch_task start failed");
|
|
63
|
+
await client.callTool({ name: "hunch_context", arguments: { target: "src/sample.ts", task_id: taskId } });
|
|
64
|
+
const finish = await client.callTool({ name: "hunch_task", arguments: { action: "finish", task_id: taskId } });
|
|
65
|
+
if (finish.isError)
|
|
66
|
+
throw new Error("hunch_task finish failed");
|
|
67
|
+
return [hostVisible("mcp.hunch_task.start", start), hostVisible("mcp.hunch_task.finish", finish)];
|
|
68
|
+
}
|
|
69
|
+
finally {
|
|
70
|
+
await client.close();
|
|
71
|
+
await server.close();
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
finally {
|
|
75
|
+
rmSync(dir, { recursive: true, force: true });
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
/** Surfaces built inline from live host/session state; not measurable in-process. */
|
|
79
|
+
const UNMEASURED = [
|
|
80
|
+
"hook.session.orientation — SessionStart text is built inline in the `hook` command action (src/cli/index.ts) from live session state",
|
|
81
|
+
"hook.pre_edit.grounding — PreToolUse grounding is built per edited file and event",
|
|
82
|
+
];
|
|
83
|
+
/** The file most decisions cite — a FILE target, so the brief carries per-record
|
|
84
|
+
* lines and omissions the way a pre-edit call does. "src" when no decision names one. */
|
|
85
|
+
function busiestFile(store) {
|
|
86
|
+
const counts = new Map();
|
|
87
|
+
for (const d of store.advisoryRecs("decisions")) {
|
|
88
|
+
for (const f of new Set(d.related_files))
|
|
89
|
+
if (!f.includes("*"))
|
|
90
|
+
counts.set(f, (counts.get(f) ?? 0) + 1);
|
|
91
|
+
}
|
|
92
|
+
return [...counts].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))[0]?.[0] ?? "src";
|
|
93
|
+
}
|
|
94
|
+
export async function measureFootprint(root, opts = {}) {
|
|
95
|
+
const surfaces = [];
|
|
96
|
+
let target = opts.target;
|
|
97
|
+
if (!target) {
|
|
98
|
+
const store = new HunchStore(hunchPaths(root));
|
|
99
|
+
try {
|
|
100
|
+
target = busiestFile(store);
|
|
101
|
+
}
|
|
102
|
+
finally {
|
|
103
|
+
store.close();
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
const server = buildServer(root);
|
|
107
|
+
const [ct, st] = InMemoryTransport.createLinkedPair();
|
|
108
|
+
const client = new Client({ name: "hunch-footprint", version: "1" });
|
|
109
|
+
await Promise.all([server.connect(st), client.connect(ct)]);
|
|
110
|
+
try {
|
|
111
|
+
const tools = (await client.listTools()).tools;
|
|
112
|
+
let outputChars = 0, inputChars = 0, descriptionChars = 0, largest = 0;
|
|
113
|
+
for (const t of tools) {
|
|
114
|
+
outputChars += jsonChars(t.outputSchema);
|
|
115
|
+
inputChars += jsonChars(t.inputSchema);
|
|
116
|
+
descriptionChars += (t.description ?? "").length;
|
|
117
|
+
largest = Math.max(largest, jsonChars(t));
|
|
118
|
+
}
|
|
119
|
+
surfaces.push(surface("mcp.tools_list", jsonChars(tools), {
|
|
120
|
+
output_schema_chars: outputChars,
|
|
121
|
+
input_schema_chars: inputChars,
|
|
122
|
+
description_chars: descriptionChars,
|
|
123
|
+
tools: tools.length,
|
|
124
|
+
largest_tool_chars: largest,
|
|
125
|
+
}));
|
|
126
|
+
// What hosts that drop outputSchema actually pay.
|
|
127
|
+
const core = tools.map(t => ({ name: t.name, description: t.description, inputSchema: t.inputSchema }));
|
|
128
|
+
surfaces.push(surface("mcp.tools_list.core", jsonChars(core)));
|
|
129
|
+
const result = await client.callTool({ name: "hunch_context", arguments: { target, budget_tokens: CONTEXT_BUDGET } });
|
|
130
|
+
// A host shows the model one channel — structuredContent when it reads it,
|
|
131
|
+
// content otherwise — so the cost it pays is the larger, not the sum.
|
|
132
|
+
const contentChars = jsonChars(result.content);
|
|
133
|
+
const structuredChars = jsonChars(result.structuredContent);
|
|
134
|
+
const chars = Math.max(contentChars, structuredChars);
|
|
135
|
+
surfaces.push(surface("mcp.hunch_context", chars, {
|
|
136
|
+
content_chars: contentChars,
|
|
137
|
+
structured_chars: structuredChars,
|
|
138
|
+
budget_tokens: CONTEXT_BUDGET,
|
|
139
|
+
// est_tokens / budget, ×100 as an integer percentage.
|
|
140
|
+
budget_ratio_pct: Math.round((estTokens(chars) / CONTEXT_BUDGET) * 100),
|
|
141
|
+
}));
|
|
142
|
+
}
|
|
143
|
+
finally {
|
|
144
|
+
await client.close();
|
|
145
|
+
await server.close();
|
|
146
|
+
}
|
|
147
|
+
surfaces.push(...await measureTaskLifecycle());
|
|
148
|
+
const store = new HunchStore(hunchPaths(root));
|
|
149
|
+
try {
|
|
150
|
+
surfaces.push(surface("grounding.block", renderHunchSection(store, root).length));
|
|
151
|
+
}
|
|
152
|
+
finally {
|
|
153
|
+
store.close();
|
|
154
|
+
}
|
|
155
|
+
surfaces.push(surface("hook.session.pipeline_loop", PIPELINE_LOOP.length));
|
|
156
|
+
surfaces.push(surface("hook.prompt.reminder", HOOK_REMINDER.length));
|
|
157
|
+
// Every prompt gets a task instruction: full once per session, compact after.
|
|
158
|
+
// Measured for a host whose stop hook closes the task, with a fixed installed
|
|
159
|
+
// launcher and cwd so the number does not move with the checkout path.
|
|
160
|
+
const sampleTask = { task_id: "htask_" + "0".repeat(24), title: "Assistant task" };
|
|
161
|
+
const sampleCwd = JSON.stringify("/home/user/repo");
|
|
162
|
+
const sampleLauncher = () => ({ shell: "node /home/user/repo/node_modules/@davesheffer/hunch/dist/cli/index.js" });
|
|
163
|
+
surfaces.push(surface("hook.prompt.task_instruction", taskInstruction(sampleTask, sampleCwd, "claude", sampleLauncher).length));
|
|
164
|
+
surfaces.push(surface("hook.prompt.task_instruction.compact", taskInstruction(sampleTask, sampleCwd, "claude", sampleLauncher, "compact").length));
|
|
165
|
+
return { schema: "hunch.footprint/1", estimate: "chars/4", surfaces, unmeasured: [...UNMEASURED] };
|
|
166
|
+
}
|
|
167
|
+
//# sourceMappingURL=footprint.js.map
|
|
@@ -25,6 +25,9 @@
|
|
|
25
25
|
* run ahead (missing record). Open findings move both ways (a finding resolved on one
|
|
26
26
|
* branch, a finding recorded on another), so a differing findings count alone is lag.
|
|
27
27
|
*/
|
|
28
|
+
/** The block's prose template version (claudemd.ts GROUNDING_TEMPLATE). A block
|
|
29
|
+
* without the stamp is template 1. */
|
|
30
|
+
export declare function groundingTemplate(block: string): number;
|
|
28
31
|
export interface GroundingCounts {
|
|
29
32
|
decisions: number;
|
|
30
33
|
bugs: number;
|
|
@@ -71,11 +74,23 @@ export type GroundingFreshness =
|
|
|
71
74
|
committed: GroundingCounts;
|
|
72
75
|
generated: GroundingCounts;
|
|
73
76
|
ahead: string[];
|
|
77
|
+
newerTemplate?: {
|
|
78
|
+
committed: number;
|
|
79
|
+
renderer: number;
|
|
80
|
+
};
|
|
74
81
|
}
|
|
75
82
|
/** The block differs outside the counts sentence (or a counts sentence is missing). */
|
|
76
83
|
| {
|
|
77
84
|
kind: "diverged";
|
|
78
85
|
reason: string;
|
|
86
|
+
}
|
|
87
|
+
/** The committed block was written by a newer renderer template than this
|
|
88
|
+
* version's: its prose is preserved (preserveNewerTemplate), never a failure. */
|
|
89
|
+
| {
|
|
90
|
+
kind: "newer";
|
|
91
|
+
committedTemplate: number;
|
|
92
|
+
rendererTemplate: number;
|
|
93
|
+
countsReadable: boolean;
|
|
79
94
|
};
|
|
80
95
|
/** Classify a committed managed block against the one the graph generates NOW.
|
|
81
96
|
* Both inputs are block CONTENT (markers stripped, trimmed). */
|
|
@@ -25,6 +25,13 @@
|
|
|
25
25
|
* run ahead (missing record). Open findings move both ways (a finding resolved on one
|
|
26
26
|
* branch, a finding recorded on another), so a differing findings count alone is lag.
|
|
27
27
|
*/
|
|
28
|
+
const TEMPLATE_RE = /<!-- hunch:template (\d+) -->/;
|
|
29
|
+
/** The block's prose template version (claudemd.ts GROUNDING_TEMPLATE). A block
|
|
30
|
+
* without the stamp is template 1. */
|
|
31
|
+
export function groundingTemplate(block) {
|
|
32
|
+
const m = TEMPLATE_RE.exec(block);
|
|
33
|
+
return m ? Number(m[1]) : 1;
|
|
34
|
+
}
|
|
28
35
|
const COUNTS_RE = /\*\*(\d+) decisions?, (\d+) bugs?, (\d+) constraints?, (\d+) components?, (\d+) polic(?:y|ies)(?:, (\d+) open findings?)?\*\*/;
|
|
29
36
|
/** Record kinds whose committed count may only ever lag behind the store. */
|
|
30
37
|
export const APPEND_ONLY_COUNT_KINDS = ["decisions", "bugs", "constraints", "components", "policies"];
|
|
@@ -64,8 +71,19 @@ export function parseGroundingCounts(block) {
|
|
|
64
71
|
export function classifyGroundingBlock(committed, generated) {
|
|
65
72
|
if (committed === generated)
|
|
66
73
|
return { kind: "fresh" };
|
|
74
|
+
const committedTemplate = groundingTemplate(committed);
|
|
75
|
+
const rendererTemplate = groundingTemplate(generated);
|
|
67
76
|
const c = parseGroundingCounts(committed);
|
|
68
77
|
const g = parseGroundingCounts(generated);
|
|
78
|
+
if (committedTemplate > rendererTemplate) {
|
|
79
|
+
// A newer template may reword the prose, but a count AHEAD of the store still
|
|
80
|
+
// means the doc knows a record this repository does not carry.
|
|
81
|
+
const ahead = c && g ? APPEND_ONLY_COUNT_KINDS.filter((k) => c.counts[k] > g.counts[k]) : [];
|
|
82
|
+
if (c && g && ahead.length) {
|
|
83
|
+
return { kind: "ahead", committed: c.counts, generated: g.counts, ahead, newerTemplate: { committed: committedTemplate, renderer: rendererTemplate } };
|
|
84
|
+
}
|
|
85
|
+
return { kind: "newer", committedTemplate, rendererTemplate, countsReadable: c !== null };
|
|
86
|
+
}
|
|
69
87
|
if (!c)
|
|
70
88
|
return { kind: "diverged", reason: "the committed block carries no record-counts sentence" };
|
|
71
89
|
if (!g)
|
|
@@ -88,9 +106,18 @@ export function describeGroundingFreshness(doc, verdict) {
|
|
|
88
106
|
case "lagging":
|
|
89
107
|
return `${doc}: counts lag the store (${delta(verdict.committed, verdict.generated, verdict.behind)}) — records merged in behind the doc; heals on the next capture or \`hunch grounding --refresh\``;
|
|
90
108
|
case "ahead":
|
|
109
|
+
// A newer Hunch may have written records this version skips as unreadable, and a
|
|
110
|
+
// plain refresh keeps a newer block's prose: only an upgrade or --force settles it.
|
|
111
|
+
if (verdict.newerTemplate) {
|
|
112
|
+
return `${doc}: counts run AHEAD of the store (${delta(verdict.committed, verdict.generated, verdict.ahead)}) in a block written by a newer Hunch (template ${verdict.newerTemplate.committed} > ${verdict.newerTemplate.renderer}) — this version may skip records it cannot read, or the record was removed; upgrade Hunch, or run \`hunch grounding --refresh --force\` and commit`;
|
|
113
|
+
}
|
|
91
114
|
return `${doc}: counts run AHEAD of the store (${delta(verdict.committed, verdict.generated, verdict.ahead)}) — the doc counted a record this repository does not carry; commit the missing .hunch/ record or regenerate`;
|
|
92
115
|
case "diverged":
|
|
93
116
|
return `${doc}: stale — ${verdict.reason}; regenerate with \`hunch grounding --refresh\` and commit`;
|
|
117
|
+
case "newer":
|
|
118
|
+
return `${doc}: written by a newer Hunch (template ${verdict.committedTemplate} > ${verdict.rendererTemplate}); ${verdict.countsReadable
|
|
119
|
+
? "its prose is kept and a refresh updates only the counts"
|
|
120
|
+
: "this version cannot read its counts sentence, so a refresh leaves the block as written"}. Upgrade Hunch, or run \`hunch grounding --refresh --force\` to re-render with this version`;
|
|
94
121
|
}
|
|
95
122
|
}
|
|
96
123
|
//# sourceMappingURL=groundingLag.js.map
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/** The UserPromptSubmit reminder `hunch hook` injects once per session (and with
|
|
2
|
+
* every correction). Lives in
|
|
3
|
+
* core so `hunch footprint` measures the exact text the hook sends. */
|
|
4
|
+
export const HOOK_REMINDER = "Hunch (engineering memory) is available for this repo. Before editing, call " +
|
|
5
|
+
"hunch_check_constraints(scope) for do-not-break invariants and hunch_why(target) " +
|
|
6
|
+
"for the rationale; use hunch_get_dependents for blast radius and hunch_bug_lineage " +
|
|
7
|
+
"for prior root causes. After a non-trivial choice, record it with hunch_record_decision.";
|
|
8
|
+
//# sourceMappingURL=hookText.js.map
|
package/dist/core/pipeline.d.ts
CHANGED
|
@@ -150,6 +150,23 @@ export interface PipelineState {
|
|
|
150
150
|
/** Activity index and count for bounded mid-flight reminders. */
|
|
151
151
|
proofReminderActivity: number;
|
|
152
152
|
proofReminders: number;
|
|
153
|
+
/** Sibling-fix lessons delivered this session (see core/siblingfix.ts). */
|
|
154
|
+
lessons: PendingLesson[];
|
|
155
|
+
}
|
|
156
|
+
/** A delivered lesson the agent has not visibly acted on yet. `hash` is the
|
|
157
|
+
* target function's body at delivery; a different body later means the agent
|
|
158
|
+
* touched it, so the lesson is no longer pending. */
|
|
159
|
+
export interface PendingLesson {
|
|
160
|
+
id: string;
|
|
161
|
+
file: string;
|
|
162
|
+
symbol: string;
|
|
163
|
+
sibling: string;
|
|
164
|
+
siblingFile: string;
|
|
165
|
+
/** "<sha8> <subject>" of the change the sibling received. */
|
|
166
|
+
change: string;
|
|
167
|
+
callers: string[];
|
|
168
|
+
hash: string | null;
|
|
169
|
+
reminded: boolean;
|
|
153
170
|
}
|
|
154
171
|
export declare const emptyState: () => PipelineState;
|
|
155
172
|
/** Validate untrusted MCP/env episode data. Invalid entries are ignored so the
|
|
@@ -273,6 +290,17 @@ export declare function proofCheckpoint(before: PipelineState, after: PipelineSt
|
|
|
273
290
|
export declare const PIPELINE_LOOP: string;
|
|
274
291
|
export declare const UNVERIFIED_NAG = "Hunch pipeline: earlier product edits are still UNVERIFIED \u2014 run the relevant test/build/typecheck before claiming anything about them.";
|
|
275
292
|
export declare function unverifiedNag(state: PipelineState): string;
|
|
293
|
+
/** Record lessons the grounding just delivered (dedup by id; first delivery keeps its baseline). */
|
|
294
|
+
export declare function onLessonsDelivered(state: PipelineState, lessons: readonly Omit<PendingLesson, "reminded">[]): PipelineState;
|
|
295
|
+
/** The one follow-up an advisory lesson gets. It fires when the agent runs a
|
|
296
|
+
* check (test/build/typecheck) while a delivered lesson's function is still
|
|
297
|
+
* untouched: the moment the agent thinks it is done, which is when an ignored
|
|
298
|
+
* lesson would otherwise ship. Once per lesson, never a block. `hashOf`
|
|
299
|
+
* returns the function's current body hash (null: cannot tell, stay silent). */
|
|
300
|
+
export declare function lessonReminder(state: PipelineState, command: string, hashOf: (lesson: PendingLesson) => string | null): {
|
|
301
|
+
state: PipelineState;
|
|
302
|
+
reminder: string;
|
|
303
|
+
};
|
|
276
304
|
/** Stop-gate verdict. Blocks only at firm/strict, only with unverified product
|
|
277
305
|
* edits, and at most twice per turn. */
|
|
278
306
|
export declare function stopVerdict(state: PipelineState, firmness: Firmness): {
|
package/dist/core/pipeline.js
CHANGED
|
@@ -63,6 +63,7 @@ export const emptyState = () => ({
|
|
|
63
63
|
proofActivity: 0,
|
|
64
64
|
proofReminderActivity: 0,
|
|
65
65
|
proofReminders: 0,
|
|
66
|
+
lessons: [],
|
|
66
67
|
});
|
|
67
68
|
const MAX_OBLIGATIONS = 12;
|
|
68
69
|
const MAX_ALTERNATIVES = 6;
|
|
@@ -1033,6 +1034,51 @@ export function unverifiedNag(state) {
|
|
|
1033
1034
|
return generic;
|
|
1034
1035
|
return `${generic} Controller obligations still pending: ${pending.slice(0, 4).map((item) => `[${item.category}] ${item.description}`).join("; ")}.`;
|
|
1035
1036
|
}
|
|
1037
|
+
const MAX_LESSONS = 6;
|
|
1038
|
+
/** Any profile's check shape: the reminder fires when the agent starts proving
|
|
1039
|
+
* its work, whatever the domain. */
|
|
1040
|
+
const CHECK_SHAPE = new RegExp([...Object.values(DEFAULT_PROFILES).map((p) => p.verify.source), "node (--test|-e\\b)"].join("|"), "i");
|
|
1041
|
+
/** Record lessons the grounding just delivered (dedup by id; first delivery keeps its baseline). */
|
|
1042
|
+
export function onLessonsDelivered(state, lessons) {
|
|
1043
|
+
const known = new Set(state.lessons.map((l) => l.id));
|
|
1044
|
+
const added = lessons.filter((l) => !known.has(l.id)).map((l) => ({ ...l, reminded: false }));
|
|
1045
|
+
return added.length ? { ...state, lessons: [...state.lessons, ...added].slice(-MAX_LESSONS) } : state;
|
|
1046
|
+
}
|
|
1047
|
+
/** The one follow-up an advisory lesson gets. It fires when the agent runs a
|
|
1048
|
+
* check (test/build/typecheck) while a delivered lesson's function is still
|
|
1049
|
+
* untouched: the moment the agent thinks it is done, which is when an ignored
|
|
1050
|
+
* lesson would otherwise ship. Once per lesson, never a block. `hashOf`
|
|
1051
|
+
* returns the function's current body hash (null: cannot tell, stay silent). */
|
|
1052
|
+
export function lessonReminder(state, command, hashOf) {
|
|
1053
|
+
if (!CHECK_SHAPE.test(command) || !state.lessons.some((l) => !l.reminded))
|
|
1054
|
+
return { state, reminder: "" };
|
|
1055
|
+
const due = [];
|
|
1056
|
+
const lessons = state.lessons.map((l) => {
|
|
1057
|
+
if (l.reminded)
|
|
1058
|
+
return l;
|
|
1059
|
+
const now = hashOf(l);
|
|
1060
|
+
if (now === null || l.hash === null)
|
|
1061
|
+
return l;
|
|
1062
|
+
if (now !== l.hash)
|
|
1063
|
+
return { ...l, reminded: true };
|
|
1064
|
+
due.push(l);
|
|
1065
|
+
return { ...l, reminded: true };
|
|
1066
|
+
});
|
|
1067
|
+
if (!due.length)
|
|
1068
|
+
return { state: { ...state, lessons }, reminder: "" };
|
|
1069
|
+
const lines = due.map((l) => {
|
|
1070
|
+
const via = l.callers.length ? ` ${l.callers.slice(0, 3).map((c) => `\`${c}\``).join(", ")} call${l.callers.length === 1 ? "s" : ""} it, so your change runs through the copy that lacks the fix.` : "";
|
|
1071
|
+
return `- \`${l.symbol}\` (${l.file}) is unchanged since Hunch showed you the change its same-shaped sibling \`${l.sibling}\` (${l.siblingFile}) received: ${l.change}.${via}`;
|
|
1072
|
+
});
|
|
1073
|
+
return {
|
|
1074
|
+
state: { ...state, lessons },
|
|
1075
|
+
reminder: [
|
|
1076
|
+
"Hunch — before you finish: a lesson delivered earlier is still open.",
|
|
1077
|
+
...lines,
|
|
1078
|
+
"The checks you are running were written before this lesson; unless one feeds this function the input the sibling's change handles, they do not show it is covered. Carry the change with a test. It does not apply only if that input cannot reach the function or is already handled another way; then say which in your final answer. \"Pre-existing\" or \"outside this task\" does not count: your change runs through this copy, so leaving it ships the gap again inside your change.",
|
|
1079
|
+
].join("\n"),
|
|
1080
|
+
};
|
|
1081
|
+
}
|
|
1036
1082
|
/** Stop-gate verdict. Blocks only at firm/strict, only with unverified product
|
|
1037
1083
|
* edits, and at most twice per turn. */
|
|
1038
1084
|
export function stopVerdict(state, firmness) {
|
|
@@ -1071,6 +1117,10 @@ export function loadPipelineState(sessionId) {
|
|
|
1071
1117
|
state.proofReminderActivity = Number.isSafeInteger(raw.proofReminderActivity) && raw.proofReminderActivity >= 0 ? raw.proofReminderActivity : 0;
|
|
1072
1118
|
state.proofReminders = Number.isSafeInteger(raw.proofReminders) && raw.proofReminders >= 0 ? raw.proofReminders : 0;
|
|
1073
1119
|
state.probeBlocks = Number.isSafeInteger(raw.probeBlocks) && raw.probeBlocks >= 0 ? raw.probeBlocks : 0;
|
|
1120
|
+
state.lessons = (Array.isArray(raw.lessons) ? raw.lessons : []).filter((l) => !!l && typeof l === "object"
|
|
1121
|
+
&& typeof l.id === "string" && typeof l.file === "string" && typeof l.symbol === "string" && typeof l.sibling === "string"
|
|
1122
|
+
&& typeof l.siblingFile === "string" && typeof l.change === "string" && Array.isArray(l.callers)
|
|
1123
|
+
&& (l.hash === null || typeof l.hash === "string") && typeof l.reminded === "boolean").slice(-MAX_LESSONS);
|
|
1074
1124
|
const specs = normalizeExecutionObligations(raw.obligations);
|
|
1075
1125
|
const tracked = new Map((Array.isArray(raw.obligations) ? raw.obligations : []).map((item) => {
|
|
1076
1126
|
const candidate = item;
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/** Record the working tree as it stands, so later shell writes are measured
|
|
2
|
+
* from here (a prompt, or any tool call that is not a shell command). */
|
|
3
|
+
export declare function refreshShellBaseline(root: string, sessionId: string | undefined, agentId?: string): void;
|
|
4
|
+
/** Repo-relative files the shell command that just ran wrote (created or
|
|
5
|
+
* modified), and the baseline moves forward. Empty without a baseline: a
|
|
6
|
+
* session's first observation cannot tell its own writes from earlier ones. */
|
|
7
|
+
export declare function shellWrittenFiles(root: string, sessionId: string | undefined, agentId?: string): string[];
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/** Files a shell command wrote.
|
|
2
|
+
*
|
|
3
|
+
* Pre-edit grounding keys on the host's edit tools (Edit/Write/apply_patch).
|
|
4
|
+
* An agent that edits through the shell — a `python` heredoc, `sed -i`, `perl
|
|
5
|
+
* -pi`, a PowerShell `Set-Content` — never passes through them, so the file it
|
|
6
|
+
* changed arrives with no grounding at all. The shell tool's post-execution
|
|
7
|
+
* hook sees every command, but not what the command touched; parsing arbitrary
|
|
8
|
+
* shell for write targets is guesswork.
|
|
9
|
+
*
|
|
10
|
+
* Instead: fingerprint the working tree's dirty files (`git status`, then
|
|
11
|
+
* mtime+size) per session, repository and agent. A prompt and every tool call refresh
|
|
12
|
+
* the fingerprint; after a shell command, a dirty file whose fingerprint moved
|
|
13
|
+
* was written by that command. Deterministic, shell-agnostic, and bounded by the
|
|
14
|
+
* dirty set (never a tree walk). Any failure yields no files — never an error. */
|
|
15
|
+
import { execFileSync } from "node:child_process";
|
|
16
|
+
import { createHash } from "node:crypto";
|
|
17
|
+
import { mkdirSync, readFileSync, statSync, writeFileSync } from "node:fs";
|
|
18
|
+
import { tmpdir } from "node:os";
|
|
19
|
+
import { join } from "node:path";
|
|
20
|
+
/** Dirty paths fingerprinted; beyond it the rest are ignored (a vendored
|
|
21
|
+
* untracked tree must not make every command slow). */
|
|
22
|
+
const MAX_PATHS = 2000;
|
|
23
|
+
/** Hunch's own state: written by the hook itself and by capture tools. */
|
|
24
|
+
const OWN_STATE = /^\.hunch(?:-cache)?\//;
|
|
25
|
+
/** Concurrent subagents share the session but not their commands: each keeps
|
|
26
|
+
* its own baseline, keyed by its agent id (hashed with the rest, never kept
|
|
27
|
+
* raw) like the pre-edit dedupe. No agent id keeps the session's baseline. */
|
|
28
|
+
function snapshotFile(root, sessionId, agentId) {
|
|
29
|
+
const key = createHash("sha256").update(`${sessionId}\u0000${root}${agentId ? `\u0000${agentId}` : ""}`).digest("hex").slice(0, 24);
|
|
30
|
+
// Same directory as the session injection cache: its sweep drops stale files.
|
|
31
|
+
return join(tmpdir(), "hunch-hookcache", `shell-${key}.json`);
|
|
32
|
+
}
|
|
33
|
+
/** Repo-relative dirty paths (tracked changes and untracked files), NUL-safe. */
|
|
34
|
+
function dirtyPaths(root) {
|
|
35
|
+
let raw;
|
|
36
|
+
try {
|
|
37
|
+
raw = execFileSync("git", ["-C", root, "status", "--porcelain=v1", "-z", "--untracked-files=all"], {
|
|
38
|
+
encoding: "utf8",
|
|
39
|
+
timeout: 2_000,
|
|
40
|
+
maxBuffer: 8_000_000,
|
|
41
|
+
stdio: ["ignore", "pipe", "ignore"],
|
|
42
|
+
env: { ...process.env, GIT_OPTIONAL_LOCKS: "0", GIT_TERMINAL_PROMPT: "0" },
|
|
43
|
+
});
|
|
44
|
+
}
|
|
45
|
+
catch {
|
|
46
|
+
return null;
|
|
47
|
+
}
|
|
48
|
+
// Porcelain paths are relative to the git toplevel, which can sit above the
|
|
49
|
+
// Hunch root (a package inside a monorepo): re-relativize, drop the outside.
|
|
50
|
+
let prefix = "";
|
|
51
|
+
try {
|
|
52
|
+
prefix = execFileSync("git", ["-C", root, "rev-parse", "--show-prefix"], { encoding: "utf8", timeout: 2_000, stdio: ["ignore", "pipe", "ignore"] }).trim();
|
|
53
|
+
}
|
|
54
|
+
catch {
|
|
55
|
+
return null;
|
|
56
|
+
}
|
|
57
|
+
const out = [];
|
|
58
|
+
const fields = raw.split("\u0000");
|
|
59
|
+
for (let i = 0; i < fields.length && out.length < MAX_PATHS; i++) {
|
|
60
|
+
const entry = fields[i];
|
|
61
|
+
if (entry.length < 4)
|
|
62
|
+
continue;
|
|
63
|
+
const status = entry.slice(0, 2);
|
|
64
|
+
const path = entry.slice(3);
|
|
65
|
+
if (path.startsWith(prefix))
|
|
66
|
+
out.push(path.slice(prefix.length));
|
|
67
|
+
// A rename/copy entry is followed by its source path.
|
|
68
|
+
if (status.includes("R") || status.includes("C"))
|
|
69
|
+
i++;
|
|
70
|
+
}
|
|
71
|
+
return out;
|
|
72
|
+
}
|
|
73
|
+
function fingerprint(root) {
|
|
74
|
+
const paths = dirtyPaths(root);
|
|
75
|
+
if (!paths)
|
|
76
|
+
return null;
|
|
77
|
+
const fp = {};
|
|
78
|
+
for (const p of paths) {
|
|
79
|
+
if (OWN_STATE.test(p))
|
|
80
|
+
continue;
|
|
81
|
+
try {
|
|
82
|
+
const st = statSync(join(root, p));
|
|
83
|
+
if (st.isFile())
|
|
84
|
+
fp[p] = `${st.mtimeMs}:${st.size}`;
|
|
85
|
+
}
|
|
86
|
+
catch { /* deleted: nothing to ground */ }
|
|
87
|
+
}
|
|
88
|
+
return fp;
|
|
89
|
+
}
|
|
90
|
+
function load(file) {
|
|
91
|
+
try {
|
|
92
|
+
const raw = JSON.parse(readFileSync(file, "utf8"));
|
|
93
|
+
return raw && typeof raw === "object" && !Array.isArray(raw) ? raw : null;
|
|
94
|
+
}
|
|
95
|
+
catch {
|
|
96
|
+
return null;
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
function save(file, fp) {
|
|
100
|
+
try {
|
|
101
|
+
mkdirSync(join(tmpdir(), "hunch-hookcache"), { recursive: true });
|
|
102
|
+
writeFileSync(file, JSON.stringify(fp));
|
|
103
|
+
}
|
|
104
|
+
catch { /* the next refresh retries */ }
|
|
105
|
+
}
|
|
106
|
+
/** Record the working tree as it stands, so later shell writes are measured
|
|
107
|
+
* from here (a prompt, or any tool call that is not a shell command). */
|
|
108
|
+
export function refreshShellBaseline(root, sessionId, agentId) {
|
|
109
|
+
if (!sessionId)
|
|
110
|
+
return;
|
|
111
|
+
const fp = fingerprint(root);
|
|
112
|
+
if (fp)
|
|
113
|
+
save(snapshotFile(root, sessionId, agentId), fp);
|
|
114
|
+
}
|
|
115
|
+
/** Repo-relative files the shell command that just ran wrote (created or
|
|
116
|
+
* modified), and the baseline moves forward. Empty without a baseline: a
|
|
117
|
+
* session's first observation cannot tell its own writes from earlier ones. */
|
|
118
|
+
export function shellWrittenFiles(root, sessionId, agentId) {
|
|
119
|
+
if (!sessionId)
|
|
120
|
+
return [];
|
|
121
|
+
const file = snapshotFile(root, sessionId, agentId);
|
|
122
|
+
const before = load(file);
|
|
123
|
+
const now = fingerprint(root);
|
|
124
|
+
if (!now)
|
|
125
|
+
return [];
|
|
126
|
+
save(file, now);
|
|
127
|
+
if (!before)
|
|
128
|
+
return [];
|
|
129
|
+
return Object.keys(now).filter((p) => before[p] !== now[p]).sort();
|
|
130
|
+
}
|
|
131
|
+
//# sourceMappingURL=shellwrites.js.map
|