@azure-id/orc 1.7.1 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +3649 -3381
- package/README-id.md +923 -844
- package/README.md +836 -788
- package/bin/build-agents.js +43 -27
- package/bin/cli.js +701 -3
- package/bin/graph-extract.js +927 -0
- package/bin/graph-notes.js +188 -0
- package/bin/graph-query.js +808 -0
- package/bin/graph-resolve.js +178 -0
- package/bin/graph-signals.js +277 -0
- package/bin/graph.js +605 -0
- package/bin/verify-contracts.js +4669 -4553
- package/bin/verify-package.js +626 -616
- package/bin/webui/api.js +1419 -1414
- package/bin/webui/fixtures/index.js +579 -576
- package/bin/webui/fixtures/knowledge.js +316 -291
- package/bin/webui/i18n/en/knowledge.json +167 -151
- package/bin/webui/i18n/en/overview.json +101 -100
- package/bin/webui/i18n/id/knowledge.json +167 -151
- package/bin/webui/i18n/id/overview.json +101 -100
- package/bin/webui/js/panels/knowledge.js +1065 -1006
- package/bin/webui/js/panels/overview.js +492 -488
- package/package.json +39 -39
- package/templates/agents/MODEL-MAPPING.md +163 -158
- package/templates/agents/orc-executor-haiku-4-5.md +133 -121
- package/templates/agents/orc-executor-opus-4-7-high.md +134 -122
- package/templates/agents/orc-executor-opus-4-7-med.md +134 -122
- package/templates/agents/orc-executor-opus-4-8-high.md +134 -122
- package/templates/agents/orc-executor-opus-5-high.md +134 -122
- package/templates/agents/orc-executor-opus-5-low.md +134 -122
- package/templates/agents/orc-executor-opus-5-med.md +134 -122
- package/templates/agents/orc-executor-sonnet-4-6-high.md +134 -122
- package/templates/agents/orc-executor-sonnet-4-6-med.md +134 -122
- package/templates/agents/orc-executor-sonnet-5-high.md +134 -122
- package/templates/agents/orc-graph-noter-sonnet-4-6-med.md +86 -0
- package/templates/hooks/README.md +444 -396
- package/templates/hooks/orc-graph-hook.js +336 -0
- package/templates/hooks/orc-statusline-render.js +922 -921
- package/templates/hooks/orc-statusline.js +1596 -1545
- package/templates/skills/_shared/README.md +4 -0
- package/templates/skills/_shared/code-graph.md +220 -0
- package/templates/skills/_shared/opus5-only.md +4 -0
- package/templates/skills/_shared/phases/execution.md +166 -147
- package/templates/skills/_shared/phases/planning.md +142 -135
- package/templates/skills/_shared/phases/preflight.md +132 -118
- package/templates/skills/_shared/phases/review.md +63 -53
- package/templates/skills/_shared/phases/ship.md +96 -88
- package/templates/skills/_shared/phases/trace.md +6 -0
- package/templates/skills/_shared/phases/wiki-consult.md +194 -189
- package/templates/skills/_shared/read-ladder.md +124 -102
- package/templates/skills/_shared/return-validation.md +259 -250
- package/templates/skills/orc/SKILL.md +255 -254
- package/templates/skills/orc-diy/references/flow-schema.md +101 -100
- package/templates/skills/orc-fast/SKILL.md +236 -229
- package/templates/skills/orc-mini/SKILL.md +267 -259
- package/templates/skills/orc-quick/SKILL.md +378 -361
- package/templates/skills/orc-quick/references/dispatch-gate.md +6 -0
- package/templates/skills/orc-wiki/references/staleness.md +294 -288
package/bin/graph.js
ADDED
|
@@ -0,0 +1,605 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
// ── orc graph (v1.8.0) — the code graph STORE ───────────────────────────────
|
|
3
|
+
//
|
|
4
|
+
// A small, local, git-ignored map of how this repository is connected. This
|
|
5
|
+
// file owns the store and change detection. Extraction is `graph-extract.js`;
|
|
6
|
+
// resolution and the read commands are `graph-query.js`. `bin/cli.js` routes
|
|
7
|
+
// `orc graph …` and resolves the config — no module here reads
|
|
8
|
+
// `.claude/orc.config.yaml`.
|
|
9
|
+
//
|
|
10
|
+
// THE RULES THIS FILE HOLDS
|
|
11
|
+
//
|
|
12
|
+
// 1. A change is found from git's OWN blob SHAs. `git ls-files -s` hashes
|
|
13
|
+
// every tracked file in one call; `git status` names the dirty and
|
|
14
|
+
// untracked ones, and `git hash-object` hashes only those. Nothing is
|
|
15
|
+
// parsed to find out whether it changed. (W0: 42 ms for 7,091 files.)
|
|
16
|
+
// 2. A record is CONTENT-ADDRESSED — `blobs/<ab>/<sha>.json`. A branch
|
|
17
|
+
// switch back, a revert, or a teammate's identical file reuses it.
|
|
18
|
+
// 3. Writes are atomic (temp file + rename) and serialized by `.lock`.
|
|
19
|
+
// Readers never take the lock: they see the old index or the new one.
|
|
20
|
+
// 4. The write ORDER is blobs → index.json → files.json → meta.json. A crash
|
|
21
|
+
// between any two leaves files.json OLD, so the next status reads DRIFTED
|
|
22
|
+
// and the next update redoes the work. Idempotent, never half-applied.
|
|
23
|
+
// 5. No background process and no timer, ever — a continuous rebuild is how
|
|
24
|
+
// the graph tools in the research froze machines. EW3 adds two ONE-SHOT
|
|
25
|
+
// triggers (a read that finds its own target stale, and an executor
|
|
26
|
+
// finishing). Both take the lock below, and the loser SKIPS rather than
|
|
27
|
+
// queues, so nothing can ever pile up.
|
|
28
|
+
|
|
29
|
+
const fs = require("fs");
|
|
30
|
+
const path = require("path");
|
|
31
|
+
const { spawnSync } = require("child_process");
|
|
32
|
+
const crypto = require("crypto");
|
|
33
|
+
const X = require("./graph-extract.js");
|
|
34
|
+
|
|
35
|
+
const SCHEMA = 1;
|
|
36
|
+
// Bumped whenever the record shape or extraction changes. A record made by an
|
|
37
|
+
// older engine is re-extracted, and status reads DRIFTED until it is.
|
|
38
|
+
// @3 (W9): route symbols + `ref` edges — an older index re-extracts once.
|
|
39
|
+
// @4 (EW1): per-file COVERAGE and a GENERATION on the index.
|
|
40
|
+
const ENGINE = "graph@4";
|
|
41
|
+
const GIT_MAX_BUFFER = 256 * 1024 * 1024;
|
|
42
|
+
const MAX_BYTES = 512 * 1024;
|
|
43
|
+
const LOCK_STALE_MS = 10 * 60 * 1000;
|
|
44
|
+
// Measured after W2 on this machine class: nestjs/nest 1.3 ms per file
|
|
45
|
+
// (heuristic), django/django 3.5 ms per file (Python ast + masking). The
|
|
46
|
+
// estimate uses the slower one — a first build that finishes early is fine, one
|
|
47
|
+
// that overruns its own estimate teaches people to ignore it.
|
|
48
|
+
const EST_MS_PER_FILE = 3.5;
|
|
49
|
+
const ESTIMATE_ABOVE = 2000;
|
|
50
|
+
|
|
51
|
+
const LANG_BY_EXT = {
|
|
52
|
+
".js": "js", ".mjs": "js", ".cjs": "js", ".jsx": "js",
|
|
53
|
+
".ts": "ts", ".tsx": "ts", ".mts": "ts", ".cts": "ts",
|
|
54
|
+
".py": "py",
|
|
55
|
+
".go": "go",
|
|
56
|
+
".java": "java",
|
|
57
|
+
".cs": "cs",
|
|
58
|
+
".php": "php",
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
// Never part of the map, whatever git says: ORC's own tree, installed
|
|
62
|
+
// dependencies (sometimes committed), and minified bundles.
|
|
63
|
+
const ALWAYS_SKIP = /^(\.claude\/|node_modules\/|(.*\/)?node_modules\/)|\.min\.js$/;
|
|
64
|
+
|
|
65
|
+
function graphPaths(claudeDir) {
|
|
66
|
+
const dir = path.join(claudeDir, "orc", "graph");
|
|
67
|
+
return {
|
|
68
|
+
dir,
|
|
69
|
+
meta: path.join(dir, "meta.json"),
|
|
70
|
+
files: path.join(dir, "files.json"),
|
|
71
|
+
index: path.join(dir, "index.json"),
|
|
72
|
+
blobs: path.join(dir, "blobs"),
|
|
73
|
+
notes: path.join(dir, "notes.jsonl"),
|
|
74
|
+
lock: path.join(dir, ".lock"),
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function globRe(g) {
|
|
79
|
+
const esc = String(g)
|
|
80
|
+
.replace(/[.+^${}()|[\]\\]/g, "\\$&")
|
|
81
|
+
.replace(/\*\*/g, "__GLOBSTAR__")
|
|
82
|
+
.replace(/\*/g, "[^/]*")
|
|
83
|
+
.replace(/\?/g, "[^/]")
|
|
84
|
+
.replace(/__GLOBSTAR__/g, ".*");
|
|
85
|
+
return new RegExp("^" + esc + "(/|$)");
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function makeFilter(ignore) {
|
|
89
|
+
const res = (Array.isArray(ignore) ? ignore : []).filter(Boolean).map(globRe);
|
|
90
|
+
return (rel) => {
|
|
91
|
+
if (ALWAYS_SKIP.test(rel)) return null;
|
|
92
|
+
if (res.some((re) => re.test(rel))) return null;
|
|
93
|
+
return LANG_BY_EXT[path.extname(rel).toLowerCase()] || null;
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function git(root, args, input) {
|
|
98
|
+
return spawnSync("git", args, { cwd: root, encoding: "utf8", maxBuffer: GIT_MAX_BUFFER, input });
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function stamp(d) {
|
|
102
|
+
const p = (n) => String(n).padStart(2, "0");
|
|
103
|
+
return `${p(d.getDate())}-${p(d.getMonth() + 1)}-${d.getFullYear()} ${p(d.getHours())}:${p(d.getMinutes())}:${p(d.getSeconds())}`;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function readJson(file) {
|
|
107
|
+
try {
|
|
108
|
+
return JSON.parse(fs.readFileSync(file, "utf8"));
|
|
109
|
+
} catch (_) {
|
|
110
|
+
return null;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// Windows can refuse a rename for a moment while another process has the
|
|
115
|
+
// target open (a reader mid-parse). A short retry is the whole remedy.
|
|
116
|
+
function atomicWrite(file, text) {
|
|
117
|
+
const tmp = `${file}.${process.pid}.tmp`;
|
|
118
|
+
fs.writeFileSync(tmp, text);
|
|
119
|
+
for (let i = 0; ; i++) {
|
|
120
|
+
try {
|
|
121
|
+
fs.renameSync(tmp, file);
|
|
122
|
+
return;
|
|
123
|
+
} catch (e) {
|
|
124
|
+
if (i >= 5 || (e.code !== "EPERM" && e.code !== "EBUSY" && e.code !== "EACCES")) {
|
|
125
|
+
try { fs.rmSync(tmp, { force: true }); } catch (_) {}
|
|
126
|
+
throw e;
|
|
127
|
+
}
|
|
128
|
+
const until = Date.now() + 20 * (i + 1);
|
|
129
|
+
while (Date.now() < until) {} // eslint-disable-line no-empty
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// ── the lock ────────────────────────────────────────────────────────────────
|
|
135
|
+
function acquireLock(p) {
|
|
136
|
+
fs.mkdirSync(p.dir, { recursive: true });
|
|
137
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
138
|
+
try {
|
|
139
|
+
const fd = fs.openSync(p.lock, "wx");
|
|
140
|
+
fs.writeSync(fd, JSON.stringify({ pid: process.pid, at: stamp(new Date()) }));
|
|
141
|
+
fs.closeSync(fd);
|
|
142
|
+
return { ok: true };
|
|
143
|
+
} catch (e) {
|
|
144
|
+
if (e.code !== "EEXIST") throw e;
|
|
145
|
+
let age = 0;
|
|
146
|
+
try {
|
|
147
|
+
age = Date.now() - fs.statSync(p.lock).mtimeMs;
|
|
148
|
+
} catch (_) {
|
|
149
|
+
continue; // released between the open and the stat
|
|
150
|
+
}
|
|
151
|
+
// A writer that died leaves its lock behind. Ten minutes is far past any
|
|
152
|
+
// real update (W0: 1.7 s for Django), so an older lock is a dead one.
|
|
153
|
+
if (age > LOCK_STALE_MS) {
|
|
154
|
+
try { fs.rmSync(p.lock, { force: true }); } catch (_) {}
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
return { ok: false, holder: readJson(p.lock), age_ms: Math.round(age) };
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
return { ok: false, holder: readJson(p.lock), age_ms: null };
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function releaseLock(p) {
|
|
164
|
+
try { fs.rmSync(p.lock, { force: true }); } catch (_) {}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
// ── detection ───────────────────────────────────────────────────────────────
|
|
168
|
+
// Returns { ok, files: Map<rel, blob>, langs, head } or { ok:false, reason }.
|
|
169
|
+
function detect(root, opts) {
|
|
170
|
+
const langOf = makeFilter(opts && opts.ignore);
|
|
171
|
+
const ls = git(root, ["ls-files", "-s", "-z"]);
|
|
172
|
+
if (ls.error || ls.status !== 0) return { ok: false, reason: "not-git" };
|
|
173
|
+
const files = new Map();
|
|
174
|
+
const langs = new Map();
|
|
175
|
+
for (const rec of ls.stdout.split("\0")) {
|
|
176
|
+
if (!rec) continue;
|
|
177
|
+
const tab = rec.indexOf("\t");
|
|
178
|
+
if (tab < 0) continue;
|
|
179
|
+
const [mode, blob] = rec.slice(0, tab).split(" ");
|
|
180
|
+
const rel = rec.slice(tab + 1);
|
|
181
|
+
if (mode === "160000") continue; // a submodule is another repository
|
|
182
|
+
const lang = langOf(rel);
|
|
183
|
+
if (!lang) continue;
|
|
184
|
+
// A conflicted file lists up to three stages; the working tree decides
|
|
185
|
+
// below, so the first stage seen is only a placeholder.
|
|
186
|
+
if (!files.has(rel)) files.set(rel, blob);
|
|
187
|
+
langs.set(rel, lang);
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
const st = git(root, ["status", "--porcelain=v1", "-z", "--untracked-files=all", "--no-renames"]);
|
|
191
|
+
if (st.error || st.status !== 0) return { ok: false, reason: "git-status-failed" };
|
|
192
|
+
const hashMe = [];
|
|
193
|
+
for (const rec of st.stdout.split("\0")) {
|
|
194
|
+
if (!rec || rec.length < 4) continue;
|
|
195
|
+
const x = rec[0];
|
|
196
|
+
const y = rec[1];
|
|
197
|
+
const rel = rec.slice(3);
|
|
198
|
+
const lang = langOf(rel);
|
|
199
|
+
if (!lang) continue;
|
|
200
|
+
if (y === "D") {
|
|
201
|
+
files.delete(rel);
|
|
202
|
+
langs.delete(rel);
|
|
203
|
+
continue;
|
|
204
|
+
}
|
|
205
|
+
// `ls-files -s` already carries the STAGED content. Only a working-tree
|
|
206
|
+
// difference (M, T, U, A with intent-to-add) or an untracked file needs its
|
|
207
|
+
// own hash.
|
|
208
|
+
if (x === "?" || x === "U" || y === "M" || y === "T" || y === "U" || y === "A") {
|
|
209
|
+
hashMe.push(rel);
|
|
210
|
+
langs.set(rel, lang);
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
if (hashMe.length) {
|
|
214
|
+
// No --no-filters on purpose: the clean filter (autocrlf on Windows) must
|
|
215
|
+
// run, or every CRLF file would hash differently from its index blob and
|
|
216
|
+
// read as changed forever.
|
|
217
|
+
const h = git(root, ["hash-object", "--stdin-paths"], hashMe.join("\n") + "\n");
|
|
218
|
+
if (h.error || h.status !== 0) return { ok: false, reason: "hash-object-failed" };
|
|
219
|
+
const shas = h.stdout.split(/\r?\n/).filter(Boolean);
|
|
220
|
+
hashMe.forEach((rel, i) => {
|
|
221
|
+
if (shas[i]) files.set(rel, shas[i]);
|
|
222
|
+
});
|
|
223
|
+
}
|
|
224
|
+
const head = git(root, ["rev-parse", "HEAD"]);
|
|
225
|
+
return {
|
|
226
|
+
ok: true,
|
|
227
|
+
files,
|
|
228
|
+
langs,
|
|
229
|
+
head: head.status === 0 ? head.stdout.trim() : null,
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
function diffFiles(prev, cur) {
|
|
234
|
+
const added = [];
|
|
235
|
+
const changed = [];
|
|
236
|
+
const deleted = [];
|
|
237
|
+
for (const [rel, blob] of cur) {
|
|
238
|
+
const was = prev[rel];
|
|
239
|
+
if (!was) added.push(rel);
|
|
240
|
+
else if (was.blob !== blob) changed.push(rel);
|
|
241
|
+
}
|
|
242
|
+
for (const rel of Object.keys(prev)) if (!cur.has(rel)) deleted.push(rel);
|
|
243
|
+
return { added, changed, deleted };
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
function blobPath(p, blob) {
|
|
247
|
+
return path.join(p.blobs, blob.slice(0, 2), `${blob}.json`);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
// ── generation (EW1) ────────────────────────────────────────────────────────
|
|
251
|
+
// `generation` counts index-changing writes; `gen_id` names the CONTENT those
|
|
252
|
+
// writes produced. A reader compares the number (cheap) and a second store —
|
|
253
|
+
// the resolution cache — pins itself to it. Two machines that index the same
|
|
254
|
+
// tree get the same `gen_id` and different `generation` numbers, so the id is
|
|
255
|
+
// what a card may quote and the number is what a cache may compare.
|
|
256
|
+
function genId(filesOut) {
|
|
257
|
+
const h = crypto.createHash("sha1");
|
|
258
|
+
for (const rel of Object.keys(filesOut).sort()) h.update(`${rel}${(filesOut[rel] || {}).blob || ""}`);
|
|
259
|
+
return h.digest("hex").slice(0, 8);
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// What the extractor saw of ONE file, from the record it already wrote. A path
|
|
263
|
+
// with no record at all is `excluded` — git does not track it, an ignore glob
|
|
264
|
+
// dropped it, or the language has no extractor.
|
|
265
|
+
function coverageOf(entry) {
|
|
266
|
+
if (!entry) return { coverage: "excluded" };
|
|
267
|
+
if (entry.skipped) return { coverage: `skipped:${entry.skipped}` };
|
|
268
|
+
if (entry.coverage === "partial") return { coverage: "partial", ranges: entry.partial || [] };
|
|
269
|
+
return { coverage: "full" };
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
// ── status ──────────────────────────────────────────────────────────────────
|
|
273
|
+
// exit 0 FRESH · 1 NONE (or unavailable) · 2 DRIFTED · 3 OFF
|
|
274
|
+
function graphStatus(claudeDir, root, opts) {
|
|
275
|
+
const p = graphPaths(claudeDir);
|
|
276
|
+
const meta = readJson(p.meta);
|
|
277
|
+
const prev = readJson(p.files);
|
|
278
|
+
const base = {
|
|
279
|
+
ok: true,
|
|
280
|
+
enabled: !!opts.enabled,
|
|
281
|
+
exists: !!(meta && prev),
|
|
282
|
+
files: meta ? meta.files : 0,
|
|
283
|
+
symbols: meta ? meta.symbols : 0,
|
|
284
|
+
updated_at: meta ? meta.updated_at : null,
|
|
285
|
+
head_commit: meta ? meta.head_commit : null,
|
|
286
|
+
engine: meta ? meta.engine : null,
|
|
287
|
+
generation: meta ? meta.generation || 0 : 0,
|
|
288
|
+
gen_id: meta ? meta.gen_id || null : null,
|
|
289
|
+
};
|
|
290
|
+
if (!opts.enabled) return { ...base, state: "off", exit: 3 };
|
|
291
|
+
if (!meta || !prev) return { ...base, state: "none", exit: 1 };
|
|
292
|
+
const d = detect(root, opts);
|
|
293
|
+
if (!d.ok) return { ...base, ok: false, state: "unavailable", reason: d.reason, exit: 1 };
|
|
294
|
+
const diff = diffFiles(prev, d.files);
|
|
295
|
+
const behind = { added: diff.added.length, changed: diff.changed.length, deleted: diff.deleted.length };
|
|
296
|
+
const engineStale = meta.engine !== ENGINE || meta.schema !== SCHEMA;
|
|
297
|
+
const n = behind.added + behind.changed + behind.deleted;
|
|
298
|
+
if (n || engineStale) return { ...base, state: "drifted", behind, engine_stale: engineStale, exit: 2 };
|
|
299
|
+
return { ...base, state: "fresh", behind, engine_stale: false, exit: 0 };
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// ── update ──────────────────────────────────────────────────────────────────
|
|
303
|
+
// exit 0 built/updated/unchanged · 1 unavailable (not git, locked, io) · 3 off
|
|
304
|
+
function graphUpdate(claudeDir, root, opts) {
|
|
305
|
+
const t0 = Date.now();
|
|
306
|
+
const say = opts.say || (() => {});
|
|
307
|
+
if (!opts.enabled && opts.ifEnabled) return { ok: true, enabled: false, state: "off", exit: 3 };
|
|
308
|
+
const p = graphPaths(claudeDir);
|
|
309
|
+
const d = detect(root, opts);
|
|
310
|
+
if (!d.ok) return { ok: false, enabled: !!opts.enabled, state: "unavailable", reason: d.reason, exit: 1 };
|
|
311
|
+
|
|
312
|
+
const lock = acquireLock(p);
|
|
313
|
+
if (!lock.ok) {
|
|
314
|
+
return { ok: false, enabled: !!opts.enabled, state: "unavailable", reason: "locked", holder: lock.holder, age_ms: lock.age_ms, exit: 1 };
|
|
315
|
+
}
|
|
316
|
+
try {
|
|
317
|
+
const prevMeta = readJson(p.meta);
|
|
318
|
+
const prevFiles = readJson(p.files);
|
|
319
|
+
const first = !prevMeta || !prevFiles;
|
|
320
|
+
const upgrade = !first && (prevMeta.engine !== ENGINE || prevMeta.schema !== SCHEMA);
|
|
321
|
+
const prev = first ? {} : prevFiles;
|
|
322
|
+
const fresh = { schema: SCHEMA, engine: ENGINE, by_file: {} };
|
|
323
|
+
const index = first || upgrade ? fresh : readJson(p.index) || fresh;
|
|
324
|
+
|
|
325
|
+
const diff = diffFiles(prev, d.files);
|
|
326
|
+
// An engine upgrade re-extracts everything. A missing index entry is
|
|
327
|
+
// repaired too — that is how a crash between two writes heals.
|
|
328
|
+
const work = new Set([...diff.added, ...diff.changed]);
|
|
329
|
+
if (upgrade) for (const rel of d.files.keys()) work.add(rel);
|
|
330
|
+
for (const rel of d.files.keys()) if (!index.by_file[rel]) work.add(rel);
|
|
331
|
+
|
|
332
|
+
if (first && work.size > ESTIMATE_ABOVE) {
|
|
333
|
+
say(`graph: building first index (~${work.size} files, est. ${Math.max(1, Math.round((work.size * EST_MS_PER_FILE) / 1000))} s)`);
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
// Phase 1 — reuse what is already on disk; read the rest.
|
|
337
|
+
let reused = 0;
|
|
338
|
+
let skipped = 0;
|
|
339
|
+
const records = new Map();
|
|
340
|
+
const toExtract = [];
|
|
341
|
+
for (const rel of work) {
|
|
342
|
+
const blob = d.files.get(rel);
|
|
343
|
+
const lang = d.langs.get(rel);
|
|
344
|
+
const have = readJson(blobPath(p, blob));
|
|
345
|
+
if (have && have.schema === SCHEMA && have.engine === ENGINE) {
|
|
346
|
+
records.set(rel, have);
|
|
347
|
+
reused++;
|
|
348
|
+
continue;
|
|
349
|
+
}
|
|
350
|
+
const abs = path.join(root, ...rel.split("/"));
|
|
351
|
+
try {
|
|
352
|
+
const size = fs.statSync(abs).size;
|
|
353
|
+
if (size > MAX_BYTES) records.set(rel, { schema: SCHEMA, engine: ENGINE, blob, lang, bytes: size, skipped: "too-large", coverage: "skipped", imports: [], symbols: [] });
|
|
354
|
+
else toExtract.push({ rel, abs, lang, blob, bytes: size, src: fs.readFileSync(abs, "utf8") });
|
|
355
|
+
} catch (_) {
|
|
356
|
+
records.set(rel, { schema: SCHEMA, engine: ENGINE, blob, lang, bytes: 0, skipped: "unreadable", coverage: "skipped", imports: [], symbols: [] });
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
// Phase 2 — extract in ONE batch (one Python process for every .py file).
|
|
361
|
+
const out = X.extractBatch(toExtract);
|
|
362
|
+
for (const it of toExtract) {
|
|
363
|
+
const r = out.get(it.rel) || { extractor: X.HEURISTIC, imports: [], symbols: [] };
|
|
364
|
+
records.set(it.rel, {
|
|
365
|
+
schema: SCHEMA,
|
|
366
|
+
engine: ENGINE,
|
|
367
|
+
blob: it.blob,
|
|
368
|
+
lang: it.lang,
|
|
369
|
+
bytes: it.bytes,
|
|
370
|
+
extractor: r.extractor,
|
|
371
|
+
...(r.error ? { skipped: r.error } : {}),
|
|
372
|
+
coverage: r.error ? "skipped" : r.coverage || "full",
|
|
373
|
+
...(r.partial ? { partial: r.partial } : {}),
|
|
374
|
+
imports: r.imports,
|
|
375
|
+
symbols: r.symbols,
|
|
376
|
+
});
|
|
377
|
+
}
|
|
378
|
+
for (const [rel, record] of records) {
|
|
379
|
+
if (!toExtract.some((t) => t.rel === rel) && reused && readJson(blobPath(p, record.blob))) continue;
|
|
380
|
+
const file = blobPath(p, record.blob);
|
|
381
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
382
|
+
atomicWrite(file, JSON.stringify(record));
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
const filesOut = {};
|
|
386
|
+
for (const [rel, blob] of d.files) {
|
|
387
|
+
const record = records.get(rel);
|
|
388
|
+
if (record) {
|
|
389
|
+
if (record.skipped) skipped++;
|
|
390
|
+
index.by_file[rel] = {
|
|
391
|
+
blob,
|
|
392
|
+
lang: record.lang,
|
|
393
|
+
...(record.skipped ? { skipped: record.skipped } : {}),
|
|
394
|
+
...(record.coverage && record.coverage !== "full" ? { coverage: record.coverage } : {}),
|
|
395
|
+
...(record.partial ? { partial: record.partial } : {}),
|
|
396
|
+
imports: record.imports,
|
|
397
|
+
symbols: record.symbols,
|
|
398
|
+
};
|
|
399
|
+
filesOut[rel] = {
|
|
400
|
+
blob,
|
|
401
|
+
lang: record.lang,
|
|
402
|
+
bytes: record.bytes,
|
|
403
|
+
...(record.extractor ? { extractor: record.extractor } : {}),
|
|
404
|
+
...(record.skipped ? { skipped: record.skipped } : {}),
|
|
405
|
+
...(record.coverage && record.coverage !== "full" ? { coverage: record.coverage } : {}),
|
|
406
|
+
...(record.partial ? { partial: record.partial } : {}),
|
|
407
|
+
};
|
|
408
|
+
} else {
|
|
409
|
+
filesOut[rel] = prev[rel];
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
for (const rel of Object.keys(index.by_file)) if (!d.files.has(rel)) delete index.by_file[rel];
|
|
413
|
+
|
|
414
|
+
let symbols = 0;
|
|
415
|
+
for (const v of Object.values(index.by_file)) symbols += (v.symbols || []).filter((s) => s.kind !== "module").length;
|
|
416
|
+
const unchanged = !first && !upgrade && work.size === 0 && diff.deleted.length === 0;
|
|
417
|
+
|
|
418
|
+
const gen = genId(filesOut);
|
|
419
|
+
const meta = {
|
|
420
|
+
schema: SCHEMA,
|
|
421
|
+
engine: ENGINE,
|
|
422
|
+
head_commit: d.head,
|
|
423
|
+
updated_at: unchanged && prevMeta ? prevMeta.updated_at : stamp(new Date()),
|
|
424
|
+
files: d.files.size,
|
|
425
|
+
symbols,
|
|
426
|
+
// The number goes up only when the index on disk changes, so a reader
|
|
427
|
+
// that saw generation N is looking at exactly the index that wrote N.
|
|
428
|
+
generation: unchanged && prevMeta && prevMeta.generation ? prevMeta.generation : ((prevMeta && prevMeta.generation) || 0) + 1,
|
|
429
|
+
gen_id: gen,
|
|
430
|
+
// EW3: how long the last real update took, end to end. A read that may
|
|
431
|
+
// heal uses it as the estimate for the next one — the only honest
|
|
432
|
+
// estimate available, because an update cannot be stopped half way.
|
|
433
|
+
update_ms: unchanged && prevMeta ? prevMeta.update_ms || 0 : 0,
|
|
434
|
+
};
|
|
435
|
+
if (!unchanged) {
|
|
436
|
+
index.schema = SCHEMA;
|
|
437
|
+
index.engine = ENGINE;
|
|
438
|
+
atomicWrite(p.index, JSON.stringify(index));
|
|
439
|
+
atomicWrite(p.files, JSON.stringify(filesOut));
|
|
440
|
+
atomicWrite(p.meta, JSON.stringify(meta, null, 2) + "\n");
|
|
441
|
+
}
|
|
442
|
+
// EW2: the DERIVED resolution cache, written inside this same lock, AFTER
|
|
443
|
+
// meta.json — it reads the generation it must pin itself to. It is an
|
|
444
|
+
// optimisation: a route of `failed` leaves the graph correct and only
|
|
445
|
+
// slower, so it never changes this function's answer.
|
|
446
|
+
const resolveRoute = unchanged && require("./graph-resolve.js").load(claudeDir, meta)
|
|
447
|
+
? { route: "unchanged", ms: 0 }
|
|
448
|
+
: require("./graph-resolve.js").build(claudeDir, root);
|
|
449
|
+
// EW3: stamp the duration. A second 200-byte write of the same file, not a
|
|
450
|
+
// second commit point — every other field is already the one just written,
|
|
451
|
+
// so a crash between the two leaves a valid meta that only lacks an
|
|
452
|
+
// estimate, and a missing estimate simply means "heal, and find out".
|
|
453
|
+
if (!unchanged) {
|
|
454
|
+
meta.update_ms = Date.now() - t0;
|
|
455
|
+
atomicWrite(p.meta, JSON.stringify(meta, null, 2) + "\n");
|
|
456
|
+
}
|
|
457
|
+
return {
|
|
458
|
+
ok: true,
|
|
459
|
+
enabled: !!opts.enabled,
|
|
460
|
+
state: first ? "built" : unchanged ? "unchanged" : "updated",
|
|
461
|
+
files: d.files.size,
|
|
462
|
+
symbols,
|
|
463
|
+
generation: meta.generation,
|
|
464
|
+
gen_id: meta.gen_id,
|
|
465
|
+
route: resolveRoute.route,
|
|
466
|
+
...(resolveRoute.reason ? { route_reason: resolveRoute.reason } : {}),
|
|
467
|
+
added: diff.added.length,
|
|
468
|
+
changed: diff.changed.length,
|
|
469
|
+
deleted: diff.deleted.length,
|
|
470
|
+
parsed: toExtract.length,
|
|
471
|
+
reused,
|
|
472
|
+
skipped,
|
|
473
|
+
engine_upgrade: upgrade,
|
|
474
|
+
head_commit: d.head,
|
|
475
|
+
ms: Date.now() - t0,
|
|
476
|
+
exit: 0,
|
|
477
|
+
};
|
|
478
|
+
} finally {
|
|
479
|
+
releaseLock(p);
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
// ── coverage ────────────────────────────────────────────────────────────────
|
|
484
|
+
// exit 0 always when a graph exists — a gap IS the answer, never an error.
|
|
485
|
+
// exit 1 no index · 3 off (with --if-enabled)
|
|
486
|
+
//
|
|
487
|
+
// It reads `files.json` (444 KB on django/django) and NEVER `index.json`
|
|
488
|
+
// (22 MB), because the only question is how much of each file was seen.
|
|
489
|
+
function graphCoverage(claudeDir, root, opts) {
|
|
490
|
+
const p = graphPaths(claudeDir);
|
|
491
|
+
const meta = readJson(p.meta);
|
|
492
|
+
const files = readJson(p.files);
|
|
493
|
+
if (!meta || !files) return { ok: false, state: "none", reason: "no-index", exit: 1 };
|
|
494
|
+
const paths = (opts.paths || []).map((x) => String(x).split("\\").join("/").replace(/^\.\//, ""));
|
|
495
|
+
// One `hash-object` for every path that is still on disk — the batch command
|
|
496
|
+
// must not pay one git process per file.
|
|
497
|
+
const onDisk = paths.filter((rel) => files[rel] && fs.existsSync(path.join(root, ...rel.split("/"))));
|
|
498
|
+
const shas = new Map();
|
|
499
|
+
if (onDisk.length) {
|
|
500
|
+
const h = git(root, ["hash-object", "--stdin-paths"], onDisk.join("\n") + "\n");
|
|
501
|
+
if (h.status === 0) {
|
|
502
|
+
const out = h.stdout.split(/\r?\n/).filter(Boolean);
|
|
503
|
+
onDisk.forEach((rel, i) => shas.set(rel, out[i]));
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
const rows = [];
|
|
507
|
+
for (const rel of paths) {
|
|
508
|
+
const entry = files[rel];
|
|
509
|
+
const c = coverageOf(entry);
|
|
510
|
+
let changed = null;
|
|
511
|
+
if (entry) {
|
|
512
|
+
if (!fs.existsSync(path.join(root, ...rel.split("/")))) changed = "deleted";
|
|
513
|
+
else if (!shas.has(rel)) changed = "unknown";
|
|
514
|
+
else changed = shas.get(rel) === entry.blob ? "current" : "changed";
|
|
515
|
+
}
|
|
516
|
+
rows.push({ path: rel, ...c, changed_since_index: changed });
|
|
517
|
+
}
|
|
518
|
+
const gaps = rows.filter((r) => r.coverage !== "full").length;
|
|
519
|
+
return {
|
|
520
|
+
ok: true,
|
|
521
|
+
state: "found",
|
|
522
|
+
generation: meta.generation || 0,
|
|
523
|
+
gen_id: meta.gen_id || null,
|
|
524
|
+
rows,
|
|
525
|
+
gaps,
|
|
526
|
+
exit: 0,
|
|
527
|
+
};
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
// ── gc ──────────────────────────────────────────────────────────────────────
|
|
531
|
+
// exit 0 done · 1 none or locked
|
|
532
|
+
function graphGc(claudeDir) {
|
|
533
|
+
const p = graphPaths(claudeDir);
|
|
534
|
+
const files = readJson(p.files);
|
|
535
|
+
if (!files) return { ok: false, state: "none", reason: "no-index", exit: 1 };
|
|
536
|
+
const lock = acquireLock(p);
|
|
537
|
+
if (!lock.ok) return { ok: false, state: "unavailable", reason: "locked", holder: lock.holder, exit: 1 };
|
|
538
|
+
try {
|
|
539
|
+
const keep = new Set(Object.values(files).map((f) => f.blob));
|
|
540
|
+
let removed = 0;
|
|
541
|
+
let kept = 0;
|
|
542
|
+
let bytes = 0;
|
|
543
|
+
let shards = [];
|
|
544
|
+
try {
|
|
545
|
+
shards = fs.readdirSync(p.blobs);
|
|
546
|
+
} catch (_) {}
|
|
547
|
+
for (const shard of shards) {
|
|
548
|
+
const dir = path.join(p.blobs, shard);
|
|
549
|
+
let names = [];
|
|
550
|
+
try {
|
|
551
|
+
names = fs.readdirSync(dir);
|
|
552
|
+
} catch (_) {
|
|
553
|
+
continue;
|
|
554
|
+
}
|
|
555
|
+
for (const name of names) {
|
|
556
|
+
const blob = name.replace(/\.json$/, "");
|
|
557
|
+
const full = path.join(dir, name);
|
|
558
|
+
if (name.endsWith(".json") && keep.has(blob)) {
|
|
559
|
+
kept++;
|
|
560
|
+
continue;
|
|
561
|
+
}
|
|
562
|
+
try {
|
|
563
|
+
bytes += fs.statSync(full).size;
|
|
564
|
+
fs.rmSync(full, { force: true });
|
|
565
|
+
removed++;
|
|
566
|
+
} catch (_) {}
|
|
567
|
+
}
|
|
568
|
+
try {
|
|
569
|
+
if (!fs.readdirSync(dir).length) fs.rmdirSync(dir);
|
|
570
|
+
} catch (_) {}
|
|
571
|
+
}
|
|
572
|
+
// A temp file left by a writer that died mid-write.
|
|
573
|
+
for (const name of fs.readdirSync(p.dir)) {
|
|
574
|
+
if (name.endsWith(".tmp")) {
|
|
575
|
+
try { fs.rmSync(path.join(p.dir, name), { force: true }); } catch (_) {}
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
// Compact the notes ledger in the same locked pass: the latest note per
|
|
579
|
+
// symbol whose body still hashes the same, and nothing else.
|
|
580
|
+
const notes = require("./graph-notes.js").compactNotes(p, readJson(p.index), atomicWrite);
|
|
581
|
+
return { ok: true, state: "done", removed, kept, bytes_freed: bytes, notes, exit: 0 };
|
|
582
|
+
} finally {
|
|
583
|
+
releaseLock(p);
|
|
584
|
+
}
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
module.exports = {
|
|
588
|
+
SCHEMA,
|
|
589
|
+
ENGINE,
|
|
590
|
+
LANG_BY_EXT,
|
|
591
|
+
graphPaths,
|
|
592
|
+
detect,
|
|
593
|
+
diffFiles,
|
|
594
|
+
graphStatus,
|
|
595
|
+
graphUpdate,
|
|
596
|
+
graphCoverage,
|
|
597
|
+
coverageOf,
|
|
598
|
+
genId,
|
|
599
|
+
graphGc,
|
|
600
|
+
// shared with graph-notes.js — one lock, one atomic writer, one date format
|
|
601
|
+
acquireLock,
|
|
602
|
+
releaseLock,
|
|
603
|
+
atomicWrite,
|
|
604
|
+
stamp,
|
|
605
|
+
};
|