@1agh/maude 0.47.0 → 0.48.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,648 @@
1
+ // ddr-to-kgai — productionized importer (port of scripts/kgai-smoke/ddr2kgai.py,
2
+ // feature-kgai-ecosystem-integration Task 11). Turns this repo's
3
+ // `.ai/archive/decisions/DDR-*.md` into a kgai `{decisions:[…]}` batch and ingests it via
4
+ // `kg ingest --file`, so an existing file-based decision store migrates into the
5
+ // graph in one pass.
6
+ //
7
+ // Reached via `maude kg import [--dry-run] [--force] [--root PATH]` (DDR-062).
8
+ // Model A (DDR-189 scope decision): every decision is tagged `repo:`/`dept:` from
9
+ // `config.knowledgeGraph.scope` so one shared store hosts many repos.
10
+ //
11
+ // Edge model (mirrors kgai's "few stable domain elements shaped by many decisions"):
12
+ // - each DDR → a `decision:DDR-NNN` element + shapes an `area:<primary-tag>`
13
+ // - remaining tags → `topic:` elements, `area —TOUCHES→ topic`
14
+ // - typed cross-refs FIRST (**Supersedes:**/**Related:**/**Extends:**/**Amends:**),
15
+ // then bare `DDR-\d+` body mentions as weak `references`, deduped — so the
16
+ // graph doesn't drown in the thousands of loose name-drops (plan caution #2).
17
+
18
+ import {
19
+ existsSync,
20
+ mkdirSync,
21
+ readdirSync,
22
+ readFileSync,
23
+ renameSync,
24
+ statSync,
25
+ writeFileSync,
26
+ } from 'node:fs';
27
+ import { tmpdir } from 'node:os';
28
+ import { join } from 'node:path';
29
+ import { parseArgs } from './argv.mjs';
30
+
31
+ const REF_RANK = { references: 0, extends: 1, overrides: 2, supersedes: 3 };
32
+
33
+ /** Pull a `**Label:** value` (or list-form `- **Label:** value`) field. */
34
+ function field(text, label) {
35
+ const m = text.match(new RegExp(`^\\s*(?:[-*]\\s*)?\\*\\*${label}:\\*\\*\\s*(.+)$`, 'm'));
36
+ return m ? m[1].trim() : '';
37
+ }
38
+
39
+ /** First paragraph under a `## Header`, whitespace-collapsed, capped. */
40
+ function firstPara(text, header) {
41
+ const m = text.match(new RegExp(`^##+\\s*${header}\\s*$([\\s\\S]+?)(^##\\s|$(?![\\s\\S]))`, 'm'));
42
+ if (!m) return '';
43
+ const para = m[1]
44
+ .trim()
45
+ .split(/\n\s*\n/)[0]
46
+ .trim();
47
+ return para.replace(/\s+/g, ' ').slice(0, 400);
48
+ }
49
+
50
+ function normDate(d) {
51
+ const m = (d || '').match(/(\d{4}-\d{2}-\d{2})/);
52
+ return m ? m[1] : '';
53
+ }
54
+
55
+ function parseTags(raw) {
56
+ return raw
57
+ .split(/[,/]/)
58
+ .map((x) =>
59
+ x
60
+ .trim()
61
+ .toLowerCase()
62
+ .replace(/[^a-z0-9\- ]/g, '')
63
+ .trim()
64
+ .replace(/ +/g, '-')
65
+ )
66
+ .filter((x) => x && x !== '—' && x !== '-')
67
+ .slice(0, 6);
68
+ }
69
+
70
+ /**
71
+ * Classify cross-refs. Typed markers win over bare mentions; strongest kind per
72
+ * target is kept. Returns { 'NNN': 'supersedes'|'overrides'|'extends'|'references' }.
73
+ */
74
+ function crossRefs(text, selfNum) {
75
+ const out = {};
76
+ const reversed = [];
77
+ const put = (tgt, kind) => {
78
+ if (tgt === selfNum) return;
79
+ if (!(tgt in out) || REF_RANK[kind] > REF_RANK[out[tgt]]) out[tgt] = kind;
80
+ };
81
+ // Typed markers first (Appendix A.3).
82
+ const marker = (label, kind) => {
83
+ const re = new RegExp(`^\\s*(?:[-*]\\s*)?\\*\\*${label}:\\*\\*(.+)$`, 'gim');
84
+ for (const line of text.matchAll(re)) {
85
+ for (const d of line[1].matchAll(/DDR-(\d+)/g)) put(d[1].padStart(3, '0'), kind);
86
+ }
87
+ };
88
+ marker('Supersedes', 'supersedes');
89
+ marker('Related', 'references');
90
+ marker('Relates', 'references');
91
+ marker('Extends', 'extends');
92
+ marker('Amends', 'extends');
93
+ // Then bare mentions. NOT blindly `references`: a supersede is often declared
94
+ // in prose rather than a typed marker — e.g. DDR-006's `**Status:** Superseded
95
+ // by [DDR-191](…)`, or DDR-191's `**Related:** [DDR-006] (superseded by this
96
+ // DDR)`. Marker-only classification silently downgrades both to `references`
97
+ // and the supersede chain — the single most useful edge in the graph — is lost.
98
+ // So sniff a ±80-char window around each bare mention for intent keywords
99
+ // (restores the behavior of the scripts/kgai-smoke prototype).
100
+ for (const d of text.matchAll(/DDR-(\d+)/g)) {
101
+ const tgt = d[1].padStart(3, '0');
102
+ // A typed marker is an EXPLICIT statement of intent — never let a keyword
103
+ // that merely happens to sit within the window override it. (Caught by test:
104
+ // "We replace the earlier approach. See … DDR-003" hijacked DDR-003's
105
+ // `**Related:**` REFERENCES into OVERRIDES.)
106
+ // Also skip a target already captured as a REVERSED edge — otherwise a
107
+ // later, non-passive mention in the same file adds the opposite direction
108
+ // too and we are back to an unusable bidirectional pair.
109
+ if (tgt in out || reversed.some(([t]) => t === tgt)) continue;
110
+ const before = text.slice(Math.max(0, d.index - 80), d.index).toLowerCase();
111
+ const ctx = text.slice(Math.max(0, d.index - 80), d.index + d[0].length + 80).toLowerCase();
112
+ // DIRECTION matters, and prose states it both ways: DDR-006 says "Superseded
113
+ // by DDR-191" (the MENTION is the superseder) while DDR-191 says it
114
+ // supersedes DDR-006 (SELF is). Without this the graph gets a bidirectional
115
+ // SUPERSEDES pair and "what replaced X" becomes unanswerable. Passive voice
116
+ // right before the mention ⇒ emit the edge reversed.
117
+ const passive = /(superseded|replaced|overridden|retired|deprecated)\s+(by|in)\s*\[?$/.test(
118
+ before
119
+ );
120
+ let kind = 'references';
121
+ if (/supersed/.test(ctx)) kind = 'supersedes';
122
+ else if (/\b(override|reverse[sd]?|replaces?|retire[sd]?|deprecat)/.test(ctx))
123
+ kind = 'overrides';
124
+ else if (/\bextend|amend/.test(ctx)) kind = 'extends';
125
+ // Self-guard applies to BOTH branches: `put()` has one, and the reversed
126
+ // path needs it too — DDR-025's own body says "partially superseded by
127
+ // DDR-025", which produced a `DDR-025 ⇒ DDR-025` self-loop and made "what
128
+ // superseded DDR-025" answer itself.
129
+ if (tgt === selfNum) continue;
130
+ if (passive && kind !== 'references') reversed.push([tgt, kind]);
131
+ else put(tgt, kind);
132
+ }
133
+ return { out, reversed };
134
+ }
135
+
136
+ /** Build the kgai batch from a decisions dir. Pure. */
137
+ export function buildDdrBatch(decisionsDir, scope = {}, only = null) {
138
+ const files = readdirSync(decisionsDir)
139
+ .filter((f) => /^DDR-\d+.*\.md$/.test(f))
140
+ // `only` = incremental mode: ingest just these (substring match on the
141
+ // filename, so `DDR-191` or a full path both work). Re-ingesting an existing
142
+ // DDR is SAFE and is how you refresh one whose file changed — deterministic
143
+ // `hash(kind:name)` converges the element and props merge on re-upsert; it
144
+ // only appends one more decision event, which is the honest record of "this
145
+ // was re-recorded". Bulk re-import is what you must not do casually.
146
+ .filter(
147
+ (f) => !only || only.some((o) => f.includes(o.replace(/^.*\//, '').replace(/\.md$/, '')))
148
+ )
149
+ .sort();
150
+ const decisions = [];
151
+ const stats = { files: 0, withDate: 0, withTags: 0, crossrefs: 0, tags: new Set() };
152
+ const scopeMuts = (name) => {
153
+ const m = [];
154
+ if (scope.repo) {
155
+ m.push({ op: 'upsert_element', kind: 'repo', name: scope.repo });
156
+ m.push({
157
+ op: 'add_link',
158
+ from: `decision:${name}`,
159
+ to: `repo:${scope.repo}`,
160
+ link: 'IN_REPO',
161
+ });
162
+ }
163
+ if (scope.dept) {
164
+ m.push({ op: 'upsert_element', kind: 'dept', name: scope.dept });
165
+ m.push({
166
+ op: 'add_link',
167
+ from: `decision:${name}`,
168
+ to: `dept:${scope.dept}`,
169
+ link: 'IN_DEPT',
170
+ });
171
+ }
172
+ return m;
173
+ };
174
+
175
+ for (const f of files) {
176
+ const t = readFileSync(join(decisionsDir, f), 'utf8');
177
+ stats.files++;
178
+ const num = (f.match(/DDR-(\d+)/) || [null, '000'])[1];
179
+ const title = (t.match(/^#\s*DDR-\d+:\s*(.+)$/m) || [null, f])[1].trim();
180
+ const date = normDate(field(t, 'Date')) || normDate(field(t, 'Status'));
181
+ const tags = parseTags(field(t, 'Tags'));
182
+ // FULL body, not an excerpt. The goal is a real switch: the graph must hold
183
+ // the whole decision — alternatives, consequences, revisit-when — so nothing
184
+ // has to be read out of the .md to understand WHY. An earlier cut stored only
185
+ // the lead paragraph (~3% of the file), which quietly made the graph an index
186
+ // that could not stand on its own. The committed log grows to ~5 MB for this
187
+ // corpus; that is the correct price for self-sufficiency.
188
+ const rationale = t.trim();
189
+ const { out: refs, reversed: revRefs } = crossRefs(t, num);
190
+ if (date) stats.withDate++;
191
+ if (tags.length) {
192
+ stats.withTags++;
193
+ for (const tg of tags) stats.tags.add(tg);
194
+ }
195
+ stats.crossrefs += Object.keys(refs).length;
196
+
197
+ const primary = tags[0] || 'general';
198
+ const self = `DDR-${num}`;
199
+ const muts = [
200
+ { op: 'upsert_element', kind: 'area', name: primary, props: { last_ddr: self } },
201
+ {
202
+ op: 'upsert_element',
203
+ kind: 'decision',
204
+ name: self,
205
+ // `path` makes the graph an INDEX INTO the archive rather than a lossy
206
+ // copy of it: the migration keeps only title + the first Decision/Context
207
+ // paragraph (~3% of the file), so a hit has to be able to say WHICH file
208
+ // holds the alternatives/consequences. `.ai/archive/decisions/*.md` IS committed,
209
+ // so the pointer always resolves.
210
+ props: {
211
+ title: title.slice(0, 120),
212
+ path: `.ai/archive/decisions/${f}`,
213
+ ...(date ? { date } : {}),
214
+ ...(tags.length ? { tags: tags.join(',') } : {}),
215
+ },
216
+ },
217
+ { op: 'add_link', from: `decision:${self}`, to: `area:${primary}`, link: 'ABOUT' },
218
+ ...scopeMuts(self),
219
+ ];
220
+ for (const tg of tags.slice(1)) {
221
+ muts.push({ op: 'upsert_element', kind: 'topic', name: tg });
222
+ muts.push({ op: 'add_link', from: `area:${primary}`, to: `topic:${tg}`, link: 'TOUCHES' });
223
+ }
224
+ for (const [tgt, kind] of Object.entries(refs)) {
225
+ muts.push({ op: 'upsert_element', kind: 'decision', name: `DDR-${tgt}` });
226
+ muts.push({
227
+ op: 'add_link',
228
+ from: `decision:${self}`,
229
+ to: `decision:DDR-${tgt}`,
230
+ link: kind.toUpperCase(),
231
+ });
232
+ }
233
+ // Passive-voice mentions ("Superseded by DDR-191") — the MENTION supersedes
234
+ // SELF, so the edge points the other way.
235
+ for (const [tgt, kind] of revRefs) {
236
+ muts.push({ op: 'upsert_element', kind: 'decision', name: `DDR-${tgt}` });
237
+ muts.push({
238
+ op: 'add_link',
239
+ from: `decision:DDR-${tgt}`,
240
+ to: `decision:${self}`,
241
+ link: kind.toUpperCase(),
242
+ });
243
+ }
244
+ const d = { title, mutations: muts };
245
+ if (date) d.date = date;
246
+ if (rationale) d.rationale = rationale;
247
+ decisions.push(d);
248
+ }
249
+ stats.tags = stats.tags.size;
250
+ return { batch: { decisions }, stats };
251
+ }
252
+
253
+ /**
254
+ * `.ai/logs/**` → dated verdict/finding decisions.
255
+ *
256
+ * These are a DIFFERENT case from DDRs and the extraction reflects it: `.ai/logs/`
257
+ * is **gitignored** (the repo files it under "AI workflow runtime" beside
258
+ * device/browser/cache), so unlike `.ai/archive/decisions/*.md` these 123 files exist only
259
+ * on the machine that produced them — while 164 committed references point AT them.
260
+ * Once ingested, the graph (whose log IS committed) becomes the only inheritable
261
+ * copy, so we keep a much larger excerpt than the DDR path does (which can afford
262
+ * to be a thin index because its prose is versioned).
263
+ */
264
+ export function buildLogBatch(logsDir, scope = {}) {
265
+ const logsRel = logsDir.includes('/archive/') ? '.ai/archive/logs' : '.ai/logs';
266
+ const KINDS = {
267
+ rca: 'rca',
268
+ 'system-reviews': 'system-review',
269
+ 'code-reviews': 'code-review',
270
+ 'security-reviews': 'security-review',
271
+ 'execution-reports': 'execution-report',
272
+ '.': 'log', // loose .md sitting at the logs root (e.g. a one-off perf note)
273
+ };
274
+ const decisions = [];
275
+ const stats = { files: 0, withDate: 0, cited: 0, byKind: {} };
276
+ const scopeMuts = (name, kind) => {
277
+ const m = [];
278
+ if (scope.repo) {
279
+ m.push({ op: 'upsert_element', kind: 'repo', name: scope.repo });
280
+ m.push({
281
+ op: 'add_link',
282
+ from: `${kind}:${name}`,
283
+ to: `repo:${scope.repo}`,
284
+ link: 'IN_REPO',
285
+ });
286
+ }
287
+ if (scope.dept) {
288
+ m.push({ op: 'upsert_element', kind: 'dept', name: scope.dept });
289
+ m.push({
290
+ op: 'add_link',
291
+ from: `${kind}:${name}`,
292
+ to: `dept:${scope.dept}`,
293
+ link: 'IN_DEPT',
294
+ });
295
+ }
296
+ return m;
297
+ };
298
+
299
+ for (const [dir, kind] of Object.entries(KINDS)) {
300
+ const abs = join(logsDir, dir);
301
+ if (!existsSync(abs)) continue;
302
+ // Recurse one level into nested dirs (a `rca/archive/` holds real verdicts —
303
+ // a flat readdir silently skipped them). `.`-rooted loose files stay flat so
304
+ // the pseudo-kind doesn't re-walk every sibling category.
305
+ const listing =
306
+ dir === '.'
307
+ ? readdirSync(abs).filter((x) => x.endsWith('.md') && x !== 'README.md')
308
+ : readdirSync(abs, { withFileTypes: true }).flatMap((e) =>
309
+ e.isDirectory()
310
+ ? readdirSync(join(abs, e.name))
311
+ .filter((x) => x.endsWith('.md'))
312
+ .map((x) => `${e.name}/${x}`)
313
+ : e.name.endsWith('.md') && e.name !== 'README.md'
314
+ ? [e.name]
315
+ : []
316
+ );
317
+ for (const f of listing.sort()) {
318
+ const path = join(abs, f);
319
+ const t = readFileSync(path, 'utf8');
320
+ stats.files++;
321
+ stats.byKind[kind] = (stats.byKind[kind] ?? 0) + 1;
322
+ const slug = f.replace(/\.md$/, '').replace(/\//g, '-');
323
+ const title = (t.match(/^#\s*(.+)$/m) || [null, slug])[1].trim();
324
+ // Only 34/123 carry a `**Date:**`; the rest are untracked so git has no
325
+ // creation date either — fall back to the file's own mtime.
326
+ const date = normDate(field(t, 'Date')) || statSync(path).mtime.toISOString().slice(0, 10);
327
+ if (normDate(field(t, 'Date'))) stats.withDate++;
328
+ // Full body — these files are gitignored, so the graph is the ONLY copy;
329
+ // truncating would destroy the evidence it exists to preserve. (An earlier
330
+ // cut extracted just the Summary/Verdict/Root-cause section.)
331
+ const rationale = t.trim();
332
+
333
+ const muts = [
334
+ {
335
+ op: 'upsert_element',
336
+ kind,
337
+ name: slug,
338
+ props: { title: title.slice(0, 160), path: `${logsRel}/${dir}/${f}`, date },
339
+ },
340
+ { op: 'upsert_element', kind: 'area', name: kind },
341
+ { op: 'add_link', from: `${kind}:${slug}`, to: `area:${kind}`, link: 'ABOUT' },
342
+ ...scopeMuts(slug, kind),
343
+ ];
344
+ // Evidence edges — a review/RCA that cites a DDR is evidence ABOUT it.
345
+ const cited = new Set([...t.matchAll(/DDR-(\d+)/g)].map((m) => m[1].padStart(3, '0')));
346
+ for (const num of cited) {
347
+ muts.push({ op: 'upsert_element', kind: 'decision', name: `DDR-${num}` });
348
+ muts.push({
349
+ op: 'add_link',
350
+ from: `${kind}:${slug}`,
351
+ to: `decision:DDR-${num}`,
352
+ link: 'EVIDENCE_FOR',
353
+ });
354
+ stats.cited++;
355
+ }
356
+ decisions.push({ title, date, rationale, mutations: muts });
357
+ }
358
+ }
359
+ return { batch: { decisions }, stats };
360
+ }
361
+
362
+ /** Entry — `maude kg import`. Dispatched from cli/commands/kg.mjs verbImport. */
363
+ export async function run({ args, state, projectRoot, runKg }) {
364
+ // `args` here is already the verb's args (kg.mjs stripped the `import` token).
365
+ const { flags } = parseArgs(args, {
366
+ booleans: ['dry-run', 'design', 'force', 'no-logs', 'no-state', 'no-docs', 'archive'],
367
+ });
368
+ const only = flags.only
369
+ ? String(flags.only)
370
+ .split(',')
371
+ .map((x) => x.trim())
372
+ .filter(Boolean)
373
+ : null;
374
+ // Prefer `.ai/archive/decisions` — where the prose moved once the graph became
375
+ // the source of truth (2026-07-28) — and fall back to the classic location so
376
+ // the importer still works on a repo that hasn't archived (every other repo).
377
+ const decisionsDir =
378
+ [join(projectRoot, '.ai', 'archive', 'decisions'), join(projectRoot, '.ai', 'decisions')].find(
379
+ (d) => existsSync(d)
380
+ ) ?? join(projectRoot, '.ai', 'decisions');
381
+ if (!existsSync(decisionsDir)) {
382
+ process.stderr.write(
383
+ `maude kg import: no .ai/archive/decisions/ under ${projectRoot}. Nothing to migrate.\n`
384
+ );
385
+ return 1;
386
+ }
387
+ if (flags.design) {
388
+ process.stderr.write(
389
+ 'maude kg import --design: the .design/ importer (canvas:/ds:/footage:/reel:) is a follow-up — not yet implemented.\n'
390
+ );
391
+ return 1;
392
+ }
393
+
394
+ const marker = join(projectRoot, '.ai', '.kgai-migrated');
395
+ if (existsSync(marker) && !flags.force && !flags['dry-run'] && !only) {
396
+ process.stderr.write(
397
+ `maude kg import: already migrated (${marker} exists). Re-run with --force to ingest again (adds duplicate decision events).\n`
398
+ );
399
+ return 1;
400
+ }
401
+
402
+ const { batch, stats } = buildDdrBatch(decisionsDir, state.scope, only);
403
+
404
+ // `.ai/logs/**` rides the same import unless --no-logs. Deliberately NOT a
405
+ // separate opt-in verb: these files are gitignored, so leaving them out is how
406
+ // the RCA/security-review knowledge stays machine-local and dies on a clone.
407
+ const logsDir = [
408
+ join(projectRoot, '.ai', 'archive', 'logs'),
409
+ join(projectRoot, '.ai', 'logs'),
410
+ ].find((d) => existsSync(d));
411
+ let logStats = null;
412
+ if (!only && !flags['no-logs'] && logsDir) {
413
+ const logs = buildLogBatch(logsDir, state.scope);
414
+ logStats = logs.stats;
415
+ batch.decisions.push(...logs.batch.decisions);
416
+ }
417
+
418
+ // Narrative docs (B-class) — node + full body so `kg search "PRD"` resolves.
419
+ let docsStats = null;
420
+ if (!only && !flags['no-docs']) {
421
+ const dcs = buildDocsBatch(join(projectRoot, '.ai'), state.scope);
422
+ if (dcs.batch.decisions.length) {
423
+ docsStats = dcs.stats;
424
+ batch.decisions.push(...dcs.batch.decisions);
425
+ }
426
+ }
427
+
428
+ // STATE.md rides the same import — it is an event stream, not a document.
429
+ let stateStats = null;
430
+ if (!only && !flags['no-state']) {
431
+ // Prefer the ARCHIVED pre-migration STATE (the live one is a pointer-stub
432
+ // once the graph took over); fall back to the live file on a repo that
433
+ // hasn't migrated yet.
434
+ const archState = join(projectRoot, '.ai', 'archive', 'state');
435
+ const statePath =
436
+ [
437
+ ...(existsSync(archState) ? readdirSync(archState, { withFileTypes: true }) : [])
438
+ .filter((e) => e.isFile() && e.name.endsWith('.md'))
439
+ .map((e) => join(archState, e.name))
440
+ .sort(),
441
+ ][0] ?? join(projectRoot, '.ai', 'state', 'STATE.md');
442
+ const st = buildStateBatch(statePath, state.scope);
443
+ if (st.batch.decisions.length) {
444
+ stateStats = st.stats;
445
+ batch.decisions.push(...st.batch.decisions);
446
+ }
447
+ }
448
+
449
+ const totalMuts = batch.decisions.reduce((n, d) => n + d.mutations.length, 0);
450
+
451
+ process.stdout.write(
452
+ `maude kg import${flags['dry-run'] ? ' (dry-run)' : ''}\n` +
453
+ ` source: ${decisionsDir}\n` +
454
+ ` scope: ${JSON.stringify(state.scope)}\n` +
455
+ ` decisions: ${batch.decisions.length}\n` +
456
+ ` mutations: ${totalMuts}\n` +
457
+ ` with date: ${stats.withDate} with tags: ${stats.withTags} cross-refs: ${stats.crossrefs} distinct tags: ${stats.tags}\n` +
458
+ (logStats
459
+ ? ` logs: ${logStats.files} (${Object.entries(logStats.byKind)
460
+ .map(([k, n]) => `${k}:${n}`)
461
+ .join(' ')}), ${logStats.cited} evidence edges\n`
462
+ : ' logs: skipped (--no-logs)\n') +
463
+ (stateStats
464
+ ? ` state: ${stateStats.progress} progress blocks + ${stateStats.history} history rows (STATE.md event stream)\n`
465
+ : ' state: skipped\n') +
466
+ (docsStats ? ` docs: ${docsStats.files} narrative docs (doc: nodes)\n` : '')
467
+ );
468
+
469
+ if (flags['dry-run']) {
470
+ const sample = batch.decisions[0];
471
+ if (sample) {
472
+ process.stdout.write(
473
+ ` sample: ${sample.title}\n ${JSON.stringify(sample.mutations.slice(0, 4))}\n`
474
+ );
475
+ }
476
+ process.stdout.write(
477
+ ' (dry-run — nothing written. `.ai/archive/decisions/` is preserved as archive.)\n'
478
+ );
479
+ return 0;
480
+ }
481
+
482
+ if (!state.active) {
483
+ process.stderr.write(
484
+ 'maude kg import: kgai is inactive here (mode/off or no kg/store). Run `maude kg doctor`.\n'
485
+ );
486
+ return 1;
487
+ }
488
+
489
+ // Write the batch to a temp file and ingest via `kg ingest --file` (avoids stdin plumbing).
490
+ const tmp = join(
491
+ tmpdir(),
492
+ `kg-import-${state.scope.repo || 'repo'}-${batch.decisions.length}.json`
493
+ );
494
+ writeFileSync(tmp, JSON.stringify(batch));
495
+ const status = runKg(['ingest', '--file', tmp]);
496
+ if (status === 0) {
497
+ writeFileSync(
498
+ marker,
499
+ `migrated ${batch.decisions.length} decisions on ${new Date().toISOString()}\n`
500
+ );
501
+ process.stdout.write(
502
+ ` ✓ ingested. Marker: ${marker} (re-import needs --force). Archive kept: ${decisionsDir}\n`
503
+ );
504
+ }
505
+ return status;
506
+ }
507
+
508
+ /**
509
+ * `.ai/state/STATE.md` → dated milestone events.
510
+ *
511
+ * STATE.md is an EVENT STREAM, not a document: 88 `## Execution Progress` blocks
512
+ * and 127 `| date | phase | note |` History rows on this repo — 930 KB that grows
513
+ * forever and that nothing but a human ever reads end-to-end. Each entry is
514
+ * already dated and already describes one movement, so it maps 1:1 onto a kgai
515
+ * decision. Once ingested, STATE.md can shrink to a pointer-stub (`maude init
516
+ * --kg` writes one) and the history is queryable instead of scrollable.
517
+ */
518
+ export function buildStateBatch(statePath, scope = {}) {
519
+ const decisions = [];
520
+ const stats = { progress: 0, history: 0 };
521
+ if (!existsSync(statePath)) return { batch: { decisions }, stats };
522
+ const t = readFileSync(statePath, 'utf8');
523
+ const scopeMuts = (ref) => {
524
+ const m = [];
525
+ if (scope.repo) {
526
+ m.push({ op: 'upsert_element', kind: 'repo', name: scope.repo });
527
+ m.push({ op: 'add_link', from: ref, to: `repo:${scope.repo}`, link: 'IN_REPO' });
528
+ }
529
+ if (scope.dept) {
530
+ m.push({ op: 'upsert_element', kind: 'dept', name: scope.dept });
531
+ m.push({ op: 'add_link', from: ref, to: `dept:${scope.dept}`, link: 'IN_DEPT' });
532
+ }
533
+ return m;
534
+ };
535
+
536
+ // `## Execution Progress — <feature> — <prose>` blocks, body = up to the next `## `.
537
+ const blocks = t.split(/^## (?=Execution Progress)/m).slice(1);
538
+ for (const raw of blocks) {
539
+ const body = raw.split(/^## /m)[0].trim();
540
+ const header = body.split('\n')[0];
541
+ const feature = (header.match(/(feature-[a-z0-9.-]+|phase-[a-z0-9.-]+)/i) || [])[1] || null;
542
+ const date = (body.match(/(\d{4}-\d{2}-\d{2})/) || [])[1];
543
+ const slug = `${feature || 'progress'}-${date || String(decisions.length).padStart(3, '0')}`;
544
+ const ref = `milestone:${slug}`;
545
+ const muts = [
546
+ {
547
+ op: 'upsert_element',
548
+ kind: 'milestone',
549
+ name: slug,
550
+ props: { source: '.ai/state/STATE.md', ...(date ? { date } : {}) },
551
+ },
552
+ ...scopeMuts(ref),
553
+ ];
554
+ if (feature) {
555
+ muts.push({ op: 'upsert_element', kind: 'plan', name: feature });
556
+ muts.push({ op: 'add_link', from: ref, to: `plan:${feature}`, link: 'PROGRESS_ON' });
557
+ }
558
+ const d = {
559
+ title: header.replace(/\*\*/g, '').slice(0, 160),
560
+ rationale: body,
561
+ mutations: muts,
562
+ };
563
+ if (date) d.date = date;
564
+ decisions.push(d);
565
+ stats.progress++;
566
+ }
567
+
568
+ // History table rows: | YYYY-MM-DD | phase | note |
569
+ for (const row of t.matchAll(/^\|\s*(\d{4}-\d{2}-\d{2})\s*\|([^|]*)\|([^|]*)\|(.*)$/gm)) {
570
+ const [, date, phase, status, note] = row;
571
+ const slug = `history-${date}-${phase
572
+ .trim()
573
+ .replace(/[^a-z0-9.-]+/gi, '-')
574
+ .toLowerCase()}`.slice(0, 90);
575
+ const ref = `milestone:${slug}`;
576
+ decisions.push({
577
+ title: `${date} · ${phase.trim()} · ${status.trim()}`.slice(0, 160),
578
+ date,
579
+ rationale: `${phase.trim()} — ${status.trim()}: ${note.trim()}`.slice(0, 4000),
580
+ mutations: [
581
+ {
582
+ op: 'upsert_element',
583
+ kind: 'milestone',
584
+ name: slug,
585
+ props: { source: '.ai/state/STATE.md', date },
586
+ },
587
+ ...scopeMuts(ref),
588
+ ],
589
+ });
590
+ stats.history++;
591
+ }
592
+ return { batch: { decisions }, stats };
593
+ }
594
+
595
+ /**
596
+ * Narrative docs (`docs/`, `dev-logs/`, `context/`) → `doc:` nodes, full body.
597
+ *
598
+ * B-class in the plan's taxonomy: prose a human reads start-to-finish (PRD,
599
+ * patterns, research notes), not a dated decision. They still belong in the graph
600
+ * — the PRD is the single most-referenced piece of context in the repo and
601
+ * `kg search "PRD"` returning nothing is a hole. Indexes (`README.md`) and
602
+ * regenerable snapshots (`codebase-map.md`) stay out: D-class.
603
+ */
604
+ export function buildDocsBatch(aiDir, scope = {}) {
605
+ const SKIP = new Set(['README.md', 'INDEX.md', 'codebase-map.md']);
606
+ const decisions = [];
607
+ const stats = { files: 0 };
608
+ for (const sub of ['docs', 'dev-logs', 'context']) {
609
+ const abs = join(aiDir, sub);
610
+ if (!existsSync(abs)) continue;
611
+ for (const f of readdirSync(abs)
612
+ .filter((x) => x.endsWith('.md') && !SKIP.has(x))
613
+ .sort()) {
614
+ const t = readFileSync(join(abs, f), 'utf8');
615
+ const slug = f.replace(/\.md$/, '');
616
+ const title = (t.match(/^#\s*(.+)$/m) || [null, slug])[1].trim();
617
+ const ref = `doc:${slug}`;
618
+ const muts = [
619
+ {
620
+ op: 'upsert_element',
621
+ kind: 'doc',
622
+ name: slug,
623
+ props: { title: title.slice(0, 160), path: `.ai/${sub}/${f}`, area: sub },
624
+ },
625
+ ];
626
+ if (scope.repo) {
627
+ muts.push({ op: 'upsert_element', kind: 'repo', name: scope.repo });
628
+ muts.push({ op: 'add_link', from: ref, to: `repo:${scope.repo}`, link: 'IN_REPO' });
629
+ }
630
+ if (scope.dept) {
631
+ muts.push({ op: 'upsert_element', kind: 'dept', name: scope.dept });
632
+ muts.push({ op: 'add_link', from: ref, to: `dept:${scope.dept}`, link: 'IN_DEPT' });
633
+ }
634
+ // A doc that cites DDRs is context ABOUT them.
635
+ for (const num of new Set([...t.matchAll(/DDR-(\d+)/g)].map((m) => m[1].padStart(3, '0')))) {
636
+ muts.push({ op: 'upsert_element', kind: 'decision', name: `DDR-${num}` });
637
+ muts.push({ op: 'add_link', from: ref, to: `decision:DDR-${num}`, link: 'REFERENCES' });
638
+ }
639
+ decisions.push({
640
+ title: `Doc: ${title}`.slice(0, 160),
641
+ rationale: t.trim(),
642
+ mutations: muts,
643
+ });
644
+ stats.files++;
645
+ }
646
+ }
647
+ return { batch: { decisions }, stats };
648
+ }