akm-cli 0.9.17-alpha.4 → 0.9.17-alpha.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +167 -3
- package/dist/commands/improve/execution.js +5 -5
- package/dist/commands/improve/improve-strategies.js +3 -0
- package/dist/commands/improve/loop-stages.js +5 -4
- package/dist/commands/lint/base-linter.js +19 -5
- package/dist/commands/read/curate.js +31 -49
- package/dist/commands/read/show.js +2 -81
- package/dist/commands/sources/bundle-config-ops.js +3 -6
- package/dist/commands/sources/source-manage.js +9 -2
- package/dist/core/asset/asset-placement.js +4 -13
- package/dist/core/config/config.js +1 -1
- package/dist/core/config/schema/improve-processes.js +4 -3
- package/dist/core/config/schema/index-config.js +4 -23
- package/dist/indexer/db/graph-db.js +8 -81
- package/dist/indexer/graph/graph-extraction.js +74 -227
- package/dist/indexer/graph/graph-related.js +5 -4
- package/dist/indexer/search/db-search.js +10 -1
- package/dist/llm/feature-gate.js +0 -3
- package/dist/llm/graph-extract.js +81 -41
- package/dist/scripts/akm-migrate-node.js +4781 -4800
- package/dist/scripts/akm-migrate.js +4781 -4800
- package/dist/storage/repositories/index-entries-repository.js +5 -7
- package/dist/storage/repositories/index-schema.js +16 -34
- package/docs/reference/cli.md +9 -1
- package/docs/reference/configuration.md +12 -0
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -6
|
@@ -42,19 +42,16 @@ import path from "node:path";
|
|
|
42
42
|
import { stashDirFor } from "../../core/asset/asset-placement.js";
|
|
43
43
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
44
44
|
import { concurrentMap } from "../../core/concurrent.js";
|
|
45
|
-
import { getIndexPassConfig,
|
|
46
|
-
import { ConfigError
|
|
45
|
+
import { getIndexPassConfig, resolveBatchSize } from "../../core/config/config.js";
|
|
46
|
+
import { ConfigError } from "../../core/errors.js";
|
|
47
47
|
import { warn, warnVerbose } from "../../core/warn.js";
|
|
48
48
|
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
49
49
|
import { isProcessEnabled } from "../../llm/feature-gate.js";
|
|
50
50
|
import * as graphExtract from "../../llm/graph-extract.js";
|
|
51
51
|
import { resolveIndexPassExecution } from "../../llm/index-passes.js";
|
|
52
52
|
import { computeBodyHash, getLlmCacheEntriesByRefs, upsertLlmCacheEntry, } from "../../storage/repositories/index-llm-cache-repository.js";
|
|
53
|
-
import {
|
|
54
|
-
import { acknowledgeExtractionQueueEntry, enqueueGraphExtraction, loadStoredGraphMeta, loadStoredGraphSnapshot, peekExtractionQueue, replaceStoredGraph, } from "../db/graph-db.js";
|
|
53
|
+
import { loadStoredGraphMeta, loadStoredGraphSnapshot, replaceStoredGraph } from "../db/graph-db.js";
|
|
55
54
|
import { walkMarkdownFiles } from "../walk/walker.js";
|
|
56
|
-
/** Schema version for the persisted artifact — bumps trigger a full rebuild. */
|
|
57
|
-
export const GRAPH_FILE_SCHEMA_VERSION = GRAPH_SCHEMA_VERSION;
|
|
58
55
|
/**
|
|
59
56
|
* The frozen execution a graph call runs under: the invocation's own runner
|
|
60
57
|
* when the caller passed one (it already passed its own gates), otherwise the
|
|
@@ -94,13 +91,7 @@ const EMPTY_RESULT = {
|
|
|
94
91
|
},
|
|
95
92
|
warnings: [],
|
|
96
93
|
};
|
|
97
|
-
|
|
98
|
-
/**
|
|
99
|
-
* Max number of lazy-extraction queue rows drained per pass (#624-P3). Bounds
|
|
100
|
-
* per-run work so a large backlog is spread across runs rather than processed
|
|
101
|
-
* all at once. Generous default — the queue is normally near-empty.
|
|
102
|
-
*/
|
|
103
|
-
const GRAPH_EXTRACTION_QUEUE_DRAIN_LIMIT = 100;
|
|
94
|
+
const DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES = ["memory", "knowledge"];
|
|
104
95
|
const SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES = new Set([
|
|
105
96
|
"memory",
|
|
106
97
|
"knowledge",
|
|
@@ -128,6 +119,31 @@ export function getGraphExtractorId(config) {
|
|
|
128
119
|
})).slice(0, 16);
|
|
129
120
|
return `${GRAPH_CACHE_VARIANT_PREFIX}:${graphExtract.GRAPH_EXTRACT_PROMPT_VERSION}:${config.model}:${fingerprint}`;
|
|
130
121
|
}
|
|
122
|
+
/**
|
|
123
|
+
* GR-D16: one notice when this run's extractor differs from the one that last
|
|
124
|
+
* wrote the graph. Cached extractions are keyed by extractor, so a config
|
|
125
|
+
* change that alters it (model, batch size, included types, prompt version)
|
|
126
|
+
* re-extracts every cached file; the notice says which change and how many.
|
|
127
|
+
*/
|
|
128
|
+
function extractorChangeNotice(args) {
|
|
129
|
+
const { previous, current, files, db } = args;
|
|
130
|
+
if (!previous?.extractorId || previous.extractorId === current.extractorId)
|
|
131
|
+
return undefined;
|
|
132
|
+
const cachedUnder = (cacheVariant) => new Set(planEligibleGraphExtractions({ eligible: files, db, reEnrich: false, cacheVariant })
|
|
133
|
+
.filter((plan) => plan.kind === "cache-hit")
|
|
134
|
+
.map((plan) => plan.candidate.absPath));
|
|
135
|
+
const stillCached = cachedUnder(current.extractorId);
|
|
136
|
+
const reextracted = [...cachedUnder(previous.extractorId)].filter((file) => !stillCached.has(file)).length;
|
|
137
|
+
const changes = [
|
|
138
|
+
previous.model !== current.model ? `model ${previous.model} -> ${current.model}` : undefined,
|
|
139
|
+
previous.batchSize !== current.batchSize ? `batch size ${previous.batchSize} -> ${current.batchSize}` : undefined,
|
|
140
|
+
previous.promptVersion !== current.promptVersion
|
|
141
|
+
? `prompt ${previous.promptVersion} -> ${current.promptVersion}`
|
|
142
|
+
: undefined,
|
|
143
|
+
].filter((change) => change !== undefined);
|
|
144
|
+
return (`graph extraction: the extractor changed (${changes.join(", ") || "included asset types"}), ` +
|
|
145
|
+
`so ${reextracted} file(s) with a cached extraction will be extracted again.`);
|
|
146
|
+
}
|
|
131
147
|
function buildLowQualityWarnings(quality, telemetry) {
|
|
132
148
|
const warnings = [];
|
|
133
149
|
if (quality.consideredFiles >= 5 && quality.extractionCoverage < 0.3) {
|
|
@@ -282,21 +298,6 @@ function mergeGraphNodes(previousNodes, refreshedNodes, keptPaths) {
|
|
|
282
298
|
merged.push(...refreshedByPath.values());
|
|
283
299
|
return merged;
|
|
284
300
|
}
|
|
285
|
-
/** A previous node (validated by {@link loadGraphFile}) for this exact body, unless it failed. */
|
|
286
|
-
function reuseGraphNode(previousNodes, candidate, bodyHash) {
|
|
287
|
-
const node = previousNodes.get(candidate.absPath);
|
|
288
|
-
if (!node || node.type !== candidate.type || node.bodyHash !== bodyHash)
|
|
289
|
-
return undefined;
|
|
290
|
-
if (isFailedExtractionStatus(node.status))
|
|
291
|
-
return undefined;
|
|
292
|
-
return {
|
|
293
|
-
entities: node.entities,
|
|
294
|
-
relations: node.relations,
|
|
295
|
-
confidence: node.confidence,
|
|
296
|
-
...(node.status ? { status: node.status } : {}),
|
|
297
|
-
...(node.reason ? { reason: node.reason } : {}),
|
|
298
|
-
};
|
|
299
|
-
}
|
|
300
301
|
/**
|
|
301
302
|
* A file is a cache hit only through `llm_enrichment_cache`, whose variant is
|
|
302
303
|
* the extractor id. A stored graph node is never reused here: the graph keeps
|
|
@@ -421,69 +422,6 @@ async function extractGraphBatches(args) {
|
|
|
421
422
|
}, llmRunner.connection.concurrency ?? 1);
|
|
422
423
|
return { results, ...(configFailure ? { configFailure } : {}) };
|
|
423
424
|
}
|
|
424
|
-
function readCurrentGraphBodyHash(filePath) {
|
|
425
|
-
try {
|
|
426
|
-
const body = parseFrontmatter(fs.readFileSync(filePath, "utf8")).content.trim();
|
|
427
|
-
return body ? computeBodyHash(body) : undefined;
|
|
428
|
-
}
|
|
429
|
-
catch {
|
|
430
|
-
return undefined;
|
|
431
|
-
}
|
|
432
|
-
}
|
|
433
|
-
function planQueuedGraphExtractions(args) {
|
|
434
|
-
const { db, stashRoot, previousNodes, signal, reEnrich } = args;
|
|
435
|
-
return peekExtractionQueue(db, stashRoot, GRAPH_EXTRACTION_QUEUE_DRAIN_LIMIT).map((queued) => {
|
|
436
|
-
const base = { filePath: queued.filePath, queuedBodyHash: queued.bodyHash, priority: queued.priority };
|
|
437
|
-
if (signal?.aborted)
|
|
438
|
-
return { kind: "deferred", ...base };
|
|
439
|
-
const currentBodyHash = readCurrentGraphBodyHash(queued.filePath);
|
|
440
|
-
if (!currentBodyHash)
|
|
441
|
-
return { kind: "discard", ...base };
|
|
442
|
-
const type = inferGraphTypeForPath(stashRoot, queued.filePath) ?? "memory";
|
|
443
|
-
const hit = !reEnrich && reuseGraphNode(previousNodes, { absPath: queued.filePath, type }, currentBodyHash);
|
|
444
|
-
return { kind: hit ? "hit" : "model", ...base, currentBodyHash };
|
|
445
|
-
});
|
|
446
|
-
}
|
|
447
|
-
async function executeQueuedGraphPlans(args) {
|
|
448
|
-
const { plans, db, stashRoot, featureConfig, signal, llmRunner, onNotices } = args;
|
|
449
|
-
let graphChanged = false;
|
|
450
|
-
const acknowledgements = [];
|
|
451
|
-
for (const plan of plans) {
|
|
452
|
-
if (signal?.aborted || plan.kind === "deferred")
|
|
453
|
-
break;
|
|
454
|
-
// The body this plan settled: none for a discard (the file was gone or empty).
|
|
455
|
-
let settledBodyHash;
|
|
456
|
-
if (plan.kind === "model") {
|
|
457
|
-
const outcome = await extractGraphForSingleFileRevision(db, stashRoot, plan.filePath, {
|
|
458
|
-
config: featureConfig,
|
|
459
|
-
signal,
|
|
460
|
-
llmRunner,
|
|
461
|
-
onNotices,
|
|
462
|
-
});
|
|
463
|
-
if (!outcome.written)
|
|
464
|
-
continue;
|
|
465
|
-
graphChanged = true;
|
|
466
|
-
settledBodyHash = outcome.bodyHash;
|
|
467
|
-
}
|
|
468
|
-
else if (plan.kind === "hit") {
|
|
469
|
-
settledBodyHash = plan.currentBodyHash;
|
|
470
|
-
}
|
|
471
|
-
// A body that changed since it was planned goes back on the queue.
|
|
472
|
-
const currentBodyHash = readCurrentGraphBodyHash(plan.filePath);
|
|
473
|
-
if (currentBodyHash !== settledBodyHash) {
|
|
474
|
-
if (currentBodyHash)
|
|
475
|
-
enqueueGraphExtraction(db, stashRoot, plan.filePath, currentBodyHash, plan.priority);
|
|
476
|
-
continue;
|
|
477
|
-
}
|
|
478
|
-
acknowledgements.push({ filePath: plan.filePath, queuedBodyHash: plan.queuedBodyHash });
|
|
479
|
-
}
|
|
480
|
-
return { graphChanged, acknowledgements };
|
|
481
|
-
}
|
|
482
|
-
function acknowledgeQueuedGraphPlans(db, stashRoot, execution) {
|
|
483
|
-
for (const intent of execution.acknowledgements) {
|
|
484
|
-
acknowledgeExtractionQueueEntry(db, stashRoot, intent.filePath, intent.queuedBodyHash);
|
|
485
|
-
}
|
|
486
|
-
}
|
|
487
425
|
/**
|
|
488
426
|
* Top-level entry point. Returns a no-op result when the pass is disabled.
|
|
489
427
|
*
|
|
@@ -546,31 +484,16 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
546
484
|
return emptyResult();
|
|
547
485
|
}
|
|
548
486
|
const includeTypes = options.includeTypes ?? getGraphExtractionIncludeTypes(config);
|
|
549
|
-
|
|
550
|
-
const previousNodes = new Map(previousGraph.files.map((node) => [node.path, node]));
|
|
487
|
+
const previousGraph = loadGraphFile(primary.path, db);
|
|
551
488
|
const batchSize = resolveBatchSize(options.batchSize ?? getIndexPassConfig(config.index, "graph")?.graphExtractionBatchSize, llmRunner.connection.contextLength);
|
|
552
489
|
const extractorId = getGraphExtractorId({ model: llmRunner.connection.model, batchSize, includeTypes });
|
|
553
|
-
const queuePlans = planQueuedGraphExtractions({
|
|
554
|
-
db,
|
|
555
|
-
stashRoot: primary.path,
|
|
556
|
-
previousNodes,
|
|
557
|
-
signal,
|
|
558
|
-
reEnrich,
|
|
559
|
-
});
|
|
560
|
-
const queuedPaths = new Set(queuePlans.map((plan) => plan.filePath));
|
|
561
490
|
const scan = collectEligibleFiles(primary.path, includeTypes);
|
|
562
491
|
// The stored nodes this run keeps without touching them: every eligible file
|
|
563
|
-
// (outside candidatePaths or topN, or never reached before an abort)
|
|
564
|
-
//
|
|
565
|
-
//
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
? new Set([
|
|
569
|
-
...scan.files.map((file) => file.absPath),
|
|
570
|
-
...queuePlans.filter((plan) => plan.kind !== "discard").map((plan) => plan.filePath),
|
|
571
|
-
])
|
|
572
|
-
: undefined;
|
|
573
|
-
let eligible = scan.files.filter((candidate) => (!options.candidatePaths || options.candidatePaths.has(candidate.absPath)) && !queuedPaths.has(candidate.absPath));
|
|
492
|
+
// (outside candidatePaths or topN, or never reached before an abort). Only a
|
|
493
|
+
// node whose file left the eligible set — gone, emptied, inferred, or of a
|
|
494
|
+
// type no longer included — is dropped, and an incomplete scan drops nothing.
|
|
495
|
+
const keptPaths = scan.complete ? new Set(scan.files.map((file) => file.absPath)) : undefined;
|
|
496
|
+
let eligible = scan.files.filter((candidate) => !options.candidatePaths || options.candidatePaths.has(candidate.absPath));
|
|
574
497
|
// P2 (#624): when topN is set, rank the (already candidate-filtered)
|
|
575
498
|
// eligible set by utility_scores DESC and keep only the top-N. Unset issues
|
|
576
499
|
// no ranking query. Ranking composes WITH the candidatePaths filter:
|
|
@@ -582,25 +505,12 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
582
505
|
const eligiblePlans = planEligibleGraphExtractions({ eligible, db, reEnrich, cacheVariant: extractorId });
|
|
583
506
|
if (signal?.aborted)
|
|
584
507
|
return emptyResult();
|
|
585
|
-
// Validate exactly once iff classification found real model work.
|
|
586
|
-
//
|
|
587
|
-
//
|
|
588
|
-
if (
|
|
508
|
+
// Validate exactly once iff classification found real model work. Cache
|
|
509
|
+
// writes and graph replacement happen after this boundary, so a missing
|
|
510
|
+
// credential cannot partially mutate a batch.
|
|
511
|
+
if (eligiblePlans.some((plan) => plan.kind === "model"))
|
|
589
512
|
assertRunnerCredentials(llmRunner);
|
|
590
|
-
const queueExecution = await executeQueuedGraphPlans({
|
|
591
|
-
plans: queuePlans,
|
|
592
|
-
db,
|
|
593
|
-
stashRoot: primary.path,
|
|
594
|
-
featureConfig,
|
|
595
|
-
signal,
|
|
596
|
-
llmRunner,
|
|
597
|
-
onNotices,
|
|
598
|
-
});
|
|
599
|
-
if (queueExecution.graphChanged) {
|
|
600
|
-
previousGraph = loadGraphFile(primary.path, db);
|
|
601
|
-
}
|
|
602
513
|
if (considered === 0) {
|
|
603
|
-
acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
|
|
604
514
|
const scoped = options.candidatePaths ? ` matching ${options.candidatePaths.size} candidate path(s)` : "";
|
|
605
515
|
warnVerbose(`graph extraction: skipped because no eligible files${scoped} were found under ${primary.path}. ` +
|
|
606
516
|
`includeTypes=${includeTypes.join(",")}`);
|
|
@@ -655,6 +565,19 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
655
565
|
nonArrayBatchFailures: 0,
|
|
656
566
|
};
|
|
657
567
|
const abortState = { attempts: 0, failures: 0, aborted: false };
|
|
568
|
+
const extractorNotice = extractorChangeNotice({
|
|
569
|
+
previous: previousGraph.telemetry,
|
|
570
|
+
current: {
|
|
571
|
+
extractorId,
|
|
572
|
+
model: llmRunner.connection.model,
|
|
573
|
+
batchSize,
|
|
574
|
+
promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
|
|
575
|
+
},
|
|
576
|
+
files: scan.files,
|
|
577
|
+
db,
|
|
578
|
+
});
|
|
579
|
+
if (extractorNotice)
|
|
580
|
+
warn(extractorNotice);
|
|
658
581
|
warnVerbose(`graph extraction: starting for ${considered} eligible file(s) under ${primary.path}; ` +
|
|
659
582
|
`includeTypes=${includeTypes.join(",")}, batchSize=${batchSize}, concurrency=${llmRunner.connection.concurrency ?? 1}, ` +
|
|
660
583
|
`reEnrich=${reEnrich === true}, candidateScoped=${options.candidatePaths ? "true" : "false"}.`);
|
|
@@ -677,7 +600,6 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
677
600
|
});
|
|
678
601
|
if (configFailure)
|
|
679
602
|
throw configFailure;
|
|
680
|
-
acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
|
|
681
603
|
// A failed attempt says nothing about the file, so a stored node for it stays
|
|
682
604
|
// as it was; only a file with no stored node records the failure.
|
|
683
605
|
const storedPaths = new Set(previousGraph.files.map((node) => node.path));
|
|
@@ -690,11 +612,17 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
690
612
|
telemetry.htmlErrorCount = runtimeTelemetry.htmlErrorCount ?? 0;
|
|
691
613
|
telemetry.retryAttempts = runtimeTelemetry.retryAttempts ?? 0;
|
|
692
614
|
telemetry.nonArrayBatchFailures = runtimeTelemetry.nonArrayBatchFailures ?? 0;
|
|
615
|
+
telemetry.filteredGenericEntities = runtimeTelemetry.filteredGenericEntities ?? 0;
|
|
616
|
+
telemetry.filteredInvalidRelations = runtimeTelemetry.filteredInvalidRelations ?? 0;
|
|
617
|
+
telemetry.filteredLowConfidenceRelations = runtimeTelemetry.filteredLowConfidenceRelations ?? 0;
|
|
618
|
+
telemetry.contextBatchRetries = runtimeTelemetry.contextBatchRetries ?? 0;
|
|
693
619
|
telemetry.aborted = abortState.aborted;
|
|
694
620
|
const graph = buildGraphFile(primary.path, mergeGraphNodes(previousGraph.files, nodes, keptPaths), telemetry);
|
|
695
621
|
const written = writeGraphFile(db, graph);
|
|
696
622
|
const quality = loadStoredGraphMeta(primary.path, db)?.quality ?? EMPTY_RESULT.quality;
|
|
697
623
|
const warnings = buildLowQualityWarnings(quality, telemetry);
|
|
624
|
+
if (extractorNotice)
|
|
625
|
+
warnings.push(extractorNotice);
|
|
698
626
|
if (abortState.message)
|
|
699
627
|
warnings.push(abortState.message);
|
|
700
628
|
for (const warning of warnings)
|
|
@@ -714,14 +642,28 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
714
642
|
...(noticesByKey.size > 0 ? { notices: Object.freeze([...noticesByKey.values()]) } : {}),
|
|
715
643
|
};
|
|
716
644
|
}
|
|
717
|
-
/**
|
|
645
|
+
/**
|
|
646
|
+
* The persisted node for one extraction outcome: entities trimmed and kept
|
|
647
|
+
* once per {@link graphExtract.normalizeEntityKey} (the first form wins),
|
|
648
|
+
* relations trimmed.
|
|
649
|
+
*/
|
|
718
650
|
function toGraphNode(record, extractionRunId) {
|
|
719
651
|
const confidence = normalizeConfidence(record.confidence);
|
|
652
|
+
const entityKeys = new Set();
|
|
653
|
+
const entities = record.entities
|
|
654
|
+
.map((entity) => entity.trim())
|
|
655
|
+
.filter((entity) => {
|
|
656
|
+
const key = graphExtract.normalizeEntityKey(entity);
|
|
657
|
+
if (!key || entityKeys.has(key))
|
|
658
|
+
return false;
|
|
659
|
+
entityKeys.add(key);
|
|
660
|
+
return true;
|
|
661
|
+
});
|
|
720
662
|
return {
|
|
721
663
|
path: record.absPath,
|
|
722
664
|
type: record.type,
|
|
723
665
|
bodyHash: record.bodyHash,
|
|
724
|
-
entities
|
|
666
|
+
entities,
|
|
725
667
|
relations: record.relations
|
|
726
668
|
.map((r) => ({
|
|
727
669
|
from: r.from.trim(),
|
|
@@ -739,107 +681,12 @@ function toGraphNode(record, extractionRunId) {
|
|
|
739
681
|
/** The graph snapshot to store for `files`; its counts are derived from the stored rows on write. */
|
|
740
682
|
function buildGraphFile(stashRoot, files, telemetry) {
|
|
741
683
|
return {
|
|
742
|
-
schemaVersion: GRAPH_FILE_SCHEMA_VERSION,
|
|
743
684
|
generatedAt: new Date().toISOString(),
|
|
744
685
|
stashRoot,
|
|
745
686
|
files,
|
|
746
687
|
...(telemetry ? { telemetry } : {}),
|
|
747
688
|
};
|
|
748
689
|
}
|
|
749
|
-
/**
|
|
750
|
-
* Infer the asset type (`memory`, `knowledge`, …) for a path from the stash
|
|
751
|
-
* directory segment it lives under. Returns the matching include-type, or
|
|
752
|
-
* `undefined` when the path is not under a known graph-eligible type dir.
|
|
753
|
-
*/
|
|
754
|
-
function inferGraphTypeForPath(stashRoot, absPath) {
|
|
755
|
-
const rel = path.relative(stashRoot, absPath);
|
|
756
|
-
const firstSeg = rel.split(path.sep)[0];
|
|
757
|
-
if (!firstSeg)
|
|
758
|
-
return undefined;
|
|
759
|
-
for (const type of SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES) {
|
|
760
|
-
if (stashDirFor(type) === firstSeg)
|
|
761
|
-
return type;
|
|
762
|
-
}
|
|
763
|
-
return undefined;
|
|
764
|
-
}
|
|
765
|
-
/**
|
|
766
|
-
* #624-P3 — extract graph data for a SINGLE file and merge it into the stored
|
|
767
|
-
* graph WITHOUT clobbering other files' rows.
|
|
768
|
-
*
|
|
769
|
-
* Re-reads the body from disk at call time (the queued body_hash is NOT trusted
|
|
770
|
-
* blindly — the file may have been deleted or changed since enqueue) and skips
|
|
771
|
-
* silently (returns `false`) when the file is gone or empty. Resolves a frozen
|
|
772
|
-
* LLM execution via {@link resolveIndexPassExecution} (model-available guard:
|
|
773
|
-
* returns `false` when no provider is configured) UNLESS a caller supplies
|
|
774
|
-
* `opts.llmRunner` or `opts.llmOverride`. Those seams let a command reuse an
|
|
775
|
-
* invocation-owned selection or provide a test extractor without re-resolving.
|
|
776
|
-
*
|
|
777
|
-
* Returns `true` when a graph row was written for the file, `false` on any
|
|
778
|
-
* skip (missing file, empty body, unknown type, no model, or extraction error).
|
|
779
|
-
*/
|
|
780
|
-
async function extractGraphForSingleFileRevision(db, stashRoot, filePath, opts) {
|
|
781
|
-
try {
|
|
782
|
-
// Re-read from disk — never trust a stale queued body.
|
|
783
|
-
let raw;
|
|
784
|
-
try {
|
|
785
|
-
raw = fs.readFileSync(filePath, "utf8");
|
|
786
|
-
}
|
|
787
|
-
catch {
|
|
788
|
-
return { written: false }; // file gone / unreadable → silent skip
|
|
789
|
-
}
|
|
790
|
-
const parsed = parseFrontmatter(raw);
|
|
791
|
-
const body = parsed.content.trim();
|
|
792
|
-
if (!body)
|
|
793
|
-
return { written: false };
|
|
794
|
-
const type = inferGraphTypeForPath(stashRoot, filePath) ?? "memory";
|
|
795
|
-
const effectiveHash = computeBodyHash(body);
|
|
796
|
-
// Extract — via the injected seam, or the real per-asset path.
|
|
797
|
-
let extraction;
|
|
798
|
-
if (opts?.llmOverride) {
|
|
799
|
-
extraction = await opts.llmOverride(body);
|
|
800
|
-
}
|
|
801
|
-
else {
|
|
802
|
-
const selection = selectGraphExecution(opts ?? {}, opts?.config ?? loadConfig());
|
|
803
|
-
if (!selection)
|
|
804
|
-
return { written: false };
|
|
805
|
-
opts?.onNotices?.(selection.execution.notices);
|
|
806
|
-
const llmRunner = selection.execution.runner;
|
|
807
|
-
if (!llmRunner)
|
|
808
|
-
return { written: false }; // model-available guard
|
|
809
|
-
extraction = await graphExtract.extractGraphFromBody(llmRunner, body, opts?.signal, selection.featureConfig, undefined, { ...(opts?.onNotices ? { onNotices: opts.onNotices } : {}) });
|
|
810
|
-
}
|
|
811
|
-
// A single-file refresh records only entities, relations and confidence;
|
|
812
|
-
// its status follows from the entities that survive trimming.
|
|
813
|
-
const entities = [...new Set(extraction.entities.map((e) => e.trim()).filter(Boolean))];
|
|
814
|
-
const node = toGraphNode({
|
|
815
|
-
absPath: filePath,
|
|
816
|
-
type,
|
|
817
|
-
bodyHash: effectiveHash,
|
|
818
|
-
entities,
|
|
819
|
-
relations: extraction.relations,
|
|
820
|
-
...(extraction.confidence !== undefined ? { confidence: extraction.confidence } : {}),
|
|
821
|
-
}, crypto.randomUUID());
|
|
822
|
-
// Merge with the previously-stored nodes, replacing JUST this path so other
|
|
823
|
-
// files' rows are preserved (and graph_meta counts refresh).
|
|
824
|
-
const previousGraph = loadGraphFile(stashRoot, db);
|
|
825
|
-
const graph = buildGraphFile(stashRoot, mergeGraphNodes(previousGraph.files, [node]), previousGraph.telemetry);
|
|
826
|
-
return writeGraphFile(db, graph) ? { written: true, bodyHash: effectiveHash } : { written: false };
|
|
827
|
-
}
|
|
828
|
-
catch (err) {
|
|
829
|
-
if (err instanceof ConfigError)
|
|
830
|
-
throw err;
|
|
831
|
-
rethrowIfTestIsolationError(err);
|
|
832
|
-
// A genuine extraction/write failure, distinct from the deliberate
|
|
833
|
-
// "nothing to do" skips above (missing file, empty body, no model). Warn
|
|
834
|
-
// so it is visible instead of looking identical to a no-op skip; the
|
|
835
|
-
// entry stays queued and is retried on the next pass.
|
|
836
|
-
warn(`graph extraction: failed to extract graph for ${filePath}: ${err instanceof Error ? err.message : String(err)}`);
|
|
837
|
-
return { written: false };
|
|
838
|
-
}
|
|
839
|
-
}
|
|
840
|
-
export async function extractGraphForSingleFile(db, stashRoot, filePath, opts) {
|
|
841
|
-
return (await extractGraphForSingleFileRevision(db, stashRoot, filePath, opts)).written;
|
|
842
|
-
}
|
|
843
690
|
// ── Eligible-file detection ─────────────────────────────────────────────────
|
|
844
691
|
/**
|
|
845
692
|
* Rank eligible graph-extraction candidates by their entry `utility_scores`,
|
|
@@ -31,13 +31,14 @@ export function listRelatedPathsForFile(stashRoot, filePath, limit = 5, db) {
|
|
|
31
31
|
if (row === undefined)
|
|
32
32
|
return [];
|
|
33
33
|
const effectiveLimit = Math.max(1, limit);
|
|
34
|
-
//
|
|
35
|
-
// rows for `filePath`; candidates are any OTHER file_path in the
|
|
36
|
-
// shares a normalized entity.
|
|
34
|
+
// Distinct shared entities per candidate file_path. The target's entities
|
|
35
|
+
// are the rows for `filePath`; candidates are any OTHER file_path in the
|
|
36
|
+
// stash that shares a normalized entity. Counting distinct keys keeps two
|
|
37
|
+
// stored forms of one entity (rows older extractors wrote) from counting twice.
|
|
37
38
|
const candidateRows = db
|
|
38
39
|
.prepare(`SELECT gf.file_path AS file_path,
|
|
39
40
|
gf.file_type AS file_type,
|
|
40
|
-
COUNT(
|
|
41
|
+
COUNT(DISTINCT e.entity_norm) AS shared
|
|
41
42
|
FROM graph_file_entities target
|
|
42
43
|
JOIN graph_file_entities e
|
|
43
44
|
ON e.stash_root = target.stash_root
|
|
@@ -239,11 +239,14 @@ async function searchDatabase(db, input) {
|
|
|
239
239
|
}
|
|
240
240
|
/**
|
|
241
241
|
* Walk the fused list in order, loading entries a batch at a time, and keep
|
|
242
|
-
* the first `limit` that survive path deduplication and the filters.
|
|
242
|
+
* the first `limit` that survive path deduplication and the filters. An entry
|
|
243
|
+
* whose indexed content is identical to a kept one's (the same body saved
|
|
244
|
+
* under another name or in another bundle) is a duplicate and is skipped.
|
|
243
245
|
*/
|
|
244
246
|
function selectFusedEntries(db, fused, limit, filterOptions) {
|
|
245
247
|
const selected = [];
|
|
246
248
|
const seenPaths = new Set();
|
|
249
|
+
const seenContent = new Set();
|
|
247
250
|
const batchSize = Math.max(limit * 2, 20);
|
|
248
251
|
for (let offset = 0; offset < fused.length && selected.length < limit; offset += batchSize) {
|
|
249
252
|
const batch = [];
|
|
@@ -256,6 +259,12 @@ function selectFusedEntries(db, fused, limit, filterOptions) {
|
|
|
256
259
|
}
|
|
257
260
|
for (const kept of applyEntryFilters(batch, filterOptions)) {
|
|
258
261
|
const { candidate, ...row } = kept;
|
|
262
|
+
const content = row.entry.content?.replace(/\s+/g, " ").trim();
|
|
263
|
+
if (content) {
|
|
264
|
+
if (seenContent.has(content))
|
|
265
|
+
continue;
|
|
266
|
+
seenContent.add(content);
|
|
267
|
+
}
|
|
259
268
|
selected.push({ candidate, row });
|
|
260
269
|
}
|
|
261
270
|
}
|
package/dist/llm/feature-gate.js
CHANGED
|
@@ -79,9 +79,6 @@ export function isProcessEnabled(section, processName, config) {
|
|
|
79
79
|
return false;
|
|
80
80
|
// Index passes are first-class 0.9 entries.
|
|
81
81
|
if (section === "index") {
|
|
82
|
-
if (processName === "metadata_enhance" || processName === "metadataEnhance") {
|
|
83
|
-
return config.index?.metadataEnhance?.enabled ?? true;
|
|
84
|
-
}
|
|
85
82
|
if (processName === "memory_inference" || processName === "memoryInference") {
|
|
86
83
|
return isLlmFeatureEnabled(config, "memory_inference");
|
|
87
84
|
}
|