akm-cli 0.9.17-alpha.4 → 0.9.17-alpha.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -42,19 +42,16 @@ import path from "node:path";
42
42
  import { stashDirFor } from "../../core/asset/asset-placement.js";
43
43
  import { parseFrontmatter } from "../../core/asset/frontmatter.js";
44
44
  import { concurrentMap } from "../../core/concurrent.js";
45
- import { getIndexPassConfig, loadConfig, resolveBatchSize } from "../../core/config/config.js";
46
- import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
45
+ import { getIndexPassConfig, resolveBatchSize } from "../../core/config/config.js";
46
+ import { ConfigError } from "../../core/errors.js";
47
47
  import { warn, warnVerbose } from "../../core/warn.js";
48
48
  import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
49
49
  import { isProcessEnabled } from "../../llm/feature-gate.js";
50
50
  import * as graphExtract from "../../llm/graph-extract.js";
51
51
  import { resolveIndexPassExecution } from "../../llm/index-passes.js";
52
52
  import { computeBodyHash, getLlmCacheEntriesByRefs, upsertLlmCacheEntry, } from "../../storage/repositories/index-llm-cache-repository.js";
53
- import { GRAPH_SCHEMA_VERSION } from "../../storage/repositories/index-schema.js";
54
- import { acknowledgeExtractionQueueEntry, enqueueGraphExtraction, loadStoredGraphMeta, loadStoredGraphSnapshot, peekExtractionQueue, replaceStoredGraph, } from "../db/graph-db.js";
53
+ import { loadStoredGraphMeta, loadStoredGraphSnapshot, replaceStoredGraph } from "../db/graph-db.js";
55
54
  import { walkMarkdownFiles } from "../walk/walker.js";
56
- /** Schema version for the persisted artifact — bumps trigger a full rebuild. */
57
- export const GRAPH_FILE_SCHEMA_VERSION = GRAPH_SCHEMA_VERSION;
58
55
  /**
59
56
  * The frozen execution a graph call runs under: the invocation's own runner
60
57
  * when the caller passed one (it already passed its own gates), otherwise the
@@ -94,13 +91,7 @@ const EMPTY_RESULT = {
94
91
  },
95
92
  warnings: [],
96
93
  };
97
- export const DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES = ["memory", "knowledge"];
98
- /**
99
- * Max number of lazy-extraction queue rows drained per pass (#624-P3). Bounds
100
- * per-run work so a large backlog is spread across runs rather than processed
101
- * all at once. Generous default — the queue is normally near-empty.
102
- */
103
- const GRAPH_EXTRACTION_QUEUE_DRAIN_LIMIT = 100;
94
+ const DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES = ["memory", "knowledge"];
104
95
  const SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES = new Set([
105
96
  "memory",
106
97
  "knowledge",
@@ -128,6 +119,31 @@ export function getGraphExtractorId(config) {
128
119
  })).slice(0, 16);
129
120
  return `${GRAPH_CACHE_VARIANT_PREFIX}:${graphExtract.GRAPH_EXTRACT_PROMPT_VERSION}:${config.model}:${fingerprint}`;
130
121
  }
122
+ /**
123
+ * GR-D16: one notice when this run's extractor differs from the one that last
124
+ * wrote the graph. Cached extractions are keyed by extractor, so a config
125
+ * change that alters it (model, batch size, included types, prompt version)
126
+ * re-extracts every cached file; the notice says which change and how many.
127
+ */
128
+ function extractorChangeNotice(args) {
129
+ const { previous, current, files, db } = args;
130
+ if (!previous?.extractorId || previous.extractorId === current.extractorId)
131
+ return undefined;
132
+ const cachedUnder = (cacheVariant) => new Set(planEligibleGraphExtractions({ eligible: files, db, reEnrich: false, cacheVariant })
133
+ .filter((plan) => plan.kind === "cache-hit")
134
+ .map((plan) => plan.candidate.absPath));
135
+ const stillCached = cachedUnder(current.extractorId);
136
+ const reextracted = [...cachedUnder(previous.extractorId)].filter((file) => !stillCached.has(file)).length;
137
+ const changes = [
138
+ previous.model !== current.model ? `model ${previous.model} -> ${current.model}` : undefined,
139
+ previous.batchSize !== current.batchSize ? `batch size ${previous.batchSize} -> ${current.batchSize}` : undefined,
140
+ previous.promptVersion !== current.promptVersion
141
+ ? `prompt ${previous.promptVersion} -> ${current.promptVersion}`
142
+ : undefined,
143
+ ].filter((change) => change !== undefined);
144
+ return (`graph extraction: the extractor changed (${changes.join(", ") || "included asset types"}), ` +
145
+ `so ${reextracted} file(s) with a cached extraction will be extracted again.`);
146
+ }
131
147
  function buildLowQualityWarnings(quality, telemetry) {
132
148
  const warnings = [];
133
149
  if (quality.consideredFiles >= 5 && quality.extractionCoverage < 0.3) {
@@ -282,21 +298,6 @@ function mergeGraphNodes(previousNodes, refreshedNodes, keptPaths) {
282
298
  merged.push(...refreshedByPath.values());
283
299
  return merged;
284
300
  }
285
- /** A previous node (validated by {@link loadGraphFile}) for this exact body, unless it failed. */
286
- function reuseGraphNode(previousNodes, candidate, bodyHash) {
287
- const node = previousNodes.get(candidate.absPath);
288
- if (!node || node.type !== candidate.type || node.bodyHash !== bodyHash)
289
- return undefined;
290
- if (isFailedExtractionStatus(node.status))
291
- return undefined;
292
- return {
293
- entities: node.entities,
294
- relations: node.relations,
295
- confidence: node.confidence,
296
- ...(node.status ? { status: node.status } : {}),
297
- ...(node.reason ? { reason: node.reason } : {}),
298
- };
299
- }
300
301
  /**
301
302
  * A file is a cache hit only through `llm_enrichment_cache`, whose variant is
302
303
  * the extractor id. A stored graph node is never reused here: the graph keeps
@@ -421,69 +422,6 @@ async function extractGraphBatches(args) {
421
422
  }, llmRunner.connection.concurrency ?? 1);
422
423
  return { results, ...(configFailure ? { configFailure } : {}) };
423
424
  }
424
- function readCurrentGraphBodyHash(filePath) {
425
- try {
426
- const body = parseFrontmatter(fs.readFileSync(filePath, "utf8")).content.trim();
427
- return body ? computeBodyHash(body) : undefined;
428
- }
429
- catch {
430
- return undefined;
431
- }
432
- }
433
- function planQueuedGraphExtractions(args) {
434
- const { db, stashRoot, previousNodes, signal, reEnrich } = args;
435
- return peekExtractionQueue(db, stashRoot, GRAPH_EXTRACTION_QUEUE_DRAIN_LIMIT).map((queued) => {
436
- const base = { filePath: queued.filePath, queuedBodyHash: queued.bodyHash, priority: queued.priority };
437
- if (signal?.aborted)
438
- return { kind: "deferred", ...base };
439
- const currentBodyHash = readCurrentGraphBodyHash(queued.filePath);
440
- if (!currentBodyHash)
441
- return { kind: "discard", ...base };
442
- const type = inferGraphTypeForPath(stashRoot, queued.filePath) ?? "memory";
443
- const hit = !reEnrich && reuseGraphNode(previousNodes, { absPath: queued.filePath, type }, currentBodyHash);
444
- return { kind: hit ? "hit" : "model", ...base, currentBodyHash };
445
- });
446
- }
447
- async function executeQueuedGraphPlans(args) {
448
- const { plans, db, stashRoot, featureConfig, signal, llmRunner, onNotices } = args;
449
- let graphChanged = false;
450
- const acknowledgements = [];
451
- for (const plan of plans) {
452
- if (signal?.aborted || plan.kind === "deferred")
453
- break;
454
- // The body this plan settled: none for a discard (the file was gone or empty).
455
- let settledBodyHash;
456
- if (plan.kind === "model") {
457
- const outcome = await extractGraphForSingleFileRevision(db, stashRoot, plan.filePath, {
458
- config: featureConfig,
459
- signal,
460
- llmRunner,
461
- onNotices,
462
- });
463
- if (!outcome.written)
464
- continue;
465
- graphChanged = true;
466
- settledBodyHash = outcome.bodyHash;
467
- }
468
- else if (plan.kind === "hit") {
469
- settledBodyHash = plan.currentBodyHash;
470
- }
471
- // A body that changed since it was planned goes back on the queue.
472
- const currentBodyHash = readCurrentGraphBodyHash(plan.filePath);
473
- if (currentBodyHash !== settledBodyHash) {
474
- if (currentBodyHash)
475
- enqueueGraphExtraction(db, stashRoot, plan.filePath, currentBodyHash, plan.priority);
476
- continue;
477
- }
478
- acknowledgements.push({ filePath: plan.filePath, queuedBodyHash: plan.queuedBodyHash });
479
- }
480
- return { graphChanged, acknowledgements };
481
- }
482
- function acknowledgeQueuedGraphPlans(db, stashRoot, execution) {
483
- for (const intent of execution.acknowledgements) {
484
- acknowledgeExtractionQueueEntry(db, stashRoot, intent.filePath, intent.queuedBodyHash);
485
- }
486
- }
487
425
  /**
488
426
  * Top-level entry point. Returns a no-op result when the pass is disabled.
489
427
  *
@@ -546,31 +484,16 @@ export async function runGraphExtractionPass(ctx) {
546
484
  return emptyResult();
547
485
  }
548
486
  const includeTypes = options.includeTypes ?? getGraphExtractionIncludeTypes(config);
549
- let previousGraph = loadGraphFile(primary.path, db);
550
- const previousNodes = new Map(previousGraph.files.map((node) => [node.path, node]));
487
+ const previousGraph = loadGraphFile(primary.path, db);
551
488
  const batchSize = resolveBatchSize(options.batchSize ?? getIndexPassConfig(config.index, "graph")?.graphExtractionBatchSize, llmRunner.connection.contextLength);
552
489
  const extractorId = getGraphExtractorId({ model: llmRunner.connection.model, batchSize, includeTypes });
553
- const queuePlans = planQueuedGraphExtractions({
554
- db,
555
- stashRoot: primary.path,
556
- previousNodes,
557
- signal,
558
- reEnrich,
559
- });
560
- const queuedPaths = new Set(queuePlans.map((plan) => plan.filePath));
561
490
  const scan = collectEligibleFiles(primary.path, includeTypes);
562
491
  // The stored nodes this run keeps without touching them: every eligible file
563
- // (outside candidatePaths or topN, or never reached before an abort) and every
564
- // file the queue handled. Only a node whose file left the eligible set — gone,
565
- // emptied, inferred, or of a type no longer included — is dropped, and an
566
- // incomplete scan drops nothing.
567
- const keptPaths = scan.complete
568
- ? new Set([
569
- ...scan.files.map((file) => file.absPath),
570
- ...queuePlans.filter((plan) => plan.kind !== "discard").map((plan) => plan.filePath),
571
- ])
572
- : undefined;
573
- let eligible = scan.files.filter((candidate) => (!options.candidatePaths || options.candidatePaths.has(candidate.absPath)) && !queuedPaths.has(candidate.absPath));
492
+ // (outside candidatePaths or topN, or never reached before an abort). Only a
493
+ // node whose file left the eligible set — gone, emptied, inferred, or of a
494
+ // type no longer included — is dropped, and an incomplete scan drops nothing.
495
+ const keptPaths = scan.complete ? new Set(scan.files.map((file) => file.absPath)) : undefined;
496
+ let eligible = scan.files.filter((candidate) => !options.candidatePaths || options.candidatePaths.has(candidate.absPath));
574
497
  // P2 (#624): when topN is set, rank the (already candidate-filtered)
575
498
  // eligible set by utility_scores DESC and keep only the top-N. Unset issues
576
499
  // no ranking query. Ranking composes WITH the candidatePaths filter:
@@ -582,25 +505,12 @@ export async function runGraphExtractionPass(ctx) {
582
505
  const eligiblePlans = planEligibleGraphExtractions({ eligible, db, reEnrich, cacheVariant: extractorId });
583
506
  if (signal?.aborted)
584
507
  return emptyResult();
585
- // Validate exactly once iff classification found real model work. Queue
586
- // acknowledgements, cache writes, and graph replacement all happen after
587
- // this boundary, so a missing credential cannot partially mutate a batch.
588
- if ([...queuePlans, ...eligiblePlans].some((plan) => plan.kind === "model"))
508
+ // Validate exactly once iff classification found real model work. Cache
509
+ // writes and graph replacement happen after this boundary, so a missing
510
+ // credential cannot partially mutate a batch.
511
+ if (eligiblePlans.some((plan) => plan.kind === "model"))
589
512
  assertRunnerCredentials(llmRunner);
590
- const queueExecution = await executeQueuedGraphPlans({
591
- plans: queuePlans,
592
- db,
593
- stashRoot: primary.path,
594
- featureConfig,
595
- signal,
596
- llmRunner,
597
- onNotices,
598
- });
599
- if (queueExecution.graphChanged) {
600
- previousGraph = loadGraphFile(primary.path, db);
601
- }
602
513
  if (considered === 0) {
603
- acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
604
514
  const scoped = options.candidatePaths ? ` matching ${options.candidatePaths.size} candidate path(s)` : "";
605
515
  warnVerbose(`graph extraction: skipped because no eligible files${scoped} were found under ${primary.path}. ` +
606
516
  `includeTypes=${includeTypes.join(",")}`);
@@ -655,6 +565,19 @@ export async function runGraphExtractionPass(ctx) {
655
565
  nonArrayBatchFailures: 0,
656
566
  };
657
567
  const abortState = { attempts: 0, failures: 0, aborted: false };
568
+ const extractorNotice = extractorChangeNotice({
569
+ previous: previousGraph.telemetry,
570
+ current: {
571
+ extractorId,
572
+ model: llmRunner.connection.model,
573
+ batchSize,
574
+ promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
575
+ },
576
+ files: scan.files,
577
+ db,
578
+ });
579
+ if (extractorNotice)
580
+ warn(extractorNotice);
658
581
  warnVerbose(`graph extraction: starting for ${considered} eligible file(s) under ${primary.path}; ` +
659
582
  `includeTypes=${includeTypes.join(",")}, batchSize=${batchSize}, concurrency=${llmRunner.connection.concurrency ?? 1}, ` +
660
583
  `reEnrich=${reEnrich === true}, candidateScoped=${options.candidatePaths ? "true" : "false"}.`);
@@ -677,7 +600,6 @@ export async function runGraphExtractionPass(ctx) {
677
600
  });
678
601
  if (configFailure)
679
602
  throw configFailure;
680
- acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
681
603
  // A failed attempt says nothing about the file, so a stored node for it stays
682
604
  // as it was; only a file with no stored node records the failure.
683
605
  const storedPaths = new Set(previousGraph.files.map((node) => node.path));
@@ -690,11 +612,17 @@ export async function runGraphExtractionPass(ctx) {
690
612
  telemetry.htmlErrorCount = runtimeTelemetry.htmlErrorCount ?? 0;
691
613
  telemetry.retryAttempts = runtimeTelemetry.retryAttempts ?? 0;
692
614
  telemetry.nonArrayBatchFailures = runtimeTelemetry.nonArrayBatchFailures ?? 0;
615
+ telemetry.filteredGenericEntities = runtimeTelemetry.filteredGenericEntities ?? 0;
616
+ telemetry.filteredInvalidRelations = runtimeTelemetry.filteredInvalidRelations ?? 0;
617
+ telemetry.filteredLowConfidenceRelations = runtimeTelemetry.filteredLowConfidenceRelations ?? 0;
618
+ telemetry.contextBatchRetries = runtimeTelemetry.contextBatchRetries ?? 0;
693
619
  telemetry.aborted = abortState.aborted;
694
620
  const graph = buildGraphFile(primary.path, mergeGraphNodes(previousGraph.files, nodes, keptPaths), telemetry);
695
621
  const written = writeGraphFile(db, graph);
696
622
  const quality = loadStoredGraphMeta(primary.path, db)?.quality ?? EMPTY_RESULT.quality;
697
623
  const warnings = buildLowQualityWarnings(quality, telemetry);
624
+ if (extractorNotice)
625
+ warnings.push(extractorNotice);
698
626
  if (abortState.message)
699
627
  warnings.push(abortState.message);
700
628
  for (const warning of warnings)
@@ -714,14 +642,28 @@ export async function runGraphExtractionPass(ctx) {
714
642
  ...(noticesByKey.size > 0 ? { notices: Object.freeze([...noticesByKey.values()]) } : {}),
715
643
  };
716
644
  }
717
- /** The persisted node for one extraction outcome (entities and relations trimmed, deduplicated). */
645
+ /**
646
+ * The persisted node for one extraction outcome: entities trimmed and kept
647
+ * once per {@link graphExtract.normalizeEntityKey} (the first form wins),
648
+ * relations trimmed.
649
+ */
718
650
  function toGraphNode(record, extractionRunId) {
719
651
  const confidence = normalizeConfidence(record.confidence);
652
+ const entityKeys = new Set();
653
+ const entities = record.entities
654
+ .map((entity) => entity.trim())
655
+ .filter((entity) => {
656
+ const key = graphExtract.normalizeEntityKey(entity);
657
+ if (!key || entityKeys.has(key))
658
+ return false;
659
+ entityKeys.add(key);
660
+ return true;
661
+ });
720
662
  return {
721
663
  path: record.absPath,
722
664
  type: record.type,
723
665
  bodyHash: record.bodyHash,
724
- entities: [...new Set(record.entities.map((entity) => entity.trim()).filter(Boolean))],
666
+ entities,
725
667
  relations: record.relations
726
668
  .map((r) => ({
727
669
  from: r.from.trim(),
@@ -739,107 +681,12 @@ function toGraphNode(record, extractionRunId) {
739
681
  /** The graph snapshot to store for `files`; its counts are derived from the stored rows on write. */
740
682
  function buildGraphFile(stashRoot, files, telemetry) {
741
683
  return {
742
- schemaVersion: GRAPH_FILE_SCHEMA_VERSION,
743
684
  generatedAt: new Date().toISOString(),
744
685
  stashRoot,
745
686
  files,
746
687
  ...(telemetry ? { telemetry } : {}),
747
688
  };
748
689
  }
749
- /**
750
- * Infer the asset type (`memory`, `knowledge`, …) for a path from the stash
751
- * directory segment it lives under. Returns the matching include-type, or
752
- * `undefined` when the path is not under a known graph-eligible type dir.
753
- */
754
- function inferGraphTypeForPath(stashRoot, absPath) {
755
- const rel = path.relative(stashRoot, absPath);
756
- const firstSeg = rel.split(path.sep)[0];
757
- if (!firstSeg)
758
- return undefined;
759
- for (const type of SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES) {
760
- if (stashDirFor(type) === firstSeg)
761
- return type;
762
- }
763
- return undefined;
764
- }
765
- /**
766
- * #624-P3 — extract graph data for a SINGLE file and merge it into the stored
767
- * graph WITHOUT clobbering other files' rows.
768
- *
769
- * Re-reads the body from disk at call time (the queued body_hash is NOT trusted
770
- * blindly — the file may have been deleted or changed since enqueue) and skips
771
- * silently (returns `false`) when the file is gone or empty. Resolves a frozen
772
- * LLM execution via {@link resolveIndexPassExecution} (model-available guard:
773
- * returns `false` when no provider is configured) UNLESS a caller supplies
774
- * `opts.llmRunner` or `opts.llmOverride`. Those seams let a command reuse an
775
- * invocation-owned selection or provide a test extractor without re-resolving.
776
- *
777
- * Returns `true` when a graph row was written for the file, `false` on any
778
- * skip (missing file, empty body, unknown type, no model, or extraction error).
779
- */
780
- async function extractGraphForSingleFileRevision(db, stashRoot, filePath, opts) {
781
- try {
782
- // Re-read from disk — never trust a stale queued body.
783
- let raw;
784
- try {
785
- raw = fs.readFileSync(filePath, "utf8");
786
- }
787
- catch {
788
- return { written: false }; // file gone / unreadable → silent skip
789
- }
790
- const parsed = parseFrontmatter(raw);
791
- const body = parsed.content.trim();
792
- if (!body)
793
- return { written: false };
794
- const type = inferGraphTypeForPath(stashRoot, filePath) ?? "memory";
795
- const effectiveHash = computeBodyHash(body);
796
- // Extract — via the injected seam, or the real per-asset path.
797
- let extraction;
798
- if (opts?.llmOverride) {
799
- extraction = await opts.llmOverride(body);
800
- }
801
- else {
802
- const selection = selectGraphExecution(opts ?? {}, opts?.config ?? loadConfig());
803
- if (!selection)
804
- return { written: false };
805
- opts?.onNotices?.(selection.execution.notices);
806
- const llmRunner = selection.execution.runner;
807
- if (!llmRunner)
808
- return { written: false }; // model-available guard
809
- extraction = await graphExtract.extractGraphFromBody(llmRunner, body, opts?.signal, selection.featureConfig, undefined, { ...(opts?.onNotices ? { onNotices: opts.onNotices } : {}) });
810
- }
811
- // A single-file refresh records only entities, relations and confidence;
812
- // its status follows from the entities that survive trimming.
813
- const entities = [...new Set(extraction.entities.map((e) => e.trim()).filter(Boolean))];
814
- const node = toGraphNode({
815
- absPath: filePath,
816
- type,
817
- bodyHash: effectiveHash,
818
- entities,
819
- relations: extraction.relations,
820
- ...(extraction.confidence !== undefined ? { confidence: extraction.confidence } : {}),
821
- }, crypto.randomUUID());
822
- // Merge with the previously-stored nodes, replacing JUST this path so other
823
- // files' rows are preserved (and graph_meta counts refresh).
824
- const previousGraph = loadGraphFile(stashRoot, db);
825
- const graph = buildGraphFile(stashRoot, mergeGraphNodes(previousGraph.files, [node]), previousGraph.telemetry);
826
- return writeGraphFile(db, graph) ? { written: true, bodyHash: effectiveHash } : { written: false };
827
- }
828
- catch (err) {
829
- if (err instanceof ConfigError)
830
- throw err;
831
- rethrowIfTestIsolationError(err);
832
- // A genuine extraction/write failure, distinct from the deliberate
833
- // "nothing to do" skips above (missing file, empty body, no model). Warn
834
- // so it is visible instead of looking identical to a no-op skip; the
835
- // entry stays queued and is retried on the next pass.
836
- warn(`graph extraction: failed to extract graph for ${filePath}: ${err instanceof Error ? err.message : String(err)}`);
837
- return { written: false };
838
- }
839
- }
840
- export async function extractGraphForSingleFile(db, stashRoot, filePath, opts) {
841
- return (await extractGraphForSingleFileRevision(db, stashRoot, filePath, opts)).written;
842
- }
843
690
  // ── Eligible-file detection ─────────────────────────────────────────────────
844
691
  /**
845
692
  * Rank eligible graph-extraction candidates by their entry `utility_scores`,
@@ -31,13 +31,14 @@ export function listRelatedPathsForFile(stashRoot, filePath, limit = 5, db) {
31
31
  if (row === undefined)
32
32
  return [];
33
33
  const effectiveLimit = Math.max(1, limit);
34
- // Shared-entity count per candidate file_path. The target's entities are the
35
- // rows for `filePath`; candidates are any OTHER file_path in the stash that
36
- // shares a normalized entity.
34
+ // Distinct shared entities per candidate file_path. The target's entities
35
+ // are the rows for `filePath`; candidates are any OTHER file_path in the
36
+ // stash that shares a normalized entity. Counting distinct keys keeps two
37
+ // stored forms of one entity (rows older extractors wrote) from counting twice.
37
38
  const candidateRows = db
38
39
  .prepare(`SELECT gf.file_path AS file_path,
39
40
  gf.file_type AS file_type,
40
- COUNT(*) AS shared
41
+ COUNT(DISTINCT e.entity_norm) AS shared
41
42
  FROM graph_file_entities target
42
43
  JOIN graph_file_entities e
43
44
  ON e.stash_root = target.stash_root
@@ -239,11 +239,14 @@ async function searchDatabase(db, input) {
239
239
  }
240
240
  /**
241
241
  * Walk the fused list in order, loading entries a batch at a time, and keep
242
- * the first `limit` that survive path deduplication and the filters.
242
+ * the first `limit` that survive path deduplication and the filters. An entry
243
+ * whose indexed content is identical to a kept one's (the same body saved
244
+ * under another name or in another bundle) is a duplicate and is skipped.
243
245
  */
244
246
  function selectFusedEntries(db, fused, limit, filterOptions) {
245
247
  const selected = [];
246
248
  const seenPaths = new Set();
249
+ const seenContent = new Set();
247
250
  const batchSize = Math.max(limit * 2, 20);
248
251
  for (let offset = 0; offset < fused.length && selected.length < limit; offset += batchSize) {
249
252
  const batch = [];
@@ -256,6 +259,12 @@ function selectFusedEntries(db, fused, limit, filterOptions) {
256
259
  }
257
260
  for (const kept of applyEntryFilters(batch, filterOptions)) {
258
261
  const { candidate, ...row } = kept;
262
+ const content = row.entry.content?.replace(/\s+/g, " ").trim();
263
+ if (content) {
264
+ if (seenContent.has(content))
265
+ continue;
266
+ seenContent.add(content);
267
+ }
259
268
  selected.push({ candidate, row });
260
269
  }
261
270
  }
@@ -79,9 +79,6 @@ export function isProcessEnabled(section, processName, config) {
79
79
  return false;
80
80
  // Index passes are first-class 0.9 entries.
81
81
  if (section === "index") {
82
- if (processName === "metadata_enhance" || processName === "metadataEnhance") {
83
- return config.index?.metadataEnhance?.enabled ?? true;
84
- }
85
82
  if (processName === "memory_inference" || processName === "memoryInference") {
86
83
  return isLlmFeatureEnabled(config, "memory_inference");
87
84
  }