@monoes/monomindcli 2.10.10 → 2.10.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. package/.claude/commands/mastermind/brain.md +14 -14
  2. package/.claude/commands/mastermind/help.md +2 -2
  3. package/.claude/commands/mastermind/master.md +24 -19
  4. package/.claude/commands/mastermind/memory.md +8 -8
  5. package/.claude/commands/mastermind/monoswarm.md +4 -4
  6. package/.claude/commands/mastermind.md +7 -7
  7. package/.claude/commands/truth/start.md +3 -3
  8. package/.claude/skills/mastermind/SKILL.md +7 -16
  9. package/.claude/skills/mastermind-debug/SKILL.md +3 -3
  10. package/.claude/skills/mastermind-design/SKILL.md +2 -0
  11. package/.claude/skills/mastermind-execute/SKILL.md +66 -11
  12. package/.claude/skills/mastermind-idea/SKILL.md +9 -2
  13. package/.claude/skills/mastermind-intake/SKILL.md +31 -7
  14. package/.claude/skills/mastermind-issue-detail/SKILL.md +70 -16
  15. package/.claude/skills/mastermind-issues/SKILL.md +111 -16
  16. package/.claude/skills/mastermind-liveness/SKILL.md +96 -26
  17. package/.claude/skills/mastermind-my-issues/SKILL.md +40 -8
  18. package/.claude/skills/mastermind-org/SKILL.md +2 -0
  19. package/.claude/skills/mastermind-plan/SKILL.md +7 -16
  20. package/.claude/skills/mastermind-plan-to-tasks/SKILL.md +132 -24
  21. package/.claude/skills/mastermind-protocol/SKILL.md +33 -22
  22. package/.claude/skills/mastermind-runorg/SKILL.md +22 -3
  23. package/.claude/skills/mastermind-skill-builder/SKILL.md +1 -1
  24. package/.claude/skills/mastermind-tasks/SKILL.md +5 -0
  25. package/.claude/skills/mastermind-techport/SKILL.md +1 -1
  26. package/.claude/skills/performance-analysis/SKILL.md +1 -1
  27. package/.claude/skills/verification-quality/SKILL.md +2 -3
  28. package/README.md +2 -2
  29. package/dist/src/commands/doc.js +2 -2
  30. package/dist/src/commands/doc.js.map +1 -1
  31. package/dist/src/commands/doctor-project-checks.d.ts.map +1 -1
  32. package/dist/src/commands/doctor-project-checks.js +20 -1
  33. package/dist/src/commands/doctor-project-checks.js.map +1 -1
  34. package/dist/src/commands/monograph.d.ts.map +1 -1
  35. package/dist/src/commands/monograph.js +11 -4
  36. package/dist/src/commands/monograph.js.map +1 -1
  37. package/dist/src/commands/org-observe.d.ts.map +1 -1
  38. package/dist/src/commands/org-observe.js +56 -6
  39. package/dist/src/commands/org-observe.js.map +1 -1
  40. package/dist/src/commands/org.d.ts +26 -0
  41. package/dist/src/commands/org.d.ts.map +1 -1
  42. package/dist/src/commands/org.js +144 -28
  43. package/dist/src/commands/org.js.map +1 -1
  44. package/dist/src/init/executor.d.ts.map +1 -1
  45. package/dist/src/init/executor.js +10 -9
  46. package/dist/src/init/executor.js.map +1 -1
  47. package/dist/src/init/settings-generator.d.ts.map +1 -1
  48. package/dist/src/init/settings-generator.js.map +1 -1
  49. package/dist/src/init/write-codex.d.ts.map +1 -1
  50. package/dist/src/init/write-codex.js.map +1 -1
  51. package/dist/src/knowledge/document-pipeline.d.ts +5 -0
  52. package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
  53. package/dist/src/knowledge/document-pipeline.js +32 -16
  54. package/dist/src/knowledge/document-pipeline.js.map +1 -1
  55. package/dist/src/mcp-tools/hooks-routing.d.ts +9 -0
  56. package/dist/src/mcp-tools/hooks-routing.d.ts.map +1 -1
  57. package/dist/src/mcp-tools/hooks-routing.js +12 -1
  58. package/dist/src/mcp-tools/hooks-routing.js.map +1 -1
  59. package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
  60. package/dist/src/mcp-tools/knowledge-tools.js +105 -13
  61. package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
  62. package/dist/src/mcp-tools/memory-tools.d.ts +13 -0
  63. package/dist/src/mcp-tools/memory-tools.d.ts.map +1 -1
  64. package/dist/src/mcp-tools/memory-tools.js +257 -31
  65. package/dist/src/mcp-tools/memory-tools.js.map +1 -1
  66. package/dist/src/mcp-tools/monograph/health-tools.d.ts.map +1 -1
  67. package/dist/src/mcp-tools/monograph/health-tools.js +99 -42
  68. package/dist/src/mcp-tools/monograph/health-tools.js.map +1 -1
  69. package/dist/src/mcp-tools/monograph/impact-tools.d.ts.map +1 -1
  70. package/dist/src/mcp-tools/monograph/impact-tools.js +123 -51
  71. package/dist/src/mcp-tools/monograph/impact-tools.js.map +1 -1
  72. package/dist/src/mcp-tools/monograph/query-tools.d.ts.map +1 -1
  73. package/dist/src/mcp-tools/monograph/query-tools.js +113 -96
  74. package/dist/src/mcp-tools/monograph/query-tools.js.map +1 -1
  75. package/dist/src/mcp-tools/monograph/shared.d.ts +37 -4
  76. package/dist/src/mcp-tools/monograph/shared.d.ts.map +1 -1
  77. package/dist/src/mcp-tools/monograph/shared.js +75 -56
  78. package/dist/src/mcp-tools/monograph/shared.js.map +1 -1
  79. package/dist/src/memory/memory-bridge.d.ts +60 -1
  80. package/dist/src/memory/memory-bridge.d.ts.map +1 -1
  81. package/dist/src/memory/memory-bridge.js +172 -42
  82. package/dist/src/memory/memory-bridge.js.map +1 -1
  83. package/dist/src/memory/memory-kg.d.ts +495 -28
  84. package/dist/src/memory/memory-kg.d.ts.map +1 -1
  85. package/dist/src/memory/memory-kg.js +2186 -251
  86. package/dist/src/memory/memory-kg.js.map +1 -1
  87. package/dist/src/memory/query-router.d.ts +51 -0
  88. package/dist/src/memory/query-router.d.ts.map +1 -1
  89. package/dist/src/memory/query-router.js +38 -2
  90. package/dist/src/memory/query-router.js.map +1 -1
  91. package/dist/src/orgrt/agent-exec.d.ts.map +1 -1
  92. package/dist/src/orgrt/agent-exec.js +7 -0
  93. package/dist/src/orgrt/agent-exec.js.map +1 -1
  94. package/dist/src/orgrt/agent-runner.d.ts +26 -1
  95. package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
  96. package/dist/src/orgrt/agent-runner.js +73 -22
  97. package/dist/src/orgrt/agent-runner.js.map +1 -1
  98. package/dist/src/orgrt/antigravity-runner.d.ts +1 -1
  99. package/dist/src/orgrt/antigravity-runner.d.ts.map +1 -1
  100. package/dist/src/orgrt/antigravity-runner.js +36 -19
  101. package/dist/src/orgrt/antigravity-runner.js.map +1 -1
  102. package/dist/src/orgrt/approvals.d.ts +26 -2
  103. package/dist/src/orgrt/approvals.d.ts.map +1 -1
  104. package/dist/src/orgrt/approvals.js +67 -7
  105. package/dist/src/orgrt/approvals.js.map +1 -1
  106. package/dist/src/orgrt/broker.d.ts +21 -2
  107. package/dist/src/orgrt/broker.d.ts.map +1 -1
  108. package/dist/src/orgrt/broker.js +56 -8
  109. package/dist/src/orgrt/broker.js.map +1 -1
  110. package/dist/src/orgrt/bus.d.ts +8 -0
  111. package/dist/src/orgrt/bus.d.ts.map +1 -1
  112. package/dist/src/orgrt/bus.js +27 -0
  113. package/dist/src/orgrt/bus.js.map +1 -1
  114. package/dist/src/orgrt/checkpoint-ops.d.ts.map +1 -1
  115. package/dist/src/orgrt/checkpoint-ops.js +15 -5
  116. package/dist/src/orgrt/checkpoint-ops.js.map +1 -1
  117. package/dist/src/orgrt/checkpoint.d.ts +42 -4
  118. package/dist/src/orgrt/checkpoint.d.ts.map +1 -1
  119. package/dist/src/orgrt/checkpoint.js +75 -4
  120. package/dist/src/orgrt/checkpoint.js.map +1 -1
  121. package/dist/src/orgrt/codex-runner.d.ts +1 -1
  122. package/dist/src/orgrt/codex-runner.d.ts.map +1 -1
  123. package/dist/src/orgrt/codex-runner.js +50 -23
  124. package/dist/src/orgrt/codex-runner.js.map +1 -1
  125. package/dist/src/orgrt/copilot-runner.d.ts +24 -2
  126. package/dist/src/orgrt/copilot-runner.d.ts.map +1 -1
  127. package/dist/src/orgrt/copilot-runner.js +257 -157
  128. package/dist/src/orgrt/copilot-runner.js.map +1 -1
  129. package/dist/src/orgrt/cross-org.d.ts +10 -3
  130. package/dist/src/orgrt/cross-org.d.ts.map +1 -1
  131. package/dist/src/orgrt/cross-org.js +225 -10
  132. package/dist/src/orgrt/cross-org.js.map +1 -1
  133. package/dist/src/orgrt/crush-runner.d.ts +22 -2
  134. package/dist/src/orgrt/crush-runner.d.ts.map +1 -1
  135. package/dist/src/orgrt/crush-runner.js +227 -120
  136. package/dist/src/orgrt/crush-runner.js.map +1 -1
  137. package/dist/src/orgrt/daemon.d.ts +77 -7
  138. package/dist/src/orgrt/daemon.d.ts.map +1 -1
  139. package/dist/src/orgrt/daemon.js +989 -385
  140. package/dist/src/orgrt/daemon.js.map +1 -1
  141. package/dist/src/orgrt/decisions.d.ts +2 -2
  142. package/dist/src/orgrt/decisions.d.ts.map +1 -1
  143. package/dist/src/orgrt/decisions.js +90 -18
  144. package/dist/src/orgrt/decisions.js.map +1 -1
  145. package/dist/src/orgrt/grok-runner.d.ts +29 -3
  146. package/dist/src/orgrt/grok-runner.d.ts.map +1 -1
  147. package/dist/src/orgrt/grok-runner.js +286 -150
  148. package/dist/src/orgrt/grok-runner.js.map +1 -1
  149. package/dist/src/orgrt/kimicode-runner.d.ts +5 -5
  150. package/dist/src/orgrt/kimicode-runner.d.ts.map +1 -1
  151. package/dist/src/orgrt/kimicode-runner.js +53 -32
  152. package/dist/src/orgrt/kimicode-runner.js.map +1 -1
  153. package/dist/src/orgrt/mailbox.d.ts +15 -0
  154. package/dist/src/orgrt/mailbox.d.ts.map +1 -1
  155. package/dist/src/orgrt/mailbox.js +29 -1
  156. package/dist/src/orgrt/mailbox.js.map +1 -1
  157. package/dist/src/orgrt/migrate.d.ts.map +1 -1
  158. package/dist/src/orgrt/migrate.js +8 -5
  159. package/dist/src/orgrt/migrate.js.map +1 -1
  160. package/dist/src/orgrt/opencode-runner.d.ts +1 -1
  161. package/dist/src/orgrt/opencode-runner.d.ts.map +1 -1
  162. package/dist/src/orgrt/opencode-runner.js +29 -3
  163. package/dist/src/orgrt/opencode-runner.js.map +1 -1
  164. package/dist/src/orgrt/org-memory.d.ts +19 -3
  165. package/dist/src/orgrt/org-memory.d.ts.map +1 -1
  166. package/dist/src/orgrt/org-memory.js +110 -38
  167. package/dist/src/orgrt/org-memory.js.map +1 -1
  168. package/dist/src/orgrt/pi-rpc-runner.d.ts +3 -1
  169. package/dist/src/orgrt/pi-rpc-runner.d.ts.map +1 -1
  170. package/dist/src/orgrt/pi-rpc-runner.js +35 -3
  171. package/dist/src/orgrt/pi-rpc-runner.js.map +1 -1
  172. package/dist/src/orgrt/pi-runner.d.ts +25 -2
  173. package/dist/src/orgrt/pi-runner.d.ts.map +1 -1
  174. package/dist/src/orgrt/pi-runner.js +271 -152
  175. package/dist/src/orgrt/pi-runner.js.map +1 -1
  176. package/dist/src/orgrt/policy.d.ts +1 -0
  177. package/dist/src/orgrt/policy.d.ts.map +1 -1
  178. package/dist/src/orgrt/policy.js +199 -39
  179. package/dist/src/orgrt/policy.js.map +1 -1
  180. package/dist/src/orgrt/provider.d.ts +4 -0
  181. package/dist/src/orgrt/provider.d.ts.map +1 -1
  182. package/dist/src/orgrt/provider.js +16 -0
  183. package/dist/src/orgrt/provider.js.map +1 -1
  184. package/dist/src/orgrt/qwen-rpc-runner.d.ts +3 -1
  185. package/dist/src/orgrt/qwen-rpc-runner.d.ts.map +1 -1
  186. package/dist/src/orgrt/qwen-rpc-runner.js +35 -3
  187. package/dist/src/orgrt/qwen-rpc-runner.js.map +1 -1
  188. package/dist/src/orgrt/qwen-runner.d.ts +30 -3
  189. package/dist/src/orgrt/qwen-runner.d.ts.map +1 -1
  190. package/dist/src/orgrt/qwen-runner.js +282 -142
  191. package/dist/src/orgrt/qwen-runner.js.map +1 -1
  192. package/dist/src/orgrt/role-slot.d.ts +88 -0
  193. package/dist/src/orgrt/role-slot.d.ts.map +1 -0
  194. package/dist/src/orgrt/role-slot.js +133 -0
  195. package/dist/src/orgrt/role-slot.js.map +1 -0
  196. package/dist/src/orgrt/runtime-options.d.ts +17 -0
  197. package/dist/src/orgrt/runtime-options.d.ts.map +1 -0
  198. package/dist/src/orgrt/runtime-options.js +32 -0
  199. package/dist/src/orgrt/runtime-options.js.map +1 -0
  200. package/dist/src/orgrt/scheduler-integration.d.ts +11 -0
  201. package/dist/src/orgrt/scheduler-integration.d.ts.map +1 -1
  202. package/dist/src/orgrt/scheduler-integration.js +75 -17
  203. package/dist/src/orgrt/scheduler-integration.js.map +1 -1
  204. package/dist/src/orgrt/scheduler.d.ts +4 -0
  205. package/dist/src/orgrt/scheduler.d.ts.map +1 -1
  206. package/dist/src/orgrt/scheduler.js +7 -1
  207. package/dist/src/orgrt/scheduler.js.map +1 -1
  208. package/dist/src/orgrt/server.d.ts +8 -3
  209. package/dist/src/orgrt/server.d.ts.map +1 -1
  210. package/dist/src/orgrt/server.js +48 -13
  211. package/dist/src/orgrt/server.js.map +1 -1
  212. package/dist/src/orgrt/session.d.ts +26 -2
  213. package/dist/src/orgrt/session.d.ts.map +1 -1
  214. package/dist/src/orgrt/session.js +98 -4
  215. package/dist/src/orgrt/session.js.map +1 -1
  216. package/dist/src/orgrt/task-dag.d.ts +5 -0
  217. package/dist/src/orgrt/task-dag.d.ts.map +1 -1
  218. package/dist/src/orgrt/task-dag.js +41 -0
  219. package/dist/src/orgrt/task-dag.js.map +1 -1
  220. package/dist/src/orgrt/test-loop.js +2 -2
  221. package/dist/src/orgrt/test-loop.js.map +1 -1
  222. package/dist/src/orgrt/types.d.ts +14 -2
  223. package/dist/src/orgrt/types.d.ts.map +1 -1
  224. package/dist/src/orgrt/types.js +15 -1
  225. package/dist/src/orgrt/types.js.map +1 -1
  226. package/dist/src/orgrt/vercel-runner.d.ts.map +1 -1
  227. package/dist/src/orgrt/vercel-runner.js +4 -0
  228. package/dist/src/orgrt/vercel-runner.js.map +1 -1
  229. package/dist/src/ui/routes-org.mjs +27 -7
  230. package/dist/src/ui/server.mjs +82 -48
  231. package/dist/tsconfig.tsbuildinfo +1 -1
  232. package/package.json +3 -2
  233. package/dist/src/ui/data/mastermind-sessions.json +0 -1
@@ -10,25 +10,302 @@
10
10
  * feedback/frequency weighting for free — KG node ranking improves with use
11
11
  * automatically.
12
12
  *
13
- * Identity is deterministic and NAME-ONLY (cognee's Entity.identity_fields):
14
- * the entry KEY is `n:<normalized-name>`, so the same entity extracted from
15
- * any session merges idempotently via upsert regardless of assigned type.
16
- * Every write carries `origin_refs` so a bad ingest can be rolled back per
17
- * run/session.
13
+ * OWNERSHIP is carried by `KgScope`, not by the store path. Every org under
14
+ * one project root shares ONE org-memory store, so those namespaces would
15
+ * otherwise be a single pool that any org can read, merge into, and roll back.
16
+ * A scope suffixes every one of them (`kg:nodes:org:<org>`, …) and stamps the
17
+ * asserting org onto every origin ref, so an org's reads, writes, glossary and
18
+ * rollback all resolve through `kgNamespaces()` and can only reach what that
19
+ * org asserted. An absent scope means PROJECT-SHARED knowledge — a scope in
20
+ * its own right, not "all scopes". Crossing from an org into the shared graph
21
+ * is `kgPromote`, an explicit operation, never a side effect of learning.
18
22
  *
19
- * // monolean: graph traversal is in-process over a full kg:edges list —
20
- * // fine to ~10k edges; upgrade path is a real SQLite edges table with
21
- * // indexed src/dst columns if orgs outgrow that.
23
+ * IDENTITY is the (type, name) tuple, hashed length-prefixed so no component
24
+ * can bleed into its neighbour and no distinction is truncated away — the same
25
+ * injective-serialization discipline monograph's `symbolId`/`fileId` use, and
26
+ * for the same reason. Name-only identity made `Person:Alex` and `Service:Alex`
27
+ * one entity, and made two names that first differ at character 250 one entity
28
+ * as well. Keys stay per-namespace, so scope still lives in the NAMESPACE: two
29
+ * orgs asserting different facts about the same name produce the same id, and
30
+ * only separate namespaces keep them from overwriting each other.
31
+ *
32
+ * Name-only merging did buy something real — it stopped the same entity forking
33
+ * when the LLM said "Module" and the heuristic said "entity" — so that benefit
34
+ * is kept by a NAME INDEX (`kg:names`) rather than by a lossy key. Resolution
35
+ * goes through the index, never through a computed key: a generic assertion
36
+ * adopts the single same-name entity if there is exactly one, a typed assertion
37
+ * adopts a lone untyped one and promotes its type, and anything genuinely
38
+ * ambiguous becomes a separate entity with the alternatives REPORTED as
39
+ * candidates rather than silently merged.
40
+ *
41
+ * KNOWLEDGE IS CLAIMS, not a summary string. Each entity/edge/rule carries a
42
+ * `claims` ledger of per-origin description contributions; the stored
43
+ * `description` is DERIVED from it — the most recent contribution wins, because
44
+ * description length measures verbosity and not truth. That is what makes a
45
+ * correction expressible ("Now PostgreSQL" supersedes a longer MySQL blurb) and
46
+ * what makes rollback reversible: withdrawing an origin drops its contribution
47
+ * and RE-DERIVES the summary from the survivors, so the previous correct
48
+ * description comes back instead of a bad one being frozen in place.
49
+ * `origin_refs` remains, derived from the ledger, so provenance readers are
50
+ * unaffected. Support is never silently truncated: past `MAX_CLAIMS` the entry
51
+ * records `origins_dropped` and `provenance_complete:false`.
52
+ *
53
+ * EVIDENCE travels with the claim. Each contribution records how it was
54
+ * obtained — `asserted` when someone stated it, `heuristic` when
55
+ * `heuristicExtract` inferred it from two names sharing a sentence — and the
56
+ * element's standing is projected from its live claims, with one asserted claim
57
+ * outranking any number of guesses. A claim written before this recording
58
+ * leaves the projection UNKNOWN rather than voting: reading an unrecorded
59
+ * method as `asserted` would relabel the entire pre-existing graph as stated
60
+ * fact. Retrieval uses exactly this and the `conflict` flag to rank, and
61
+ * nothing else — source credibility and claim freshness need an evaluation set
62
+ * to tune against, and guessing at them is the overclaim this replaced.
63
+ *
64
+ * MIGRATION: nothing is re-keyed, deleted, or orphaned. An entry written under
65
+ * the old `n:<normalized-name>` scheme is still found — resolution falls back to
66
+ * probing the legacy key — and is then adopted IN PLACE under its existing key
67
+ * and registered in the name index. Legacy rows therefore keep working for
68
+ * search, stats, glossary and rollback, and keep their edges (whose keys embed
69
+ * the endpoint keys) intact. Only a genuinely NEW identity distinction — a
70
+ * second type for a name, or a name that differs only past the old truncation
71
+ * point — mints a new hashed id.
72
+ *
73
+ * // monolean: graph traversal is in-process over a paged kg:edges scan. The
74
+ * // bridge exposes no indexed adjacency (src/dst) or origin lookup, so every
75
+ * // neighbourhood/provenance question is a namespace scan; the upgrade path is
76
+ * // a real SQLite edges table with indexed src/dst/origin columns, which turns
77
+ * // these O(namespace) scans into O(matches). `kgStats` uses a real
78
+ * // `bridgeCountEntries` (SELECT COUNT(*) WHERE namespace = ?) instead of a
79
+ * // scan — that needs no history, a count is correct regardless of when a row
80
+ * // was written. Adjacency/origin, in contrast, is NOT filled in here: an
81
+ * // index built only from now on would silently miss every edge/claim written
82
+ * // before it existed (there is no legacy-key probe for an arbitrary historical
83
+ * // edge, unlike resolveEntity's single fallback key), so kgSearch/kgRollback
84
+ * // would go from an honest, complete scan to an INcomplete indexed answer —
85
+ * // a regression, not the fix. Closing this needs a real backfill/migration
86
+ * // decision (how to populate the index for existing namespaces, and how to
87
+ * // know when one is complete enough to trust), not more code here.
22
88
  *
23
89
  * @module v1/cli/memory/memory-kg
24
90
  */
25
- import { bridgeDeleteEntry, bridgeGetEntry, bridgeListEntries, bridgeSearchEntries, bridgeStoreEntry, } from './memory-bridge.js';
91
+ import { createHash } from 'node:crypto';
92
+ import { bridgeCountEntries, bridgeDeleteEntry, bridgeGetEntry, bridgeListEntries, bridgeSearchEntries, bridgeStoreEntry, } from './memory-bridge.js';
26
93
  export const KG_NODES_NS = 'kg:nodes';
27
94
  export const KG_EDGES_NS = 'kg:edges';
28
95
  export const RULES_NS = 'rules';
96
+ /** Name → entity index. Index rows are NOT claims, so they live outside the
97
+ * three claim namespaces: putting them in `kg:nodes` would make them seed
98
+ * candidates for `kgSearch` and rows in `kgStats`. */
99
+ export const KG_NAMES_NS = 'kg:names';
100
+ /** Derived adjacency index namespace (K7). See `KgNamespaces.adj`. */
101
+ export const KG_ADJ_NS = 'kg:adj';
102
+ /** Derived origin-support index namespace (K7). See `KgNamespaces.originIdx`. */
103
+ export const KG_ORIGIN_IDX_NS = 'kg:origin-idx';
104
+ /** Derived-index status namespace (K7). See `KgNamespaces.indexStatus`. */
105
+ export const KG_INDEX_STATUS_NS = 'kg:index-status';
29
106
  const MAX_NAME_LEN = 200;
30
107
  const MAX_DESC_LEN = 2000;
31
- const MAX_LIST = 10_000;
108
+ /** Nodes/edges/rules accepted per call. Overflow is REPORTED, never sliced away
109
+ * in silence — see `KgIngestResult.nodesTruncated`. */
110
+ const MAX_NODES_PER_CALL = 500;
111
+ const MAX_EDGES_PER_CALL = 1000;
112
+ const MAX_RULES_PER_CALL = 50;
113
+ /** Per-origin description contributions retained on one element.
114
+ *
115
+ * A cap has to exist — the bridge caps a stored value, and an element asserted
116
+ * by ten thousand runs would otherwise stop being writable. What must NOT
117
+ * happen is the old `origin_refs.slice(-100)`, which dropped the oldest
118
+ * provenance and left the entry claiming complete history. Past this cap the
119
+ * oldest contributions are dropped AND the entry records `origins_dropped`
120
+ * with `provenance_complete: false`, so a reader can tell that a rollback of an
121
+ * old origin may find nothing to withdraw. */
122
+ const MAX_CLAIMS = 200;
123
+ // ── Identity ────────────────────────────────────────────────────────
124
+ /**
125
+ * Version of the identity scheme implemented by `entityId`/`edgeKey`/`ruleKey`.
126
+ *
127
+ * Bump when a derivation changes. Entries written under an older scheme are not
128
+ * orphaned: resolution probes the previous key shape and adopts the row in
129
+ * place (see `resolveEntity`), so a bump costs a second keyed lookup on the
130
+ * miss path rather than a migration.
131
+ *
132
+ * Version 1 was the name-only scheme (`n:<normalized-name>`, truncated to 200
133
+ * characters), under which `Person:Alex` and `Service:Alex` were one entity and
134
+ * two names first differing at character 250 were one entity.
135
+ */
136
+ export const KG_ID_VERSION = 2;
137
+ /** Identity-grade name normalization: the same folding as `normalizeName` but
138
+ * WITHOUT its 200-character truncation, which silently merged long names that
139
+ * differ only past the cut. `normalizeName` keeps truncating because it feeds
140
+ * tags and display, where length matters and collisions do not. */
141
+ function canonicalName(name) {
142
+ return String(name).trim().toLowerCase().replace(/['’]/g, '').replace(/\s+/g, '_');
143
+ }
144
+ /** Types that assert nothing. A generic label must not fork an entity away from
145
+ * its typed self, so it maps to the empty discriminator and is resolved by
146
+ * name (see `resolveEntity`). */
147
+ const GENERIC_TYPES = new Set(['', 'entity', 'unknown']);
148
+ /** The identity-bearing part of a type: '' when the caller told us nothing. */
149
+ function typeBucket(type) {
150
+ const t = canonicalName(type ?? '');
151
+ return GENERIC_TYPES.has(t) ? '' : t;
152
+ }
153
+ /** Hash a tuple injectively: every component is length-prefixed, so no
154
+ * component's content can masquerade as a delimiter or bleed into its
155
+ * neighbour, and no two distinct tuples share an input string. Same discipline
156
+ * as monograph's `symbolId`/`fileId` (packages/@monomind/monograph/src/types.ts),
157
+ * and adopted here for the same reason — the previous scheme lost distinctions
158
+ * to truncation and to an unescaped separator. */
159
+ function hashTuple(components) {
160
+ const canonical = components.map((c) => `${Buffer.byteLength(c, 'utf8')}:${c}`).join('');
161
+ return createHash('sha256').update(canonical, 'utf8').digest('hex').slice(0, 32);
162
+ }
163
+ /** Mint an entity id from the full identity tuple.
164
+ *
165
+ * This MINTS; it does not RESOLVE. A caller that computes an id and writes to
166
+ * it bypasses the name index and re-forks the entities the index exists to
167
+ * keep together — every ingest path goes through `resolveEntity` instead. */
168
+ export function nodeKey(type, name) {
169
+ return `n:${hashTuple([String(KG_ID_VERSION), typeBucket(type), canonicalName(name)])}`;
170
+ }
171
+ /** The pre-`KG_ID_VERSION` key for a name: name-only, truncated at 200. Probed
172
+ * on a miss so existing graphs keep working without being re-keyed. */
173
+ function legacyNodeKey(name) {
174
+ return `n:${normalizeName(name)}`;
175
+ }
176
+ /** Rows per `bridgeListEntries` call while scanning a namespace.
177
+ *
178
+ * A single list call cannot return more than the backend's own 10,000-row
179
+ * ceiling (`MAX_QUERY_LIMIT` in sql-backend.ts), so the old `limit: MAX_LIST`
180
+ * scans were not "large enough to be safe" — they were exactly the point past
181
+ * which the graph goes silently invisible. Everything below pages instead.
182
+ *
183
+ * 1,000 balances the two costs: one page is at most ~1 MB even at the bridge's
184
+ * 16 KB value cap (typical KG entries are far smaller), and it takes a tenth
185
+ * of the round trips a 100-row page would. */
186
+ const SCAN_PAGE = 1_000;
187
+ /** Edge rows `kgSearch` will read before giving up and reporting `truncated`.
188
+ * Search is interactive and runs a full scan per query, so unlike rollback it
189
+ * keeps a ceiling — five times the old silent one, and now stated in the
190
+ * result rather than hidden. */
191
+ const SEARCH_EDGE_SCAN_MAX = 50_000;
192
+ /** Page through an entire namespace, handing each page to `onPage` so callers
193
+ * fold as they go instead of materializing the namespace.
194
+ *
195
+ * `onPage` returning exactly `false` stops the scan early (only `kgSearch`
196
+ * does that); any other return value continues. Returns false when the backend
197
+ * was unavailable — an unreadable namespace must never be mistaken for an
198
+ * empty one.
199
+ *
200
+ * Callers that MUTATE what they find must collect during the scan and mutate
201
+ * afterwards: deleting a row pulls every later row back one position, so under
202
+ * an advancing offset the row that slid into the gap is never read. (The
203
+ * bridge's upsert no longer reorders — it reuses the existing entry id and
204
+ * preserves `createdAt` — so rewrites alone are safe; deletes are not, and
205
+ * both callers here delete.) */
206
+ async function scanNamespace(namespace, dbPath, onPage) {
207
+ for (let offset = 0;; offset += SCAN_PAGE) {
208
+ const res = await bridgeListEntries({ namespace, limit: SCAN_PAGE, offset, dbPath });
209
+ if (!res)
210
+ return false;
211
+ if (res.entries.length && onPage(res.entries) === false)
212
+ return true;
213
+ if (res.entries.length < SCAN_PAGE)
214
+ return true;
215
+ }
216
+ }
217
+ /** Delete every entry in `namespace`. Used only to reset a DERIVED-index
218
+ * namespace (`kg:adj`/`kg:origin-idx`) before a fresh `kgRebuildIndex` —
219
+ * never a canonical one. Collect-then-delete, same reason `kgRollback`
220
+ * does: deleting mid-scan pulls later rows back under an advancing offset.
221
+ * Returns false (nothing deleted) on an unreadable or partially-deletable
222
+ * namespace, so the caller never proceeds as if a clear that didn't fully
223
+ * happen did. */
224
+ async function clearNamespace(namespace, dbPath) {
225
+ const doomed = [];
226
+ const covered = await scanNamespace(namespace, dbPath, (page) => {
227
+ for (const e of page)
228
+ doomed.push(e);
229
+ });
230
+ if (!covered)
231
+ return false;
232
+ let ok = true;
233
+ for (const e of doomed) {
234
+ const del = await bridgeDeleteEntry({ id: e.id, namespace, dbPath });
235
+ if (!del?.deleted)
236
+ ok = false;
237
+ }
238
+ return ok;
239
+ }
240
+ /** Cap on reported failure messages — a dead backend fails every write, and a
241
+ * 500-entry failure list is noise, not signal. `error` carries the true count. */
242
+ const MAX_FAILURES = 20;
243
+ /** `bridgeStoreEntry` never throws: it returns `null` when no backend is
244
+ * reachable and `{ success: false, error }` when the write itself failed.
245
+ * Both look like success to an `await` that ignores the result, which is how
246
+ * the graph came to claim knowledge it had not persisted. Every write in this
247
+ * module goes through here.
248
+ *
249
+ * @returns a failure message, or null when the write landed. */
250
+ function storeFailure(res, what) {
251
+ if (!res)
252
+ return `${what}: memory backend unavailable`;
253
+ if (!res.success)
254
+ return `${what}: ${res.error ?? 'store rejected'}`;
255
+ return null;
256
+ }
257
+ /** Accumulates write failures across a multi-write operation. The memory
258
+ * bridge exposes no transaction primitive, so ingest CANNOT be atomic: some
259
+ * writes land and some do not. Rather than hide that, callers get exact
260
+ * counters for what persisted plus the failure list for what did not. */
261
+ class FailureLog {
262
+ messages = [];
263
+ count = 0;
264
+ /** @returns true when the write failed (caller should not count it). */
265
+ add(res, what) {
266
+ const msg = storeFailure(res, what);
267
+ if (!msg)
268
+ return false;
269
+ this.count++;
270
+ if (this.messages.length < MAX_FAILURES)
271
+ this.messages.push(msg);
272
+ return true;
273
+ }
274
+ note(message) {
275
+ this.count++;
276
+ if (this.messages.length < MAX_FAILURES)
277
+ this.messages.push(message);
278
+ }
279
+ get failed() {
280
+ return this.count > 0;
281
+ }
282
+ /** Summary for the result's `error` field; undefined when everything landed. */
283
+ summary() {
284
+ if (!this.count)
285
+ return undefined;
286
+ return `${this.count} bridge operation(s) failed; graph state is partial`;
287
+ }
288
+ }
289
+ /** Attempts a compare-and-swap claim write, retrying while the bridge reports
290
+ * a version conflict (K5): a concurrent writer applied its own claim to the
291
+ * same row between our read and our write. `attempt()` must re-read,
292
+ * re-merge (via `applyClaim`), and re-attempt the write itself on every call
293
+ * — retrying with the SAME stale write would just lose the update again,
294
+ * the exact bug this exists to close.
295
+ *
296
+ * Exhausting `maxAttempts` under sustained contention is returned as-is (the
297
+ * last conflict response) rather than retried forever: the caller's
298
+ * `FailureLog` reports it as a real, visible failure — never a silent lost
299
+ * update — same as any other write the bridge refused. */
300
+ async function withCasRetry(attempt, maxAttempts = 3) {
301
+ let res = null;
302
+ for (let i = 0; i < maxAttempts; i++) {
303
+ res = await attempt();
304
+ if (!res?.conflict)
305
+ return res;
306
+ }
307
+ return res;
308
+ }
32
309
  /** cognee DataPoint normalization: lowercase, spaces→_, strip apostrophes. */
33
310
  export function normalizeName(name) {
34
311
  return String(name)
@@ -38,140 +315,498 @@ export function normalizeName(name) {
38
315
  .replace(/\s+/g, '_')
39
316
  .slice(0, MAX_NAME_LEN);
40
317
  }
41
- /** Identity is NAME-ONLY (cognee's Entity.identity_fields = ["name"]) — type
42
- * lives in metadata. Including type in the key forked the same entity when
43
- * the LLM said "Module" and the heuristic said "entity". */
44
- export function nodeKey(_type, name) {
45
- return `n:${normalizeName(name)}`;
46
- }
318
+ /** Edge identity: the endpoint IDs (already collision-free) plus the relation,
319
+ * hashed injectively so a relation containing the old `|` separator can no
320
+ * longer forge a different edge's key. */
47
321
  function edgeKey(srcKey, relation, dstKey) {
322
+ return `e:${hashTuple([String(KG_ID_VERSION), srcKey, canonicalName(relation), dstKey])}`;
323
+ }
324
+ /** The pre-`KG_ID_VERSION` edge key, for the same in-place adoption as
325
+ * `legacyNodeKey`. Composed from the RESOLVED endpoint keys, so it names the
326
+ * legacy edge exactly when both endpoints are themselves legacy rows. */
327
+ function legacyEdgeKey(srcKey, relation, dstKey) {
48
328
  return `e:${srcKey}|${normalizeName(relation)}|${dstKey}`;
49
329
  }
330
+ /** Rule identity: the full rule text. The old key truncated the normalized rule
331
+ * at 120 characters, so two rules sharing a long preamble were one rule. */
332
+ function ruleKey(rule) {
333
+ return `rule:${hashTuple([String(KG_ID_VERSION), canonicalName(rule)])}`;
334
+ }
335
+ function legacyRuleKey(rule) {
336
+ return `rule:${normalizeName(rule).slice(0, 120)}`;
337
+ }
338
+ /** The namespaces a scope owns. Every read and write in this module resolves
339
+ * through here, which is what makes ownership enforced rather than advisory:
340
+ * there is no code path that reaches an org's facts without naming that org,
341
+ * and none that reaches every org at once. */
342
+ export function kgNamespaces(scope) {
343
+ const org = scope?.org?.trim();
344
+ if (!org)
345
+ return {
346
+ nodes: KG_NODES_NS,
347
+ edges: KG_EDGES_NS,
348
+ rules: RULES_NS,
349
+ names: KG_NAMES_NS,
350
+ adj: KG_ADJ_NS,
351
+ originIdx: KG_ORIGIN_IDX_NS,
352
+ indexStatus: KG_INDEX_STATUS_NS,
353
+ };
354
+ const suffix = `:org:${normalizeName(org)}`;
355
+ return {
356
+ nodes: KG_NODES_NS + suffix,
357
+ edges: KG_EDGES_NS + suffix,
358
+ rules: RULES_NS + suffix,
359
+ names: KG_NAMES_NS + suffix,
360
+ adj: KG_ADJ_NS + suffix,
361
+ originIdx: KG_ORIGIN_IDX_NS + suffix,
362
+ indexStatus: KG_INDEX_STATUS_NS + suffix,
363
+ };
364
+ }
365
+ /** Stamp the asserting org onto a provenance ref, so a claim's origin says WHO
366
+ * asserted it and not merely which run id — `run:m4x2` alone is ambiguous
367
+ * across orgs, and a promoted claim in the shared graph would otherwise carry
368
+ * an origin no one owns.
369
+ *
370
+ * Applied by the ingest/rollback entry points rather than by callers: a
371
+ * caller that forgets is exactly how ownership stopped being enforced. */
372
+ export function kgQualifyOrigin(originRef, scope) {
373
+ const org = scope?.org?.trim();
374
+ return org ? `org:${normalizeName(org)}/${originRef}` : originRef;
375
+ }
376
+ function claimsOf(md) {
377
+ const raw = md.claims;
378
+ if (Array.isArray(raw))
379
+ return raw
380
+ .filter((c) => c && typeof c.origin === 'string')
381
+ .map((c) => ({
382
+ origin: c.origin,
383
+ description: typeof c.description === 'string' ? c.description : '',
384
+ at: typeof c.at === 'number' ? c.at : 0,
385
+ ...(c.method === 'asserted' || c.method === 'heuristic' ? { method: c.method } : {}),
386
+ }));
387
+ // Pre-ledger row (including every entry written before KG_ID_VERSION 2):
388
+ // seed one contribution per recorded origin, all carrying the one description
389
+ // the merge left behind. That is the honest reconstruction — the old merge
390
+ // destroyed which origin said what — and it preserves current rollback
391
+ // behaviour exactly: withdrawing one of several origins leaves the summary
392
+ // unchanged, withdrawing the last deletes the element.
393
+ const origins = Array.isArray(md.origin_refs) ? md.origin_refs : [];
394
+ const description = typeof md.description === 'string' ? md.description : '';
395
+ const at = typeof md.valid_from === 'number' ? md.valid_from : 0;
396
+ return origins.map((origin) => ({ origin, description, at }));
397
+ }
398
+ /** Add or replace `origin`'s contribution and re-derive everything that hangs
399
+ * off the ledger. */
400
+ function applyClaim(md, origin, description, now, method = 'asserted') {
401
+ const existing = claimsOf(md);
402
+ const dropped = typeof md.origins_dropped === 'number' ? md.origins_dropped : 0;
403
+ // An EMPTY description is support without content — "this entity exists",
404
+ // which is what naming an edge endpoint asserts. It must never overwrite what
405
+ // the same origin already said, or an edge listing a node the same payload
406
+ // described would blank that description out.
407
+ const prior = existing.find((c) => c.origin === origin);
408
+ if (!description.trim() && prior?.description.trim())
409
+ return deriveClaims(existing, dropped);
410
+ const claims = existing.filter((c) => c.origin !== origin);
411
+ claims.push({ origin, description, at: now, method });
412
+ return deriveClaims(claims, dropped);
413
+ }
414
+ /** The element's standing, from its surviving claims.
415
+ *
416
+ * One asserted claim outranks any number of co-occurrence guesses: a fact
417
+ * someone stated does not become less stated because a heuristic also stumbled
418
+ * onto it. A claim with no recorded method makes the whole projection UNKNOWN
419
+ * rather than voting — an old row cannot be read as evidence either way. */
420
+ function deriveMethod(claims) {
421
+ if (!claims.length || claims.some((c) => !c.method))
422
+ return undefined;
423
+ return claims.some((c) => c.method === 'asserted') ? 'asserted' : 'heuristic';
424
+ }
425
+ function deriveClaims(claims, alreadyDropped) {
426
+ let dropped = alreadyDropped;
427
+ if (claims.length > MAX_CLAIMS) {
428
+ // Oldest support goes first, and the loss is RECORDED. The scheme this
429
+ // replaced did the same drop silently and left the entry claiming a
430
+ // complete history it no longer had.
431
+ dropped += claims.length - MAX_CLAIMS;
432
+ claims = claims.slice(-MAX_CLAIMS);
433
+ }
434
+ return {
435
+ claims,
436
+ description: currentDescription(claims),
437
+ origin_refs: claims.map((c) => c.origin),
438
+ provenance_complete: dropped === 0,
439
+ origins_dropped: dropped,
440
+ conflict: new Set(claims.map((c) => c.description.trim()).filter(Boolean)).size > 1,
441
+ method: deriveMethod(claims),
442
+ };
443
+ }
444
+ /** The current summary: the most recent contribution that actually says
445
+ * something. LATEST wins, not longest — length measures verbosity, and
446
+ * "longest wins" is why re-ingesting a corrected "Now PostgreSQL" left the
447
+ * stale MySQL blurb in place. */
448
+ function currentDescription(claims) {
449
+ let best;
450
+ for (const c of claims) {
451
+ if (!c.description.trim())
452
+ continue;
453
+ if (!best || c.at >= best.at)
454
+ best = c;
455
+ }
456
+ return best?.description ?? '';
457
+ }
458
+ /** Withdraw one origin. Returns null when nothing supports the element any
459
+ * more (the caller deletes it); otherwise the RE-DERIVED state, which is what
460
+ * restores a previous correct description after a bad update is rolled back. */
461
+ function withoutOrigin(md, origin) {
462
+ const remaining = claimsOf(md).filter((c) => c.origin !== origin);
463
+ if (!remaining.length)
464
+ return null;
465
+ return deriveClaims(remaining, typeof md.origins_dropped === 'number' ? md.origins_dropped : 0);
466
+ }
467
+ function nameIndexKey(name) {
468
+ return `nm:${hashTuple([String(KG_ID_VERSION), canonicalName(name)])}`;
469
+ }
470
+ /** @returns the indexed entities, or null when the backend could not be read.
471
+ *
472
+ * The null is load-bearing: an unreadable index looks exactly like an empty
473
+ * one, and treating a read failure as "no entities with this name" would make
474
+ * the next write REPLACE the row with a single entry, erasing every other
475
+ * same-name entity from the index. A later generic assertion would then see one
476
+ * candidate and adopt it — a silent wrong merge caused by a read hiccup. */
477
+ async function readNameIndex(name, ns, dbPath) {
478
+ const res = await bridgeGetEntry({ key: nameIndexKey(name), namespace: ns.names, dbPath });
479
+ if (!res)
480
+ return null;
481
+ const raw = res.entry?.metadata?.entities;
482
+ if (!Array.isArray(raw))
483
+ return [];
484
+ return raw.filter((c) => c && typeof c.id === 'string');
485
+ }
486
+ /** Persist (or clear) the index row for one name.
487
+ *
488
+ * The index is a HINT, not a source of truth: it is written after the entity
489
+ * write lands, and a row that goes stale — pointing at an entity a rollback
490
+ * removed — self-heals, because resolving onto a missing id makes the next
491
+ * ingest create that entity fresh with an empty claim ledger, which is exactly
492
+ * what a new entity is. Nothing reads the index to decide what the graph
493
+ * contains; `kgStats`, `kgSearch` and `kgRollback` all read the claim
494
+ * namespaces. */
495
+ async function writeNameIndex(name, entities, ns, dbPath, failures) {
496
+ const key = nameIndexKey(name);
497
+ if (!entities.length) {
498
+ await bridgeDeleteEntry({ key, namespace: ns.names, dbPath });
499
+ return;
500
+ }
501
+ const res = await bridgeStoreEntry({
502
+ key,
503
+ value: name,
504
+ namespace: ns.names,
505
+ dbPath,
506
+ upsert: true,
507
+ generateEmbeddingFlag: false,
508
+ tags: ['kg', 'name-index'],
509
+ metadata: { kg: 'name', name, entities },
510
+ });
511
+ failures.add(res, `name index ${key}`);
512
+ }
513
+ /**
514
+ * Resolve (type, name) to the entity that should carry the assertion.
515
+ *
516
+ * Identity is the (type, name) tuple, so distinct types are distinct entities.
517
+ * The one thing name-only identity got right — a generic label must not fork an
518
+ * entity away from its typed self — is preserved here instead of in the key:
519
+ *
520
+ * - exact type-bucket match → that entity
521
+ * - generic assertion, exactly one same-name entity → adopt it
522
+ * - typed assertion, exactly one same-name entity and it is untyped → adopt it
523
+ * and promote its type (its ID does not change, so references stay valid)
524
+ * - anything else → a new entity, with the alternatives as candidates
525
+ *
526
+ * @returns null when the name index could not be read — the caller must refuse
527
+ * the assertion rather than resolve it against an index it could not see.
528
+ */
529
+ async function resolveEntity(name, type, ns, dbPath) {
530
+ const bucket = typeBucket(type);
531
+ let known = await readNameIndex(name, ns, dbPath);
532
+ if (known === null)
533
+ return null;
534
+ /** True when the index row already holds exactly `known` — the only case in
535
+ * which an unchanged resolution needs no index write. */
536
+ let indexed = known.length > 0;
537
+ if (!indexed) {
538
+ // Nothing indexed. Before minting, probe the pre-KG_ID_VERSION key: an
539
+ // existing graph's rows are adopted IN PLACE (keeping their key, and so
540
+ // keeping the edges whose keys embed it) rather than re-keyed or orphaned.
541
+ const legacyKey = legacyNodeKey(name);
542
+ const legacy = await bridgeGetEntry({ key: legacyKey, namespace: ns.nodes, dbPath });
543
+ if (legacy?.found && legacy.entry) {
544
+ const md = legacy.entry.metadata;
545
+ known = [{ id: legacyKey, type: typeBucket(typeof md.type === 'string' ? md.type : '') }];
546
+ indexed = false;
547
+ }
548
+ }
549
+ const exact = known.find((c) => c.type === bucket);
550
+ if (exact)
551
+ return { id: exact.id, candidates: others(known, exact.id), index: indexed ? null : known };
552
+ // A generic assertion adopts the one entity that bears this name; a typed
553
+ // assertion adopts a lone UNTYPED entity and PROMOTES it — same id, so every
554
+ // edge and returned reference to it stays valid. A typed assertion never
555
+ // absorbs a differently-typed entity, and neither adopts when the name is
556
+ // already ambiguous.
557
+ if (known.length === 1 && (!bucket || !known[0].type)) {
558
+ const only = known[0];
559
+ const promoted = { id: only.id, type: bucket || only.type };
560
+ return { id: only.id, candidates: [], index: [promoted] };
561
+ }
562
+ const minted = { id: mintEntityId(type, name, known), type: bucket };
563
+ return { id: minted.id, candidates: known, index: [...known, minted] };
564
+ }
565
+ /** An ID for a new entity that no entity under this name already uses.
566
+ *
567
+ * `nodeKey(type, name)` alone is not sufficient, because promotion decouples
568
+ * an entity's ID from its current type: an entity minted untyped keeps the
569
+ * `''`-bucket ID after a typed assertion promotes it, so a LATER untyped
570
+ * assertion would re-derive that same ID and write its claims into the
571
+ * promoted entity — while simultaneously reporting that entity as one it had
572
+ * "kept separate from". The discriminator is bumped only on collision, so the
573
+ * first entity of a (type, name) still gets exactly `nodeKey(type, name)`. */
574
+ function mintEntityId(type, name, known) {
575
+ const taken = new Set(known.map((c) => c.id));
576
+ let id = nodeKey(type, name);
577
+ for (let n = 1; taken.has(id); n++)
578
+ id = `n:${hashTuple([String(KG_ID_VERSION), typeBucket(type), canonicalName(name), String(n)])}`;
579
+ return id;
580
+ }
581
+ function others(known, id) {
582
+ return known.filter((c) => c.id !== id);
583
+ }
50
584
  // ── Ingest ──────────────────────────────────────────────────────────
51
- /** Idempotently merge extracted nodes/edges into the KG. Same-name entities
52
- * collapse onto one node (deterministic key + upsert); origin_refs accumulate
53
- * so rollback can undo a single run's contribution. */
585
+ /** Merge extracted nodes/edges into the KG.
586
+ *
587
+ * Identity is the (type, name) tuple resolved through the name index, so an
588
+ * entity is idempotent under re-extraction but two different things sharing a
589
+ * name stay two things. Each write adds this origin's CONTRIBUTION to the
590
+ * element's claim ledger, from which the description and `origin_refs` are
591
+ * derived — which is what lets a later ingest correct an earlier one and lets
592
+ * rollback put the earlier one back.
593
+ *
594
+ * The COMPLETE payload is validated before anything is written, and every edge
595
+ * endpoint is made to exist (as a placeholder entity when the caller named one
596
+ * that does not) before the edge lands. An edge whose endpoint could not be
597
+ * created is rejected rather than written: retrieval is seeded from nodes, so
598
+ * an edge with a missing endpoint is a fact that cannot be found through
599
+ * either of the things it is about.
600
+ *
601
+ * NOT ATOMIC — the memory bridge has no transaction primitive, so a failure
602
+ * part-way through leaves earlier writes persisted. The counters therefore
603
+ * report only what actually landed, `failures` lists what did not, and
604
+ * `success` is false whenever anything was refused. A partial ingest is safe
605
+ * to retry: every write is a keyed upsert. */
54
606
  export async function kgIngest(options) {
55
- const { originRef, dbPath } = options;
607
+ const { dbPath } = options;
608
+ const method = options.method ?? 'asserted';
609
+ const ns = kgNamespaces(options.scope);
610
+ const originRef = kgQualifyOrigin(options.originRef, options.scope);
611
+ const failures = new FailureLog();
612
+ const report = new IngestReport();
56
613
  let nodesAdded = 0, nodesMerged = 0, edgesAdded = 0, edgesMerged = 0;
57
614
  try {
58
- const keyByName = new Map();
59
- for (const n of (options.nodes ?? []).slice(0, 500)) {
615
+ // ── Validate the COMPLETE payload before mutating anything ──
616
+ // An invalid item that surfaces halfway through leaves the graph holding
617
+ // part of a payload the caller was told was rejected.
618
+ const allNodes = options.nodes ?? [];
619
+ const allEdges = options.edges ?? [];
620
+ const nodes = allNodes.slice(0, MAX_NODES_PER_CALL);
621
+ const edges = allEdges.slice(0, MAX_EDGES_PER_CALL);
622
+ report.nodesTruncated = allNodes.length - nodes.length;
623
+ report.edgesTruncated = allEdges.length - edges.length;
624
+ const validNodes = [];
625
+ nodes.forEach((n, i) => {
60
626
  if (!n?.name?.trim())
627
+ return report.rejectNode(`node[${i}]: missing name`);
628
+ validNodes.push({
629
+ input: n,
630
+ type: n.type?.trim() || 'entity',
631
+ desc: (n.description ?? '').slice(0, MAX_DESC_LEN),
632
+ });
633
+ });
634
+ const validEdges = [];
635
+ edges.forEach((e, i) => {
636
+ const missing = !e?.source?.trim()
637
+ ? 'source'
638
+ : !e?.target?.trim()
639
+ ? 'target'
640
+ : !e?.relation?.trim()
641
+ ? 'relation'
642
+ : '';
643
+ if (missing)
644
+ return report.rejectEdge(`edge[${i}]: missing ${missing}`);
645
+ validEdges.push(e);
646
+ });
647
+ /** Entity IDs this call has already resolved AND confirmed to exist, so an
648
+ * edge does not re-resolve an endpoint the node loop just wrote. */
649
+ const resolved = new Map();
650
+ // Length-prefix the type so the split point is unambiguous regardless of
651
+ // what characters `name` contains (it can hold arbitrary text, including
652
+ // whatever separator a naive join might pick) — same discipline
653
+ // hashTuple() above uses for the same reason. A bare `\0`-joined string
654
+ // here was written as an actual NUL byte, not the literal two-character
655
+ // escape text, which is harmless at runtime (this key is an in-memory
656
+ // Map key only, never persisted) but made git treat this file as binary.
657
+ const memoKey = (name, type) => {
658
+ const t = typeBucket(type);
659
+ return `${t.length}:${t}${canonicalName(name)}`;
660
+ };
661
+ for (const { input: n, type, desc } of validNodes) {
662
+ const target = await resolveEntity(n.name, type, ns, dbPath);
663
+ if (!target) {
664
+ failures.note(`node ${n.name}: name index unreadable`);
61
665
  continue;
62
- const type = n.type?.trim() || 'entity';
63
- const key = nodeKey(type, n.name);
64
- keyByName.set(normalizeName(n.name), key);
65
- const desc = (n.description ?? '').slice(0, MAX_DESC_LEN);
66
- const existing = await bridgeGetEntry({ key, namespace: KG_NODES_NS, dbPath });
67
- if (existing?.found && existing.entry) {
68
- const md = existing.entry.metadata;
69
- const origins = Array.isArray(md.origin_refs) ? md.origin_refs : [];
70
- if (!origins.includes(originRef))
71
- origins.push(originRef);
72
- // Prefer the richer description; never let a terse re-extraction erase detail.
73
- const prevDesc = typeof md.description === 'string' ? md.description : '';
74
- const bestDesc = desc.length > prevDesc.length ? desc : prevDesc;
75
- // Keep the most specific type: a generic heuristic 'entity' never
76
- // overwrites an LLM-assigned type.
77
- const prevType = typeof md.type === 'string' ? md.type : 'entity';
78
- const bestType = prevType.toLowerCase() !== 'entity' ? prevType : type;
79
- await bridgeStoreEntry({
80
- key,
81
- value: `${n.name} — ${bestDesc || bestType}`,
82
- namespace: KG_NODES_NS,
83
- dbPath,
84
- upsert: true,
85
- tags: ['kg', normalizeName(bestType), ...(n.nodeSet ? [normalizeName(n.nodeSet)] : [])],
86
- metadata: {
87
- ...md,
88
- kg: 'node',
89
- type: bestType,
90
- name: n.name,
91
- description: bestDesc,
92
- node_set: n.nodeSet ?? md.node_set ?? null,
93
- origin_refs: origins.slice(-100),
94
- version: (typeof md.version === 'number' ? md.version : 1) + 1,
95
- valid_from: md.valid_from ?? Date.now(),
96
- valid_to: null,
97
- },
98
- });
99
- nodesMerged++;
100
666
  }
101
- else {
102
- await bridgeStoreEntry({
103
- key,
104
- value: `${n.name} — ${desc || type}`,
105
- namespace: KG_NODES_NS,
106
- dbPath,
107
- upsert: true,
108
- tags: ['kg', normalizeName(type), ...(n.nodeSet ? [normalizeName(n.nodeSet)] : [])],
109
- metadata: {
110
- kg: 'node',
111
- type,
112
- name: n.name,
113
- description: desc,
114
- node_set: n.nodeSet ?? null,
115
- origin_refs: [originRef],
116
- version: 1,
117
- valid_from: Date.now(),
118
- valid_to: null,
119
- },
120
- });
667
+ report.noteAmbiguity(n.name, target.candidates);
668
+ const wrote = await writeEntity({
669
+ id: target.id,
670
+ name: n.name,
671
+ type,
672
+ description: desc,
673
+ nodeSet: n.nodeSet,
674
+ originRef,
675
+ method,
676
+ ns,
677
+ dbPath,
678
+ failures,
679
+ report,
680
+ });
681
+ if (wrote === null)
682
+ continue;
683
+ if (wrote)
121
684
  nodesAdded++;
122
- }
685
+ else
686
+ nodesMerged++;
687
+ resolved.set(memoKey(n.name, type), target.id);
688
+ if (target.index)
689
+ await writeNameIndex(n.name, target.index, ns, dbPath, failures);
690
+ await onEntrySupported(ns, ns.nodes, target.id, originRef, dbPath);
123
691
  }
124
- for (const e of (options.edges ?? []).slice(0, 1000)) {
125
- if (!e?.source?.trim() || !e?.target?.trim() || !e?.relation?.trim())
692
+ /** Resolve an edge endpoint, creating an explicit placeholder entity when
693
+ * the caller named something that does not exist. Returns null when the
694
+ * endpoint could not be made to exist — the edge is then rejected rather
695
+ * than written without it. */
696
+ const endpoint = async (name, type) => {
697
+ const memo = memoKey(name, type ?? 'entity');
698
+ const hit = resolved.get(memo);
699
+ if (hit)
700
+ return hit;
701
+ const target = await resolveEntity(name, type ?? 'entity', ns, dbPath);
702
+ if (!target) {
703
+ failures.note(`endpoint ${name}: name index unreadable`);
704
+ return null;
705
+ }
706
+ report.noteAmbiguity(name, target.candidates);
707
+ // Always write: naming an endpoint IS an assertion that it exists, so
708
+ // this origin joins the entity's support either way. A write that CREATES
709
+ // the entity is the placeholder case worth reporting.
710
+ const wrote = await writeEntity({
711
+ id: target.id,
712
+ name,
713
+ type: type ?? 'entity',
714
+ description: '',
715
+ placeholder: true,
716
+ originRef,
717
+ method,
718
+ ns,
719
+ dbPath,
720
+ failures,
721
+ report,
722
+ });
723
+ if (wrote === null)
724
+ return null;
725
+ if (wrote)
726
+ report.placeholders++;
727
+ if (target.index)
728
+ await writeNameIndex(name, target.index, ns, dbPath, failures);
729
+ resolved.set(memo, target.id);
730
+ await onEntrySupported(ns, ns.nodes, target.id, originRef, dbPath);
731
+ return target.id;
732
+ };
733
+ for (const e of validEdges) {
734
+ const srcKey = await endpoint(e.source, e.sourceType);
735
+ const dstKey = await endpoint(e.target, e.targetType);
736
+ if (!srcKey || !dstKey) {
737
+ report.rejectEdge(`edge ${e.source}-${e.relation}->${e.target}: endpoint not persisted`);
126
738
  continue;
127
- const srcKey = keyByName.get(normalizeName(e.source)) ?? nodeKey(e.sourceType ?? 'entity', e.source);
128
- const dstKey = keyByName.get(normalizeName(e.target)) ?? nodeKey(e.targetType ?? 'entity', e.target);
129
- const key = edgeKey(srcKey, e.relation, dstKey);
739
+ }
130
740
  const desc = (e.description ?? '').slice(0, MAX_DESC_LEN);
131
- const existing = await bridgeGetEntry({ key, namespace: KG_EDGES_NS, dbPath });
132
- if (existing?.found && existing.entry) {
133
- const md = existing.entry.metadata;
134
- const origins = Array.isArray(md.origin_refs) ? md.origin_refs : [];
135
- if (!origins.includes(originRef))
136
- origins.push(originRef);
137
- await bridgeStoreEntry({
138
- key,
139
- value: desc || `${e.source} ${e.relation} ${e.target}`,
140
- namespace: KG_EDGES_NS,
141
- dbPath,
142
- upsert: true,
143
- generateEmbeddingFlag: false,
144
- tags: ['kg', normalizeName(e.relation)],
145
- metadata: { ...md, origin_refs: origins.slice(-100) },
146
- });
147
- edgesMerged++;
741
+ const fallbackFact = `${e.source} ${e.relation} ${e.target}`;
742
+ // New key first, then the pre-KG_ID_VERSION shape: a legacy edge between
743
+ // two adopted legacy endpoints keeps its own key rather than being
744
+ // duplicated under a new one. Resolved once — a CAS retry re-reads this
745
+ // same key, it does not re-run legacy resolution.
746
+ let key = edgeKey(srcKey, e.relation, dstKey);
747
+ const probe = await bridgeGetEntry({ key, namespace: ns.edges, dbPath });
748
+ if (!probe?.found) {
749
+ const legacy = legacyEdgeKey(srcKey, e.relation, dstKey);
750
+ const hit = await bridgeGetEntry({ key: legacy, namespace: ns.edges, dbPath });
751
+ if (hit?.found && hit.entry)
752
+ key = legacy;
148
753
  }
149
- else {
150
- await bridgeStoreEntry({
754
+ // Re-read and re-merge on every CAS attempt (see withCasRetry): a stale
755
+ // `md`/`derived` retried against a fresh row would silently re-lose
756
+ // whatever a concurrent writer just added (K5).
757
+ let md = {};
758
+ let derived;
759
+ let isNew = false;
760
+ const res = await withCasRetry(async () => {
761
+ const existing = await bridgeGetEntry({ key, namespace: ns.edges, dbPath });
762
+ isNew = !(existing?.found && existing.entry);
763
+ md = (existing?.entry?.metadata ?? {});
764
+ derived = applyClaim(md, originRef, desc, Date.now(), method);
765
+ const ver = existing?.entry?.version;
766
+ return bridgeStoreEntry({
151
767
  key,
152
- value: desc || `${e.source} ${e.relation} ${e.target}`,
153
- namespace: KG_EDGES_NS,
768
+ value: derived.description || fallbackFact,
769
+ namespace: ns.edges,
154
770
  dbPath,
155
771
  upsert: true,
156
772
  generateEmbeddingFlag: false,
157
773
  tags: ['kg', normalizeName(e.relation)],
158
774
  metadata: {
775
+ ...md,
159
776
  kg: 'edge',
777
+ id_version: KG_ID_VERSION,
160
778
  src: srcKey,
161
779
  dst: dstKey,
162
780
  relation: normalizeName(e.relation),
163
781
  source_name: e.source,
164
782
  target_name: e.target,
165
- description: desc,
166
- origin_refs: [originRef],
167
- valid_from: Date.now(),
783
+ ...derived,
784
+ valid_from: md.valid_from ?? Date.now(),
168
785
  valid_to: null,
169
786
  },
787
+ ifVersion: isNew ? 'absent' : typeof ver === 'number' ? ver : undefined,
170
788
  });
789
+ });
790
+ if (failures.add(res, `edge ${key}`))
791
+ continue;
792
+ report.noteProvenanceLoss(md, derived);
793
+ if (isNew)
171
794
  edgesAdded++;
172
- }
795
+ else
796
+ edgesMerged++;
797
+ await onEntrySupported(ns, ns.edges, key, originRef, dbPath);
798
+ if (isNew)
799
+ await onEdgeWritten(ns, key, srcKey, dstKey, dbPath);
173
800
  }
174
- return { success: true, nodesAdded, nodesMerged, edgesAdded, edgesMerged };
801
+ return {
802
+ success: !failures.failed,
803
+ nodesAdded,
804
+ nodesMerged,
805
+ edgesAdded,
806
+ edgesMerged,
807
+ ...(failures.failed ? { failures: failures.messages, error: failures.summary() } : {}),
808
+ ...report.fields(),
809
+ };
175
810
  }
176
811
  catch (err) {
177
812
  return {
@@ -180,22 +815,149 @@ export async function kgIngest(options) {
180
815
  nodesMerged,
181
816
  edgesAdded,
182
817
  edgesMerged,
818
+ ...(failures.messages.length ? { failures: failures.messages } : {}),
819
+ ...report.fields(),
183
820
  error: err instanceof Error ? err.message : String(err),
184
821
  };
185
822
  }
186
823
  }
824
+ /** Write one entity's claim contribution.
825
+ *
826
+ * @returns true when the entity was created, false when an existing one was
827
+ * merged into, and null when the bridge refused the write (the caller must not
828
+ * count it, and must not treat the entity as existing). */
829
+ async function writeEntity(o) {
830
+ // Re-read and re-merge on every CAS attempt (see withCasRetry) rather than
831
+ // once up front: a stale `md`/`derived` retried against a fresh row would
832
+ // silently re-lose whatever a concurrent writer just added (K5).
833
+ let md = {};
834
+ let derived;
835
+ let isNew = false;
836
+ const res = await withCasRetry(async () => {
837
+ const existing = await bridgeGetEntry({ key: o.id, namespace: o.ns.nodes, dbPath: o.dbPath });
838
+ isNew = !(existing?.found && existing.entry);
839
+ md = (existing?.entry?.metadata ?? {});
840
+ derived = applyClaim(md, o.originRef, o.description, Date.now(), o.method);
841
+ // Keep the most specific type: a generic heuristic 'entity' never
842
+ // overwrites an LLM-assigned one, and a specific one promotes an untyped
843
+ // entity.
844
+ const prevType = typeof md.type === 'string' ? md.type : '';
845
+ const bestType = typeBucket(prevType) ? prevType : o.type;
846
+ const nodeSet = o.nodeSet ?? (typeof md.node_set === 'string' ? md.node_set : null);
847
+ const ver = existing?.entry?.version;
848
+ return bridgeStoreEntry({
849
+ key: o.id,
850
+ value: `${o.name} — ${derived.description || bestType}`,
851
+ namespace: o.ns.nodes,
852
+ dbPath: o.dbPath,
853
+ upsert: true,
854
+ tags: ['kg', normalizeName(bestType), ...(nodeSet ? [normalizeName(nodeSet)] : [])],
855
+ metadata: {
856
+ ...md,
857
+ kg: 'node',
858
+ id_version: KG_ID_VERSION,
859
+ type: bestType,
860
+ name: o.name,
861
+ node_set: nodeSet,
862
+ ...derived,
863
+ // A placeholder stops being one the moment a real assertion describes it.
864
+ placeholder: o.placeholder === true && !derived.description ? true : undefined,
865
+ version: (typeof md.version === 'number' ? md.version : 0) + 1,
866
+ valid_from: md.valid_from ?? Date.now(),
867
+ valid_to: null,
868
+ },
869
+ // isNew ⇒ this row must not already exist; otherwise it must still be at
870
+ // the version we just read. Absent when the loaded backend/test double
871
+ // does not report a version, so behaviour is unchanged there (see
872
+ // bridgeGetEntry's `version` field).
873
+ ifVersion: isNew ? 'absent' : typeof ver === 'number' ? ver : undefined,
874
+ });
875
+ });
876
+ // Name first: the ID is a digest, and a failure message a human cannot map
877
+ // back to the thing that failed is not a diagnosis.
878
+ if (o.failures.add(res, `node ${o.name} (${o.id})`))
879
+ return null;
880
+ // Only after the write landed: a refused write dropped no provenance, because
881
+ // it changed nothing.
882
+ o.report.noteProvenanceLoss(md, derived);
883
+ return isNew;
884
+ }
885
+ /** The accepted/rejected/truncated bookkeeping a caller needs in order to know
886
+ * that what it sent is what the graph holds. */
887
+ class IngestReport {
888
+ nodesRejected = 0;
889
+ edgesRejected = 0;
890
+ nodesTruncated = 0;
891
+ edgesTruncated = 0;
892
+ placeholders = 0;
893
+ provenanceTruncated = 0;
894
+ rejections = [];
895
+ ambiguities = [];
896
+ rejectNode(message) {
897
+ this.nodesRejected++;
898
+ if (this.rejections.length < MAX_FAILURES)
899
+ this.rejections.push(message);
900
+ }
901
+ rejectEdge(message) {
902
+ this.edgesRejected++;
903
+ if (this.rejections.length < MAX_FAILURES)
904
+ this.rejections.push(message);
905
+ }
906
+ noteAmbiguity(name, candidates) {
907
+ if (!candidates.length || this.ambiguities.length >= MAX_FAILURES)
908
+ return;
909
+ const types = candidates.map((c) => c.type || 'untyped').join(', ');
910
+ this.ambiguities.push(`${name}: kept separate from ${candidates.length} same-name (${types})`);
911
+ }
912
+ noteProvenanceLoss(md, derived) {
913
+ const before = typeof md.origins_dropped === 'number' ? md.origins_dropped : 0;
914
+ if (derived.origins_dropped > before)
915
+ this.provenanceTruncated++;
916
+ }
917
+ /** Only non-zero counters are emitted, so a clean payload yields a clean
918
+ * result and the existing `KgIngestResult` shape is unchanged for callers
919
+ * that construct it themselves. */
920
+ fields() {
921
+ return {
922
+ ...(this.nodesRejected ? { nodesRejected: this.nodesRejected } : {}),
923
+ ...(this.edgesRejected ? { edgesRejected: this.edgesRejected } : {}),
924
+ ...(this.nodesTruncated ? { nodesTruncated: this.nodesTruncated } : {}),
925
+ ...(this.edgesTruncated ? { edgesTruncated: this.edgesTruncated } : {}),
926
+ ...(this.rejections.length ? { rejections: this.rejections } : {}),
927
+ ...(this.placeholders ? { placeholders: this.placeholders } : {}),
928
+ ...(this.ambiguities.length ? { ambiguities: this.ambiguities } : {}),
929
+ ...(this.provenanceTruncated ? { provenanceTruncated: this.provenanceTruncated } : {}),
930
+ };
931
+ }
932
+ }
187
933
  /** Stage-2 of cognee's curator/writer distillation: the CALLER (an LLM agent)
188
934
  * proposes candidate rules; this accepts each unless a semantically
189
935
  * near-identical rule exists (embedding dedup — deterministic keys can't
190
936
  * collapse paraphrases). Accepted rules are stored both as KG nodes
191
937
  * (node_set=rules) and as plain `rules`-namespace entries so the existing
192
- * injection/search surfaces pick them up with zero new plumbing. */
938
+ * injection/search surfaces pick them up with zero new plumbing.
939
+ *
940
+ * A candidate that dedups against an existing rule still ADDS its origin to
941
+ * that rule's support set: two independent runs asserting the same rule mean
942
+ * the rule survives either one being rolled back. Dropping the second origin
943
+ * (as this used to) made rollback of the FIRST run delete knowledge the
944
+ * second run independently vouched for.
945
+ *
946
+ * Like `kgIngest`, NOT atomic — see that function's note. */
193
947
  export async function kgIngestRules(options) {
194
948
  const verdicts = [];
195
949
  const threshold = options.dedupThreshold ?? 0.78;
950
+ const failures = new FailureLog();
951
+ const ns = kgNamespaces(options.scope);
952
+ // Direct writes here store the qualified ref; nested kgIngest calls get the
953
+ // RAW ref plus the scope and qualify it themselves, so it is stamped once.
954
+ const originRef = kgQualifyOrigin(options.originRef, options.scope);
196
955
  let accepted = 0;
956
+ const all = options.rules ?? [];
957
+ const batch = all.slice(0, MAX_RULES_PER_CALL);
958
+ const rulesTruncated = all.length - batch.length;
197
959
  try {
198
- for (const r of (options.rules ?? []).slice(0, 50)) {
960
+ for (const r of batch) {
199
961
  const rule = r?.rule?.trim();
200
962
  if (!rule || rule.length < 8) {
201
963
  verdicts.push({ rule: r?.rule ?? '', verdict: 'invalid' });
@@ -203,7 +965,7 @@ export async function kgIngestRules(options) {
203
965
  }
204
966
  const similar = await bridgeSearchEntries({
205
967
  query: rule,
206
- namespace: RULES_NS,
968
+ namespace: ns.rules,
207
969
  limit: 1,
208
970
  threshold,
209
971
  dbPath: options.dbPath,
@@ -229,111 +991,347 @@ export async function kgIngestRules(options) {
229
991
  const candidate = rule.replace(/\s+/g, ' ').trim().toLowerCase();
230
992
  isDuplicate = existing === candidate;
231
993
  }
232
- if (isDuplicate) {
233
- verdicts.push({ rule, verdict: 'already_known', similarTo: top?.key });
994
+ if (isDuplicate && top?.key) {
995
+ // `already_known` is a claim that this origin now supports the existing
996
+ // rule. If reinforcement did not land, it does not — and a later
997
+ // rollback of the OTHER origin would delete a rule this run believes it
998
+ // vouched for.
999
+ const reinforced = await reinforceRuleOrigin(top.key, top.content, options.originRef, options.scope, options.dbPath, failures);
1000
+ verdicts.push({
1001
+ rule,
1002
+ verdict: reinforced ? 'already_known' : 'failed',
1003
+ similarTo: top.key,
1004
+ });
234
1005
  continue;
235
1006
  }
236
- const key = `rule:${normalizeName(rule).slice(0, 120)}`;
237
- await bridgeStoreEntry({
1007
+ // Identity is the full rule text. The old key truncated at 120 normalized
1008
+ // characters, so two rules sharing a long preamble were the same rule.
1009
+ // A rule stored under that scheme is adopted in place rather than
1010
+ // duplicated under the new key.
1011
+ let key = ruleKey(rule);
1012
+ let priorMd = {};
1013
+ const current = await bridgeGetEntry({ key, namespace: ns.rules, dbPath: options.dbPath });
1014
+ if (current?.found && current.entry)
1015
+ priorMd = current.entry.metadata;
1016
+ else {
1017
+ const legacy = legacyRuleKey(rule);
1018
+ const hit = await bridgeGetEntry({
1019
+ key: legacy,
1020
+ namespace: ns.rules,
1021
+ dbPath: options.dbPath,
1022
+ });
1023
+ if (hit?.found && hit.entry) {
1024
+ key = legacy;
1025
+ priorMd = hit.entry.metadata;
1026
+ }
1027
+ }
1028
+ // The mirrored KG node is named by the FULL rule text. Truncating it to
1029
+ // 200 characters put two rules sharing a long preamble on one node — the
1030
+ // same collision `ruleKey` was just widened to prevent, reintroduced one
1031
+ // namespace over.
1032
+ const ruleName = rule;
1033
+ const stored = await bridgeStoreEntry({
238
1034
  key,
239
1035
  value: rule + (r.context ? `\n(context: ${r.context.slice(0, 500)})` : ''),
240
- namespace: RULES_NS,
1036
+ namespace: ns.rules,
241
1037
  dbPath: options.dbPath,
242
1038
  upsert: true,
243
1039
  tags: ['rule'],
244
- metadata: { origin_refs: [options.originRef], derived_from: options.originRef },
1040
+ metadata: {
1041
+ ...priorMd,
1042
+ ...applyClaim(priorMd, originRef, rule, Date.now()),
1043
+ id_version: KG_ID_VERSION,
1044
+ derived_from: originRef,
1045
+ // Lets the dedup path reinforce the matching KG node without having
1046
+ // to re-derive the node name from the stored value (which may carry
1047
+ // an appended context block).
1048
+ rule: ruleName,
1049
+ },
245
1050
  });
246
- await kgIngest({
247
- nodes: [
248
- { name: rule.slice(0, MAX_NAME_LEN), type: 'Rule', description: rule, nodeSet: 'rules' },
249
- ],
1051
+ const nodeRes = await kgIngest({
1052
+ nodes: [{ name: ruleName, type: 'Rule', description: rule, nodeSet: 'rules' }],
250
1053
  originRef: options.originRef,
1054
+ scope: options.scope,
251
1055
  dbPath: options.dbPath,
252
1056
  });
253
- verdicts.push({ rule, verdict: 'accepted' });
254
- accepted++;
1057
+ const ruleFailed = failures.add(stored, `rule ${key}`);
1058
+ if (!ruleFailed)
1059
+ await onEntrySupported(ns, ns.rules, key, originRef, options.dbPath);
1060
+ if (nodeRes.failures?.length)
1061
+ for (const m of nodeRes.failures)
1062
+ failures.note(m);
1063
+ // Only count a rule as accepted when BOTH of its writes landed; a rule
1064
+ // present in one namespace only is not the state the caller was told
1065
+ // about. The per-rule verdict now says the same thing — it read
1066
+ // `accepted` unconditionally, so the two halves of one result disagreed.
1067
+ const landed = !ruleFailed && nodeRes.success;
1068
+ if (landed)
1069
+ accepted++;
1070
+ verdicts.push({ rule, verdict: landed ? 'accepted' : 'failed' });
255
1071
  }
256
- return { success: true, verdicts, accepted };
1072
+ return {
1073
+ success: !failures.failed,
1074
+ verdicts,
1075
+ accepted,
1076
+ ...(failures.failed ? { failures: failures.messages, error: failures.summary() } : {}),
1077
+ ...(rulesTruncated ? { rulesTruncated } : {}),
1078
+ };
257
1079
  }
258
1080
  catch (err) {
259
1081
  return {
260
1082
  success: false,
261
1083
  verdicts,
262
1084
  accepted,
1085
+ ...(failures.messages.length ? { failures: failures.messages } : {}),
1086
+ ...(rulesTruncated ? { rulesTruncated } : {}),
263
1087
  error: err instanceof Error ? err.message : String(err),
264
1088
  };
265
1089
  }
266
1090
  }
1091
+ /** Add `originRef` to an already-stored rule's support set — both the
1092
+ * `rules`-namespace entry and its `node_set=rules` KG node. Idempotent in
1093
+ * effect: re-asserting an origin the rule already carries leaves the support
1094
+ * set unchanged.
1095
+ *
1096
+ * @returns true when the rule genuinely carries this origin afterwards —
1097
+ * including the no-op case where it already did. False means the support set
1098
+ * is not what the caller is about to be told it is. */
1099
+ async function reinforceRuleOrigin(matchedKey, matchedContent,
1100
+ /** RAW ref — qualified here for the rules entry, and passed on unqualified
1101
+ * to `kgIngest`, which qualifies it once with the same scope. */
1102
+ originRef, scope, dbPath, failures) {
1103
+ const ns = kgNamespaces(scope);
1104
+ const qualified = kgQualifyOrigin(originRef, scope);
1105
+ const existing = await bridgeGetEntry({ key: matchedKey, namespace: ns.rules, dbPath });
1106
+ if (!existing?.found || !existing.entry) {
1107
+ // The dedup hit came from search; if the entry can't be re-read by key the
1108
+ // support set cannot be updated, and silently proceeding is exactly the
1109
+ // provenance loss this function exists to prevent.
1110
+ failures.note(`rule ${matchedKey}: matched by dedup but not readable by key`);
1111
+ return false;
1112
+ }
1113
+ const entry = existing.entry;
1114
+ const md = entry.metadata;
1115
+ const origins = Array.isArray(md.origin_refs) ? md.origin_refs : [];
1116
+ // The KG node's name: recorded at accept time, else the first line of the
1117
+ // stored value (any context block is appended after a newline).
1118
+ const ruleName = typeof md.rule === 'string' && md.rule
1119
+ ? md.rule
1120
+ : (matchedContent || entry.content).split('\n')[0];
1121
+ // Both writes are attempted regardless: skipping the node write because the
1122
+ // entry write failed would leave the two halves disagreeing about who
1123
+ // supports this rule. Only the REPORT changes.
1124
+ let ok = true;
1125
+ if (!origins.includes(qualified)) {
1126
+ const res = await bridgeStoreEntry({
1127
+ key: entry.key,
1128
+ value: entry.content,
1129
+ namespace: ns.rules,
1130
+ dbPath,
1131
+ upsert: true,
1132
+ generateEmbeddingFlag: entry.hasEmbedding,
1133
+ tags: entry.tags,
1134
+ metadata: {
1135
+ ...md,
1136
+ rule: ruleName,
1137
+ // A dedup hit is SUPPORT, not a correction: the candidate asserts the
1138
+ // rule the entry already holds. Its contribution therefore carries no
1139
+ // description of its own — passing one here (the truncated `rule` name,
1140
+ // as this did) made the newest claim disagree with the stored text and
1141
+ // flagged the rule as conflicted with a truncation of itself.
1142
+ ...applyClaim(md, qualified, '', Date.now()),
1143
+ },
1144
+ });
1145
+ if (failures.add(res, `rule ${entry.key}`))
1146
+ ok = false;
1147
+ else
1148
+ await onEntrySupported(ns, ns.rules, entry.key, qualified, dbPath);
1149
+ }
1150
+ // The rule's KG node needs the same origin — rollback walks nodes separately.
1151
+ const nodeRes = await kgIngest({
1152
+ nodes: [{ name: ruleName, type: 'Rule', description: '', nodeSet: 'rules' }],
1153
+ originRef,
1154
+ scope,
1155
+ dbPath,
1156
+ });
1157
+ if (nodeRes.failures?.length)
1158
+ for (const m of nodeRes.failures)
1159
+ failures.note(m);
1160
+ return ok && nodeRes.success;
1161
+ }
267
1162
  /** List stored rules (for injection or review). */
268
1163
  export async function kgListRules(options) {
269
1164
  const res = await bridgeListEntries({
270
- namespace: RULES_NS,
1165
+ namespace: kgNamespaces(options?.scope).rules,
271
1166
  limit: options?.limit ?? 50,
272
1167
  dbPath: options?.dbPath,
273
1168
  });
274
1169
  return (res?.entries ?? []).map((e) => ({ rule: e.content, key: e.key }));
275
1170
  }
276
- /** Vector-seed → neighborhood → triplet ranking (cognee's brute-force triplet
277
- * search, scaled down). Seed scores already carry the Phase 1 feedback blend. */
1171
+ /** Seed candidates pulled before filtering and ranking. */
1172
+ const SEARCH_SEED_LIMIT = 15;
1173
+ /** Extra candidates fetched when a `nodeSet` narrows the graph.
1174
+ *
1175
+ * The bridge has no tag filter, so set membership can only be tested after
1176
+ * retrieval. Filtering the unfiltered top-15 meant a node that IS in the set
1177
+ * but ranks 16th overall was missed — the set made results scarcer instead of
1178
+ * more precise. Over-fetching moves the cutoff after the filter. */
1179
+ const SEARCH_NODE_SET_OVERFETCH = 4;
1180
+ /** How far a co-occurrence guess drops below an equally-seeded stated fact.
1181
+ * Enough to lose a tie, not enough to hide it: `mentioned_with` between two
1182
+ * strong seeds is still worth surfacing when nothing better was asserted. */
1183
+ const HEURISTIC_PENALTY = 0.15;
1184
+ /** Live origins disagree about what this edge says. Still returned — a disputed
1185
+ * fact is information — but it does not outrank a settled one. */
1186
+ const CONFLICT_PENALTY = 0.1;
1187
+ /** Seeded retrieval → neighborhood → triplet ranking (cognee's brute-force
1188
+ * triplet search, scaled down). Seed scores already carry the Phase 1 feedback
1189
+ * blend, and the seed retrieval may be vector or keyword — `method` on the
1190
+ * result says which actually ran.
1191
+ *
1192
+ * Ranking weighs exactly two evidence signals, both read off the claim ledger:
1193
+ * extraction method and description conflict. It deliberately does NOT model
1194
+ * source credibility, claim freshness, or whether the relation itself answers
1195
+ * the query — those need an evaluation set to tune against, and guessing at
1196
+ * them would be the same overclaim this weighting exists to correct. */
278
1197
  export async function kgSearch(options) {
279
1198
  try {
280
1199
  const limit = options.limit ?? 8;
1200
+ const ns = kgNamespaces(options.scope);
281
1201
  const seedsRes = await bridgeSearchEntries({
282
1202
  query: options.query,
283
- namespace: KG_NODES_NS,
284
- limit: 15,
1203
+ namespace: ns.nodes,
1204
+ // Over-fetch when a set filter follows, so the cutoff lands AFTER it.
1205
+ limit: options.nodeSet ? SEARCH_SEED_LIMIT * SEARCH_NODE_SET_OVERFETCH : SEARCH_SEED_LIMIT,
285
1206
  threshold: 0.25,
286
1207
  dbPath: options.dbPath,
287
1208
  });
1209
+ // What the retrieval actually was, carried on every return below: a keyword
1210
+ // fallback presented as vector-seeded search is the overclaim B5 names.
1211
+ const retrieval = {
1212
+ ...(seedsRes?.searchMethod ? { method: seedsRes.searchMethod } : {}),
1213
+ ...(seedsRes?.fallbackReason ? { fallbackReason: seedsRes.fallbackReason } : {}),
1214
+ };
288
1215
  let seedResults = seedsRes?.results ?? [];
289
1216
  if (options.nodeSet) {
290
- const ns = normalizeName(options.nodeSet);
291
- seedResults = seedResults.filter((r) => (r.tags ?? []).includes(ns));
1217
+ const setTag = normalizeName(options.nodeSet);
1218
+ seedResults = seedResults.filter((r) => (r.tags ?? []).includes(setTag));
292
1219
  }
1220
+ seedResults = seedResults.slice(0, SEARCH_SEED_LIMIT);
293
1221
  if (!seedResults.length)
294
- return { success: true, context: '', triplets: [], seeds: [] };
1222
+ return { success: true, context: '', triplets: [], seeds: [], ...retrieval };
295
1223
  const seedScore = new Map();
296
1224
  for (const s of seedResults)
297
1225
  seedScore.set(s.key, s.score);
298
- // Full edge scan (see monolean note in module header).
299
- const edgesRes = await bridgeListEntries({
300
- namespace: KG_EDGES_NS,
301
- limit: MAX_LIST,
302
- dbPath: options.dbPath,
303
- });
304
- const edges = (edgesRes?.entries ?? []).filter((e) => {
305
- const md = e.metadata;
306
- return md?.kg === 'edge' && md.valid_to == null;
307
- });
308
- const triplets = edges
309
- .map((e) => {
310
- const md = e.metadata;
1226
+ const triplets = [];
1227
+ let scannedEdges = 0;
1228
+ let truncated = false;
1229
+ /** Score one edge against the seeded entities and, if relevant, push its
1230
+ * triplet — shared by the indexed and exhaustive gathering paths below
1231
+ * so ranking never depends on which one ran (K7). */
1232
+ const considerEdge = (e) => {
1233
+ scannedEdges++;
1234
+ const md = (e.metadata ?? {});
1235
+ if (md?.kg !== 'edge' || md.valid_to != null)
1236
+ return;
311
1237
  const src = String(md.src ?? '');
312
1238
  const dst = String(md.dst ?? '');
313
1239
  const sSrc = seedScore.get(src) ?? 0;
314
1240
  const sDst = seedScore.get(dst) ?? 0;
315
1241
  if (sSrc === 0 && sDst === 0)
316
- return null;
1242
+ return;
317
1243
  // Both endpoints seeded beats one; the unseeded endpoint contributes a
318
1244
  // neutral 0.35 so bridging edges from a strong seed still surface.
319
- const score = (Math.max(sSrc, 0.35) + Math.max(sDst, 0.35)) / 2 + (sSrc > 0 && sDst > 0 ? 0.1 : 0);
320
- return {
1245
+ const relevance = (Math.max(sSrc, 0.35) + Math.max(sDst, 0.35)) / 2 + (sSrc > 0 && sDst > 0 ? 0.1 : 0);
1246
+ // Evidence, from the claim ledger. An edge whose method was never
1247
+ // recorded is left at its relevance score — unknown is not evidence
1248
+ // against it, and penalizing it would demote the entire pre-existing
1249
+ // graph relative to anything written today.
1250
+ const method = md.method === 'asserted' || md.method === 'heuristic'
1251
+ ? md.method
1252
+ : undefined;
1253
+ const conflict = md.conflict === true;
1254
+ const score = Math.max(0, relevance -
1255
+ (method === 'heuristic' ? HEURISTIC_PENALTY : 0) -
1256
+ (conflict ? CONFLICT_PENALTY : 0));
1257
+ triplets.push({
321
1258
  source: String(md.source_name ?? src),
322
1259
  relation: String(md.relation ?? 'related_to'),
323
1260
  target: String(md.target_name ?? dst),
324
1261
  fact: e.content,
325
1262
  score,
326
- };
327
- })
328
- .filter((t) => !!t)
329
- .sort((a, b) => b.score - a.score)
330
- .slice(0, limit);
1263
+ ...(method ? { method } : {}),
1264
+ ...(conflict ? { conflict } : {}),
1265
+ id: e.id,
1266
+ key: e.key,
1267
+ });
1268
+ };
1269
+ // Scores are per-edge, so pruning to the running top-`limit` after every
1270
+ // batch yields exactly the same result as sorting the whole set at the end.
1271
+ const pruneToLimit = () => {
1272
+ if (triplets.length > limit) {
1273
+ triplets.sort((a, b) => b.score - a.score);
1274
+ triplets.length = limit;
1275
+ }
1276
+ };
1277
+ // K7: gather candidate edges via each seed's adjacency entry — O(seeds ×
1278
+ // degree) instead of a full namespace scan — when the scope's index is
1279
+ // ready and every seed's adjacency entry resolves. Any seed that misses
1280
+ // (index not ready, or an unresolvable ref) falls the WHOLE query back to
1281
+ // the exhaustive scan rather than silently searching only some seeds.
1282
+ let covered = false;
1283
+ let usedIndex = false;
1284
+ if ((await readIndexStatus(ns, options.dbPath)).state === 'ready') {
1285
+ const candidateKeys = new Set();
1286
+ let indexOk = true;
1287
+ for (const s of seedResults) {
1288
+ const adj = await readAdj(ns, s.key, options.dbPath);
1289
+ if (adj === null) {
1290
+ indexOk = false;
1291
+ break;
1292
+ }
1293
+ for (const key of adj.edgeKeys)
1294
+ candidateKeys.add(key);
1295
+ }
1296
+ if (indexOk) {
1297
+ for (const key of candidateKeys) {
1298
+ const res = await bridgeGetEntry({ key, namespace: ns.edges, dbPath: options.dbPath });
1299
+ if (res?.found && res.entry)
1300
+ considerEdge(res.entry);
1301
+ }
1302
+ pruneToLimit();
1303
+ usedIndex = true;
1304
+ covered = true;
1305
+ }
1306
+ }
1307
+ // Paged edge scan (see monolean note in module header). Each page is folded
1308
+ // into the running top-`limit` immediately, so memory stays at one page
1309
+ // regardless of how many edges the namespace holds.
1310
+ if (!usedIndex) {
1311
+ covered = await scanNamespace(ns.edges, options.dbPath, (page) => {
1312
+ for (const e of page)
1313
+ considerEdge(e);
1314
+ pruneToLimit();
1315
+ if (scannedEdges >= SEARCH_EDGE_SCAN_MAX) {
1316
+ truncated = true;
1317
+ return false;
1318
+ }
1319
+ return true;
1320
+ });
1321
+ }
1322
+ // An unreadable namespace is an incomplete answer, not an empty graph.
1323
+ if (!covered)
1324
+ truncated = true;
1325
+ triplets.sort((a, b) => b.score - a.score);
331
1326
  const seeds = seedResults.slice(0, limit).map((s) => {
332
1327
  // metadata is not in search results; parse from rendered content "name — description"
333
1328
  const dash = s.content.indexOf(' — ');
334
1329
  return {
335
1330
  name: dash > 0 ? s.content.slice(0, dash) : s.key,
336
- type: s.key.split(':')[1] ?? 'entity',
1331
+ // Tags carry the stored type (`['kg', <type>, …]`). The key never did:
1332
+ // reading `key.split(':')[1]` reported the type of `n:shared_service`
1333
+ // as `shared_service`, and under hashed IDs would report a digest.
1334
+ type: (s.tags ?? [])[1] ?? 'entity',
337
1335
  description: dash > 0 ? s.content.slice(dash + 3) : s.content,
338
1336
  score: s.score,
339
1337
  id: s.id,
@@ -343,7 +1341,15 @@ export async function kgSearch(options) {
343
1341
  ...triplets.map((t) => `${t.source} —${t.relation}→ ${t.target}${t.fact && t.fact !== `${t.source} ${t.relation} ${t.target}` ? ` (${t.fact})` : ''}`),
344
1342
  ...(triplets.length ? [] : seeds.map((s) => `${s.name}: ${s.description}`)),
345
1343
  ].join('\n');
346
- return { success: true, context, triplets, seeds };
1344
+ return {
1345
+ success: true,
1346
+ context,
1347
+ triplets,
1348
+ seeds,
1349
+ scannedEdges,
1350
+ ...(truncated && { truncated }),
1351
+ ...retrieval,
1352
+ };
347
1353
  }
348
1354
  catch (err) {
349
1355
  return {
@@ -357,135 +1363,1061 @@ export async function kgSearch(options) {
357
1363
  }
358
1364
  // ── Glossary (anti-duplicate-entity injection for extraction prompts) ──
359
1365
  export async function kgGlossary(options) {
360
- const res = await bridgeListEntries({
361
- namespace: KG_NODES_NS,
362
- limit: MAX_LIST,
363
- dbPath: options?.dbPath,
1366
+ const limit = options?.limit ?? 40;
1367
+ // Running top-`limit` by rank, deduplicated by normalized name. Folding each
1368
+ // page in and pruning keeps the whole node namespace in scope without ever
1369
+ // holding more than a page plus `limit` names.
1370
+ let top = [];
1371
+ await scanNamespace(kgNamespaces(options?.scope).nodes, options?.dbPath, (page) => {
1372
+ for (const e of page) {
1373
+ const md = e.metadata;
1374
+ // Glossary is for ENTITY name reuse — rule prose and extraction-source
1375
+ // Session nodes would drown it.
1376
+ const t = String(md?.type ?? '').toLowerCase();
1377
+ if (md?.node_set === 'rules' || t === 'rule' || t === 'session')
1378
+ continue;
1379
+ const fw = typeof md.feedback_weight === 'number' ? md.feedback_weight : 0.5;
1380
+ const freq = typeof md.frequency_weight === 'number' ? md.frequency_weight : 0;
1381
+ const version = typeof md.version === 'number' ? md.version : 1;
1382
+ const name = String(md.name ?? e.key);
1383
+ top.push({ name, norm: normalizeName(name), rank: version + freq + fw });
1384
+ }
1385
+ top.sort((a, b) => b.rank - a.rank);
1386
+ const seen = new Set();
1387
+ const pruned = [];
1388
+ for (const n of top) {
1389
+ if (seen.has(n.norm))
1390
+ continue;
1391
+ seen.add(n.norm);
1392
+ pruned.push(n);
1393
+ if (pruned.length >= limit)
1394
+ break;
1395
+ }
1396
+ top = pruned;
364
1397
  });
365
- const nodes = (res?.entries ?? [])
366
- // Glossary is for ENTITY name reuse — rule prose and extraction-source
367
- // Session nodes would drown it.
368
- .filter((e) => {
369
- const md = e.metadata;
370
- const t = String(md?.type ?? '').toLowerCase();
371
- return md?.node_set !== 'rules' && t !== 'rule' && t !== 'session';
372
- })
373
- .map((e) => {
374
- const md = e.metadata;
375
- const fw = typeof md.feedback_weight === 'number' ? md.feedback_weight : 0.5;
376
- const freq = typeof md.frequency_weight === 'number' ? md.frequency_weight : 0;
377
- const version = typeof md.version === 'number' ? md.version : 1;
378
- return { name: String(md.name ?? e.key), rank: version + freq + fw };
379
- })
380
- .sort((a, b) => b.rank - a.rank);
381
- const seen = new Set();
382
- const out = [];
383
- for (const n of nodes) {
384
- const norm = normalizeName(n.name);
385
- if (seen.has(norm))
386
- continue;
387
- seen.add(norm);
388
- out.push(n.name);
389
- if (out.length >= (options?.limit ?? 40))
390
- break;
391
- }
392
- return out;
1398
+ return top.map((n) => n.name);
393
1399
  }
394
1400
  // ── Rollback (per-origin bad-ingest recovery) ───────────────────────
395
- /** Delete every node/edge/rule whose ONLY origin is `originRef`. Elements with
396
- * other origins survive (shared knowledge isn't destroyed by one bad run);
397
- * their origin lists retain the ref — acceptable residue.
398
- * // monolean: no origin-list rewrite — needs an update-by-id bridge API */
1401
+ function originsOf(entry) {
1402
+ const origins = (entry.metadata ?? {}).origin_refs;
1403
+ return Array.isArray(origins) ? origins : [];
1404
+ }
1405
+ /** Every entry in `namespace` that `originRef` supports, collected by an
1406
+ * EXHAUSTIVE paged scan. `covered` is false when the backend became
1407
+ * unreadable partway — an incomplete answer, never an empty one.
1408
+ *
1409
+ * Collecting rather than streaming is required by the callers that mutate:
1410
+ * deleting a row pulls later rows back, so an advancing offset would skip
1411
+ * whatever slid into the gap. Only origin-carrying entries are retained, so
1412
+ * memory tracks the operation's own footprint, not the namespace size. */
1413
+ async function collectByOrigin(namespace, originRef, dbPath) {
1414
+ const entries = [];
1415
+ const covered = await scanNamespace(namespace, dbPath, (page) => {
1416
+ for (const e of page)
1417
+ if (originsOf(e).includes(originRef))
1418
+ entries.push(e);
1419
+ });
1420
+ return { entries, covered };
1421
+ }
1422
+ /** Withdraw `originRef`'s support from the graph: remove it from every
1423
+ * node/edge/rule it backs, and delete the element once no origin remains.
1424
+ *
1425
+ * The withdrawn ref is REWRITTEN out of the surviving elements' origin lists,
1426
+ * not left behind. Retaining it (as this used to) meant a second rollback saw
1427
+ * a two-entry list and retained again — so an element could outlive the
1428
+ * withdrawal of every origin that ever supported it.
1429
+ *
1430
+ * The scan is EXHAUSTIVE: it pages each namespace to the end rather than
1431
+ * reading one capped list. A capped scan let an element past the cap keep a
1432
+ * withdrawn origin while the caller was told the rollback succeeded.
1433
+ *
1434
+ * Collect-then-mutate is deliberate: deleting a row pulls every later row back
1435
+ * one position, so a delete inside the page loop would make an advancing
1436
+ * offset skip whatever slid into the gap. (Rewrites are no longer a hazard —
1437
+ * the bridge's upsert reuses the existing entry id and preserves `createdAt`
1438
+ * rather than re-inserting at the head of the default ordering — but this
1439
+ * function deletes as well as rewrites.) Only origin-carrying entries are
1440
+ * retained during the scan, so memory tracks the rollback's own footprint, not
1441
+ * the namespace size. */
399
1442
  export async function kgRollback(options) {
1443
+ const failures = new FailureLog();
1444
+ const namespaces = kgNamespaces(options.scope);
1445
+ const originRef = kgQualifyOrigin(options.originRef, options.scope);
400
1446
  let deleted = 0, retained = 0;
1447
+ /** Entity IDs this rollback removed, and the names that pointed at them. */
1448
+ const removedIds = new Set();
1449
+ const removedNames = new Set();
401
1450
  try {
402
- for (const ns of [KG_NODES_NS, KG_EDGES_NS, RULES_NS]) {
403
- const res = await bridgeListEntries({
404
- namespace: ns,
405
- limit: MAX_LIST,
406
- dbPath: options.dbPath,
407
- });
408
- for (const e of res?.entries ?? []) {
409
- const origins = e.metadata?.origin_refs;
410
- if (!Array.isArray(origins) || !origins.includes(options.originRef))
411
- continue;
412
- if (origins.length <= 1) {
1451
+ // K7: one indexed read across all three namespaces, tried before the
1452
+ // exhaustive per-namespace scan. Only trusted when the scope's index is
1453
+ // `ready` AND every ref it names still resolves — `kgIndexedByOrigin`
1454
+ // returns null otherwise, and this falls straight back to the scan.
1455
+ let indexedByNs = null;
1456
+ if ((await readIndexStatus(namespaces, options.dbPath)).state === 'ready') {
1457
+ const indexed = await kgIndexedByOrigin(namespaces, originRef, options.dbPath);
1458
+ if (indexed !== null) {
1459
+ indexedByNs = new Map([
1460
+ [namespaces.nodes, []],
1461
+ [namespaces.edges, []],
1462
+ [namespaces.rules, []],
1463
+ ]);
1464
+ for (const { ns, entry } of indexed)
1465
+ indexedByNs.get(ns)?.push(entry);
1466
+ }
1467
+ }
1468
+ for (const ns of [namespaces.nodes, namespaces.edges, namespaces.rules]) {
1469
+ const found = indexedByNs
1470
+ ? { entries: indexedByNs.get(ns) ?? [], covered: true }
1471
+ : await collectByOrigin(ns, originRef, options.dbPath);
1472
+ // A partial scan cannot be reported as a completed withdrawal.
1473
+ if (!found.covered) {
1474
+ failures.note(`${ns}: memory backend unavailable`);
1475
+ continue;
1476
+ }
1477
+ const withdrawn = found.entries.map((e) => ({
1478
+ entry: e,
1479
+ // Re-derive from the claim ledger rather than editing an origin list:
1480
+ // this is what puts back the description a withdrawn origin overwrote.
1481
+ remaining: withoutOrigin((e.metadata ?? {}), originRef),
1482
+ }));
1483
+ for (const { entry: e, remaining } of withdrawn) {
1484
+ const md = (e.metadata ?? {});
1485
+ if (remaining === null) {
413
1486
  const del = await bridgeDeleteEntry({ id: e.id, namespace: ns, dbPath: options.dbPath });
414
- if (del?.deleted)
1487
+ if (del?.deleted) {
415
1488
  deleted++;
1489
+ await onOriginWithdrawn(namespaces, ns, e.key, originRef, options.dbPath);
1490
+ await onEntryDeleted(namespaces, e.key, options.dbPath, ns === namespaces.edges
1491
+ ? { src: String(md.src ?? ''), dst: String(md.dst ?? '') }
1492
+ : undefined);
1493
+ }
1494
+ else
1495
+ failures.note(`${ns}/${e.key}: delete failed`);
1496
+ if (ns === namespaces.nodes && del?.deleted) {
1497
+ removedIds.add(e.key);
1498
+ if (typeof md.name === 'string')
1499
+ removedNames.add(md.name);
1500
+ }
1501
+ continue;
416
1502
  }
417
- else {
1503
+ const store = await bridgeStoreEntry({
1504
+ key: e.key,
1505
+ value: rerender(e.content, md, remaining.description, ns === namespaces.rules),
1506
+ namespace: ns,
1507
+ dbPath: options.dbPath,
1508
+ upsert: true,
1509
+ // Edges are stored without embeddings; re-deriving one here would
1510
+ // silently change how the entry behaves in search.
1511
+ generateEmbeddingFlag: e.hasEmbedding,
1512
+ tags: e.tags,
1513
+ metadata: { ...md, ...remaining },
1514
+ });
1515
+ if (!failures.add(store, `${ns}/${e.key}: origin withdrawal`)) {
418
1516
  retained++;
1517
+ await onOriginWithdrawn(namespaces, ns, e.key, originRef, options.dbPath);
419
1518
  }
420
1519
  }
421
1520
  }
422
- return { success: true, deleted, retained };
1521
+ // Drop the removed entities out of the name index, so a later ingest of the
1522
+ // same name does not resolve onto an entity that no longer exists.
1523
+ for (const name of removedNames) {
1524
+ const known = await readNameIndex(name, namespaces, options.dbPath);
1525
+ if (known === null) {
1526
+ // An unreadable index must not be rewritten from a guess — doing so
1527
+ // would erase every same-name entity this rollback did NOT remove.
1528
+ failures.note(`${namespaces.names}: name index unreadable for ${name}`);
1529
+ continue;
1530
+ }
1531
+ const survivors = known.filter((c) => !removedIds.has(c.id));
1532
+ if (survivors.length !== known.length)
1533
+ await writeNameIndex(name, survivors, namespaces, options.dbPath, failures);
1534
+ }
1535
+ const danglingEdgesRemoved = removedIds.size
1536
+ ? await removeEdgesMissingEndpoints(namespaces, removedIds, options.dbPath, failures)
1537
+ : 0;
1538
+ return {
1539
+ success: !failures.failed,
1540
+ deleted,
1541
+ retained,
1542
+ ...(failures.failed ? { failures: failures.messages, error: failures.summary() } : {}),
1543
+ ...(danglingEdgesRemoved ? { danglingEdgesRemoved } : {}),
1544
+ };
423
1545
  }
424
1546
  catch (err) {
425
1547
  return {
426
1548
  success: false,
427
1549
  deleted,
428
1550
  retained,
1551
+ ...(failures.messages.length ? { failures: failures.messages } : {}),
1552
+ error: err instanceof Error ? err.message : String(err),
1553
+ };
1554
+ }
1555
+ }
1556
+ /** Re-render a stored value after its derived description changed.
1557
+ *
1558
+ * Rules are left alone: their value is the rule text plus an optional context
1559
+ * block, not a rendering of a description. */
1560
+ function rerender(content, md, description, isRule) {
1561
+ if (isRule || description === md.description)
1562
+ return content;
1563
+ if (md.kg === 'edge')
1564
+ return description || `${md.source_name} ${md.relation} ${md.target_name}`;
1565
+ return `${md.name} — ${description || md.type}`;
1566
+ }
1567
+ /** Delete every edge incident to one of `removedIds`. Collect-then-mutate for
1568
+ * the same reason `kgRollback` does. */
1569
+ async function removeEdgesMissingEndpoints(ns, removedIds, dbPath, failures) {
1570
+ const doomed = [];
1571
+ const covered = await scanNamespace(ns.edges, dbPath, (page) => {
1572
+ for (const e of page) {
1573
+ const md = (e.metadata ?? {});
1574
+ const src = String(md.src ?? '');
1575
+ const dst = String(md.dst ?? '');
1576
+ if (removedIds.has(src) || removedIds.has(dst))
1577
+ doomed.push({ entry: e, src, dst });
1578
+ }
1579
+ });
1580
+ if (!covered) {
1581
+ failures.note(`${ns.edges}: memory backend unavailable during dangling-edge sweep`);
1582
+ return 0;
1583
+ }
1584
+ let removed = 0;
1585
+ for (const { entry: e, src, dst } of doomed) {
1586
+ const del = await bridgeDeleteEntry({ id: e.id, namespace: ns.edges, dbPath });
1587
+ if (del?.deleted) {
1588
+ removed++;
1589
+ // Only the surviving endpoint needs its adjacency entry pruned — the
1590
+ // removed one's own entry is moot, its entity is gone too.
1591
+ await onEntryDeleted(ns, e.key, dbPath, {
1592
+ src: removedIds.has(src) ? dst : src,
1593
+ dst: removedIds.has(src) ? dst : src,
1594
+ });
1595
+ }
1596
+ else
1597
+ failures.note(`${ns.edges}/${e.key}: dangling-edge delete failed`);
1598
+ }
1599
+ return removed;
1600
+ }
1601
+ /** Copy one origin's claims out of an org's scope into project-shared
1602
+ * knowledge.
1603
+ *
1604
+ * Sharing is deliberate, never a side effect of learning: `kgIngest` under a
1605
+ * scope only ever writes that org's namespaces, and this is the one path a
1606
+ * claim takes across the boundary. The org keeps its own copy untouched — the
1607
+ * shared copy is an INDEPENDENT assertion under `promoted:<org-ref>`, so
1608
+ * rolling back either side leaves the other standing, and a shared claim
1609
+ * always names the org that vouched for it.
1610
+ *
1611
+ * Like ingest, NOT atomic: the counters report what actually landed. */
1612
+ export async function kgPromote(options) {
1613
+ const empty = { nodes: 0, edges: 0, rules: 0 };
1614
+ if (!options.from?.org?.trim())
1615
+ return {
1616
+ success: false,
1617
+ ...empty,
1618
+ promotedAs: '',
1619
+ error: 'promotion needs an owning org to promote from',
1620
+ };
1621
+ const ns = kgNamespaces(options.from);
1622
+ const sourceRef = kgQualifyOrigin(options.originRef, options.from);
1623
+ const promotedAs = `promoted:${sourceRef}`;
1624
+ const failures = new FailureLog();
1625
+ try {
1626
+ const [nodeHits, edgeHits, ruleHits] = await Promise.all([
1627
+ collectByOrigin(ns.nodes, sourceRef, options.dbPath),
1628
+ collectByOrigin(ns.edges, sourceRef, options.dbPath),
1629
+ collectByOrigin(ns.rules, sourceRef, options.dbPath),
1630
+ ]);
1631
+ // A partial read would promote a partial claim set while reporting the
1632
+ // whole origin as shared.
1633
+ for (const [name, hit] of [
1634
+ ['nodes', nodeHits],
1635
+ ['edges', edgeHits],
1636
+ ['rules', ruleHits],
1637
+ ])
1638
+ if (!hit.covered)
1639
+ failures.note(`${name}: memory backend unavailable`);
1640
+ if (failures.failed)
1641
+ return {
1642
+ success: false,
1643
+ ...empty,
1644
+ promotedAs,
1645
+ failures: failures.messages,
1646
+ error: failures.summary(),
1647
+ };
1648
+ // Rule NODES are re-created by kgIngestRules; promoting them again through
1649
+ // kgIngest would double-count and strip their rules-namespace entry.
1650
+ const nodes = nodeHits.entries
1651
+ .filter((e) => {
1652
+ const md = (e.metadata ?? {});
1653
+ return md.node_set !== 'rules' && String(md.type ?? '').toLowerCase() !== 'rule';
1654
+ })
1655
+ .map((e) => {
1656
+ const md = (e.metadata ?? {});
1657
+ return {
1658
+ name: String(md.name ?? e.key),
1659
+ type: String(md.type ?? 'entity'),
1660
+ description: String(md.description ?? ''),
1661
+ nodeSet: typeof md.node_set === 'string' ? md.node_set : undefined,
1662
+ };
1663
+ });
1664
+ const edges = edgeHits.entries.map((e) => {
1665
+ const md = (e.metadata ?? {});
1666
+ return {
1667
+ source: String(md.source_name ?? md.src ?? ''),
1668
+ target: String(md.target_name ?? md.dst ?? ''),
1669
+ relation: String(md.relation ?? 'related_to'),
1670
+ description: String(md.description ?? ''),
1671
+ };
1672
+ });
1673
+ const rules = ruleHits.entries.map((e) => {
1674
+ const md = (e.metadata ?? {});
1675
+ return { rule: String(md.rule ?? e.content.split('\n')[0]) };
1676
+ });
1677
+ let promotedNodes = 0, promotedEdges = 0, promotedRules = 0;
1678
+ /** A batch that REJECTED part of what it was handed did not promote the
1679
+ * whole origin, so it cannot be reported as a clean share. Rejections are
1680
+ * not in `failures` — they are the caller's payload being refused, not the
1681
+ * bridge failing — so they have to be carried across explicitly. */
1682
+ const carry = (res, what) => {
1683
+ if (res.failures?.length)
1684
+ for (const m of res.failures)
1685
+ failures.note(m);
1686
+ const rejected = (res.nodesRejected ?? 0) + (res.edgesRejected ?? 0);
1687
+ if (rejected)
1688
+ failures.note(`${what}: ${rejected} item(s) rejected — ${res.rejections?.[0]}`);
1689
+ };
1690
+ // kgIngest caps a call at 500 nodes / 1000 edges, so a large origin has to
1691
+ // be promoted in batches rather than silently truncated. Nodes go first so
1692
+ // every edge endpoint already exists when the edges land.
1693
+ for (let i = 0; i < nodes.length; i += MAX_NODES_PER_CALL) {
1694
+ const res = await kgIngest({
1695
+ nodes: nodes.slice(i, i + MAX_NODES_PER_CALL),
1696
+ originRef: promotedAs,
1697
+ dbPath: options.dbPath,
1698
+ });
1699
+ promotedNodes += res.nodesAdded + res.nodesMerged;
1700
+ carry(res, 'promoted nodes');
1701
+ }
1702
+ for (let i = 0; i < edges.length; i += MAX_EDGES_PER_CALL) {
1703
+ const res = await kgIngest({
1704
+ nodes: [],
1705
+ edges: edges.slice(i, i + MAX_EDGES_PER_CALL),
1706
+ originRef: promotedAs,
1707
+ dbPath: options.dbPath,
1708
+ });
1709
+ promotedEdges += res.edgesAdded + res.edgesMerged;
1710
+ carry(res, 'promoted edges');
1711
+ }
1712
+ for (let i = 0; i < rules.length; i += MAX_RULES_PER_CALL) {
1713
+ const res = await kgIngestRules({
1714
+ rules: rules.slice(i, i + MAX_RULES_PER_CALL),
1715
+ originRef: promotedAs,
1716
+ dbPath: options.dbPath,
1717
+ });
1718
+ promotedRules += res.accepted;
1719
+ if (res.failures?.length)
1720
+ for (const m of res.failures)
1721
+ failures.note(m);
1722
+ const invalid = res.verdicts.filter((v) => v.verdict === 'invalid').length;
1723
+ if (invalid)
1724
+ failures.note(`promoted rules: ${invalid} rejected as invalid`);
1725
+ }
1726
+ return {
1727
+ success: !failures.failed,
1728
+ nodes: promotedNodes,
1729
+ edges: promotedEdges,
1730
+ rules: promotedRules,
1731
+ promotedAs,
1732
+ ...(failures.failed ? { failures: failures.messages, error: failures.summary() } : {}),
1733
+ };
1734
+ }
1735
+ catch (err) {
1736
+ return {
1737
+ success: false,
1738
+ ...empty,
1739
+ promotedAs,
1740
+ ...(failures.messages.length ? { failures: failures.messages } : {}),
429
1741
  error: err instanceof Error ? err.message : String(err),
430
1742
  };
431
1743
  }
432
1744
  }
433
1745
  /** Entities whose descriptions are stale relative to their connectivity —
434
1746
  * the LLM half runs in the LIVE agent: it rewrites each candidate's
435
- * description from the neighborhood facts and resubmits via memory_kg_ingest
436
- * (longer descriptions win on merge). No LLM here (fully local constraint). */
1747
+ * description from the neighborhood facts and resubmits via memory_kg_ingest.
1748
+ * A resubmission is a NEW contribution and therefore the current one, so a
1749
+ * shorter but better-supported summary now wins; under "longest description
1750
+ * wins" a consolidation that tightened the prose was silently discarded.
1751
+ * No LLM here (fully local constraint). */
437
1752
  export async function kgConsolidateCandidates(options) {
438
1753
  const minEdges = options?.minEdges ?? 3;
439
- const [nodesRes, edgesRes] = await Promise.all([
440
- bridgeListEntries({ namespace: KG_NODES_NS, limit: MAX_LIST, dbPath: options?.dbPath }),
441
- bridgeListEntries({ namespace: KG_EDGES_NS, limit: MAX_LIST, dbPath: options?.dbPath }),
442
- ]);
443
- const edgesByNode = new Map();
444
- for (const e of edgesRes?.entries ?? []) {
445
- const md = e.metadata;
446
- for (const end of [String(md.src ?? ''), String(md.dst ?? '')]) {
447
- if (!end)
1754
+ const limit = options?.limit ?? 10;
1755
+ const ns = kgNamespaces(options?.scope);
1756
+ // Degree index over the FULL edge namespace. Only the first 12 facts per node
1757
+ // are kept (that is all the result exposes), so the index costs a bounded
1758
+ // amount per node rather than one string per edge.
1759
+ const degree = new Map();
1760
+ await scanNamespace(ns.edges, options?.dbPath, (page) => {
1761
+ for (const e of page) {
1762
+ const md = e.metadata;
1763
+ for (const end of [String(md.src ?? ''), String(md.dst ?? '')]) {
1764
+ if (!end)
1765
+ continue;
1766
+ const slot = degree.get(end) ?? { count: 0, facts: [] };
1767
+ slot.count++;
1768
+ if (slot.facts.length < 12)
1769
+ slot.facts.push(e.content);
1770
+ degree.set(end, slot);
1771
+ }
1772
+ }
1773
+ });
1774
+ let candidates = [];
1775
+ await scanNamespace(ns.nodes, options?.dbPath, (page) => {
1776
+ for (const n of page) {
1777
+ const md = n.metadata;
1778
+ const slot = degree.get(n.key);
1779
+ if (!slot || slot.count < minEdges)
448
1780
  continue;
449
- const list = edgesByNode.get(end) ?? [];
450
- list.push(e.content);
451
- edgesByNode.set(end, list);
1781
+ const description = String(md.description ?? '');
1782
+ // Cap the growth target at MAX_DESC_LEN — a very-high-degree node whose
1783
+ // description is already at the cap can never "grow out" of candidacy and
1784
+ // would otherwise permanently occupy a slot.
1785
+ if (description.length >= Math.min(40 * slot.count, MAX_DESC_LEN))
1786
+ continue;
1787
+ candidates.push({
1788
+ name: String(md.name ?? n.key),
1789
+ type: String(md.type ?? 'entity'),
1790
+ description,
1791
+ edgeCount: slot.count,
1792
+ neighborhood: slot.facts,
1793
+ });
452
1794
  }
453
- }
454
- return ((nodesRes?.entries ?? [])
455
- .map((n) => {
456
- const md = n.metadata;
457
- const facts = edgesByNode.get(n.key) ?? [];
458
- return {
459
- name: String(md.name ?? n.key),
460
- type: String(md.type ?? 'entity'),
461
- description: String(md.description ?? ''),
462
- edgeCount: facts.length,
463
- neighborhood: facts.slice(0, 12),
464
- };
465
- })
466
- // Cap the growth target at MAX_DESC_LEN — a very-high-degree node whose
467
- // description is already at the cap can never "grow out" of candidacy and
468
- // would otherwise permanently occupy a slot.
469
- .filter((c) => c.edgeCount >= minEdges &&
470
- c.description.length < Math.min(40 * c.edgeCount, MAX_DESC_LEN))
471
- .sort((a, b) => b.edgeCount - a.edgeCount)
472
- .slice(0, options?.limit ?? 10));
1795
+ // Ranking is per-node, so keeping only the running top-`limit` after each
1796
+ // page gives the same answer as ranking every node at the end.
1797
+ candidates.sort((a, b) => b.edgeCount - a.edgeCount);
1798
+ candidates = candidates.slice(0, limit);
1799
+ });
1800
+ return candidates;
473
1801
  }
474
1802
  // ── Stats ───────────────────────────────────────────────────────────
1803
+ /** Real counts, not page lengths. `bridgeListEntries.total` reports how many
1804
+ * rows that one call returned, so the old capped list made a 10,001-node graph
1805
+ * report exactly 10,000 forever.
1806
+ *
1807
+ * Tries a real indexed `SELECT COUNT(*) WHERE namespace = ?` first
1808
+ * (`bridgeCountEntries`, K7) — cheap and exact, since every row in a KG
1809
+ * namespace is exactly one node/edge/rule (the name index lives in its own
1810
+ * namespace). Falls back to the paginated scan — one query per 1,000 rows,
1811
+ * still exact — when the loaded bridge predates `bridgeCountEntries`, or a
1812
+ * test double doesn't stub it. */
475
1813
  export async function kgStats(options) {
476
- const [n, e, r] = await Promise.all([
477
- bridgeListEntries({ namespace: KG_NODES_NS, limit: MAX_LIST, dbPath: options?.dbPath }),
478
- bridgeListEntries({ namespace: KG_EDGES_NS, limit: MAX_LIST, dbPath: options?.dbPath }),
479
- bridgeListEntries({ namespace: RULES_NS, limit: MAX_LIST, dbPath: options?.dbPath }),
1814
+ const ns = kgNamespaces(options?.scope);
1815
+ const count = async (namespace) => {
1816
+ try {
1817
+ const real = await bridgeCountEntries(namespace, options?.dbPath);
1818
+ if (real !== null)
1819
+ return real;
1820
+ }
1821
+ catch {
1822
+ /* bridgeCountEntries unavailable on this loaded bridge (or a test
1823
+ double that doesn't stub it) — fall back to the exhaustive scan. */
1824
+ }
1825
+ let n = 0;
1826
+ await scanNamespace(namespace, options?.dbPath, (page) => {
1827
+ n += page.length;
1828
+ });
1829
+ return n;
1830
+ };
1831
+ const [nodes, edges, rules] = await Promise.all([
1832
+ count(ns.nodes),
1833
+ count(ns.edges),
1834
+ count(ns.rules),
480
1835
  ]);
481
- return { nodes: n?.total ?? 0, edges: e?.total ?? 0, rules: r?.total ?? 0 };
1836
+ return { nodes, edges, rules };
1837
+ }
1838
+ /** Complete, uncapped read of every edge touching `endpointId` (as either
1839
+ * src or dst) and/or asserted by `originRef` — at least one filter is
1840
+ * required. This is the ground truth an indexed adjacency/origin lookup
1841
+ * (K7) must agree with once one exists: an exhaustive paged scan, the same
1842
+ * mechanism `kgRollback` already trusts for origin withdrawal, so it never
1843
+ * depends on an index and never inherits the old first-page cap.
1844
+ *
1845
+ * `truncated: true` means the scan did not finish — an incomplete answer,
1846
+ * never an empty one. Callers comparing this against a future index must
1847
+ * treat a truncated reference read as "unknown", not as "no matches". */
1848
+ export async function kgReferenceEdges(options) {
1849
+ if (!options.endpointId && !options.originRef) {
1850
+ return { success: false, edges: [], error: 'endpointId or originRef is required' };
1851
+ }
1852
+ const ns = kgNamespaces(options.scope);
1853
+ const originRef = options.originRef
1854
+ ? kgQualifyOrigin(options.originRef, options.scope)
1855
+ : undefined;
1856
+ const edges = [];
1857
+ try {
1858
+ const covered = await scanNamespace(ns.edges, options.dbPath, (page) => {
1859
+ for (const e of page) {
1860
+ const md = (e.metadata ?? {});
1861
+ if (md.kg !== 'edge')
1862
+ continue;
1863
+ const src = String(md.src ?? '');
1864
+ const dst = String(md.dst ?? '');
1865
+ if (options.endpointId && src !== options.endpointId && dst !== options.endpointId)
1866
+ continue;
1867
+ const originRefs = originsOf(e);
1868
+ if (originRef && !originRefs.includes(originRef))
1869
+ continue;
1870
+ edges.push({
1871
+ key: e.key,
1872
+ src,
1873
+ dst,
1874
+ relation: String(md.relation ?? 'related_to'),
1875
+ originRefs,
1876
+ });
1877
+ }
1878
+ });
1879
+ return { success: true, edges, ...(covered ? {} : { truncated: true }) };
1880
+ }
1881
+ catch (err) {
1882
+ return { success: false, edges, error: err instanceof Error ? err.message : String(err) };
1883
+ }
1884
+ }
1885
+ const KG_INDEX_SCHEMA_VERSION = 1;
1886
+ const INDEX_STATUS_KEY = 'status';
1887
+ /** Edge keys recorded per adjacency entry before the index refuses to grow it
1888
+ * further and fails the SCOPE's index rather than risk an oversized or
1889
+ * silently truncated row. A hub node past this reads via the exhaustive
1890
+ * scan, exactly as before the index existed — this only gates the fast
1891
+ * path, never correctness. */
1892
+ const MAX_ADJ_EDGES_PER_NODE = 2000;
1893
+ /** Refs recorded per origin-support entry before the same refusal applies. */
1894
+ const MAX_ORIGIN_INDEX_REFS = 5000;
1895
+ async function readIndexStatus(ns, dbPath) {
1896
+ const res = await bridgeGetEntry({ key: INDEX_STATUS_KEY, namespace: ns.indexStatus, dbPath });
1897
+ if (!res?.found || !res.entry)
1898
+ return { state: 'absent', schemaVersion: KG_INDEX_SCHEMA_VERSION };
1899
+ const md = (res.entry.metadata ?? {});
1900
+ // An older/foreign schema is not a resumable build — start over rather than
1901
+ // trust rows shaped by a version this code no longer understands.
1902
+ if (md.schemaVersion !== KG_INDEX_SCHEMA_VERSION) {
1903
+ return { state: 'absent', schemaVersion: KG_INDEX_SCHEMA_VERSION };
1904
+ }
1905
+ return { ...md, schemaVersion: KG_INDEX_SCHEMA_VERSION };
1906
+ }
1907
+ async function writeIndexStatus(ns, status, dbPath) {
1908
+ const res = await bridgeStoreEntry({
1909
+ key: INDEX_STATUS_KEY,
1910
+ value: `kg index: ${status.state}`,
1911
+ namespace: ns.indexStatus,
1912
+ dbPath,
1913
+ upsert: true,
1914
+ generateEmbeddingFlag: false,
1915
+ metadata: { ...status, updatedAt: Date.now() },
1916
+ });
1917
+ return Boolean(res?.success);
1918
+ }
1919
+ /** Downgrade a scope's index to `failed` after a dual-write hiccup, so reads
1920
+ * fall back to the exhaustive scan instead of silently drifting. Never
1921
+ * throws — this runs from inside a best-effort maintenance path. */
1922
+ async function markIndexFailed(ns, dbPath, error) {
1923
+ try {
1924
+ const current = await readIndexStatus(ns, dbPath);
1925
+ if (current.state === 'absent' || current.state === 'failed')
1926
+ return; // nothing to protect
1927
+ // Not resumable: the cursor here is stale (either a completed build's end
1928
+ // position, or mid-build), and a dual-write hiccup means something is
1929
+ // missing from an UNKNOWN part of the index — only a fresh rescan finds it.
1930
+ await writeIndexStatus(ns, { ...current, state: 'failed', resumable: false, error }, dbPath);
1931
+ }
1932
+ catch {
1933
+ /* best-effort: if even the downgrade write fails, the next rebuild's
1934
+ validation step still catches an inconsistent index before it is
1935
+ ever trusted for a read. */
1936
+ }
1937
+ }
1938
+ /** This scope's derived-index build/readiness state. `absent` means no one
1939
+ * has ever called `kgRebuildIndex` for it — every read behaves exactly as
1940
+ * it did before this index existed. */
1941
+ export async function kgIndexStatus(options) {
1942
+ return readIndexStatus(kgNamespaces(options?.scope), options?.dbPath);
1943
+ }
1944
+ async function readAdj(ns, entityId, dbPath) {
1945
+ const res = await bridgeGetEntry({ key: entityId, namespace: ns.adj, dbPath });
1946
+ if (!res)
1947
+ return null; // backend unavailable, distinct from "no entry yet"
1948
+ if (!res.found || !res.entry)
1949
+ return { edgeKeys: [] };
1950
+ const md = (res.entry.metadata ?? {});
1951
+ return { edgeKeys: Array.isArray(md.edgeKeys) ? md.edgeKeys : [] };
1952
+ }
1953
+ async function addToAdj(ns, entityId, edgeKey, dbPath) {
1954
+ const current = await readAdj(ns, entityId, dbPath);
1955
+ if (current === null)
1956
+ return false;
1957
+ if (current.edgeKeys.includes(edgeKey))
1958
+ return true; // already present, idempotent
1959
+ if (current.edgeKeys.length >= MAX_ADJ_EDGES_PER_NODE)
1960
+ return false;
1961
+ const res = await bridgeStoreEntry({
1962
+ key: entityId,
1963
+ value: `kg adjacency: ${entityId}`,
1964
+ namespace: ns.adj,
1965
+ dbPath,
1966
+ upsert: true,
1967
+ generateEmbeddingFlag: false,
1968
+ metadata: { edgeKeys: [...current.edgeKeys, edgeKey] },
1969
+ });
1970
+ return Boolean(res?.success);
1971
+ }
1972
+ async function removeFromAdj(ns, entityId, edgeKey, dbPath) {
1973
+ const current = await readAdj(ns, entityId, dbPath);
1974
+ if (current === null)
1975
+ return false;
1976
+ if (!current.edgeKeys.includes(edgeKey))
1977
+ return true; // already absent
1978
+ const res = await bridgeStoreEntry({
1979
+ key: entityId,
1980
+ value: `kg adjacency: ${entityId}`,
1981
+ namespace: ns.adj,
1982
+ dbPath,
1983
+ upsert: true,
1984
+ generateEmbeddingFlag: false,
1985
+ metadata: { edgeKeys: current.edgeKeys.filter((k) => k !== edgeKey) },
1986
+ });
1987
+ return Boolean(res?.success);
1988
+ }
1989
+ async function readOriginIndex(ns, originRef, dbPath) {
1990
+ const res = await bridgeGetEntry({ key: originRef, namespace: ns.originIdx, dbPath });
1991
+ if (!res)
1992
+ return null;
1993
+ if (!res.found || !res.entry)
1994
+ return { refs: [] };
1995
+ const md = (res.entry.metadata ?? {});
1996
+ return { refs: Array.isArray(md.refs) ? md.refs : [] };
1997
+ }
1998
+ async function addToOriginIndex(ns, originRef, ref, dbPath) {
1999
+ const current = await readOriginIndex(ns, originRef, dbPath);
2000
+ if (current === null)
2001
+ return false;
2002
+ if (current.refs.some((r) => r.ns === ref.ns && r.key === ref.key))
2003
+ return true;
2004
+ if (current.refs.length >= MAX_ORIGIN_INDEX_REFS)
2005
+ return false;
2006
+ const res = await bridgeStoreEntry({
2007
+ key: originRef,
2008
+ value: `kg origin index: ${originRef}`,
2009
+ namespace: ns.originIdx,
2010
+ dbPath,
2011
+ upsert: true,
2012
+ generateEmbeddingFlag: false,
2013
+ metadata: { refs: [...current.refs, ref] },
2014
+ });
2015
+ return Boolean(res?.success);
2016
+ }
2017
+ async function removeFromOriginIndex(ns, originRef, ref, dbPath) {
2018
+ const current = await readOriginIndex(ns, originRef, dbPath);
2019
+ if (current === null)
2020
+ return false;
2021
+ const next = current.refs.filter((r) => !(r.ns === ref.ns && r.key === ref.key));
2022
+ if (next.length === current.refs.length)
2023
+ return true; // already absent
2024
+ const res = await bridgeStoreEntry({
2025
+ key: originRef,
2026
+ value: `kg origin index: ${originRef}`,
2027
+ namespace: ns.originIdx,
2028
+ dbPath,
2029
+ upsert: true,
2030
+ generateEmbeddingFlag: false,
2031
+ metadata: { refs: next },
2032
+ });
2033
+ return Boolean(res?.success);
2034
+ }
2035
+ // ── Dual-write hooks ──────────────────────────────────────────────────
2036
+ //
2037
+ // Best-effort and self-gating: a no-op (one status read) while the scope's
2038
+ // index is `absent`, so ordinary ingest/rollback pay nothing extra until
2039
+ // someone opts in by calling `kgRebuildIndex`. A failure here fails the
2040
+ // INDEX (downgrades the scope to `failed`), never the canonical write or
2041
+ // delete it accompanies — the index is a cache, not a second source of truth.
2042
+ /** After a node/edge/rule write lands, record that `originRef` supports it. */
2043
+ async function onEntrySupported(ns, entryNs, entryKey, originRef, dbPath) {
2044
+ const status = await readIndexStatus(ns, dbPath);
2045
+ if (status.state === 'absent' || status.state === 'failed')
2046
+ return;
2047
+ const ok = await addToOriginIndex(ns, originRef, { ns: entryNs, key: entryKey }, dbPath);
2048
+ if (!ok)
2049
+ await markIndexFailed(ns, dbPath, `origin index write failed for ${originRef}`);
2050
+ }
2051
+ /** After an edge write lands, record it in both endpoints' adjacency. */
2052
+ async function onEdgeWritten(ns, edgeKey, src, dst, dbPath) {
2053
+ const status = await readIndexStatus(ns, dbPath);
2054
+ if (status.state === 'absent' || status.state === 'failed')
2055
+ return;
2056
+ const okSrc = await addToAdj(ns, src, edgeKey, dbPath);
2057
+ const okDst = src === dst ? true : await addToAdj(ns, dst, edgeKey, dbPath);
2058
+ if (!okSrc || !okDst)
2059
+ await markIndexFailed(ns, dbPath, `adjacency write failed for ${edgeKey}`);
2060
+ }
2061
+ /** After `kgRollback` deletes an entry outright (no origin left to support
2062
+ * it), remove it from every index it could appear in. */
2063
+ async function onEntryDeleted(ns, entryKey, dbPath, endpoints) {
2064
+ const status = await readIndexStatus(ns, dbPath);
2065
+ if (status.state === 'absent' || status.state === 'failed')
2066
+ return;
2067
+ let ok = true;
2068
+ if (endpoints) {
2069
+ ok = (await removeFromAdj(ns, endpoints.src, entryKey, dbPath)) && ok;
2070
+ if (endpoints.dst !== endpoints.src)
2071
+ ok = (await removeFromAdj(ns, endpoints.dst, entryKey, dbPath)) && ok;
2072
+ }
2073
+ if (!ok)
2074
+ await markIndexFailed(ns, dbPath, `adjacency removal failed for ${entryKey}`);
2075
+ }
2076
+ /** After `kgRollback` withdraws one origin's support from an entry that
2077
+ * survives (another origin still supports it), drop just that origin's ref. */
2078
+ async function onOriginWithdrawn(ns, entryNs, entryKey, originRef, dbPath) {
2079
+ const status = await readIndexStatus(ns, dbPath);
2080
+ if (status.state === 'absent' || status.state === 'failed')
2081
+ return;
2082
+ const ok = await removeFromOriginIndex(ns, originRef, { ns: entryNs, key: entryKey }, dbPath);
2083
+ if (!ok)
2084
+ await markIndexFailed(ns, dbPath, `origin index removal failed for ${originRef}`);
2085
+ }
2086
+ // ── Indexed reads (used by kgSearch/kgRollback only when state === 'ready') ─
2087
+ /** The indexed equivalent of `kgReferenceEdges({ endpointId })`: edges
2088
+ * touching one entity, read via its adjacency entry instead of a namespace
2089
+ * scan. Returns `null` when any edge key it names cannot be resolved — a
2090
+ * torn index must never be presented as a complete answer. */
2091
+ async function kgIndexedEdgesByEndpoint(ns, entityId, dbPath) {
2092
+ const adj = await readAdj(ns, entityId, dbPath);
2093
+ if (adj === null)
2094
+ return null;
2095
+ const edges = [];
2096
+ for (const key of adj.edgeKeys) {
2097
+ const res = await bridgeGetEntry({ key, namespace: ns.edges, dbPath });
2098
+ if (!res?.found || !res.entry)
2099
+ return null; // stale ref — do not half-answer
2100
+ const md = (res.entry.metadata ?? {});
2101
+ edges.push({
2102
+ key,
2103
+ src: String(md.src ?? ''),
2104
+ dst: String(md.dst ?? ''),
2105
+ relation: String(md.relation ?? 'related_to'),
2106
+ originRefs: originsOf(res.entry),
2107
+ });
2108
+ }
2109
+ return edges;
2110
+ }
2111
+ /** The indexed equivalent of `collectByOrigin`: every {namespace,key} entry
2112
+ * one origin supports, read via its origin-index entry instead of scanning
2113
+ * every namespace. Returns `null` on any unresolvable ref, same reasoning
2114
+ * as `kgIndexedEdgesByEndpoint`. */
2115
+ /** Returns each entry paired with the namespace it was read from (from the
2116
+ * index's own ref, not trusted from the entry itself) so a caller can sort
2117
+ * results back into per-namespace buckets without guessing. */
2118
+ async function kgIndexedByOrigin(ns, originRef, dbPath) {
2119
+ const idx = await readOriginIndex(ns, originRef, dbPath);
2120
+ if (idx === null)
2121
+ return null;
2122
+ const entries = [];
2123
+ for (const ref of idx.refs) {
2124
+ const res = await bridgeGetEntry({ key: ref.key, namespace: ref.ns, dbPath });
2125
+ if (!res?.found || !res.entry)
2126
+ return null;
2127
+ entries.push({ ns: ref.ns, entry: res.entry });
2128
+ }
2129
+ return entries;
2130
+ }
2131
+ // ── Rebuild (K7): the one function that builds/repairs the derived index ──
2132
+ /** Entities/origins sampled for post-build validation. A scan that saw fewer
2133
+ * than this many distinct entities AND origins gets FULL validation, not a
2134
+ * sample — most real scopes will. `KgRebuildResult.status.validation` says
2135
+ * which happened, honestly, rather than letting "validated" imply "all". */
2136
+ const VALIDATE_SAMPLE = 200;
2137
+ function phaseOrder(phase) {
2138
+ return phase === 'edges' ? 1 : phase === 'rules' ? 2 : 0;
2139
+ }
2140
+ function sameEdgeKeySet(a, b) {
2141
+ const ak = new Set(a.map((e) => e.key));
2142
+ const bk = new Set(b.map((e) => e.key));
2143
+ if (ak.size !== bk.size)
2144
+ return false;
2145
+ for (const k of ak)
2146
+ if (!bk.has(k))
2147
+ return false;
2148
+ return true;
2149
+ }
2150
+ /** (Re)build a scope's derived index from canonical data — nodes, then
2151
+ * edges, then rules, that fixed order, resuming from the last checkpointed
2152
+ * `{phase, offset}` rather than restarting when a prior call was
2153
+ * interrupted mid-build. Every write here (`addToAdj`/`addToOriginIndex`)
2154
+ * is idempotent, so a page reprocessed after an interruption cannot
2155
+ * duplicate an entry.
2156
+ *
2157
+ * A concurrent ingest/rollback during the build is safe, not just tolerated:
2158
+ * the dual-write hooks run whenever state is not `absent`/`failed`, so a
2159
+ * write made mid-build is captured whether or not the scan has reached that
2160
+ * row yet — at worst twice, which idempotency absorbs for free. The one
2161
+ * residual race (a row deleted between the scan reading it and the scan's
2162
+ * own write landing) can leave a dangling ref in the index; it is never
2163
+ * observable as wrong data, because every indexed READ
2164
+ * (`kgIndexedEdgesByEndpoint`/`kgIndexedByOrigin`) returns `null` — and the
2165
+ * caller falls back to the exhaustive scan — the instant it cannot resolve
2166
+ * a ref it holds. This is the backend's real capability (single-row CAS,
2167
+ * no cross-row transaction), used honestly rather than claiming atomicity
2168
+ * it cannot provide.
2169
+ *
2170
+ * Ends in `validating`: samples up to `VALIDATE_SAMPLE` of the entities and
2171
+ * origins the scan actually saw, and re-reads them through the just-built
2172
+ * index, comparing against a FRESH, independent reference read
2173
+ * (`kgReferenceEdges`/`collectByOrigin`) — not the in-memory data the build
2174
+ * itself computed, which would only prove the build agrees with itself. A
2175
+ * write that silently failed, or a concurrent change the dual-write hooks
2176
+ * missed, is exactly what this catches before the index is ever trusted. */
2177
+ export async function kgRebuildIndex(options) {
2178
+ const ns = kgNamespaces(options?.scope);
2179
+ const dbPath = options?.dbPath;
2180
+ let status = await readIndexStatus(ns, dbPath);
2181
+ // Resume from the checkpointed cursor for an interrupted build or a
2182
+ // resumable (build-phase) failure; a fresh scan for everything else —
2183
+ // `absent`, `ready` (this call IS the deliberate re-verify), and a
2184
+ // validation-phase failure, where something already written was wrong.
2185
+ const resume = status.state === 'building' || (status.state === 'failed' && status.resumable);
2186
+ if (!resume) {
2187
+ status = {
2188
+ state: 'building',
2189
+ schemaVersion: KG_INDEX_SCHEMA_VERSION,
2190
+ cursor: { phase: 'nodes', offset: 0 },
2191
+ counts: { nodes: 0, edges: 0, rules: 0 },
2192
+ startedAt: Date.now(),
2193
+ };
2194
+ if (!(await writeIndexStatus(ns, status, dbPath))) {
2195
+ return { success: false, status, error: 'could not persist initial build status' };
2196
+ }
2197
+ }
2198
+ const seenEntities = [];
2199
+ const seenOrigins = [];
2200
+ const noteEntity = (id) => {
2201
+ if (id && !seenEntities.includes(id) && seenEntities.length < VALIDATE_SAMPLE) {
2202
+ seenEntities.push(id);
2203
+ }
2204
+ };
2205
+ const noteOrigin = (ref) => {
2206
+ if (!seenOrigins.includes(ref) && seenOrigins.length < VALIDATE_SAMPLE)
2207
+ seenOrigins.push(ref);
2208
+ };
2209
+ // `addToAdj`/`addToOriginIndex` only APPEND onto whatever is already
2210
+ // there. That's exactly right for a RESUME (everything present was
2211
+ // written earlier in this same build attempt), but wrong for a FRESH
2212
+ // build: a stale or corrupted entry left over from a PRIOR build (e.g.
2213
+ // the exact thing a failed validation just caught) would never be
2214
+ // cleared, only added to. So a fresh build clears both derived-index
2215
+ // namespaces up front, once, before touching any phase — every write for
2216
+ // the rest of THIS build attempt, in this call or a later one resuming
2217
+ // it, can then safely append, because the namespace is known to hold only
2218
+ // rows this attempt wrote.
2219
+ if (!resume) {
2220
+ const adjCleared = await clearNamespace(ns.adj, dbPath);
2221
+ const originCleared = await clearNamespace(ns.originIdx, dbPath);
2222
+ if (!adjCleared || !originCleared) {
2223
+ status = { ...status, state: 'failed', error: 'could not clear prior index before rebuild' };
2224
+ await writeIndexStatus(ns, status, dbPath);
2225
+ return { success: false, status, error: status.error };
2226
+ }
2227
+ }
2228
+ const phases = [
2229
+ { phase: 'nodes', namespace: ns.nodes },
2230
+ { phase: 'edges', namespace: ns.edges },
2231
+ { phase: 'rules', namespace: ns.rules },
2232
+ ];
2233
+ try {
2234
+ for (const { phase, namespace } of phases) {
2235
+ if (phaseOrder(status.cursor?.phase) > phaseOrder(phase))
2236
+ continue; // already scanned
2237
+ let offset = status.cursor?.phase === phase ? (status.cursor.offset ?? 0) : 0;
2238
+ let count = status.counts?.[phase] ?? 0;
2239
+ for (;;) {
2240
+ const page = await bridgeListEntries({ namespace, limit: SCAN_PAGE, offset, dbPath });
2241
+ if (!page) {
2242
+ // The cursor stays exactly where it was: everything indexed before
2243
+ // this page is still correct, only incomplete, so a retry resumes.
2244
+ status = {
2245
+ ...status,
2246
+ state: 'failed',
2247
+ resumable: true,
2248
+ error: `${namespace}: backend unavailable during rebuild`,
2249
+ };
2250
+ await writeIndexStatus(ns, status, dbPath);
2251
+ return { success: false, status, error: status.error };
2252
+ }
2253
+ for (const e of page.entries) {
2254
+ const md = (e.metadata ?? {});
2255
+ count++;
2256
+ if (phase === 'edges' && md.kg === 'edge') {
2257
+ const src = String(md.src ?? '');
2258
+ const dst = String(md.dst ?? '');
2259
+ if (src) {
2260
+ await addToAdj(ns, src, e.key, dbPath);
2261
+ noteEntity(src);
2262
+ }
2263
+ if (dst && dst !== src) {
2264
+ await addToAdj(ns, dst, e.key, dbPath);
2265
+ noteEntity(dst);
2266
+ }
2267
+ }
2268
+ if (phase === 'nodes')
2269
+ noteEntity(e.key);
2270
+ for (const originRef of originsOf(e)) {
2271
+ await addToOriginIndex(ns, originRef, { ns: namespace, key: e.key }, dbPath);
2272
+ noteOrigin(originRef);
2273
+ }
2274
+ }
2275
+ offset += page.entries.length;
2276
+ status = {
2277
+ ...status,
2278
+ cursor: { phase, offset },
2279
+ counts: { ...status.counts, [phase]: count },
2280
+ };
2281
+ await writeIndexStatus(ns, status, dbPath);
2282
+ if (page.entries.length < SCAN_PAGE)
2283
+ break; // last page of this namespace
2284
+ }
2285
+ }
2286
+ status = { ...status, state: 'validating' };
2287
+ await writeIndexStatus(ns, status, dbPath);
2288
+ const mismatches = [];
2289
+ for (const id of seenEntities) {
2290
+ const indexed = await kgIndexedEdgesByEndpoint(ns, id, dbPath);
2291
+ const reference = await kgReferenceEdges({ endpointId: id, scope: options?.scope, dbPath });
2292
+ if (indexed === null || !reference.success || reference.truncated) {
2293
+ mismatches.push(`entity ${id}: reference read incomplete`);
2294
+ continue;
2295
+ }
2296
+ if (!sameEdgeKeySet(indexed, reference.edges))
2297
+ mismatches.push(`entity ${id}: adjacency mismatch`);
2298
+ }
2299
+ for (const originRef of seenOrigins) {
2300
+ const indexed = await kgIndexedByOrigin(ns, originRef, dbPath);
2301
+ if (indexed === null) {
2302
+ mismatches.push(`origin ${originRef}: index unreadable`);
2303
+ continue;
2304
+ }
2305
+ const refA = await collectByOrigin(ns.nodes, originRef, dbPath);
2306
+ const refB = await collectByOrigin(ns.edges, originRef, dbPath);
2307
+ const refC = await collectByOrigin(ns.rules, originRef, dbPath);
2308
+ if (!refA.covered || !refB.covered || !refC.covered) {
2309
+ mismatches.push(`origin ${originRef}: reference scan incomplete`);
2310
+ continue;
2311
+ }
2312
+ const expected = new Set([
2313
+ ...refA.entries.map((e) => `${ns.nodes}|${e.key}`),
2314
+ ...refB.entries.map((e) => `${ns.edges}|${e.key}`),
2315
+ ...refC.entries.map((e) => `${ns.rules}|${e.key}`),
2316
+ ]);
2317
+ const indexedKeys = new Set(indexed.map((e) => `${e.ns}|${e.entry.key}`));
2318
+ if (expected.size !== indexedKeys.size || [...expected].some((k) => !indexedKeys.has(k))) {
2319
+ mismatches.push(`origin ${originRef}: support-index mismatch`);
2320
+ }
2321
+ }
2322
+ const full = seenEntities.length < VALIDATE_SAMPLE && seenOrigins.length < VALIDATE_SAMPLE;
2323
+ const validation = { sampledEntities: seenEntities.length, sampledOrigins: seenOrigins.length, full };
2324
+ if (mismatches.length) {
2325
+ const summary = mismatches.slice(0, 5).join('; ');
2326
+ status = {
2327
+ ...status,
2328
+ state: 'failed',
2329
+ error: `validation failed: ${summary}${mismatches.length > 5 ? ` (+${mismatches.length - 5} more)` : ''}`,
2330
+ };
2331
+ await writeIndexStatus(ns, status, dbPath);
2332
+ return { success: false, status, validation, error: status.error };
2333
+ }
2334
+ status = { ...status, state: 'ready', error: undefined };
2335
+ await writeIndexStatus(ns, status, dbPath);
2336
+ return { success: true, status, validation };
2337
+ }
2338
+ catch (err) {
2339
+ status = { ...status, state: 'failed', error: err instanceof Error ? err.message : String(err) };
2340
+ try {
2341
+ await writeIndexStatus(ns, status, dbPath);
2342
+ }
2343
+ catch {
2344
+ /* best-effort — see markIndexFailed */
2345
+ }
2346
+ return { success: false, status, error: status.error };
2347
+ }
2348
+ }
2349
+ /** Verify that every edge's endpoints exist.
2350
+ *
2351
+ * `kgIngest` now creates missing endpoints before writing an edge and
2352
+ * `kgRollback` removes edges whose endpoints it deleted, so a healthy graph
2353
+ * reports nothing. This exists for graphs written before either rule, and as
2354
+ * the check that says so rather than assuming it. */
2355
+ export async function kgIntegrityCheck(options) {
2356
+ const ns = kgNamespaces(options?.scope);
2357
+ const limit = options?.limit ?? 100;
2358
+ const dangling = [];
2359
+ let edges = 0;
2360
+ try {
2361
+ /** Endpoint id → exists. One keyed lookup per DISTINCT endpoint. */
2362
+ const seen = new Map();
2363
+ const exists = async (id) => {
2364
+ const hit = seen.get(id);
2365
+ if (hit !== undefined)
2366
+ return hit;
2367
+ const res = await bridgeGetEntry({ key: id, namespace: ns.nodes, dbPath: options?.dbPath });
2368
+ const found = Boolean(res?.found && res.entry);
2369
+ seen.set(id, found);
2370
+ return found;
2371
+ };
2372
+ // Collect first, then probe: the probes are reads, but interleaving many
2373
+ // per page would hold the page open far longer than the scan needs.
2374
+ const rows = [];
2375
+ const covered = await scanNamespace(ns.edges, options?.dbPath, (page) => {
2376
+ for (const e of page) {
2377
+ const md = (e.metadata ?? {});
2378
+ if (md.kg !== 'edge')
2379
+ continue;
2380
+ edges++;
2381
+ rows.push({ key: e.key, src: String(md.src ?? ''), dst: String(md.dst ?? '') });
2382
+ }
2383
+ });
2384
+ for (const row of rows) {
2385
+ if (dangling.length >= limit)
2386
+ break;
2387
+ const missing = [];
2388
+ if (!row.src || !(await exists(row.src)))
2389
+ missing.push(row.src || '<no src>');
2390
+ if (!row.dst || !(await exists(row.dst)))
2391
+ missing.push(row.dst || '<no dst>');
2392
+ if (missing.length)
2393
+ dangling.push({ key: row.key, missing });
2394
+ }
2395
+ return {
2396
+ success: true,
2397
+ edges,
2398
+ dangling,
2399
+ ...(covered && dangling.length < limit ? {} : { truncated: true }),
2400
+ };
2401
+ }
2402
+ catch (err) {
2403
+ return {
2404
+ success: false,
2405
+ edges,
2406
+ dangling,
2407
+ error: err instanceof Error ? err.message : String(err),
2408
+ };
2409
+ }
482
2410
  }
483
2411
  // ── Heuristic extraction (LLM-less fallback) ────────────────────────
484
2412
  /** Regex extraction for when no LLM is in the loop (memory-palace lineage):
485
2413
  * proper-noun phrases and `code identifiers` become entities, sentence
486
- * co-occurrence becomes relates_to edges. Lower-trust by design — real
2414
+ * co-occurrence becomes `mentioned_with` edges. Lower-trust by design — real
487
2415
  * entity/relation quality comes from the LLM path (memory_kg_ingest called
488
- * by the live agent, or the org coordinator's org_learn tool). */
2416
+ * by the live agent, or the org coordinator's org_learn tool).
2417
+ *
2418
+ * Callers MUST ingest this with `method: 'heuristic'`. "Lower-trust by design"
2419
+ * was true and unenforced: nothing downstream could tell these edges from
2420
+ * facts an agent stated, so ranking treated them identically. */
489
2421
  export function heuristicExtract(text, opts) {
490
2422
  const nodes = new Map();
491
2423
  const edges = [];
@@ -566,11 +2498,14 @@ export function heuristicExtract(text, opts) {
566
2498
  }
567
2499
  }
568
2500
  // Co-occurrence edges within a sentence (first mention chains to the rest).
2501
+ // `mentioned_with`, not `relates_to`: all this observed is two names in one
2502
+ // sentence. `relates_to` reads as an asserted relation, and a reader cannot
2503
+ // tell one that an agent stated from one this regex inferred.
569
2504
  for (let i = 1; i < uniq.length && i < 4; i++) {
570
2505
  edges.push({
571
2506
  source: uniq[0],
572
2507
  target: uniq[i],
573
- relation: 'relates_to',
2508
+ relation: 'mentioned_with',
574
2509
  description: sentence.trim().slice(0, 300),
575
2510
  });
576
2511
  }