akm-cli 0.9.16 → 0.9.17-alpha.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (403) hide show
  1. package/CHANGELOG.md +2101 -0
  2. package/STABILITY.md +11 -10
  3. package/dist/akm +124 -193
  4. package/dist/akm-migrate +38 -19
  5. package/dist/assets/hints/cli-hints-full.md +6 -7
  6. package/dist/assets/improve-strategies/catchup.json +0 -3
  7. package/dist/assets/improve-strategies/consolidate.json +0 -1
  8. package/dist/assets/improve-strategies/default.json +1 -2
  9. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  10. package/dist/assets/improve-strategies/quick.json +1 -2
  11. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  12. package/dist/assets/improve-strategies/thorough.json +0 -3
  13. package/dist/assets/prompts/consolidate-pair.md +20 -0
  14. package/dist/assets/prompts/consolidate-system.md +4 -11
  15. package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
  16. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +20 -20
  17. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  18. package/dist/assets/templates/html/health.html +3 -5
  19. package/dist/cli/retired-commands.js +1 -1
  20. package/dist/cli/shared.js +6 -2
  21. package/dist/cli/unknown-flags.js +24 -1
  22. package/dist/cli.js +68 -10
  23. package/dist/commands/agent/agent-dispatch.js +1 -1
  24. package/dist/commands/command/command-execution.js +24 -62
  25. package/dist/commands/feedback-cli.js +0 -1
  26. package/dist/commands/health/accept-rate.js +6 -0
  27. package/dist/commands/health/archive-usage.js +92 -0
  28. package/dist/commands/health/checks.js +83 -74
  29. package/dist/commands/health/config-skew.js +38 -0
  30. package/dist/commands/health/data-dir-usage.js +25 -13
  31. package/dist/commands/health/egress.js +54 -0
  32. package/dist/commands/health/html-report.js +1 -42
  33. package/dist/commands/health/improve-metrics.js +136 -591
  34. package/dist/commands/health/md-report.js +1 -6
  35. package/dist/commands/health/plugin-staleness.js +53 -3
  36. package/dist/commands/health/renderers.js +12 -4
  37. package/dist/commands/health/report-view-model.js +14 -120
  38. package/dist/commands/health/types-improve.js +4 -19
  39. package/dist/commands/health/windows.js +64 -74
  40. package/dist/commands/health.js +145 -143
  41. package/dist/commands/improve/consolidate/chunking.js +26 -117
  42. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  43. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  44. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  45. package/dist/commands/improve/consolidate.js +589 -1127
  46. package/dist/commands/improve/content-hash.js +16 -24
  47. package/dist/commands/improve/distill/content-repair.js +18 -100
  48. package/dist/commands/improve/distill-guards.js +20 -81
  49. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  50. package/dist/commands/improve/distill.js +608 -1041
  51. package/dist/commands/improve/eligibility.js +126 -390
  52. package/dist/commands/improve/execution.js +8 -10
  53. package/dist/commands/improve/extract-prompt.js +1 -2
  54. package/dist/commands/improve/extract.js +487 -1046
  55. package/dist/commands/improve/feedback-valence.js +0 -25
  56. package/dist/commands/improve/improve-cli.js +75 -169
  57. package/dist/commands/improve/improve-result-file.js +10 -66
  58. package/dist/commands/improve/improve-strategies.js +52 -4
  59. package/dist/commands/improve/improve-usage-report.js +18 -64
  60. package/dist/commands/improve/improve.js +480 -1074
  61. package/dist/commands/improve/ledger.js +119 -0
  62. package/dist/commands/improve/locks.js +2 -8
  63. package/dist/commands/improve/loop-stages.js +415 -1073
  64. package/dist/commands/improve/memory/derived-ref.js +12 -77
  65. package/dist/commands/improve/memory/memory-belief.js +16 -118
  66. package/dist/commands/improve/memory/memory-improve.js +266 -14
  67. package/dist/commands/improve/outcome-loop.js +28 -156
  68. package/dist/commands/improve/planner.js +5 -15
  69. package/dist/commands/improve/preparation.js +779 -2319
  70. package/dist/commands/improve/proactive-maintenance.js +34 -101
  71. package/dist/commands/improve/reflect-noise.js +104 -280
  72. package/dist/commands/improve/reflect.js +642 -1353
  73. package/dist/commands/improve/retrieval-gate.js +127 -0
  74. package/dist/commands/improve/retrieval-scope.js +92 -0
  75. package/dist/commands/improve/salience.js +41 -240
  76. package/dist/commands/improve/session-asset.js +19 -100
  77. package/dist/commands/improve/stage.js +322 -0
  78. package/dist/commands/lint/base-linter.js +37 -15
  79. package/dist/commands/proposal/drain.js +261 -578
  80. package/dist/commands/proposal/proposal-cli.js +19 -20
  81. package/dist/commands/proposal/proposal-types.js +31 -24
  82. package/dist/commands/proposal/proposal.js +38 -8
  83. package/dist/commands/proposal/propose.js +134 -160
  84. package/dist/commands/proposal/repository.js +1097 -1394
  85. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  86. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  87. package/dist/commands/proposal/validators/proposals.js +22 -89
  88. package/dist/commands/read/curate.js +105 -462
  89. package/dist/commands/read/knowledge.js +3 -2
  90. package/dist/commands/read/search-cli.js +16 -33
  91. package/dist/commands/read/search.js +17 -23
  92. package/dist/commands/read/show.js +57 -108
  93. package/dist/commands/sources/bundle-cli.js +25 -2
  94. package/dist/commands/sources/bundle-config-ops.js +4 -0
  95. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  96. package/dist/commands/sources/info.js +127 -29
  97. package/dist/commands/sources/installed-stashes.js +197 -746
  98. package/dist/commands/sources/schema-repair.js +98 -129
  99. package/dist/commands/sources/source-add.js +62 -12
  100. package/dist/commands/sources/source-manage.js +9 -2
  101. package/dist/commands/sources/stash-cli.js +24 -4
  102. package/dist/commands/tasks/explain.js +10 -13
  103. package/dist/commands/tasks/tasks-cli.js +12 -13
  104. package/dist/commands/tasks/tasks.js +350 -936
  105. package/dist/commands/tasks/validate.js +26 -24
  106. package/dist/commands/workflow/plan.js +22 -29
  107. package/dist/commands/workflow-cli.js +4 -4
  108. package/dist/core/adapter/adapters/akm-adapter.js +2 -1
  109. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  110. package/dist/core/adapter/adapters/akm-metadata.js +42 -12
  111. package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
  112. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  113. package/dist/core/adapter/execution-source.js +17 -29
  114. package/dist/core/asset/asset-placement.js +4 -13
  115. package/dist/core/asset/frontmatter.js +106 -1
  116. package/dist/core/asset/resolve-ref.js +1 -1
  117. package/dist/core/bundle-id.js +42 -5
  118. package/dist/core/bundle-rename.js +285 -0
  119. package/dist/core/config/config-io.js +1 -2
  120. package/dist/core/config/config-schema.js +9 -34
  121. package/dist/core/config/config-walker.js +1 -1
  122. package/dist/core/config/config.js +184 -111
  123. package/dist/core/config/engine-semantics.js +0 -2
  124. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  125. package/dist/core/config/schema/embedding.js +20 -5
  126. package/dist/core/config/schema/engines.js +5 -0
  127. package/dist/core/config/schema/execution.js +1 -1
  128. package/dist/core/config/schema/experimental.js +1 -1
  129. package/dist/core/config/schema/improve-processes.js +54 -125
  130. package/dist/core/config/schema/improve.js +4 -42
  131. package/dist/core/config/schema/index-config.js +9 -48
  132. package/dist/core/config/schema/scheduler.js +12 -12
  133. package/dist/core/config/schema/search.js +6 -22
  134. package/dist/core/env-secret-ref.js +0 -1
  135. package/dist/core/errors.js +8 -9
  136. package/dist/core/file-change.js +13 -5
  137. package/dist/core/file-lock.js +76 -173
  138. package/dist/core/improve-result.js +35 -7
  139. package/dist/core/improve-types.js +0 -1
  140. package/dist/core/logs-db.js +2 -2
  141. package/dist/core/loopback.js +7 -12
  142. package/dist/core/non-task-input.js +20 -0
  143. package/dist/core/parse.js +13 -16
  144. package/dist/core/paths.js +0 -24
  145. package/dist/core/redaction.js +109 -2
  146. package/dist/core/run-lock.js +2 -5
  147. package/dist/core/spawn-env.js +1 -1
  148. package/dist/core/state/migrations.js +123 -61
  149. package/dist/core/state-db-scope.js +2 -4
  150. package/dist/core/state-db.js +126 -692
  151. package/dist/core/time.js +0 -20
  152. package/dist/core/type-presentation.js +1 -9
  153. package/dist/core/write-source.js +294 -1005
  154. package/dist/execution/input-contract.js +1 -1
  155. package/dist/execution/resolved-request.js +135 -689
  156. package/dist/execution/source.js +63 -257
  157. package/dist/execution/target-ref.js +1 -1
  158. package/dist/indexer/bundle-identity-guard.js +2 -2
  159. package/dist/indexer/db/llm-cache.js +2 -2
  160. package/dist/indexer/ensure-index.js +77 -73
  161. package/dist/indexer/index-rebuild-lock.js +3 -11
  162. package/dist/indexer/index-writer-lock.js +8 -17
  163. package/dist/indexer/index-written-assets.js +141 -154
  164. package/dist/indexer/indexer.js +400 -1124
  165. package/dist/indexer/links/declared-links.js +90 -0
  166. package/dist/indexer/materialize-embeddings.js +60 -397
  167. package/dist/indexer/passes/memory-inference.js +96 -90
  168. package/dist/indexer/passes/metadata.js +132 -219
  169. package/dist/indexer/read-preflight.js +0 -7
  170. package/dist/indexer/scan/doc-to-entry.js +2 -3
  171. package/dist/indexer/scan/drain-dir.js +1 -1
  172. package/dist/indexer/search/db-search.js +190 -590
  173. package/dist/indexer/search/fts-query.js +30 -41
  174. package/dist/indexer/search/ranking.js +28 -154
  175. package/dist/indexer/search/search-attribution.js +12 -32
  176. package/dist/indexer/search/search-fields.js +11 -15
  177. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  178. package/dist/indexer/search/search-source.js +1 -4
  179. package/dist/indexer/usage/usage-events.js +36 -7
  180. package/dist/indexer/walk/walker.js +3 -4
  181. package/dist/integrations/agent/engine-fallback.js +23 -40
  182. package/dist/integrations/agent/engine-resolution.js +93 -183
  183. package/dist/integrations/agent/execution.js +507 -0
  184. package/dist/integrations/agent/model-map.js +28 -156
  185. package/dist/integrations/agent/request-lowering.js +66 -141
  186. package/dist/integrations/agent/runner-dispatch.js +143 -321
  187. package/dist/integrations/agent/runner.js +54 -14
  188. package/dist/integrations/lockfile.js +53 -101
  189. package/dist/llm/client.js +18 -6
  190. package/dist/llm/embedders/deterministic.js +2 -3
  191. package/dist/llm/embedders/profile.js +71 -0
  192. package/dist/llm/embedders/remote.js +11 -17
  193. package/dist/llm/feature-gate.js +0 -8
  194. package/dist/llm/index-passes.js +3 -5
  195. package/dist/llm/memory-infer.js +1 -2
  196. package/dist/llm/structured-call.js +5 -24
  197. package/dist/output/generic-render.js +23 -11
  198. package/dist/output/html-render.js +13 -10
  199. package/dist/output/render-registry.js +3 -32
  200. package/dist/output/shapes/helpers.js +25 -38
  201. package/dist/output/shapes/passthrough.js +1 -9
  202. package/dist/{indexer/graph/graph-types.js → output/text/bundle-rename.js} +4 -1
  203. package/dist/output/text/command-format.js +69 -31
  204. package/dist/output/text/helpers.js +1 -1
  205. package/dist/output/text/migrate.js +5 -14
  206. package/dist/output/text/proposal-format.js +48 -3
  207. package/dist/output/text/show-format.js +13 -17
  208. package/dist/output/text/workflow-format.js +0 -32
  209. package/dist/output/text.js +2 -0
  210. package/dist/registry/factory.js +4 -19
  211. package/dist/registry/network.js +66 -220
  212. package/dist/registry/providers/index.js +0 -2
  213. package/dist/registry/providers/skills-sh.js +3 -14
  214. package/dist/registry/providers/static-index.js +24 -26
  215. package/dist/registry/resolve.js +55 -131
  216. package/dist/scripts/akm-migrate-node.js +42948 -92369
  217. package/dist/scripts/akm-migrate.js +42935 -92354
  218. package/dist/setup/registry-stash-loader.js +4 -13
  219. package/dist/setup/semantic-assets.js +3 -44
  220. package/dist/setup/setup.js +1 -1
  221. package/dist/setup/steps/connection.js +5 -6
  222. package/dist/setup/steps/platforms.js +2 -2
  223. package/dist/setup/steps/tasks.js +25 -15
  224. package/dist/sources/provider-factory.js +17 -18
  225. package/dist/sources/providers/filesystem.js +2 -3
  226. package/dist/sources/providers/git-install.js +7 -1
  227. package/dist/sources/providers/git-provider.js +0 -3
  228. package/dist/sources/providers/git-stash.js +83 -21
  229. package/dist/sources/providers/npm.js +2 -4
  230. package/dist/sources/providers/provider-utils.js +5 -10
  231. package/dist/sources/providers/website.js +0 -2
  232. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  233. package/dist/sources/website-url.js +2 -2
  234. package/dist/storage/database.js +9 -35
  235. package/dist/storage/repositories/improve-ledger-repository.js +209 -0
  236. package/dist/storage/repositories/index-connection.js +39 -72
  237. package/dist/storage/repositories/index-entries-repository.js +131 -129
  238. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  239. package/dist/storage/repositories/index-entry-schema.js +101 -268
  240. package/dist/storage/repositories/index-fts-repository.js +86 -256
  241. package/dist/storage/repositories/index-links-repository.js +143 -0
  242. package/dist/storage/repositories/index-llm-cache-repository.js +7 -9
  243. package/dist/storage/repositories/index-meta-repository.js +6 -4
  244. package/dist/storage/repositories/index-schema.js +257 -325
  245. package/dist/storage/repositories/index-utility-repository.js +8 -29
  246. package/dist/storage/repositories/index-vec-repository.js +133 -414
  247. package/dist/storage/repositories/outcome-repository.js +2 -1
  248. package/dist/storage/repositories/proposals-repository.js +104 -1
  249. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  250. package/dist/storage/repositories/salience-repository.js +1 -19
  251. package/dist/storage/repositories/task-history-repository.js +26 -4
  252. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  253. package/dist/storage/sqlite-migrations.js +136 -0
  254. package/dist/storage/sqlite-pragmas.js +11 -9
  255. package/dist/storage/sqlite-transaction.js +170 -0
  256. package/dist/storage/state-db-integrity.js +130 -0
  257. package/dist/tasks/activation-config.js +134 -62
  258. package/dist/tasks/backends/cron.js +191 -302
  259. package/dist/tasks/backends/exec-utils.js +2 -5
  260. package/dist/tasks/backends/launchd.js +141 -748
  261. package/dist/tasks/backends/schtasks.js +119 -623
  262. package/dist/tasks/prepare/prepare-support.js +5 -15
  263. package/dist/tasks/prepare/prepare.js +0 -2
  264. package/dist/tasks/resolve-akm-bin.js +20 -79
  265. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  266. package/dist/tasks/run/load-task.js +1 -1
  267. package/dist/tasks/scheduler-binding.js +20 -238
  268. package/dist/tasks/scheduler-invocation.js +136 -244
  269. package/dist/tasks/scheduler-lock.js +53 -0
  270. package/dist/tasks/scheduler-sync.js +368 -679
  271. package/dist/tasks/source/parse-task-source.js +55 -9
  272. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  273. package/dist/tasks/source/task-to-v4.js +464 -88
  274. package/dist/workflows/authoring/authoring.js +3 -12
  275. package/dist/workflows/compile.js +211 -0
  276. package/dist/workflows/concurrency-policy.js +13 -74
  277. package/dist/workflows/exec/child-invocation.js +3 -17
  278. package/dist/workflows/exec/child-workflow.js +32 -141
  279. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  280. package/dist/workflows/exec/environment.js +98 -0
  281. package/dist/workflows/exec/exec-unit.js +33 -140
  282. package/dist/workflows/exec/frozen-judge.js +7 -59
  283. package/dist/workflows/exec/native-executor.js +82 -341
  284. package/dist/workflows/exec/param-secrets.js +29 -47
  285. package/dist/workflows/exec/run-workflow.js +154 -387
  286. package/dist/workflows/exec/scheduler.js +9 -36
  287. package/dist/workflows/exec/step-work.js +127 -430
  288. package/dist/workflows/exec/unit-dispatch.js +11 -63
  289. package/dist/workflows/exec/unit-writer.js +8 -52
  290. package/dist/workflows/exec/worktree.js +39 -273
  291. package/dist/workflows/freeze/child-output-references.js +4 -15
  292. package/dist/workflows/freeze/environment.js +99 -92
  293. package/dist/workflows/freeze/freeze.js +172 -0
  294. package/dist/workflows/freeze/step-values.js +19 -21
  295. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  296. package/dist/workflows/freeze/targets/command.js +10 -33
  297. package/dist/workflows/freeze/targets/script.js +5 -12
  298. package/dist/workflows/freeze/targets/shell.js +3 -6
  299. package/dist/workflows/freeze/targets/task.js +25 -80
  300. package/dist/workflows/freeze/task-bindings.js +20 -67
  301. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  302. package/dist/workflows/ir/params.js +6 -51
  303. package/dist/workflows/ir/plan-hash.js +2 -34
  304. package/dist/workflows/parser.js +140 -43
  305. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  306. package/dist/workflows/renderer.js +36 -69
  307. package/dist/workflows/resource-limits.js +12 -120
  308. package/dist/workflows/runtime/agent-identity.js +8 -40
  309. package/dist/workflows/runtime/run-outputs.js +3 -6
  310. package/dist/workflows/runtime/run-plan.js +316 -0
  311. package/dist/workflows/runtime/runs.js +48 -200
  312. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  313. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  314. package/dist/workflows/validate-summary.js +2 -7
  315. package/docs/integration/bundling-akm.md +49 -42
  316. package/docs/migration/README.md +1 -0
  317. package/docs/migration/release-notes/0.9.17.md +43 -0
  318. package/docs/migration/v0.9.1-to-v0.9.2.md +23 -7
  319. package/docs/reference/cli.md +232 -135
  320. package/docs/reference/configuration.md +71 -57
  321. package/docs/reference/data-and-telemetry.md +20 -21
  322. package/docs/reference/tasks.md +105 -39
  323. package/docs/reference/workflow-schema.md +14 -18
  324. package/docs/reference/workflows.md +6 -9
  325. package/package.json +1 -1
  326. package/schemas/akm-config.json +115 -738
  327. package/schemas/akm-workflow.json +1 -0
  328. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  329. package/dist/assets/prompts/contradiction-judge.md +0 -33
  330. package/dist/assets/prompts/graph-extract-system.md +0 -1
  331. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  332. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  333. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  334. package/dist/commands/health/advisories.js +0 -150
  335. package/dist/commands/health/metrics.js +0 -329
  336. package/dist/commands/health/surfaces.js +0 -102
  337. package/dist/commands/improve/anti-collapse.js +0 -83
  338. package/dist/commands/improve/collapse-detector.js +0 -432
  339. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  340. package/dist/commands/improve/consolidate/merge.js +0 -149
  341. package/dist/commands/improve/distill/promote-memory.js +0 -291
  342. package/dist/commands/improve/distill/quality-gate.js +0 -337
  343. package/dist/commands/improve/eval-cases.js +0 -52
  344. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  345. package/dist/commands/improve/proposal-envelope.js +0 -31
  346. package/dist/commands/improve/run-context.js +0 -123
  347. package/dist/commands/improve/shared.js +0 -31
  348. package/dist/commands/improve/source-identity.js +0 -28
  349. package/dist/commands/improve/triage.js +0 -96
  350. package/dist/commands/proposal/drain-policies.js +0 -151
  351. package/dist/commands/sources/update-transaction.js +0 -220
  352. package/dist/core/action-contributors.js +0 -28
  353. package/dist/core/config/config-version-shim.js +0 -101
  354. package/dist/core/fs-txn.js +0 -405
  355. package/dist/core/lexical-score.js +0 -25
  356. package/dist/core/maintenance-barrier.js +0 -167
  357. package/dist/execution/executable-identity.js +0 -105
  358. package/dist/execution/guarded-source.js +0 -427
  359. package/dist/indexer/db/graph-db.js +0 -444
  360. package/dist/indexer/graph/graph-boost.js +0 -427
  361. package/dist/indexer/graph/graph-dedup.js +0 -95
  362. package/dist/indexer/graph/graph-extraction.js +0 -1108
  363. package/dist/indexer/search/name-match.js +0 -35
  364. package/dist/indexer/search/ranking-contributors.js +0 -515
  365. package/dist/indexer/search/ranking-types.js +0 -4
  366. package/dist/indexer/walk/project-context.js +0 -192
  367. package/dist/integrations/agent/execution-cascade.js +0 -566
  368. package/dist/integrations/agent/execution-definitions.js +0 -202
  369. package/dist/integrations/agent/execution-lowering.js +0 -841
  370. package/dist/integrations/agent/execution-preparation.js +0 -98
  371. package/dist/integrations/agent/inline-execution.js +0 -74
  372. package/dist/llm/graph-extract.js +0 -728
  373. package/dist/llm/metadata-enhance.js +0 -96
  374. package/dist/registry/create-provider-registry.js +0 -29
  375. package/dist/registry/pinned-request-helper.js +0 -247
  376. package/dist/registry/pinned-transport.js +0 -717
  377. package/dist/sources/providers/index.js +0 -14
  378. package/dist/storage/engines/sqlite-migrations.js +0 -271
  379. package/dist/storage/repositories/canaries-repository.js +0 -107
  380. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  381. package/dist/storage/repositories/registry-cache.js +0 -113
  382. package/dist/tasks/scheduler-sync-preview.js +0 -52
  383. package/dist/tasks/source/task-to-v3.js +0 -507
  384. package/dist/workflows/freeze/resolve-steps.js +0 -86
  385. package/dist/workflows/freeze/source-freeze.js +0 -64
  386. package/dist/workflows/ir/compile.js +0 -321
  387. package/dist/workflows/ir/environment-v4.js +0 -330
  388. package/dist/workflows/ir/freeze-v4.js +0 -153
  389. package/dist/workflows/ir/schema-v4.js +0 -745
  390. package/dist/workflows/ir/schema.js +0 -354
  391. package/dist/workflows/program/schema.js +0 -77
  392. package/dist/workflows/runtime/checkin.js +0 -57
  393. package/dist/workflows/runtime/plan-classifier.js +0 -196
  394. package/dist/workflows/runtime/unit-checkin.js +0 -45
  395. package/dist/workflows/runtime/unit-phases.js +0 -20
  396. package/dist/workflows/schema.js +0 -4
  397. package/dist/workflows/source-ir/compile.js +0 -200
  398. package/dist/workflows/source-ir/program.js +0 -50
  399. package/dist/workflows/source-ir/result.js +0 -26
  400. package/dist/workflows/source-ir/schema.js +0 -786
  401. package/dist/workflows/source-ir/triggers.js +0 -79
  402. package/dist/workflows/source-ir/uses.js +0 -40
  403. package/dist/workflows/validator.js +0 -60
@@ -1,7 +1,23 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- import { createHash } from "node:crypto";
4
+ /**
5
+ * `akm consolidate` — show the model the memory pool in chunks of similar
6
+ * memories and queue a knowledge proposal for each memory it says should be
7
+ * promoted. Promotion emits a reviewable proposal and never touches the
8
+ * memory directly; accepting it later retires the source memory (O1, in
9
+ * `proposal/repository.ts`). Memories the improve ledger judged recently and
10
+ * that have not changed since are not judged again.
11
+ *
12
+ * Accounting invariant (the promote pass only): `processed == promoted +
13
+ * judgedNoAction + Σ(skipReasons) + failedChunkMemories`.
14
+ *
15
+ * A second pass, the pair pass (`consolidate/pair-pass.ts`, alpha.9), runs
16
+ * alongside this one and keeps its own separate counters (`pairPass` on the
17
+ * result) — it judges near-duplicate and superseding pairs across the wider
18
+ * memory tier and mints `retire` proposals; see that module's own doc
19
+ * comment.
20
+ */
5
21
  import fs from "node:fs";
6
22
  import path from "node:path";
7
23
  import consolidateSystemPrompt from "../../assets/prompts/consolidate-system.md" with { type: "text" };
@@ -11,52 +27,64 @@ import { parseFrontmatter } from "../../core/asset/frontmatter.js";
11
27
  import { conceptIdFromTypeName, displayRef, parseRefInput } from "../../core/asset/resolve-ref.js";
12
28
  import { getImproveProcessConfig, loadConfig } from "../../core/config/config.js";
13
29
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
14
- import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
15
30
  import { openStateDatabase } from "../../core/state-db.js";
16
- import { parseSinceToIsoLenient } from "../../core/time.js";
17
31
  import { warn, warnVerbose } from "../../core/warn.js";
18
32
  import { resolveWriteTarget } from "../../core/write-source.js";
19
33
  import { deriveInstallations } from "../../indexer/installations.js";
20
34
  import { resolveSourceEntries } from "../../indexer/search/search-source.js";
21
- import { disposeLoweredExecutionDispatchLease, } from "../../integrations/agent/execution-lowering.js";
35
+ import { USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
36
+ import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
22
37
  import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
23
- import { callStructured, preflightStructuredLlmRunner } from "../../llm/structured-call.js";
24
38
  import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
25
39
  import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../storage/repositories/index-connection.js";
26
- import { findEntryIdByRef, getAllEntries, getEntryById } from "../../storage/repositories/index-entries-repository.js";
27
- import { getNeighborsByEntryId } from "../../storage/repositories/index-vec-repository.js";
28
- import { isProposalSkipped, listProposals, listProposalsReadOnly, proposalContent, } from "../proposal/repository.js";
29
- import { hasSupersededStatus, validateProposalFrontmatter } from "../proposal/validators/proposal-quality-validators.js";
30
- import { DEFAULT_RANDOM_CLUSTER_FRACTION } from "./anti-collapse.js";
31
- import { cacheHash } from "./content-hash.js";
32
- import { resolveImproveLlmExecution } from "./execution.js";
33
- import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
34
- import { emitProposal } from "./proposal-envelope.js";
35
- import { createRunContext } from "./run-context.js";
36
- // Chunk sizing + per-chunk prompt assembly live in ./consolidate/chunking.
40
+ import { getAllEntries } from "../../storage/repositories/index-entries-repository.js";
41
+ import { listProposals, listProposalsReadOnly, proposalContent } from "../proposal/repository.js";
42
+ import { hasHotCaptureMode, hasSupersededStatus, validateProposalFrontmatter, } from "../proposal/validators/proposal-quality-validators.js";
37
43
  import { buildChunkPrompt, computeSafeChunkSize, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
38
- // Eligibility / safety predicates live in ./consolidate/eligibility.
39
- import { isConsolidationEligibleMemoryName, isHotCapturedMemory } from "./consolidate/eligibility.js";
40
- // Plan parsing / merging (pure op-reconciliation algebra) lives in
41
- // ./consolidate/merge.
42
- import { isValidOp, mergePlans } from "./consolidate/merge.js";
43
- // LLM-output sanitization (pure string/frontmatter transforms) lives in
44
- // ./consolidate/sanitize.
44
+ import { runConsolidatePairPass } from "./consolidate/pair-pass.js";
45
45
  import { sanitizeMergedContent } from "./consolidate/sanitize.js";
46
- // ── Prompts ─────────────────────────────────────────────────────────────────
47
- const CONSOLIDATE_SYSTEM_PROMPT = consolidateSystemPrompt;
46
+ import { contentHash } from "./content-hash.js";
47
+ import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
48
+ import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot, recordLedgerAttempt } from "./ledger.js";
49
+ import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
50
+ import { callStage, mintProposal, noticeSet, stageRunner } from "./stage.js";
51
+ /** A plan op worth acting on. Retired advisory ops (merge/delete/contradict) are dropped, never thrown on. */
52
+ export function isValidOp(op) {
53
+ if (typeof op !== "object" || op === null)
54
+ return false;
55
+ const o = op;
56
+ return o.op === "promote" && typeof o.ref === "string" && typeof o.knowledgeRef === "string";
57
+ }
58
+ /** Reconcile the per-chunk plans: one promotion per source memory, the last chunk's wins. */
59
+ export function mergePlans(chunks) {
60
+ const byRef = new Map();
61
+ for (const chunk of chunks)
62
+ for (const op of chunk)
63
+ byRef.set(op.ref, op);
64
+ return [...byRef.values()];
65
+ }
66
+ export function isConsolidationEligibleMemoryName(name) {
67
+ return !name.endsWith(".derived");
68
+ }
48
69
  /**
49
- * JSON Schema for structured consolidate plans (PR 1 of the asset-writers
50
- * decision — see knowledge/projects/akm/asset-writers-investigation/00-synthesis).
51
- * Mirrors the {ops[], warnings?[]} shape currently described in
52
- * CONSOLIDATE_SYSTEM_PROMPT. Providers with `supportsJsonSchema: true` enforce
53
- * the shape upstream so the chunk-level "invalid plan from AI — skipping"
54
- * branch in `runConsolidate` becomes unreachable on schema-honouring providers.
55
- *
56
- * The four operation variants (merge / delete / promote / contradict) are
57
- * modeled as a oneOf so a structured-output provider can still tell them apart
58
- * by the required `op` discriminator. `parseEmbeddedJsonResponse` keeps
59
- * working as a fallback parser for providers that ignore the schema.
70
+ * A `captureMode: hot` memory (written deliberately with `akm remember`). A
71
+ * missing file is not hot; an unreadable one is treated as hot — the check is
72
+ * a protection and must not fail open.
73
+ */
74
+ export function isHotCapturedMemory(filePath) {
75
+ if (!fs.existsSync(filePath))
76
+ return false;
77
+ try {
78
+ return hasHotCaptureMode(parseFrontmatter(fs.readFileSync(filePath, "utf8")).data);
79
+ }
80
+ catch {
81
+ return true;
82
+ }
83
+ }
84
+ /**
85
+ * Structured-output schema for a plan. Promote-only: merge/delete/contradict
86
+ * were removed in 0.9.17-alpha.1 (`e82eec811`) after running in production —
87
+ * they cost thousands of completion tokens.
60
88
  */
61
89
  export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
62
90
  type: "object",
@@ -65,191 +93,91 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
65
93
  properties: {
66
94
  operations: {
67
95
  type: "array",
68
- description: "Ordered list of consolidate operations the planner proposes.",
96
+ description: "Ordered list of promote operations the planner proposes.",
69
97
  items: {
70
- oneOf: [
71
- {
72
- type: "object",
73
- required: ["op", "primary", "secondaries", "mergeStrategy"],
74
- additionalProperties: false,
75
- properties: {
76
- op: { type: "string", enum: ["merge"] },
77
- primary: { type: "string", minLength: 1 },
78
- secondaries: {
79
- type: "array",
80
- minItems: 1,
81
- maxItems: 1,
82
- items: { type: "string", minLength: 1 },
83
- },
84
- mergeStrategy: { type: "string", minLength: 1 },
85
- confidence: { type: "number", minimum: 0, maximum: 1 },
86
- },
87
- },
88
- {
89
- type: "object",
90
- required: ["op", "ref", "reason"],
91
- additionalProperties: false,
92
- properties: {
93
- op: { type: "string", enum: ["delete"] },
94
- ref: { type: "string", minLength: 1 },
95
- reason: { type: "string", minLength: 1 },
96
- confidence: { type: "number", minimum: 0, maximum: 1 },
97
- },
98
- },
99
- {
100
- type: "object",
101
- required: ["op", "ref", "knowledgeRef", "reason"],
102
- additionalProperties: false,
103
- properties: {
104
- op: { type: "string", enum: ["promote"] },
105
- ref: { type: "string", minLength: 1 },
106
- knowledgeRef: { type: "string", minLength: 1 },
107
- reason: { type: "string", minLength: 1 },
108
- description: { type: "string" },
109
- confidence: { type: "number", minimum: 0, maximum: 1 },
110
- },
111
- },
112
- {
113
- type: "object",
114
- required: ["op", "ref", "contradictedByRef", "reason"],
115
- additionalProperties: false,
116
- properties: {
117
- op: { type: "string", enum: ["contradict"] },
118
- ref: { type: "string", minLength: 1 },
119
- contradictedByRef: { type: "string", minLength: 1 },
120
- reason: { type: "string", minLength: 1 },
121
- confidence: { type: "number", minimum: 0, maximum: 1 },
122
- },
123
- },
124
- ],
98
+ type: "object",
99
+ required: ["op", "ref", "knowledgeRef", "reason"],
100
+ additionalProperties: false,
101
+ properties: {
102
+ op: { type: "string", enum: ["promote"] },
103
+ ref: { type: "string", minLength: 1 },
104
+ knowledgeRef: { type: "string", minLength: 1 },
105
+ reason: { type: "string", minLength: 1, maxLength: 200 },
106
+ description: { type: "string" },
107
+ confidence: { type: "number", minimum: 0, maximum: 1 },
108
+ },
125
109
  },
126
110
  },
127
- warnings: {
128
- type: "array",
129
- description: "Optional list of human-readable concerns the planner wants to surface.",
130
- items: { type: "string" },
131
- },
132
111
  },
133
112
  };
113
+ /**
114
+ * Order memories so similar ones sit together and land in the same chunk:
115
+ * a greedy nearest-neighbour chain over description+tag embeddings (cached in
116
+ * `body_embeddings` under the text's hash). Keeps the original order without
117
+ * an embedding config, for fewer than three memories, or when embedding fails.
118
+ */
134
119
  async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
135
- const noTelemetry = { embedMs: 0, cacheHits: 0, cacheMisses: 0 };
120
+ const telemetry = { embedMs: 0, cacheHits: 0, cacheMisses: 0 };
136
121
  if (memories.length < 3 || !config.embedding)
137
- return { ordered: memories, embedTelemetry: noTelemetry };
138
- // WS-3a: cluster uses description+tags as the embedding input (NOT the raw
139
- // body) — this is intentionally different from the dedup/body cache because
140
- // the clustering goal is semantic grouping, not dedup twin detection.
141
- // The body_embeddings cache is keyed by cacheHash(body); clustering inputs
142
- // are keyed by cacheHash(description+tags text). Re-use the same table with
143
- // a distinct hash so the two lookup sets never collide.
122
+ return { ordered: memories, embedTelemetry: telemetry };
144
123
  const modelId = resolveEmbeddingModelId(config.embedding);
145
- const texts = memories.map((m) => {
146
- const parts = [];
147
- if (m.description)
148
- parts.push(m.description);
149
- if (m.tags.length > 0)
150
- parts.push(m.tags.join(" "));
151
- return parts.join(". ") || m.name;
152
- });
153
- // Compute content hashes for the cluster texts (not bodies — different input).
154
- const contentHashes = texts.map((t) => createHash("sha256").update(t, "utf8").digest("hex"));
155
- // WS-5: track embed cache hits/misses for perf telemetry.
156
- let embedMs = 0;
157
- let cacheHits = 0;
158
- let cacheMisses = 0;
159
- let cachedVecs = new Map();
124
+ const texts = memories.map((m) => [m.description, m.tags.join(" ")].filter(Boolean).join(". ") || m.name);
125
+ const hashes = texts.map((t) => contentHash(t));
126
+ let cached = new Map();
160
127
  if (stateDb) {
161
128
  try {
162
- cachedVecs = getBodyEmbeddings(stateDb, contentHashes, modelId);
129
+ cached = getBodyEmbeddings(stateDb, hashes, modelId);
163
130
  }
164
131
  catch {
165
- // Fail open.
166
- cachedVecs = new Map();
132
+ cached = new Map();
167
133
  }
168
134
  }
169
- const missIndices = [];
170
- const missTexts = [];
171
- for (let i = 0; i < texts.length; i++) {
172
- if (!cachedVecs.has(contentHashes[i])) {
173
- missIndices.push(i);
174
- missTexts.push(texts[i]);
175
- cacheMisses++;
176
- }
177
- else {
178
- cacheHits++;
179
- }
180
- }
181
- let missVecs = [];
182
- if (missTexts.length > 0) {
135
+ const missIndices = hashes.flatMap((hash, i) => (cached.has(hash) ? [] : [i]));
136
+ telemetry.cacheHits = memories.length - missIndices.length;
137
+ telemetry.cacheMisses = missIndices.length;
138
+ const vectors = new Map(cached);
139
+ if (missIndices.length > 0) {
183
140
  const embedStart = Date.now();
141
+ let missVecs;
184
142
  try {
185
- missVecs = await embedBatch(missTexts, config.embedding, signal);
143
+ missVecs = await embedBatch(missIndices.map((i) => texts[i]), config.embedding, signal);
186
144
  }
187
145
  catch {
188
- // Fail open: embedding failures degrade gracefully to original order.
189
- return { ordered: memories, embedTelemetry: { embedMs, cacheHits, cacheMisses } };
146
+ return { ordered: memories, embedTelemetry: telemetry };
190
147
  }
191
148
  finally {
192
- embedMs += Date.now() - embedStart;
149
+ telemetry.embedMs += Date.now() - embedStart;
193
150
  }
194
- // Upsert newly computed vectors into the cache. A skipped document
195
- // (embedBatch reports it via `undefined` rather than throwing, #874) has
196
- // no vector to cache — omit it rather than writing a bogus embedding.
197
- if (stateDb && missVecs.length === missTexts.length) {
151
+ const fresh = missIndices.flatMap((idx, pos) => {
152
+ const embedding = missVecs[pos];
153
+ return embedding ? [{ contentHash: hashes[idx], embedding, modelId }] : [];
154
+ });
155
+ for (const entry of fresh)
156
+ vectors.set(entry.contentHash, entry.embedding);
157
+ // A document the embedder skipped has no vector to cache.
158
+ if (stateDb && missVecs.length === missIndices.length) {
198
159
  try {
199
- const toUpsert = missIndices.flatMap((idx, pos) => {
200
- const embedding = missVecs[pos];
201
- return embedding ? [{ contentHash: contentHashes[idx], embedding, modelId }] : [];
202
- });
203
- upsertBodyEmbeddings(stateDb, toUpsert);
160
+ upsertBodyEmbeddings(stateDb, fresh);
204
161
  }
205
162
  catch {
206
- // Fail open: cache write errors are non-fatal.
207
- }
208
- }
209
- }
210
- // Assemble the full embedding array in memories order.
211
- let embeddings = null;
212
- {
213
- const assembled = [];
214
- let ok = true;
215
- for (let i = 0; i < memories.length; i++) {
216
- const hash = contentHashes[i];
217
- const cached = cachedVecs.get(hash);
218
- if (cached) {
219
- assembled.push(cached);
220
- continue;
221
- }
222
- const missPos = missIndices.indexOf(i);
223
- const vec = missPos >= 0 ? missVecs[missPos] : undefined;
224
- if (vec) {
225
- assembled.push(vec);
163
+ // Cache writes are best-effort.
226
164
  }
227
- else {
228
- ok = false;
229
- break;
230
- }
231
- }
232
- if (ok && assembled.length === memories.length) {
233
- embeddings = assembled;
234
165
  }
235
166
  }
236
- const embedTelemetry = { embedMs, cacheHits, cacheMisses };
237
- if (!embeddings || embeddings.length !== memories.length)
238
- return { ordered: memories, embedTelemetry };
239
- // Greedy nearest-neighbour chain.
167
+ const embeddings = hashes.map((hash) => vectors.get(hash));
168
+ if (embeddings.some((vec) => !vec))
169
+ return { ordered: memories, embedTelemetry: telemetry };
240
170
  const used = new Array(memories.length).fill(false);
241
- const ordered = [];
242
- let current = 0; // start from the first memory
243
- ordered.push(memories[current]);
244
- used[current] = true;
171
+ const ordered = [memories[0]];
172
+ used[0] = true;
173
+ let current = 0;
245
174
  for (let step = 1; step < memories.length; step++) {
246
- const currentEmb = embeddings[current];
247
175
  let bestIdx = -1;
248
176
  let bestSim = -Infinity;
249
177
  for (let j = 0; j < memories.length; j++) {
250
178
  if (used[j])
251
179
  continue;
252
- const sim = cosineSimilarity(currentEmb, embeddings[j]);
180
+ const sim = cosineSimilarity(embeddings[current], embeddings[j]);
253
181
  if (sim > bestSim) {
254
182
  bestSim = sim;
255
183
  bestIdx = j;
@@ -261,49 +189,64 @@ async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
261
189
  used[bestIdx] = true;
262
190
  current = bestIdx;
263
191
  }
264
- return { ordered, embedTelemetry };
192
+ return { ordered, embedTelemetry: telemetry };
265
193
  }
266
- // ── Chunk helpers ────────────────────────────────────────────────────────────
267
194
  /**
268
- * Precompute body-hashes of all currently-pending consolidate proposals so
269
- * the per-chunk prompt can annotate memories whose body would just produce
270
- * a deterministic `dedup_pending_proposal` skip. Uses `cacheHash` (case-
271
- * preserving stripped body) — the same domain used by the body-embedding
272
- * cache. Empty set on any read/parse error — fail-safe to "annotate nothing"
273
- * so the LLM still proposes.
195
+ * Anti-collapse (default on, `antiCollapse.enabled: false` opts out): a small
196
+ * deterministic sample of the pool is spread through the similarity order so
197
+ * consolidation is not purely similarity-driven.
274
198
  */
199
+ function injectRandomClusterMembers(memories, profile, warnings) {
200
+ const config = getImproveProcessConfig("consolidate", profile)?.antiCollapse ?? {};
201
+ if (config.enabled === false || memories.length <= 2)
202
+ return memories;
203
+ const fraction = config.randomClusterFraction ?? 0.05;
204
+ const randomCount = Math.max(1, Math.floor(memories.length * fraction));
205
+ const sample = [...memories]
206
+ .sort((a, b) => contentHash(a.name).localeCompare(contentHash(b.name)))
207
+ .slice(0, randomCount);
208
+ const sampled = new Set(sample.map((m) => m.name));
209
+ const interval = Math.max(2, Math.floor(memories.length / randomCount));
210
+ const out = [];
211
+ let next = 0;
212
+ for (let i = 0; i < memories.length; i++) {
213
+ const m = memories[i];
214
+ if (m && !sampled.has(m.name))
215
+ out.push(m);
216
+ if (i > 0 && i % interval === 0 && next < sample.length)
217
+ out.push(sample[next++]);
218
+ }
219
+ while (next < sample.length)
220
+ out.push(sample[next++]);
221
+ warnings.push(`Anti-collapse: injected ${randomCount} random (non-similarity-driven) cluster member(s) into consolidation pool (fraction=${fraction}).`);
222
+ return out;
223
+ }
224
+ /** Body hashes of pending consolidate proposals, so the prompt can mark memories already queued. */
275
225
  function loadPendingConsolidateProposalHashes(stashDir) {
276
226
  const hashes = new Set();
277
227
  try {
278
- const pending = listProposalsReadOnly(stashDir, { status: "pending" }).filter((proposal) => proposal.source === "consolidate");
279
- for (const p of pending) {
228
+ for (const p of listProposalsReadOnly(stashDir, { status: "pending" })) {
229
+ if (p.source !== "consolidate")
230
+ continue;
280
231
  try {
281
- hashes.add(cacheHash(proposalContent(p)));
232
+ hashes.add(contentHash(proposalContent(p), "body"));
282
233
  }
283
234
  catch {
284
- // skip malformed payloads — they can't dedup anyway
235
+ // A malformed payload cannot dedup anyway.
285
236
  }
286
237
  }
287
238
  }
288
239
  catch {
289
- // listProposals throws on missing stash dir during tests — empty set is safe
240
+ // Annotate nothing; the model still proposes.
290
241
  }
291
242
  return hashes;
292
243
  }
293
244
  /**
294
- * Hash the bodies of live knowledge assets once per consolidation run.
295
- *
296
- * Pending-proposal dedup prevents repeated queue entries, but accepted
297
- * proposals leave that set. Without a live-asset guard, the next run can copy
298
- * the same memory body into a new knowledge slug indefinitely. Scan the target
299
- * tree directly (rather than trusting the asynchronously refreshed index) so
300
- * an already-written asset suppresses recurrence immediately.
245
+ * Body hashes of the live knowledge assets, read from disk (the index may lag
246
+ * a just-written asset), so an accepted promotion is not proposed again.
301
247
  */
302
248
  export function loadExistingKnowledgeBodyHashes(targetRoot) {
303
249
  const hashes = new Set();
304
- const knowledgeRoot = path.join(targetRoot, "knowledge");
305
- if (!fs.existsSync(knowledgeRoot))
306
- return hashes;
307
250
  const visit = (dir) => {
308
251
  let entries;
309
252
  try {
@@ -314,84 +257,31 @@ export function loadExistingKnowledgeBodyHashes(targetRoot) {
314
257
  }
315
258
  for (const entry of entries) {
316
259
  const entryPath = path.join(dir, entry.name);
317
- if (entry.isDirectory()) {
260
+ if (entry.isDirectory())
318
261
  visit(entryPath);
319
- }
320
262
  else if (entry.isFile() && entry.name.endsWith(".md")) {
321
263
  try {
322
- hashes.add(cacheHash(fs.readFileSync(entryPath, "utf8")));
264
+ hashes.add(contentHash(fs.readFileSync(entryPath, "utf8"), "body"));
323
265
  }
324
266
  catch {
325
- // An unreadable asset cannot provide reliable duplicate evidence.
267
+ // An unreadable asset is no duplicate evidence.
326
268
  }
327
269
  }
328
270
  }
329
271
  };
330
- visit(knowledgeRoot);
272
+ visit(path.join(targetRoot, "knowledge"));
331
273
  return hashes;
332
274
  }
333
- /** Parse a stored provenance ref and emit its canonical D-R5 display spelling. */
334
- function canonicalStoredXref(ref) {
275
+ /** A provenance ref in its canonical display spelling. */
276
+ function canonicalXref(ref) {
335
277
  try {
336
278
  const p = parseRefInput(ref);
337
279
  return displayRef({ type: p.type, name: p.name, bundleId: p.origin });
338
280
  }
339
281
  catch {
340
- return undefined;
282
+ return ref;
341
283
  }
342
284
  }
343
- function canonicalXref(ref) {
344
- return canonicalStoredXref(ref) ?? ref;
345
- }
346
- /**
347
- * The promoted asset's provenance xref set: existing body-frontmatter xrefs +
348
- * the promoted source ref, deduped after canonicalization (WI-8.5b: emitted in
349
- * the D-R5 new grammar via {@link canonicalXref}).
350
- */
351
- function promoteProvenanceXrefs(existing, sourceRef) {
352
- const priors = Array.isArray(existing) ? existing.map(String) : [];
353
- return [...new Set([...priors, sourceRef].map(canonicalXref))];
354
- }
355
- // ── LLM resolution ──────────────────────────────────────────────────────────
356
- /**
357
- * Resolve the symbolic LLM runner for the consolidate pass.
358
- *
359
- * Priority order (mirrors extract / reflect / distill — see
360
- * `resolveExtractRunConfig` in `src/commands/improve/extract.ts` and the
361
- * canonical improve execution-cascade pattern):
362
- *
363
- * 1. `improve.strategies.<name>.processes.consolidate.engine`
364
- * via the common execution planner. Lets the user pin
365
- * a dedicated model (e.g. `ministral-3b`) for consolidation instead of
366
- * whatever `defaults.llmEngine` happens to be.
367
- * 2. the baseline default LLM engine.
368
- *
369
- * All consolidate execution crosses the same improve engine-resolution
370
- * boundary as extract, reflect, and distill.
371
- */
372
- function resolveConsolidateLlmRunner(config, activeProfile) {
373
- return resolveImproveLlmExecution({
374
- config,
375
- profile: activeProfile,
376
- process: getImproveProcessConfig("consolidate", activeProfile),
377
- processName: "consolidate",
378
- });
379
- }
380
- function consolidateRunnerFromOptions(opts, config) {
381
- if (Object.hasOwn(opts, "llmRunner"))
382
- return opts.llmRunner ?? undefined;
383
- const resolved = resolveConsolidateLlmRunner(config, opts.improveProfile);
384
- if (resolved)
385
- opts.onNotices?.(resolved.notices);
386
- return resolved?.runner;
387
- }
388
- /**
389
- * Build a {@link ConsolidateResult} from partial overrides, filling the envelope
390
- * defaults (schemaVersion / ok / shape + the zeroed counters). Collapses the
391
- * ~7 near-identical result literals that previously appeared verbatim at every
392
- * early-return site and the final return of `akmConsolidateInner`. Callers pass
393
- * only the fields that differ from the all-zero, ok, non-preview baseline.
394
- */
395
285
  export function makeConsolidateResult(overrides) {
396
286
  return {
397
287
  schemaVersion: 1,
@@ -408,7 +298,6 @@ export function makeConsolidateResult(overrides) {
408
298
  ...overrides,
409
299
  };
410
300
  }
411
- // ── Main entry point ─────────────────────────────────────────────────────────
412
301
  function resolveConsolidationWriteTarget(opts, config) {
413
302
  if (opts.writeTarget) {
414
303
  const root = path.resolve(opts.writeTarget.source.path);
@@ -421,140 +310,78 @@ function resolveConsolidationWriteTarget(opts, config) {
421
310
  },
422
311
  };
423
312
  }
424
- if (opts.target) {
425
- const target = resolveWriteTarget(config, opts.target);
426
- return { ...target, source: { ...target.source, path: path.resolve(target.source.path) } };
427
- }
428
- if (opts.stashDir) {
313
+ if (!opts.target && opts.stashDir) {
429
314
  const root = path.resolve(opts.stashDir);
430
315
  return {
431
316
  source: { kind: "filesystem", name: "stash", path: root, adapterId: detectAdapterId(root) },
432
317
  config: { type: "filesystem", name: "stash", path: root, writable: true },
433
318
  };
434
319
  }
435
- const target = resolveWriteTarget(config);
320
+ const target = resolveWriteTarget(config, opts.target);
436
321
  return { ...target, source: { ...target.source, path: path.resolve(target.source.path) } };
437
322
  }
438
323
  export async function akmConsolidate(opts = {}) {
439
324
  const startMs = Date.now();
440
- // Derive a stable PROV-DM token for this run. Callers (e.g. akmImprove)
441
- // should pass opts.sourceRun to tie proposals back to the parent run;
442
- // standalone `akm consolidate` gets a self-contained token.
443
- const sourceRun = opts.sourceRun ?? `consolidate-${startMs}`;
444
325
  const config = opts.config ?? loadConfig();
445
326
  const writeTarget = resolveConsolidationWriteTarget(opts, config);
446
- opts = { ...opts, target: writeTarget.source.name, writeTarget };
447
- const activeProfile = opts.improveProfile ?? resolveImproveStrategy(undefined, config).config;
448
- opts = { ...opts, improveProfile: activeProfile };
327
+ const profile = opts.improveProfile ?? resolveImproveStrategy(undefined, config).config;
449
328
  const stashDir = writeTarget.source.path;
450
- const executionNotices = new Map();
451
- const externalOnNotices = opts.onNotices;
452
- const collectNotices = (notices) => {
453
- for (const notice of notices)
454
- executionNotices.set(JSON.stringify(notice), notice);
455
- externalOnNotices?.(notices);
329
+ const notices = noticeSet(opts.onNotices);
330
+ const enabled = resolveProcessEnabled("consolidate", profile);
331
+ const runner = enabled ? stageRunner(opts, config, profile, "consolidate", notices.add) : undefined;
332
+ opts = {
333
+ ...opts,
334
+ target: writeTarget.source.name,
335
+ writeTarget,
336
+ improveProfile: profile,
337
+ onNotices: notices.add,
338
+ sourceRun: opts.sourceRun ?? `consolidate-${startMs}`,
339
+ // Every later reader sees this one runner snapshot.
340
+ llmRunner: runner ?? null,
456
341
  };
457
- opts = { ...opts, onNotices: collectNotices };
458
- const consolidateEnabled = resolveProcessEnabled("consolidate", activeProfile);
459
- const frozenLlmRunner = consolidateEnabled ? consolidateRunnerFromOptions(opts, config) : undefined;
460
- // Own the field even when no runner exists. Every downstream reader now
461
- // observes this one symbolic snapshot instead of re-running config/model-map
462
- // selection during the same invocation.
463
- opts = { ...opts, llmRunner: frozenLlmRunner ?? null };
464
- const withNotices = (result) => executionNotices.size > 0 ? { ...result, notices: Object.freeze([...executionNotices.values()]) } : result;
465
- // WI-9.10: construct this run's RunContext from values already resolved
466
- // above (sourceRun, config, stashDir) — no second config load, no new db
467
- // handle. consolidate.ts has no `eventsCtx`/proposals-`ctx` option at all
468
- // (WS-3a retired its only appendEvent usage; `emitProposal` here is always
469
- // called with the default, seam-less ProposalsContext — see
470
- // emitPromotionProposal below), so both get the safe empty-object default,
471
- // behaviorally identical to `undefined` (EventsContext/ProposalsContext
472
- // fields are all optional-chained by their consumers). LLM work uses the
473
- // already-frozen symbolic runner through the shared dispatch seam.
474
- const runContext = createRunContext({
475
- stashDir,
476
- config,
477
- eventsCtx: {},
478
- proposalsCtx: {},
479
- getLlmRunner: () => opts.llmRunner ?? null,
480
- sourceRun,
481
- dryRun: opts.dryRun ?? false,
482
- signal: opts.signal,
483
- });
484
- const warnings = [];
485
- if (!consolidateEnabled) {
486
- return withNotices(makeConsolidateResult({
487
- // Sourced from runContext (identical value to `opts.dryRun ?? false`)
488
- // so the constructed RunContext has a genuine downstream reference —
489
- // consolidate's own content-read sites are out of this stage's stated
490
- // item-2 scope (reflect + distill only; see the WI-9.10c report).
491
- dryRun: runContext.dryRun,
492
- target: opts.target ?? stashDir,
493
- durationMs: Date.now() - startMs,
494
- warnings,
495
- }));
342
+ if (!enabled) {
343
+ const target = opts.target ?? stashDir;
344
+ return {
345
+ ...makeConsolidateResult({ dryRun: opts.dryRun ?? false, target, durationMs: Date.now() - startMs }),
346
+ ...notices.fields(),
347
+ };
496
348
  }
497
- // WS-3a: open one state.db handle shared by the body-embedding cache (dedup
498
- // + cluster) and the judged-state cache. All callers in the function body
499
- // receive this handle; it is closed in the `finally` block below.
500
- // Fail-open: any open error leaves it `undefined` and all cache paths skip.
501
- let sharedStateDb;
349
+ // One state.db handle for the embedding cache; unavailable means no cache.
350
+ let stateDb;
502
351
  if (config.embedding) {
503
352
  try {
504
- sharedStateDb = openStateDatabase();
353
+ stateDb = openStateDatabase();
505
354
  }
506
355
  catch {
507
- // State DB unavailable → skip the embedding cache for this run.
356
+ stateDb = undefined;
508
357
  }
509
358
  }
510
359
  try {
511
- return withNotices(await akmConsolidateInner(opts, config, stashDir, startMs, warnings, sharedStateDb));
360
+ return { ...(await consolidate(opts, config, stashDir, startMs, stateDb)), ...notices.fields() };
512
361
  }
513
362
  finally {
514
- sharedStateDb?.close();
363
+ stateDb?.close();
515
364
  }
516
365
  }
517
- /** Fresh, zeroed accounting accumulators for one consolidate run. */
518
- function createConsolidateAccounting() {
519
- const acc = {
520
- judgedNoAction: 0,
521
- failedChunkMemories: 0,
522
- totalChunksFailed: 0,
523
- skipReasons: [],
524
- skipReasonByRef: new Map(),
525
- judgedNoActionRefs: new Set(),
526
- pushSkipReason: () => { },
527
- };
528
- acc.pushSkipReason = (op, ref, reason) => {
529
- // 2026-05-27 cross-chunk double-count fix: if `ref` already contributed
530
- // to judgedNoAction in its own chunk (a different chunk proposed an op
531
- // for it that is now being rejected here), promote it from the
532
- // judgedNoAction bucket into the more specific skipReason bucket.
533
- // Preserves the invariant: processed == actioned + judgedNoAction +
534
- // Σ(skipReasons) + failedChunkMemories.
535
- if (acc.judgedNoActionRefs.delete(ref))
536
- acc.judgedNoAction--;
537
- const existing = acc.skipReasonByRef.get(ref);
538
- if (existing) {
539
- // Already counted once for accounting. Append the extra skip to the
540
- // ref's grouped entry for observability without adding a new array
541
- // entry (which would break the accounting invariant).
542
- existing.skips.push({ op, reason });
543
- return;
544
- }
545
- const entry = { ref, skips: [{ op, reason }] };
546
- acc.skipReasonByRef.set(ref, entry);
547
- acc.skipReasons.push(entry);
548
- };
549
- return acc;
366
+ function pushSkipReason(acc, op, ref, reason) {
367
+ if (acc.judgedNoActionRefs.delete(ref))
368
+ acc.judgedNoAction--;
369
+ const existing = acc.skipReasonByRef.get(ref);
370
+ if (existing) {
371
+ // One entry per ref keeps the invariant; the extra reason is kept for observability.
372
+ existing.skips.push({ op, reason });
373
+ return;
374
+ }
375
+ const entry = { ref, skips: [{ op, reason }] };
376
+ acc.skipReasonByRef.set(ref, entry);
377
+ acc.skipReasons.push(entry);
550
378
  }
551
379
  function resolveConsolidationSourceOwner(opts, stashDir) {
552
380
  const targetRoot = path.resolve(opts.writeTarget?.source.path ?? stashDir);
553
381
  try {
554
382
  const sources = resolveSourceEntries(stashDir, opts.config);
555
- const installations = deriveInstallations(sources);
556
383
  const targetIndex = sources.findIndex((source) => path.resolve(source.path) === targetRoot);
557
- const target = installations[targetIndex];
384
+ const target = deriveInstallations(sources)[targetIndex];
558
385
  if (!target)
559
386
  return undefined;
560
387
  return {
@@ -570,872 +397,511 @@ function resolveConsolidationSourceOwner(opts, stashDir) {
570
397
  return undefined;
571
398
  }
572
399
  }
400
+ const mtimeMsOf = (memory) => {
401
+ try {
402
+ return fs.statSync(memory.filePath).mtimeMs;
403
+ }
404
+ catch {
405
+ return 0;
406
+ }
407
+ };
573
408
  /**
574
- * Read and narrow the exact pool the live pass consumes, without embedding,
575
- * LLM, proposal, event, or asset writes. Used by both preview and execution.
409
+ * The exact pool the live pass consumes, with no embedding, model call or
410
+ * write: on-disk eligible memories, minus those the ledger holds, narrowed
411
+ * incrementally, minus bodies already in `knowledge/`, capped to `limit`
412
+ * (oldest-modified first). Shared by preview and execution.
576
413
  */
577
- export function inspectConsolidationPool(opts, stashDir, warnings, access) {
414
+ export function inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes = new Set(), access) {
578
415
  const readOnly = access?.readOnly === true;
579
- const sourceOwner = resolveConsolidationSourceOwner(opts, stashDir);
580
- let memories = loadMemoriesForSource(sourceOwner, warnings, readOnly);
416
+ let memories = loadMemoriesForSource(resolveConsolidationSourceOwner(opts, stashDir), warnings, readOnly);
581
417
  const staleCount = memories.filter((memory) => !fs.existsSync(memory.filePath)).length;
582
418
  if (staleCount > 0) {
583
419
  warnings.push(`Pre-flight: filtered ${staleCount} stale DB entr${staleCount === 1 ? "y" : "ies"} (file absent on disk) from memory pool before chunking.`);
584
420
  }
585
421
  memories = memories.filter((memory) => fs.existsSync(memory.filePath));
586
422
  const poolSize = memories.length;
587
- if (opts.incrementalSince && memories.length > 0) {
588
- memories = narrowToIncrementalCandidates(memories, opts.incrementalSince, warnings, opts.neighborsPerChanged, readOnly);
423
+ // A memory judged within its revisit window comes back once it is edited.
424
+ const ledger = loadLedgerSnapshot({ proposalsCtx: opts.proposalsCtx, readOnly }, stashDir, ["consolidate"]);
425
+ if (ledger.size > 0) {
426
+ const nowIso = new Date().toISOString();
427
+ memories = memories.filter((memory) => {
428
+ const row = ledger.get(ledgerKey("consolidate", conceptIdFromTypeName("memory", memory.name)));
429
+ let changedAt;
430
+ try {
431
+ changedAt = fs.statSync(memory.filePath).mtime.toISOString();
432
+ }
433
+ catch {
434
+ changedAt = undefined;
435
+ }
436
+ return !row || !isLedgerBlocked(row, nowIso, changedAt);
437
+ });
589
438
  }
439
+ const judgedUnchanged = poolSize - memories.length;
440
+ // Only what retrieval returned or new material improve never processed (#986).
441
+ const retrievalScope = loadRetrievalScope({ proposalsCtx: opts.proposalsCtx, readOnly }, stashDir);
442
+ const beforeScope = memories.length;
443
+ memories = memories.filter((memory) => isInRetrievalScope(retrievalScope, conceptIdFromTypeName("memory", memory.name), memory.filePath));
444
+ const outsideRetrievalScope = beforeScope - memories.length;
590
445
  const dedupPoolSize = memories.length;
591
446
  if (opts.limit === undefined && memories.length > 150) {
592
447
  warnings.push(`Consolidation: pool has ${memories.length} memories and no limit is set. Consider adding a limit to your consolidate config to prevent timeouts on slow LLM endpoints.`);
593
448
  }
594
- if (opts.limit !== undefined && memories.length > opts.limit) {
595
- const mtimeOf = (memory) => {
449
+ // Before the limit, so the cap picks from memories the run can act on.
450
+ let prefilteredAlreadyPromoted = 0;
451
+ if (existingKnowledgeBodyHashes.size > 0) {
452
+ memories = memories.filter((memory) => {
453
+ let raw;
596
454
  try {
597
- return fs.statSync(memory.filePath).mtimeMs;
455
+ raw = fs.readFileSync(memory.filePath, "utf8");
598
456
  }
599
457
  catch {
600
- return 0;
458
+ return true;
601
459
  }
602
- };
603
- const mtimeCache = new Map(memories.map((memory) => [memory.filePath, mtimeOf(memory)]));
604
- memories = [...memories].sort((a, b) => (mtimeCache.get(a.filePath) ?? 0) - (mtimeCache.get(b.filePath) ?? 0));
460
+ const duplicate = existingKnowledgeBodyHashes.has(contentHash(raw, "body"));
461
+ if (duplicate)
462
+ prefilteredAlreadyPromoted++;
463
+ return !duplicate;
464
+ });
465
+ }
466
+ if (opts.limit !== undefined && memories.length > opts.limit) {
467
+ const mtimes = new Map(memories.map((memory) => [memory.filePath, mtimeMsOf(memory)]));
468
+ memories = [...memories].sort((a, b) => (mtimes.get(a.filePath) ?? 0) - (mtimes.get(b.filePath) ?? 0));
605
469
  warnings.push(`Consolidation: pool capped at ${opts.limit} of ${memories.length} memories (limit option, oldest-modified first).`);
606
470
  memories = memories.slice(0, opts.limit);
607
471
  }
608
- return { poolSize, candidatePoolSize: memories.length, dedupPoolSize, memories };
609
- }
610
- /**
611
- * Pass 1 — narrow the memory pool before any LLM work: drop stale DB entries,
612
- * apply incremental-since narrowing, and cap to `opts.limit` (oldest-modified
613
- * first). Returns an early envelope when the pool empties at any stage;
614
- * otherwise returns the narrowed pool and the state the plan/apply passes
615
- * consume. Behavior-identical to the former inlined narrowing block.
616
- */
617
- async function narrowConsolidationPool(opts, stashDir, startMs, warnings) {
618
- const snapshot = inspectConsolidationPool(opts, stashDir, warnings);
619
- const memories = snapshot.memories;
620
- // (The former WS-3b Step 0a homeostatic demotion pass was removed — R4:
621
- // it was default-off and self-undoing (the next salience recompute
622
- // unconditionally overwrote the demoted values). Continuous decay now lives
623
- // in computeSalience's recency term, whose floor decays on a long half-life.)
624
- if (memories.length === 0) {
625
- return {
626
- done: true,
627
- result: makeConsolidateResult({
628
- dryRun: opts.dryRun ?? false,
629
- target: opts.target ?? stashDir,
630
- warnings,
631
- durationMs: Date.now() - startMs,
632
- }),
633
- };
634
- }
635
- return { done: false, memories, dedupPoolSize: snapshot.dedupPoolSize };
636
- }
637
- /**
638
- * Pass 2 — turn the narrowed pool into an executable plan. Sizes chunks to the
639
- * model context window, clusters by embedding similarity, injects the
640
- * anti-collapse random fraction, applies the cold-start budget cap, runs the
641
- * per-chunk LLM calls (with retry + failure-rate abort), and reconciles the
642
- * per-chunk op arrays via {@link mergePlans}. Populates `accounting` in place.
643
- * Behavior-identical to the former inlined plan-generation block.
644
- */
645
- /**
646
- * Per-chunk judgedNoAction accounting: count memories the LLM saw inside a chunk
647
- * but proposed no op for. Membership is by `memory:<name>` ref against the
648
- * targets of each op (primary + secondaries for merge; ref otherwise). 2026-05-26:
649
- * pre-fix this was a 78/119 (66%) silent drop in the cron run — no warning,
650
- * event, or counter. See tuning investigation §Q2. Moved verbatim.
651
- */
652
- function recordChunkJudgedNoAction(chunk, ops, accounting) {
653
- const targetRefs = new Set();
654
- for (const op of ops) {
655
- if (op.op === "merge") {
656
- targetRefs.add(op.primary);
657
- for (const s of op.secondaries)
658
- targetRefs.add(s);
659
- }
660
- else {
661
- targetRefs.add(op.ref);
662
- }
663
- }
664
- let chunkNoAction = 0;
665
- for (const m of chunk) {
666
- const memRef = conceptIdFromTypeName("memory", m.name);
667
- if (!targetRefs.has(memRef)) {
668
- chunkNoAction++;
669
- accounting.judgedNoActionRefs.add(memRef);
670
- }
671
- }
672
- accounting.judgedNoAction += chunkNoAction;
472
+ return {
473
+ poolSize,
474
+ candidatePoolSize: memories.length,
475
+ dedupPoolSize,
476
+ memories,
477
+ prefilteredAlreadyPromoted,
478
+ judgedUnchanged,
479
+ outsideRetrievalScope,
480
+ };
673
481
  }
482
+ const ABORT_MIN_CHUNKS = 4;
483
+ const ABORT_FAILURE_RATE = 0.5;
674
484
  /**
675
- * Per-chunk LLM judge loop — the heart of plan generation. Iterates the sized
676
- * chunks, applies the budget-abort/failure-rate/all-hot guards, calls the model
677
- * (with one retry), validates the returned ops, and accumulates the per-chunk
678
- * judgedNoAction accounting. Extracted verbatim from `planConsolidation`: the
679
- * abort-rate policy, all-hot early-exit, and the 2026-05-26 accounting invariant
680
- * (`processed == actioned + judgedNoAction + Σ(skipReasons) + failedChunkMemories`)
681
- * are byte-identical, and every counter-increment point is unmoved.
485
+ * The chunk loop: stop cleanly on the budget signal, abort once ≥50% of at
486
+ * least 4 chunks failed (the model is likely down), skip an all-hot chunk
487
+ * without a call (the only thing the model could do with it is refused), and
488
+ * count every memory into exactly one accounting bucket.
682
489
  */
683
490
  async function judgeConsolidationChunks(args) {
684
- const { chunks, opts, config, llmRunner, lease, sourceName, bodyTruncation, pendingProposalBodyHashes, standardsContext, warnings, accounting, } = args;
685
- const chunkOpsArrays = [];
686
- // judgedNoAction tracks memories the LLM saw inside a chunk but proposed
687
- // no op for. Computed per chunk as `chunk.length − unique(targetRefs in ops)`.
688
- // The structured skip-reason histogram (2026-05-26) plus the cross-chunk
689
- // double-count fixes now live on `accounting`; every deterministic post-LLM
690
- // op rejection site calls `accounting.pushSkipReason`. See
691
- // `/tmp/akm-health-investigations/tuning-reasons-investigation.md` §Q2.
692
- // C-6 / #392: Replace two-consecutive-failures abort with failure-rate threshold.
693
- // Consecutive-count policies are brittle against transient LM Studio reloads:
694
- // two transient failures abort the run even though the next chunk would succeed.
695
- // Rate-based abort (≥50% failure over ≥4 chunks) is more robust.
696
- // Tanenbaum, Distributed Systems §8 — rate-based policies with minimum sample sizes.
697
- let totalChunksProcessed = 0;
698
- const ABORT_MIN_CHUNKS = 4;
699
- const ABORT_FAILURE_RATE = 0.5;
491
+ const { chunks, opts, config, warnings, acc } = args;
492
+ const llmRunner = opts.llmRunner ?? undefined;
493
+ const memRef = (m) => conceptIdFromTypeName("memory", m.name);
494
+ const failChunk = (message, chunk) => {
495
+ warn(message);
496
+ warnings.push(message);
497
+ acc.totalChunksFailed++;
498
+ acc.failedChunkMemories += chunk.length;
499
+ };
500
+ const skipRemaining = (from) => {
501
+ for (let i = from; i < chunks.length; i++)
502
+ acc.failedChunkMemories += chunks[i].length;
503
+ };
504
+ const planned = [];
505
+ let processed = 0;
700
506
  for (let chunkIdx = 0; chunkIdx < chunks.length; chunkIdx++) {
701
- // Budget-signal check: break cleanly before the next LLM call if the
702
- // caller's budget has been exhausted. Commits work done so far.
507
+ const label = `chunk ${chunkIdx + 1}`;
703
508
  if (opts.signal?.aborted) {
704
- const skipped = chunks.length - chunkIdx;
705
- const msg = `[consolidate] budget signal aborted before chunk ${chunkIdx + 1}/${chunks.length}; ${skipped} chunk(s) not processed (partial_timeout — work done so far committed).`;
509
+ const msg = `[consolidate] budget signal aborted before chunk ${chunkIdx + 1}/${chunks.length}; ${chunks.length - chunkIdx} chunk(s) not processed (partial_timeout — work done so far committed).`;
706
510
  warn(msg);
707
511
  warnings.push(msg);
708
- // Account for memories in unprocessed chunks.
709
- for (let i = chunkIdx; i < chunks.length; i++) {
710
- accounting.failedChunkMemories += chunks[i].length;
711
- }
512
+ skipRemaining(chunkIdx);
712
513
  break;
713
514
  }
714
- // Abort if failure rate >= 50% over at least 4 processed chunks.
715
- if (totalChunksProcessed >= ABORT_MIN_CHUNKS) {
716
- const failureRate = accounting.totalChunksFailed / totalChunksProcessed;
515
+ if (processed >= ABORT_MIN_CHUNKS) {
516
+ const failureRate = acc.totalChunksFailed / processed;
717
517
  if (failureRate >= ABORT_FAILURE_RATE) {
718
- const skipped = chunks.length - chunkIdx;
719
- const abortMsg = `Consolidation aborted — failure rate ${(failureRate * 100).toFixed(0)}% over ${totalChunksProcessed} chunks (>= ${ABORT_FAILURE_RATE * 100}% threshold). LLM may be unavailable. ${skipped} chunk(s) skipped.`;
720
- warn(abortMsg);
721
- warnings.push(abortMsg);
722
- // Account for memories in chunks we never attempted: they are
723
- // neither judgedNoAction (no plan parsed) nor skipReason (no op
724
- // rejected). Without this, the accounting invariant fails by
725
- // `Σ(unattempted_chunk.length)` whenever the abort fires.
726
- for (let i = chunkIdx; i < chunks.length; i++) {
727
- accounting.failedChunkMemories += chunks[i].length;
728
- }
518
+ const msg = `Consolidation aborted — failure rate ${(failureRate * 100).toFixed(0)}% over ${processed} chunks (>= ${ABORT_FAILURE_RATE * 100}% threshold). LLM may be unavailable. ${chunks.length - chunkIdx} chunk(s) skipped.`;
519
+ warn(msg);
520
+ warnings.push(msg);
521
+ skipRemaining(chunkIdx);
729
522
  break;
730
523
  }
731
524
  }
732
525
  const chunk = chunks[chunkIdx];
733
- // All-hot chunk early-exit. The per-prompt hot-list block (see
734
- // buildChunkPrompt) only *discourages* delete proposals on a mixed chunk;
735
- // when EVERY memory in the chunk is captureMode: hot, the only ops the LLM
736
- // could ever propose are deletes — all of which the downstream guard
737
- // refuses unconditionally. Calling the model is therefore pure token waste.
738
- // Skip the request entirely and bucket every memory as judgedNoAction (we
739
- // judged "no action" without spending an LLM call), preserving the
740
- // accounting invariant `processed == actioned + judgedNoAction +
741
- // Σ(skipReasons) + failedChunkMemories`. Not counted toward the
742
- // LLM-failure-rate abort policy — no request was attempted.
743
526
  if (chunk.length > 0 && chunk.every((m) => isHotCapturedMemory(m.filePath))) {
744
- for (const m of chunk)
745
- accounting.judgedNoActionRefs.add(conceptIdFromTypeName("memory", m.name));
746
- accounting.judgedNoAction += chunk.length;
527
+ for (const m of chunk) {
528
+ acc.judgedNoActionRefs.add(memRef(m));
529
+ acc.judgedRefs.add(memRef(m));
530
+ }
531
+ acc.judgedNoAction += chunk.length;
747
532
  warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length}: all ${chunk.length} memories are captureMode: hot — skipping LLM (judged no-action).`);
748
533
  continue;
749
534
  }
750
535
  warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length} (${chunk.length} memories) …`);
751
- const userPrompt = buildChunkPrompt(sourceName, chunk, chunkIdx, chunks.length, bodyTruncation, pendingProposalBodyHashes, standardsContext);
752
- // Single chunk LLM call, wrapped in the feature gate. Deduplicated across
753
- // the first attempt and the retry below (the two blocks were byte-identical
754
- // apart from their fallback error string). responseSchema lift (PR 1,
755
- // asset-writers-investigation §5): providers with `supportsJsonSchema: true`
756
- // enforce the shape upstream; others fall through to
757
- // `parseEmbeddedJsonResponse` on the response side.
758
- const callChunkLlm = async (fallbackError) => {
759
- // The gate runs with enabled:true (always open), so this guard is
760
- // exactly the envelope the gated fn used to return first thing.
761
- if (!llmRunner)
762
- return { ok: false, error: "No LLM configured for consolidation" };
763
- return callStructured({
764
- feature: "memory_consolidation",
765
- akmConfig: config,
766
- enabled: true,
767
- runner: llmRunner,
768
- ...(lease ? { lease } : {}),
769
- messages: [
770
- { role: "system", content: CONSOLIDATE_SYSTEM_PROMPT },
771
- { role: "user", content: userPrompt },
772
- ],
773
- request: {
774
- responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
775
- enableThinking: false,
776
- timeoutMs: llmRunner.timeoutMs,
777
- signal: opts.signal,
778
- },
779
- parse: (raw) => ({ ok: true, content: raw ?? "" }),
780
- // A transport throw was caught INSIDE the gated fn and returned as an
781
- // {ok:false} envelope (never reaching the gate's fallback); onError
782
- // reproduces that. The fallback fires only on wrapper timeout.
783
- onError: (_cls, e) => ({ ok: false, error: String(e) }),
784
- fallback: { ok: false, error: fallbackError },
785
- ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
786
- });
787
- };
788
- // callChunkLlm already retries once internally (llm/client.ts's
789
- // chatCompletion, jittered 200-800ms backoff) — a second, outer retry
790
- // here stacked an uncoordinated fixed 2s backoff on top of it. Removed;
791
- // only mark the chunk failed once the single retry the client already
792
- // performs has been exhausted.
793
- const raw = await callChunkLlm(`chunk ${chunkIdx + 1} failed`);
794
- if (!raw.ok) {
795
- warn(raw.error ?? `chunk ${chunkIdx + 1} failed`);
796
- warnings.push(raw.error ?? `chunk ${chunkIdx + 1} failed`);
797
- totalChunksProcessed++;
798
- accounting.totalChunksFailed++;
799
- // Account for the chunk's memories under the failed-chunk bucket.
800
- // judgedNoAction does NOT run on this path (it's after the success
801
- // guards) so without this the accounting invariant breaks on every
802
- // chunk-level transport/parse failure.
803
- accounting.failedChunkMemories += chunk.length;
536
+ processed++;
537
+ if (!llmRunner) {
538
+ failChunk("No LLM configured for consolidation", chunk);
804
539
  continue;
805
540
  }
806
- // C9 action 1: AKM_DEBUG_LLM was a separate, undocumented env var for this
807
- // one diagnostic; folded into the standard AKM_VERBOSE gate (warnVerbose)
808
- // rather than kept as its own toggle.
809
- {
810
- const preview = (raw.content ?? "").slice(0, 500);
811
- warnVerbose(`[akm:consolidate] chunk ${chunkIdx + 1} raw response (first 500 chars): ${preview}`);
541
+ // The transport already retries once; a failed chunk is not retried here.
542
+ const outcome = await callStage({
543
+ feature: "memory_consolidation",
544
+ runner: llmRunner,
545
+ system: consolidateSystemPrompt,
546
+ prompt: buildChunkPrompt(args.sourceName, chunk, chunkIdx, chunks.length, args.bodyTruncation, args.pendingProposalBodyHashes),
547
+ gate: { config, enabled: true },
548
+ request: {
549
+ responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
550
+ enableThinking: false,
551
+ timeoutMs: llmRunner.timeoutMs,
552
+ signal: opts.signal,
553
+ },
554
+ ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
555
+ });
556
+ if (!outcome.ok) {
557
+ failChunk(outcome.reason === "error" && outcome.error ? outcome.error : `${label} failed`, chunk);
558
+ continue;
812
559
  }
813
- const parsed = parseEmbeddedJsonResponse(raw.content);
560
+ warnVerbose(`[akm:consolidate] ${label} raw response (first 500 chars): ${outcome.raw.slice(0, 500)}`);
561
+ const parsed = parseEmbeddedJsonResponse(outcome.raw);
814
562
  if (!parsed || !Array.isArray(parsed.operations)) {
815
- const hint = raw.content !== undefined && raw.content.trim() === ""
816
- ? " (empty response — if using a thinking model, disable thinking mode)"
817
- : "";
818
- warn(`Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`);
819
- warnings.push(`Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`);
820
- totalChunksProcessed++;
821
- accounting.totalChunksFailed++;
822
- accounting.failedChunkMemories += chunk.length;
563
+ const hint = outcome.raw.trim() === "" ? " (empty response — if using a thinking model, disable thinking mode)" : "";
564
+ const msg = `Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`;
565
+ warn(msg);
566
+ warnings.push(msg);
567
+ acc.totalChunksFailed++;
568
+ acc.failedChunkMemories += chunk.length;
823
569
  continue;
824
570
  }
825
- totalChunksProcessed++; // success
826
571
  const ops = [];
827
572
  for (const op of parsed.operations) {
828
- if (isValidOp(op)) {
573
+ if (isValidOp(op))
829
574
  ops.push(op);
830
- }
831
- else {
575
+ else
832
576
  warnings.push(`Chunk ${chunkIdx + 1}: skipping invalid operation: ${JSON.stringify(op)}`);
833
- }
834
577
  }
835
- if (Array.isArray(parsed.warnings)) {
836
- for (const w of parsed.warnings) {
837
- if (typeof w === "string")
838
- warnings.push(w);
839
- }
578
+ for (const w of Array.isArray(parsed.warnings) ? parsed.warnings : [])
579
+ if (typeof w === "string")
580
+ warnings.push(w);
581
+ // Memories the model saw but proposed nothing for.
582
+ const targeted = new Set(ops.map((op) => op.ref));
583
+ for (const m of chunk) {
584
+ acc.judgedRefs.add(memRef(m));
585
+ if (targeted.has(memRef(m)))
586
+ continue;
587
+ acc.judgedNoAction++;
588
+ acc.judgedNoActionRefs.add(memRef(m));
840
589
  }
841
- recordChunkJudgedNoAction(chunk, ops, accounting);
842
- chunkOpsArrays.push(ops);
590
+ planned.push(ops);
843
591
  }
844
- return chunkOpsArrays;
592
+ return planned;
845
593
  }
846
- async function planConsolidation(opts, config, stashDir, _startMs, memories, warnings, sharedStateDb, accounting) {
847
- // Consolidation always uses the HTTP LLM client directly — never the agent
848
- // CLI. The agent CLI is for interactive agent sessions (reflect, propose);
849
- // structured JSON generation works better and faster via HTTP.
850
- //
851
- // The outer invocation freezes standalone and improve-owned selection once.
594
+ /**
595
+ * The model's plan for the narrowed pool: chunk size from the context window,
596
+ * an up-front cap when the remaining budget cannot cover every chunk (oldest
597
+ * first, the rest deferred), similarity clustering, anti-collapse, then the
598
+ * chunk loop.
599
+ */
600
+ async function planConsolidation(opts, config, stashDir, memories, warnings, stateDb, acc) {
852
601
  const llmRunner = opts.llmRunner ?? undefined;
853
- // Chunk sizing: derive a safe chunk size from the configured model context
854
- // window so that the full prompt (system prompt + chunk user prompt) never
855
- // exceeds the model's n_ctx limit. When no context length is configured we
856
- // fall back to DEFAULT_CONTEXT_LENGTH_TOKENS (8 000) which is conservative
857
- // enough for most 8K–16K local models.
858
- //
859
- // bodyTruncation caps the body excerpt included per memory in the prompt.
860
- // Reducing it further than 500 chars degrades consolidation quality, so we
861
- // keep it fixed and let computeSafeChunkSize vary the number of memories
862
- // per chunk instead.
602
+ // 500 body chars per memory keep the judgement useful; chunk size varies instead.
863
603
  const bodyTruncation = 500;
864
- const modelContextLength = llmRunner?.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS;
865
- const chunkSize = computeSafeChunkSize(modelContextLength, bodyTruncation, opts.maxChunkSize);
866
- // -- Phase A: plan generation -----------------------------------------------
604
+ const chunkSize = computeSafeChunkSize(llmRunner?.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
867
605
  const sourceName = opts.target ?? stashDir;
868
- let budgetedMemories = memories;
869
- if (opts.signal) {
870
- const budgetMs = opts.signal.remainingBudgetMs;
871
- if (budgetMs !== undefined) {
872
- const p90Chunk = opts.p90ChunkSecondsDefault ?? 30;
873
- const safeChunks = Math.max(0, Math.floor((Math.max(0, budgetMs) / 1000 / p90Chunk) * 0.6));
874
- const cap = safeChunks * chunkSize;
875
- if (cap < memories.length) {
876
- budgetedMemories = memories
877
- .map((entry) => {
878
- let mtimeMs = 0;
879
- try {
880
- mtimeMs = fs.statSync(entry.filePath).mtimeMs;
881
- }
882
- catch {
883
- // Missing files sort first and are filtered by the existing guards.
884
- }
885
- return { entry, mtimeMs };
886
- })
887
- .sort((a, b) => a.mtimeMs - b.mtimeMs || a.entry.name.localeCompare(b.entry.name))
888
- .map(({ entry }) => entry)
889
- .slice(0, cap);
890
- const msg = `[consolidate] cold-start budget: reducing pool from ${memories.length} to ${budgetedMemories.length} memories (${safeChunks} safe chunks; remainder deferred).`;
891
- warn(msg);
892
- warnings.push(msg);
893
- }
894
- }
895
- }
896
- // WS-5: capture llmPoolSize after every pre-LLM cap.
897
- const llmPoolSize = budgetedMemories.length;
898
- const dispatchingChunks = [];
899
- for (let i = 0; i < budgetedMemories.length; i += chunkSize) {
900
- const chunk = budgetedMemories.slice(i, i + chunkSize);
901
- if (chunk.length > 0 && !chunk.every((memory) => isHotCapturedMemory(memory.filePath))) {
902
- dispatchingChunks.push(chunk);
606
+ let budgeted = memories;
607
+ const budgetMs = opts.signal?.remainingBudgetMs;
608
+ if (opts.signal && budgetMs !== undefined) {
609
+ const safeChunks = Math.max(0, Math.floor((Math.max(0, budgetMs) / 1000 / (opts.p90ChunkSecondsDefault ?? 30)) * 0.6));
610
+ if (safeChunks * chunkSize < memories.length) {
611
+ budgeted = [...memories]
612
+ .sort((a, b) => mtimeMsOf(a) - mtimeMsOf(b) || a.name.localeCompare(b.name))
613
+ .slice(0, safeChunks * chunkSize);
614
+ const msg = `[consolidate] cold-start budget: reducing pool from ${memories.length} to ${budgeted.length} memories (${safeChunks} safe chunks; remainder deferred).`;
615
+ warn(msg);
616
+ warnings.push(msg);
903
617
  }
904
618
  }
905
- const dispatchLease = llmRunner && dispatchingChunks.length > 0 ? await preflightStructuredLlmRunner(llmRunner) : undefined;
906
- try {
907
- // C-1 / #380: Pre-cluster memories by embedding similarity before chunking.
908
- // This ensures that semantically similar memories land in the same LLM
909
- // context window, allowing the model to detect and merge duplicates that
910
- // would otherwise be split across chunks and survive indefinitely.
911
- // mem0 arXiv:2504.19413, A-MEM arXiv:2502.12110.
912
- // Fails open: if embeddings are unavailable or fail, original order is used.
913
- const { ordered: clusteredMemories, embedTelemetry } = await clusterMemoriesBySimilarity(budgetedMemories, config, sharedStateDb, opts.signal);
914
- // WS-3b Anti-collapse step 8c: inject random (non-similar) clusters.
915
- // A small fraction (default 5%) of the pool is shuffled into random positions
916
- // so the pipeline isn't PURELY similarity-driven. This prevents rich-get-richer
917
- // entrenchment where only the most-retrieved assets ever get consolidated.
918
- // DEFAULT ON since R5 — opt out via antiCollapse.enabled: false.
919
- let finalClusteredMemories = clusteredMemories;
920
- {
921
- const antiCollapseForCluster = getImproveProcessConfig("consolidate", opts.improveProfile)?.antiCollapse ??
922
- {};
923
- if (antiCollapseForCluster.enabled !== false && clusteredMemories.length > 2) {
924
- const fraction = antiCollapseForCluster.randomClusterFraction ?? DEFAULT_RANDOM_CLUSTER_FRACTION;
925
- const randomCount = Math.max(1, Math.floor(clusteredMemories.length * fraction));
926
- // Pick `randomCount` positions to inject random (un-clustered) members.
927
- // Use a seeded-ish shuffle: sort by hash of the name so it's deterministic
928
- // per run but not strictly similarity-driven.
929
- const shuffled = [...clusteredMemories].sort((a, b) => {
930
- // Deterministic shuffle: compare sha256-ish (use name hash as proxy).
931
- const ha = a.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
932
- const hb = b.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
933
- return ha - hb;
934
- });
935
- const randomSlice = shuffled.slice(0, randomCount);
936
- const randomSet = new Set(randomSlice.map((m) => m.name));
937
- // Insert random members at intervals through the clustered sequence.
938
- const withRandom = [];
939
- const interval = Math.max(2, Math.floor(clusteredMemories.length / randomCount));
940
- let randomIdx = 0;
941
- for (let i = 0; i < clusteredMemories.length; i++) {
942
- const m = clusteredMemories[i];
943
- if (m && !randomSet.has(m.name))
944
- withRandom.push(m);
945
- if (i > 0 && i % interval === 0 && randomIdx < randomSlice.length) {
946
- const r = randomSlice[randomIdx++];
947
- if (r)
948
- withRandom.push(r);
949
- }
950
- }
951
- // Append any remaining random members not yet inserted.
952
- while (randomIdx < randomSlice.length) {
953
- const r = randomSlice[randomIdx++];
954
- if (r)
955
- withRandom.push(r);
956
- }
957
- finalClusteredMemories = withRandom;
958
- warnings.push(`Anti-collapse: injected ${randomCount} random (non-similarity-driven) cluster member(s) into consolidation pool (fraction=${fraction}).`);
959
- }
960
- }
961
- const chunks = [];
962
- for (let i = 0; i < finalClusteredMemories.length; i += chunkSize) {
963
- chunks.push(finalClusteredMemories.slice(i, i + chunkSize));
964
- }
965
- // 2026-05-27 prompt-context fix: precompute body-hashes of pending
966
- // consolidate proposals once, so the per-chunk prompt can annotate
967
- // memories whose body would just produce a deterministic
968
- // `dedup_pending_proposal` skip. Cuts ~110 wasted LLM proposals per
969
- // 4h on this user's stack. See
970
- // /tmp/akm-health-investigations/tuning-reasons-investigation.md §Q3.
971
- const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
972
- warn(`[consolidate] ${budgetedMemories.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
973
- ` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
974
- // Consolidate output merges memories (non-wiki) → stash authoring standards.
975
- // Resolved ONCE per run and passed to each chunk prompt (facts not re-read
976
- // per chunk).
977
- const standardsContext = resolveStandardsContext("memories/_consolidated", stashDir);
978
- const chunkOpsArrays = await judgeConsolidationChunks({
979
- chunks,
980
- opts,
981
- config,
982
- llmRunner,
983
- lease: dispatchLease,
984
- sourceName,
985
- bodyTruncation,
986
- pendingProposalBodyHashes,
987
- standardsContext,
988
- warnings,
989
- accounting,
990
- });
991
- // Build the known-refs set from the already-filtered memory pool so
992
- // mergePlans() can reject LLM-hallucinated primary refs before execution.
993
- const knownRefs = new Set(budgetedMemories.map((m) => conceptIdFromTypeName("memory", m.name)));
994
- const { ops: allOps, warnings: mergeWarnings } = mergePlans(chunkOpsArrays, knownRefs);
995
- warnings.push(...mergeWarnings);
996
- return {
997
- allOps,
998
- totalChunks: chunks.length,
999
- llmPoolSize,
1000
- deferredMemories: memories.length - budgetedMemories.length,
1001
- embedTelemetry,
1002
- sourceName,
1003
- };
1004
- }
1005
- finally {
1006
- if (dispatchLease)
1007
- disposeLoweredExecutionDispatchLease(dispatchLease);
1008
- }
619
+ const slice = (list) => Array.from({ length: Math.ceil(list.length / chunkSize) }, (_, i) => list.slice(i * chunkSize, (i + 1) * chunkSize));
620
+ const willDispatch = slice(budgeted).some((chunk) => chunk.length > 0 && !chunk.every((memory) => isHotCapturedMemory(memory.filePath)));
621
+ if (llmRunner && willDispatch)
622
+ assertRunnerCredentials(llmRunner);
623
+ const { ordered, embedTelemetry } = await clusterMemoriesBySimilarity(budgeted, config, stateDb, opts.signal);
624
+ const chunks = slice(injectRandomClusterMembers(ordered, opts.improveProfile, warnings));
625
+ const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
626
+ warn(`[consolidate] ${budgeted.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
627
+ ` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
628
+ const planned = await judgeConsolidationChunks({
629
+ chunks,
630
+ opts,
631
+ config,
632
+ sourceName,
633
+ bodyTruncation,
634
+ pendingProposalBodyHashes,
635
+ warnings,
636
+ acc,
637
+ });
638
+ return {
639
+ allOps: mergePlans(planned),
640
+ totalChunks: chunks.length,
641
+ llmPoolSize: budgeted.length,
642
+ deferredMemories: memories.length - budgeted.length,
643
+ embedTelemetry,
644
+ sourceName,
645
+ };
1009
646
  }
1010
- async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sharedStateDb) {
1011
- // -- Pass 1: narrow the memory pool (may early-return an envelope) ----------
1012
- const narrowed = await narrowConsolidationPool(opts, stashDir, startMs, warnings);
1013
- if (narrowed.done)
1014
- return narrowed.result;
1015
- const { memories, dedupPoolSize } = narrowed;
1016
- // -- Pass 2: build the LLM plan (populates the shared accounting counters) ---
1017
- const accounting = createConsolidateAccounting();
1018
- const { allOps, totalChunks, llmPoolSize, deferredMemories, embedTelemetry, sourceName } = await planConsolidation(opts, config, stashDir, startMs, memories, warnings, sharedStateDb, accounting);
1019
- // -- Dry-run: show AI plan without executing any writes --------------------
1020
- if (opts.dryRun) {
647
+ async function consolidate(opts, config, stashDir, startMs, stateDb) {
648
+ const warnings = [];
649
+ const existingKnowledgeBodyHashes = opts.existingKnowledgeBodyHashes ?? loadExistingKnowledgeBodyHashes(stashDir);
650
+ const pool = inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes);
651
+ const { memories, prefilteredAlreadyPromoted } = pool;
652
+ const plural = (n) => `memor${n === 1 ? "y" : "ies"}`;
653
+ if (pool.judgedUnchanged > 0) {
654
+ warnings.push(`Consolidation: skipped ${pool.judgedUnchanged} ${plural(pool.judgedUnchanged)} judged within the revisit window and unchanged since.`);
655
+ }
656
+ if (pool.outsideRetrievalScope > 0) {
657
+ warnings.push(`Consolidation: skipped ${pool.outsideRetrievalScope} ${plural(pool.outsideRetrievalScope)} already judged that retrieval has not returned in the last ${USAGE_EVENT_RETENTION_DAYS} days.`);
658
+ }
659
+ if (prefilteredAlreadyPromoted > 0) {
660
+ warnings.push(`Consolidation: pre-filtered ${prefilteredAlreadyPromoted} ${plural(prefilteredAlreadyPromoted)} whose body already exists verbatim in knowledge/ before chunking.`);
661
+ }
662
+ const target = opts.target ?? stashDir;
663
+ // The pair pass (alpha.9) has its own initiator/candidate selection (it
664
+ // sees .derived memories, flat knowledge and lessons, not just the
665
+ // promote pool above), so it runs regardless of whether the promote pool
666
+ // is empty — every return path below carries its result.
667
+ const pairPassBundleId = resolveConsolidationSourceOwner(opts, stashDir)?.bundleId;
668
+ const pairPass = await runConsolidatePairPass(opts, config, stashDir, pairPassBundleId, warnings);
669
+ if (memories.length === 0) {
1021
670
  return makeConsolidateResult({
1022
- dryRun: true,
1023
- previewOnly: true,
1024
- target: sourceName,
1025
- processed: llmPoolSize,
1026
- failedChunks: accounting.totalChunksFailed,
1027
- totalChunks,
1028
- judgedNoAction: accounting.judgedNoAction,
1029
- skipReasons: accounting.skipReasons,
1030
- // No merge has executed on the preview path — the per-secondary tally is
1031
- // provably still 0 here (it only increments in the op-execution loop).
1032
- mergedSecondaries: 0,
1033
- failedChunkMemories: accounting.failedChunkMemories,
1034
- deferredMemories,
1035
- planned: allOps,
671
+ dryRun: opts.dryRun ?? false,
672
+ target,
1036
673
  warnings,
1037
674
  durationMs: Date.now() - startMs,
675
+ prefilteredAlreadyPromoted,
676
+ pairPass,
1038
677
  });
1039
678
  }
1040
- warn(`[consolidate] plan: ${allOps.length} operation(s)`);
1041
- // Destructive operations remain advisory. Promote is safe to execute because
1042
- // it emits a reviewable proposal rather than mutating an asset.
1043
- const promoted = [];
1044
- const promotionFailures = { count: 0 };
1045
- const memoryByRef = new Map(memories.map((memory) => [conceptIdFromTypeName("memory", memory.name), memory]));
1046
- const promoteContext = {
679
+ const acc = {
680
+ judgedNoAction: 0,
681
+ failedChunkMemories: 0,
682
+ totalChunksFailed: 0,
683
+ skipReasons: [],
684
+ skipReasonByRef: new Map(),
685
+ judgedNoActionRefs: new Set(),
686
+ judgedRefs: new Set(),
687
+ };
688
+ const plan = await planConsolidation(opts, config, stashDir, memories, warnings, stateDb, acc);
689
+ // Evaluated at return time: a promotion skip can move a ref out of judgedNoAction.
690
+ const summary = () => ({
691
+ target: plan.sourceName,
692
+ processed: plan.llmPoolSize,
693
+ failedChunks: acc.totalChunksFailed,
694
+ totalChunks: plan.totalChunks,
695
+ judgedNoAction: acc.judgedNoAction,
696
+ skipReasons: acc.skipReasons,
697
+ mergedSecondaries: 0,
698
+ failedChunkMemories: acc.failedChunkMemories,
699
+ deferredMemories: plan.deferredMemories,
700
+ planned: plan.allOps,
701
+ warnings,
702
+ prefilteredAlreadyPromoted,
703
+ durationMs: Date.now() - startMs,
704
+ });
705
+ if (opts.dryRun)
706
+ return makeConsolidateResult({ ...summary(), dryRun: true, previewOnly: true, pairPass });
707
+ warn(`[consolidate] plan: ${plan.allOps.length} operation(s)`);
708
+ const ctx = {
1047
709
  config,
1048
710
  stashDir,
1049
711
  sourceRun: opts.sourceRun ?? `consolidate-${startMs}`,
1050
712
  proposalsCtx: opts.proposalsCtx,
1051
713
  target: opts.writeTarget,
1052
- memoryByRef,
1053
- promoted,
714
+ memoryByRef: new Map(memories.map((memory) => [conceptIdFromTypeName("memory", memory.name), memory])),
715
+ promoted: [],
1054
716
  promotedSourceRefs: new Set(),
1055
- existingKnowledgeBodyHashes: loadExistingKnowledgeBodyHashes(opts.writeTarget.source.path),
1056
- promotionFailures,
717
+ existingKnowledgeBodyHashes,
718
+ promotionFailures: { count: 0 },
1057
719
  warnings,
1058
- pushSkipReason: accounting.pushSkipReason,
1059
- llmRunner: opts.llmRunner ?? null,
720
+ pushSkipReason: (op, ref, reason) => pushSkipReason(acc, op, ref, reason),
1060
721
  };
1061
- for (const op of allOps) {
1062
- if (op.op === "promote")
1063
- await emitPromotionProposal(op, promoteContext);
1064
- }
722
+ for (const op of plan.allOps)
723
+ await emitPromotionProposal(op, ctx);
724
+ // Every other judged memory waits out its revisit window (or its next edit);
725
+ // a promotion that failed to persist is retried next run.
726
+ recordLedgerAttempt({ proposalsCtx: opts.proposalsCtx }, [...acc.judgedRefs]
727
+ .filter((ref) => !ctx.promotedSourceRefs.has(ref) &&
728
+ !acc.skipReasonByRef.get(ref)?.skips.some((skip) => skip.reason === "promote_create_failed"))
729
+ .map((ref) => ({ stashDir, ref, source: "consolidate", outcome: "judged_no_action" })));
1065
730
  return makeConsolidateResult({
1066
- target: sourceName,
1067
- processed: llmPoolSize,
1068
- failedChunks: accounting.totalChunksFailed,
1069
- totalChunks,
1070
- judgedNoAction: accounting.judgedNoAction,
1071
- skipReasons: accounting.skipReasons,
1072
- mergedSecondaries: 0,
1073
- failedChunkMemories: accounting.failedChunkMemories,
1074
- deferredMemories,
1075
- promoted,
1076
- failedPromotions: promotionFailures.count,
1077
- planned: allOps,
1078
- warnings,
1079
- durationMs: Date.now() - startMs,
731
+ ...summary(),
732
+ promoted: ctx.promoted,
733
+ failedPromotions: ctx.promotionFailures.count,
734
+ pairPass,
1080
735
  perfTelemetry: {
1081
- dedupPoolSize,
1082
- llmPoolSize,
1083
- embedMs: embedTelemetry.embedMs,
1084
- embedCacheHits: embedTelemetry.cacheHits,
1085
- embedCacheMisses: embedTelemetry.cacheMisses,
736
+ dedupPoolSize: pool.dedupPoolSize,
737
+ llmPoolSize: plan.llmPoolSize,
738
+ embedMs: plan.embedTelemetry.embedMs,
739
+ embedCacheHits: plan.embedTelemetry.cacheHits,
740
+ embedCacheMisses: plan.embedTelemetry.cacheMisses,
1086
741
  },
1087
742
  });
1088
743
  }
1089
- /** Reject a promotion when its body already exists in knowledge or the queue. */
1090
- function shouldSkipPromotionBodyDuplicate(args) {
1091
- const { bodyHash, op, knowledgeRef, ctx } = args;
1092
- if (ctx.existingKnowledgeBodyHashes.has(bodyHash)) {
1093
- ctx.warnings.push(`Skipping promote: identical body already exists in knowledge; skipping duplicate for ${op.ref} → ${knowledgeRef}`);
1094
- ctx.pushSkipReason("promote", op.ref, "dedup_existing_knowledge");
1095
- return true;
744
+ /** The conceptId a ref maps to, or undefined for an invalid ref. */
745
+ function conceptIdForRef(ref) {
746
+ try {
747
+ const p = parseRefInput(ref);
748
+ return conceptIdFromTypeName(p.type, p.name);
749
+ }
750
+ catch {
751
+ return undefined;
1096
752
  }
1097
- const contentDupProposal = listProposals(ctx.stashDir, { status: "pending" })
1098
- .filter((proposal) => proposal.source === "consolidate")
1099
- .find((proposal) => cacheHash(proposalContent(proposal)) === bodyHash);
1100
- if (!contentDupProposal)
1101
- return false;
1102
- ctx.warnings.push(`Skipping promote: identical body already pending as proposal ${contentDupProposal.id} (ref: ${contentDupProposal.ref}); skipping duplicate for ${op.ref} → ${knowledgeRef}`);
1103
- ctx.pushSkipReason("promote", op.ref, "dedup_pending_proposal");
1104
- return true;
1105
753
  }
1106
- /** Execute one reconciled promotion by emitting a reviewable proposal. */
1107
- /** @internal Executes the real proposal-emission path for one promote operation. */
754
+ /** A slug with dates, counters and word order folded away, for spotting variants. */
755
+ function normalizeSlugForDedup(ref) {
756
+ const monthRe = /(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)/i;
757
+ return parseRefInput(ref)
758
+ .name.toLowerCase()
759
+ .split("-")
760
+ .filter((tok) => tok.length > 0 && !/^\d+$/.test(tok) && !monthRe.test(tok))
761
+ .sort()
762
+ .join("-");
763
+ }
764
+ const PROMOTE_BODY_MIN_CHARS = 100;
765
+ /**
766
+ * Queue one promotion as a proposal. Refused (with a skip reason) when the
767
+ * memory is unknown, already promoted this run, already pending or present
768
+ * as knowledge (by concept, body hash or slug variant), unreadable, fails
769
+ * sanitization, is superseded, has a body too small to be knowledge, or has
770
+ * no valid description.
771
+ * @internal Exported for promotion-path integration tests.
772
+ */
1108
773
  export async function emitPromotionProposal(op, ctx) {
1109
- const { config, stashDir, sourceRun, target, memoryByRef, warnings, pushSkipReason, promoted, promotedSourceRefs } = ctx;
1110
- const entry = memoryByRef.get(op.ref);
774
+ const { stashDir, target, warnings, pushSkipReason } = ctx;
775
+ const entry = ctx.memoryByRef.get(op.ref);
1111
776
  if (!entry) {
777
+ // A phantom ref was never counted as processed, so it gets no skip reason.
1112
778
  warnings.push(`Promote: ${op.ref} not found in loaded memories — skipping.`);
1113
- // Phantom ref: not in processed, so no skipReason (same rationale as
1114
- // delete_ref_missing above).
1115
779
  return;
1116
780
  }
1117
- // Within-run source-ref dedup: skip if this source memory was already
1118
- // promoted earlier in this run (safety belt — mergePlans already
1119
- // deduplicates promote ops by source ref via Map, but this guard also
1120
- // catches any future code paths that bypass mergePlans).
1121
- if (promotedSourceRefs.has(op.ref)) {
1122
- warnings.push(`Skipping promote: ${op.ref} already promoted in this run`);
1123
- pushSkipReason("promote", op.ref, "promote_already_promoted_this_run");
1124
- return;
781
+ const skip = (reason, message) => {
782
+ warnings.push(message);
783
+ pushSkipReason("promote", op.ref, reason);
784
+ };
785
+ if (ctx.promotedSourceRefs.has(op.ref)) {
786
+ return skip("promote_already_promoted_this_run", `Skipping promote: ${op.ref} already promoted in this run`);
1125
787
  }
1126
- const proposedName = op.knowledgeRef.split("/").filter(Boolean).at(-1) ??
788
+ const slug = (op.knowledgeRef.split("/").filter(Boolean).at(-1) ??
1127
789
  entry.name.split("/").filter(Boolean).at(-1) ??
1128
- "promoted-memory";
1129
- const slug = proposedName
790
+ "promoted-memory")
1130
791
  .replace(/[^a-z0-9-]/gi, "-")
1131
792
  .replace(/-+/g, "-")
1132
793
  .replace(/^-|-$/g, "")
1133
794
  .toLowerCase();
1134
795
  const knowledgeRef = conceptIdFromTypeName("knowledge", slug);
1135
- parseRefInput(knowledgeRef);
1136
- if (knowledgeRef !== op.knowledgeRef) {
796
+ const parsedKnowledgeRef = parseRefInput(knowledgeRef);
797
+ if (knowledgeRef !== op.knowledgeRef)
1137
798
  warnings.push(`Normalized generated ref "${op.knowledgeRef}" → "${knowledgeRef}"`);
799
+ const pending = listProposals(stashDir, { status: "pending" });
800
+ const wantConcept = conceptIdForRef(knowledgeRef);
801
+ if (wantConcept !== undefined && pending.some((p) => conceptIdForRef(p.ref) === wantConcept)) {
802
+ return skip("promote_pending_proposal_exists", `Skipping promote: pending proposal already exists for ${knowledgeRef}`);
1138
803
  }
1139
- // A pending proposal may carry a qualified item_ref, so compare its parsed
1140
- // conceptId rather than exact display spelling.
1141
- if (hasPendingProposalForConcept(stashDir, knowledgeRef)) {
1142
- warnings.push(`Skipping promote: pending proposal already exists for ${knowledgeRef}`);
1143
- pushSkipReason("promote", op.ref, "promote_pending_proposal_exists");
1144
- return;
1145
- }
1146
- // Idempotency: check if knowledge asset already exists
1147
- const parsedKnowledgeRef = parseRefInput(knowledgeRef);
1148
- const destPath = path.join(target.source.path, "knowledge", `${parsedKnowledgeRef.name}.md`);
1149
- if (fs.existsSync(destPath)) {
1150
- warnings.push(`Skipping promote: ${knowledgeRef} already exists in source`);
1151
- pushSkipReason("promote", op.ref, "promote_already_exists");
1152
- return;
804
+ if (fs.existsSync(path.join(target.source.path, "knowledge", `${parsedKnowledgeRef.name}.md`))) {
805
+ return skip("promote_already_exists", `Skipping promote: ${knowledgeRef} already exists in source`);
1153
806
  }
1154
- let memoryContent = "";
807
+ let memoryContent;
1155
808
  try {
1156
809
  memoryContent = fs.readFileSync(entry.filePath, "utf8");
1157
810
  }
1158
811
  catch (e) {
1159
- warnings.push(`Promote: could not read ${op.ref}: ${String(e)}`);
1160
- pushSkipReason("promote", op.ref, "promote_read_failed");
1161
- return;
812
+ return skip("promote_read_failed", `Promote: could not read ${op.ref}: ${String(e)}`);
813
+ }
814
+ // O1 hash nit: the RAW body, before sanitization — accept re-reads the
815
+ // source with a plain fs.readFileSync and never re-sanitizes, so hashing
816
+ // anything else here would compare two different representations of the
817
+ // same unedited memory and report a false "changed since mint" (measured:
818
+ // 3 of 767 real memories sanitize to different bytes than their raw body).
819
+ const sourceRawBodyHash = contentHash(memoryContent, "body");
820
+ const sanitized = sanitizeMergedContent(memoryContent);
821
+ if (!sanitized.ok) {
822
+ return skip("promote_sanitization_failed", `Promote: rejected ${op.ref} — source memory failed sanitization (${sanitized.reason}).`);
823
+ }
824
+ memoryContent = sanitized.result.content;
825
+ if (hasSupersededStatus(sanitized.result.frontmatter)) {
826
+ return skip("promote_superseded", `Promote: refused for ${op.ref} → ${knowledgeRef} — source memory has status:superseded; superseded memories are not promotable knowledge.`);
1162
827
  }
1163
- // Validate and normalize source content before proposing a promoted asset.
1164
- const promoteSanitized = sanitizeMergedContent(memoryContent);
1165
- if (!promoteSanitized.ok) {
1166
- warnings.push(`Promote: rejected ${op.ref} — source memory failed sanitization (${promoteSanitized.reason}).`);
1167
- pushSkipReason("promote", op.ref, "promote_sanitization_failed");
1168
- return;
1169
- }
1170
- memoryContent = promoteSanitized.result.content;
1171
- // SOURCE_SUPERSEDED guard: refuse to promote a memory whose source
1172
- // frontmatter carries `status: superseded`. Predicate at module top
1173
- // (`hasSupersededStatus`) so tests can exercise it directly.
1174
- if (hasSupersededStatus(promoteSanitized.result.frontmatter)) {
1175
- warnings.push(`Promote: refused for ${op.ref} → ${knowledgeRef} — source memory has status:superseded; superseded memories are not promotable knowledge.`);
1176
- pushSkipReason("promote", op.ref, "promote_superseded");
1177
- return;
1178
- }
1179
- // Parse the source memory up-front so the body/frontmatter checks below
1180
- // share the same parsed view.
1181
828
  const parsedMemory = parseFrontmatter(memoryContent);
1182
- // Reject sources whose body is too small to make useful knowledge.
1183
- // Observed failure: memory files whose body is literally a tags string
1184
- // ("discord,notification,send-notification") get promoted to knowledge
1185
- // proposals that no reviewer would accept. Threshold is conservative —
1186
- // 100 chars catches single-line tag dumps without rejecting genuinely
1187
- // terse but valid notes.
1188
- const PROMOTE_BODY_MIN_CHARS = 100;
1189
829
  const sourceBody = parsedMemory.content.trim();
1190
830
  if (sourceBody.length < PROMOTE_BODY_MIN_CHARS) {
1191
- warnings.push(`Promote: rejected ${op.ref} → ${knowledgeRef} — source memory body is too small (${sourceBody.length} chars; need ≥${PROMOTE_BODY_MIN_CHARS}) to make useful knowledge.`);
1192
- pushSkipReason("promote", op.ref, "promote_source_too_small");
1193
- return;
831
+ return skip("promote_source_too_small", `Promote: rejected ${op.ref} → ${knowledgeRef} — source memory body is too small (${sourceBody.length} chars; need ≥${PROMOTE_BODY_MIN_CHARS}) to make useful knowledge.`);
832
+ }
833
+ // The body is the load-bearing content: twins that differ only in
834
+ // bookkeeping frontmatter, or an earlier run's differently-slugged proposal,
835
+ // are the same promotion.
836
+ const bodyHash = contentHash(memoryContent, "body");
837
+ if (ctx.existingKnowledgeBodyHashes.has(bodyHash)) {
838
+ return skip("dedup_existing_knowledge", `Skipping promote: identical body already exists in knowledge; skipping duplicate for ${op.ref} → ${knowledgeRef}`);
839
+ }
840
+ const pendingConsolidate = listProposals(stashDir, { status: "pending" }).filter((p) => p.source === "consolidate");
841
+ const sameBody = pendingConsolidate.find((p) => contentHash(proposalContent(p), "body") === bodyHash);
842
+ if (sameBody) {
843
+ return skip("dedup_pending_proposal", `Skipping promote: identical body already pending as proposal ${sameBody.id} (ref: ${sameBody.ref}); skipping duplicate for ${op.ref} → ${knowledgeRef}`);
1194
844
  }
1195
- // Cross-run + within-run content dedup: if an identical body already
1196
- // exists in ANY pending consolidate proposal (regardless of target ref),
1197
- // skip. This prevents duplicate proposals when:
1198
- // (a) Multiple source memories have identical bodies but differ only
1199
- // in noise frontmatter (`inferenceProcessed: true` twin alongside
1200
- // the original; differing `updated:` timestamps; etc.) — the body
1201
- // is the load-bearing content, so dedup must hash on body only.
1202
- // (b) A prior run created a proposal for the same body under a
1203
- // different knowledgeRef slug.
1204
- // Use cacheHash (case-preserving stripped body) to match the canonical
1205
- // hash domain used by the body-embedding cache and pending-proposal set.
1206
- const bodyHash = cacheHash(sourceBody);
1207
- if (shouldSkipPromotionBodyDuplicate({ bodyHash, op, knowledgeRef, ctx }))
1208
- return;
1209
845
  try {
1210
- // Use LLM-provided description; fall back to memory's own description
1211
- // (post-sanitization frontmatter is authoritative).
1212
846
  const description = (typeof op.description === "string" && op.description.trim()
1213
847
  ? op.description.trim()
1214
848
  : parsedMemory.data?.description?.trim()) ?? "";
1215
- // Validate the resolved frontmatter before emitting a proposal.
1216
- // Required field: non-empty description. Reject obvious truncation
1217
- // markers (description ends with `,`/`;`/`:`/`...`/hanging connector)
1218
- // so the queue never sees half-formed metadata that the reviewer
1219
- // would only reject.
1220
849
  const fmCheck = validateProposalFrontmatter({ description });
1221
850
  if (!fmCheck.ok) {
1222
- warnings.push(`Promote: rejected ${op.ref} → ${knowledgeRef} — ${fmCheck.reason}.`);
1223
- pushSkipReason("promote", op.ref, "promote_invalid_frontmatter");
1224
- return;
851
+ return skip("promote_invalid_frontmatter", `Promote: rejected ${op.ref} → ${knowledgeRef} — ${fmCheck.reason}.`);
1225
852
  }
1226
- // Merge `description` INTO the body's YAML frontmatter so it lands in
1227
- // the on-disk asset when the proposal is accepted. The descriptionQuality
1228
- // validator parses `payload.content` body (not the envelope
1229
- // `payload.frontmatter`), and a memory's native frontmatter has
1230
- // `captureMode`/`beliefState`/etc. but never `description` — without
1231
- // this merge, 60+ pending proposals were blocked at accept-time with
1232
- // MISSING_FRONTMATTER_DESCRIPTION even though the envelope had it.
1233
- // (The body-frontmatter assumption baked into the 2026-05-20 comment
1234
- // below was wrong: body fm and envelope fm only converge when the
1235
- // writer explicitly merges them, which it now does.)
1236
- const mergedBodyFm = {
853
+ // The description goes into the body frontmatter, which accept-time validation reads.
854
+ const xrefs = Array.isArray(parsedMemory.data?.xrefs) ? parsedMemory.data.xrefs.map(String) : [];
855
+ const mergedFrontmatter = {
1237
856
  ...(parsedMemory.data ?? {}),
1238
857
  description,
1239
- xrefs: promoteProvenanceXrefs(parsedMemory.data?.xrefs, op.ref),
858
+ xrefs: [...new Set([...xrefs, op.ref].map(canonicalXref))],
1240
859
  };
1241
- const serializedMergedFm = serializeFrontmatter(mergedBodyFm);
1242
- const promotedAssetContent = assembleAssetFromString(serializedMergedFm, parsedMemory.content);
1243
- // Pre-emit dedup against pending consolidate proposals from the
1244
- // same improve run (slug-variant match). The cross-run content-hash
1245
- // dedup inside `mergePlans` handles duplicates against existing
1246
- // stash assets — see commit history for the deletion of the
1247
- // unbounded embedding + cross-type slug branches.
1248
- const dedup = await checkPreEmitDedup({
1249
- candidateRef: knowledgeRef,
1250
- candidateText: `${description}. ${memoryContent}`,
1251
- stashDir,
1252
- config,
1253
- });
1254
- if (dedup.duplicate) {
1255
- warnings.push(`Promote: skipped ${op.ref} → ${knowledgeRef} — ${dedup.reason}.`);
1256
- pushSkipReason("promote", op.ref, "promote_dedup_window");
1257
- return;
860
+ const normalized = normalizeSlugForDedup(knowledgeRef);
861
+ const variant = pendingConsolidate.find((p) => normalizeSlugForDedup(p.ref) === normalized);
862
+ if (variant) {
863
+ return skip("promote_dedup_window", `Promote: skipped ${op.ref} → ${knowledgeRef} — slug-variant of pending proposal ${variant.id} (${variant.ref}).`);
1258
864
  }
1259
- const proposalResult = emitProposal({ stashDir, proposalsCtx: ctx.proposalsCtx }, {
865
+ const proposal = mintProposal(stashDir, ctx.proposalsCtx, {
1260
866
  ref: knowledgeRef,
1261
867
  target: { source: target.source.name, root: target.source.path },
1262
868
  source: "consolidate",
1263
- sourceRun,
1264
- // §23.6 fingerprint model-id term (WI-6.4).
1265
- ...(ctx.llmRunner?.connection.model ? { modelId: ctx.llmRunner.connection.model } : {}),
869
+ sourceRun: ctx.sourceRun,
1266
870
  payload: {
1267
- content: promotedAssetContent,
871
+ content: assembleAssetFromString(serializeFrontmatter(mergedFrontmatter), parsedMemory.content),
1268
872
  frontmatter: { description, xrefs: [canonicalXref(op.ref)] },
1269
873
  },
1270
874
  ...(typeof op.confidence === "number" ? { confidence: op.confidence } : {}),
875
+ // The ledger keys the attempt by the source memory.
876
+ attemptedRefs: [op.ref],
877
+ // O1 (alpha.9): on accept, promoteProposal retires this source memory
878
+ // (and its .derived twin) so promotion no longer leaves a duplicate.
879
+ promotionSource: op.ref,
880
+ // B3: recorded so accept can refuse to archive a source that was
881
+ // edited after this promotion was queued — the RAW body hash (see
882
+ // sourceRawBodyHash's own comment above), not `bodyHash`, which is the
883
+ // sanitized-for-knowledge representation accept never re-derives.
884
+ promotionSourceHash: sourceRawBodyHash,
1271
885
  });
1272
- if (isProposalSkipped(proposalResult)) {
1273
- warnings.push(`Promote: skipped proposal for ${op.ref} (${proposalResult.reason}): ${proposalResult.message}`);
1274
- pushSkipReason("promote", op.ref, `promote_proposal_${proposalResult.reason}`);
1275
- }
1276
- else {
1277
- promoted.push(proposalResult.id);
1278
- promotedSourceRefs.add(op.ref);
1279
- }
886
+ ctx.promoted.push(proposal.id);
887
+ ctx.promotedSourceRefs.add(op.ref);
1280
888
  }
1281
889
  catch (e) {
1282
890
  ctx.promotionFailures.count++;
1283
- warnings.push(`Promote: createProposal failed for ${op.ref}: ${String(e)}`);
1284
- pushSkipReason("promote", op.ref, "promote_create_failed");
891
+ skip("promote_create_failed", `Promote: createProposal failed for ${op.ref}: ${String(e)}`);
1285
892
  }
1286
893
  }
1287
- // ── Helpers ─────────────────────────────────────────────────────────────────
1288
- /**
1289
- * Normalise a knowledge slug for variant-aware deduplication. Collapses:
1290
- * - date suffixes (`-may-2026`, `-2026-05-03`, `-2026`)
1291
- * - numeric counter suffixes (`-2`, `-3`)
1292
- * - trailing -patterns / -2026-05-03 styles
1293
- * - word reorderings via alphabetical sort of the remaining tokens.
1294
- *
1295
- * Two slugs that normalise to the same string are considered the same asset
1296
- * for dedup purposes even if they don't share an exact ref.
1297
- */
1298
- /** The conceptId a proposal ref maps to, or undefined for an invalid ref. */
1299
- function conceptIdForRef(ref) {
1300
- try {
1301
- const p = parseRefInput(ref);
1302
- return conceptIdFromTypeName(p.type, p.name);
1303
- }
1304
- catch {
1305
- return undefined;
1306
- }
1307
- }
1308
- /** Is a pending proposal already queued for `conceptRef`'s concept? */
1309
- function hasPendingProposalForConcept(stashDir, conceptRef) {
1310
- const want = conceptIdForRef(conceptRef);
1311
- return (want !== undefined && listProposals(stashDir, { status: "pending" }).some((p) => conceptIdForRef(p.ref) === want));
1312
- }
1313
- function normalizeSlugForDedup(ref) {
1314
- const slug = parseRefInput(ref).name;
1315
- const monthRe = /(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)/i;
1316
- const tokens = slug
1317
- .toLowerCase()
1318
- .split("-")
1319
- .filter((tok) => tok.length > 0)
1320
- // Strip purely-numeric tokens (years, dates, counter suffixes like -2 / -3).
1321
- // Numbers carry no semantic information for our dedup purposes — every
1322
- // observed defective slug variant differs only in dates or counters.
1323
- .filter((tok) => !/^\d+$/.test(tok))
1324
- .filter((tok) => !monthRe.test(tok));
1325
- // Sort to absorb word reorderings.
1326
- tokens.sort();
1327
- return tokens.join("-");
1328
- }
1329
- /**
1330
- * Pre-emit dedup check: compare the candidate ref against pending consolidate
1331
- * proposals only. Returns a reason string if a slug-variant match is found,
1332
- * else null.
1333
- *
1334
- * Historical context (REMOVED 2026-05-20): this function previously also ran
1335
- * (a) a normalised-slug match against existing knowledge AND memory entries
1336
- * in the DB, and
1337
- * (b) an embedding cosine-similarity check (>= 0.85) against ALL knowledge
1338
- * and non-derived memory entries.
1339
- * Both branches had ZERO observed fires across 30 sampled runs in the
1340
- * post-fix window. The 29 actual dedup catches all came from the SEPARATE
1341
- * content-hash dedup inside `mergePlans` (the older SHA-256 helper). The
1342
- * embedding branch in particular had unbounded cost per promote (embedded
1343
- * every knowledge + non-derived memory entry, every time) with no observed
1344
- * benefit. Empirical signal → deleted.
1345
- *
1346
- * What remains: a check against pending consolidate proposals in the SAME
1347
- * improve run. This catches duplicates queued back-to-back within a single
1348
- * improve invocation — a different concern from the cross-run content-hash
1349
- * dedup, and cheap (no embeddings, no DB query).
1350
- */
1351
- async function checkPreEmitDedup(opts) {
1352
- const normCandidate = normalizeSlugForDedup(opts.candidateRef);
1353
- // Pending consolidate proposals (slug match) — within the same improve run.
1354
- const pendingConsolidate = listProposals(opts.stashDir, { status: "pending" }).filter((p) => p.source === "consolidate");
1355
- for (const p of pendingConsolidate) {
1356
- if (normalizeSlugForDedup(p.ref) === normCandidate) {
1357
- return { duplicate: true, reason: `slug-variant of pending proposal ${p.id} (${p.ref})` };
1358
- }
1359
- }
1360
- return { duplicate: false };
1361
- }
1362
- /**
1363
- * Incremental candidate set: {changed} ∪ {top-k persisted-vector neighbours of
1364
- * each changed memory}, intersected with the loaded pool. Returns [] when
1365
- * nothing changed (caller emits a no-op envelope), the full pool when
1366
- * everything changed or the index can't answer (fail-open to preserve merge
1367
- * correctness). `since` is an ISO timestamp.
1368
- */
1369
- export function narrowToIncrementalCandidates(memories, since, warnings, neighborsPerChanged = 5, readOnly = false) {
1370
- // Lenient by design: garbage `since` passes through unchanged and the ISO
1371
- // string comparison below then selects nothing (see core/time.ts doc).
1372
- const sinceIso = parseSinceToIsoLenient(since);
1373
- const isChanged = (m) => {
1374
- try {
1375
- return fs.statSync(m.filePath).mtime.toISOString() > sinceIso;
1376
- }
1377
- catch {
1378
- return true; // never silently drop a memory we cannot stat
1379
- }
1380
- };
1381
- const changed = memories.filter(isChanged);
1382
- if (changed.length === 0)
1383
- return [];
1384
- if (changed.length === memories.length)
1385
- return memories;
1386
- const byName = new Map(memories.map((m) => [m.name, m]));
1387
- const keep = new Set(changed.map((m) => m.name));
1388
- let db;
1389
- try {
1390
- db = readOnly ? openReadonlyExistingDatabase(undefined, { isolatedSnapshot: true }) : openExistingDatabase();
1391
- if (!db)
1392
- return memories;
1393
- for (const m of changed) {
1394
- const id = findEntryIdByRef(db, conceptIdFromTypeName("memory", m.name));
1395
- if (id === undefined)
1396
- continue;
1397
- for (const hit of getNeighborsByEntryId(db, id, neighborsPerChanged + 1)) {
1398
- if (hit.id === id)
1399
- continue;
1400
- const entry = getEntryById(db, hit.id);
1401
- if (!entry)
1402
- continue;
1403
- const name = entry.entry.name;
1404
- if (byName.has(name))
1405
- keep.add(name); // only neighbours present in the loaded pool
1406
- }
1407
- }
1408
- }
1409
- catch {
1410
- warnings.push("Incremental consolidation: index unavailable — processing full pool.");
1411
- return memories;
1412
- }
1413
- finally {
1414
- if (db)
1415
- closeDatabase(db);
1416
- }
1417
- const candidates = memories.filter((m) => keep.has(m.name));
1418
- warnings.push(`Incremental consolidation: ${changed.length} changed + neighbours → ${candidates.length}/${memories.length} memories considered (since ${since}${sinceIso !== since ? ` = ${sinceIso}` : ""}).`);
1419
- return candidates;
1420
- }
894
+ /** The target bundle's eligible memories from the index, else walked from disk. */
1421
895
  function loadMemoriesForSource(source, warnings, readOnly) {
1422
- // Load from DB first
1423
896
  let memories = [];
1424
897
  let db;
1425
898
  try {
1426
899
  db = readOnly ? openReadonlyExistingDatabase(undefined, { isolatedSnapshot: true }) : openExistingDatabase();
1427
900
  if (!db)
1428
901
  throw new Error("index unavailable");
1429
- const entries = getAllEntries(db, "memory");
1430
- memories = entries
1431
- .filter((entry) => source !== undefined && entry.bundleId === source.bundleId)
1432
- .filter((e) => isConsolidationEligibleMemoryName(e.entry.name))
1433
- // Skip stale DB entries whose file was deleted by a prior run but not yet
1434
- // re-indexed. Without this guard the deleted file's ref appears in chunks
1435
- // sent to the LLM, which then proposes a second delete → delete_failed
1436
- // because the file is already gone. Re-indexing runs on a cron cadence so
1437
- // several successful deletes can accumulate before the DB catches up.
1438
- .filter((e) => fs.existsSync(e.filePath))
902
+ memories = getAllEntries(db, "memory")
903
+ .filter((e) => source !== undefined && e.bundleId === source.bundleId)
904
+ .filter((e) => isConsolidationEligibleMemoryName(e.entry.name) && fs.existsSync(e.filePath))
1439
905
  .map((e) => ({
1440
906
  name: e.entry.name,
1441
907
  filePath: e.filePath,
@@ -1451,34 +917,30 @@ function loadMemoriesForSource(source, warnings, readOnly) {
1451
917
  if (db)
1452
918
  closeDatabase(db);
1453
919
  }
1454
- if (memories.length === 0 && source) {
1455
- // DB fallback: walk filesystem
1456
- const memoriesDir = path.join(source.sourceRoot, "memories");
1457
- const fsStashDir = source.sourceRoot;
1458
- if (fs.existsSync(memoriesDir)) {
1459
- const pending = [memoriesDir];
1460
- while (pending.length > 0) {
1461
- const current = pending.pop();
1462
- for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
1463
- const filePath = path.join(current, entry.name);
1464
- if (entry.isDirectory()) {
1465
- if (source.excludedSourceRoots.has(path.resolve(filePath)))
1466
- continue;
920
+ if (memories.length > 0 || !source)
921
+ return memories;
922
+ const memoriesDir = path.join(source.sourceRoot, "memories");
923
+ if (fs.existsSync(memoriesDir)) {
924
+ const pending = [memoriesDir];
925
+ while (pending.length > 0) {
926
+ const current = pending.pop();
927
+ for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
928
+ const filePath = path.join(current, entry.name);
929
+ if (entry.isDirectory()) {
930
+ if (!source.excludedSourceRoots.has(path.resolve(filePath)))
1467
931
  pending.push(filePath);
1468
- continue;
1469
- }
1470
- if (!entry.isFile() || !entry.name.endsWith(".md"))
1471
- continue;
1472
- const name = path.relative(memoriesDir, filePath).replace(/\.md$/, "").split(path.sep).join("/");
1473
- if (!isConsolidationEligibleMemoryName(name))
1474
- continue;
1475
- memories.push({ name, filePath, description: "", tags: [], stashDir: fsStashDir });
932
+ continue;
933
+ }
934
+ if (!entry.isFile() || !entry.name.endsWith(".md"))
935
+ continue;
936
+ const name = path.relative(memoriesDir, filePath).replace(/\.md$/, "").split(path.sep).join("/");
937
+ if (isConsolidationEligibleMemoryName(name)) {
938
+ memories.push({ name, filePath, description: "", tags: [], stashDir: source.sourceRoot });
1476
939
  }
1477
940
  }
1478
941
  }
1479
- if (memories.length > 0) {
1480
- warnings.push("DB not found or empty — loaded memories directly from filesystem.");
1481
- }
1482
942
  }
943
+ if (memories.length > 0)
944
+ warnings.push("DB not found or empty — loaded memories directly from filesystem.");
1483
945
  return memories;
1484
946
  }